mnemonad-cli 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +113 -9
- package/bin/mnemonad.js +111 -5
- package/lib/bridgeCodec.js +12 -0
- package/lib/chainClient.js +21 -7
- package/lib/commands/buildIndex.js +171 -0
- package/lib/commands/burn.js +1 -1
- package/lib/commands/compact.js +1 -1
- package/lib/commands/diff.js +1 -1
- package/lib/commands/index.js +2 -0
- package/lib/commands/info.js +1 -1
- package/lib/commands/pull.js +1 -1
- package/lib/commands/push.js +44 -1
- package/lib/commands/search.js +91 -0
- package/lib/commands/shared.js +3 -2
- package/lib/commands/watch.js +58 -9
- package/lib/passkeyBridge.js +203 -0
- package/lib/search/Indexer.js +204 -0
- package/lib/search/Searcher.js +68 -0
- package/lib/search/VectorIndex.js +225 -0
- package/lib/search/browser/BrowserSearcher.js +131 -0
- package/lib/search/browser/WasmIndexReader.js +93 -0
- package/lib/search/browser.js +13 -0
- package/lib/search/chunking/ChunkingStrategy.js +23 -0
- package/lib/search/chunking/TextWindowChunkingStrategy.js +69 -0
- package/lib/search/embeddings/EmbeddingProvider.js +40 -0
- package/lib/search/embeddings/StaticEmbeddingProvider.js +180 -0
- package/lib/search/embeddings/modelFiles.browser.js +28 -0
- package/lib/search/embeddings/modelFiles.node.js +41 -0
- package/lib/search/embeddings/modelFiles.shared.js +44 -0
- package/lib/search/index.js +18 -0
- package/lib/search/providers.js +153 -0
- package/lib/search/schema.js +103 -0
- package/mnemonad.config.js +19 -5
- package/package.json +13 -3
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
import { randomBytes } from 'node:crypto';
|
|
2
|
+
import { createServer } from 'node:http';
|
|
3
|
+
import { WebSocketServer } from 'ws';
|
|
4
|
+
import open from 'open';
|
|
5
|
+
import { toAccount } from 'viem/accounts';
|
|
6
|
+
import { userError } from './commands/shared.js';
|
|
7
|
+
import { encodeForBridge } from './bridgeCodec.js';
|
|
8
|
+
|
|
9
|
+
const CONNECT_TIMEOUT_MS = 3 * 60 * 1000;
|
|
10
|
+
|
|
11
|
+
// One bridge per CLI process — `--passkey` resolves at most one signing account per
|
|
12
|
+
// invocation (see resolveAccount in chainClient.js), so there's nothing to key this by.
|
|
13
|
+
let active = null;
|
|
14
|
+
|
|
15
|
+
/** A one-line "what's this for" the browser tab shows while it waits — just the command
|
|
16
|
+
* shape, not the flags, so it stays short. */
|
|
17
|
+
function describeCommand(args) {
|
|
18
|
+
const parts = ['mnemonad', args.command];
|
|
19
|
+
if (args.streamId) parts.push(args.streamId);
|
|
20
|
+
if (args.path) parts.push(args.path);
|
|
21
|
+
return parts.join(' ');
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* The CLI half of `--passkey`: opens the explorer's `/cli-auth` page in the system browser
|
|
26
|
+
* to run the WebAuthn ceremony there (this process has no `navigator.credentials` — passkeys
|
|
27
|
+
* only exist in a browser), then relays every signing request to that tab for the rest of
|
|
28
|
+
* this command. The raw key never enters this process, only signatures do — see
|
|
29
|
+
* docs/wallet/passkey-accounts.md's CLI section for why it has to be that domain specifically
|
|
30
|
+
* (rp.id) and why this is a one-shot-per-command bridge rather than something that persists
|
|
31
|
+
* across separate CLI invocations.
|
|
32
|
+
*
|
|
33
|
+
* @param {Object} args - parsed CLI args: `authOrigin` (the explorer deployment to open, e.g.
|
|
34
|
+
* https://mnemonad.vercel.app), plus `command`/`streamId`/`path` to build the one-line
|
|
35
|
+
* description the page shows while it waits.
|
|
36
|
+
* @returns {Promise<Object>} a viem Account whose signMessage/signTransaction proxy to the tab
|
|
37
|
+
*/
|
|
38
|
+
export async function openPasskeyBridge(args) {
|
|
39
|
+
if (active) return active.account;
|
|
40
|
+
|
|
41
|
+
const token = randomBytes(24).toString('hex');
|
|
42
|
+
// Read by /ping below — the one thing the page can't get from the WS (it doesn't exist
|
|
43
|
+
// yet at that point, see docs/wallet/passkey-accounts.md's CLI section) but still wants
|
|
44
|
+
// live: how much of the connect-timeout is actually left, and confirmation the CLI is
|
|
45
|
+
// still there at all.
|
|
46
|
+
const deadlineAt = Date.now() + CONNECT_TIMEOUT_MS;
|
|
47
|
+
const httpServer = createServer((req, res) => {
|
|
48
|
+
const url = new URL(req.url, 'http://127.0.0.1');
|
|
49
|
+
if (url.pathname === '/ping' && url.searchParams.get('token') === token) {
|
|
50
|
+
res.writeHead(200, {
|
|
51
|
+
'Content-Type': 'application/json',
|
|
52
|
+
// The page is served from the explorer's own origin, not this loopback one —
|
|
53
|
+
// an ordinary fetch() response is otherwise unreadable cross-origin. The token
|
|
54
|
+
// above is what actually gates this, same as the WS connection itself; the
|
|
55
|
+
// wildcard just matches that (no meaningful "origin" to restrict to when the
|
|
56
|
+
// caller could be any explorer deployment this token was ever handed to).
|
|
57
|
+
'Access-Control-Allow-Origin': '*',
|
|
58
|
+
});
|
|
59
|
+
res.end(JSON.stringify({ ttl: Math.max(0, Math.round((deadlineAt - Date.now()) / 1000)) }));
|
|
60
|
+
return;
|
|
61
|
+
}
|
|
62
|
+
res.writeHead(404);
|
|
63
|
+
res.end();
|
|
64
|
+
});
|
|
65
|
+
const wss = new WebSocketServer({ server: httpServer });
|
|
66
|
+
|
|
67
|
+
await new Promise((resolve, reject) => {
|
|
68
|
+
httpServer.once('error', reject);
|
|
69
|
+
httpServer.listen(0, '127.0.0.1', resolve);
|
|
70
|
+
});
|
|
71
|
+
const port = httpServer.address().port;
|
|
72
|
+
|
|
73
|
+
const pending = new Map();
|
|
74
|
+
let nextId = 1;
|
|
75
|
+
let socket = null;
|
|
76
|
+
|
|
77
|
+
const connected = new Promise((resolve, reject) => {
|
|
78
|
+
const timer = setTimeout(() => {
|
|
79
|
+
// Nothing ever reached the point (below) that assigns `active`, so
|
|
80
|
+
// closePasskeyBridge() has nothing to close on this path — without this, the
|
|
81
|
+
// server/socket this function opened above would sit open forever, and the CLI
|
|
82
|
+
// process would hang right after printing the error below rather than exiting.
|
|
83
|
+
wss.close();
|
|
84
|
+
httpServer.close();
|
|
85
|
+
reject(userError(
|
|
86
|
+
'Timed out waiting for the browser passkey sign-in. Complete the prompt in the\n' +
|
|
87
|
+
' tab that opened, or re-run with --passkey to try again.'
|
|
88
|
+
));
|
|
89
|
+
}, CONNECT_TIMEOUT_MS);
|
|
90
|
+
|
|
91
|
+
wss.on('connection', (ws, req) => {
|
|
92
|
+
const url = new URL(req.url, 'http://127.0.0.1');
|
|
93
|
+
if (url.searchParams.get('token') !== token) {
|
|
94
|
+
ws.close(4001, 'bad token');
|
|
95
|
+
return;
|
|
96
|
+
}
|
|
97
|
+
socket = ws;
|
|
98
|
+
|
|
99
|
+
ws.on('message', (raw) => {
|
|
100
|
+
let msg;
|
|
101
|
+
try {
|
|
102
|
+
msg = JSON.parse(raw.toString());
|
|
103
|
+
} catch {
|
|
104
|
+
return;
|
|
105
|
+
}
|
|
106
|
+
if (msg.type === 'hello') {
|
|
107
|
+
clearTimeout(timer);
|
|
108
|
+
resolve(msg.address);
|
|
109
|
+
} else if (msg.type === 'sign-result' || msg.type === 'sign-error') {
|
|
110
|
+
const p = pending.get(msg.id);
|
|
111
|
+
if (!p) return;
|
|
112
|
+
pending.delete(msg.id);
|
|
113
|
+
if (msg.type === 'sign-result') p.resolve(msg.result);
|
|
114
|
+
else p.reject(new Error(msg.error || 'signing failed'));
|
|
115
|
+
}
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
ws.on('close', () => {
|
|
119
|
+
if (socket === ws) socket = null;
|
|
120
|
+
for (const p of pending.values()) p.reject(userError('Browser tab disconnected before signing finished.'));
|
|
121
|
+
pending.clear();
|
|
122
|
+
});
|
|
123
|
+
});
|
|
124
|
+
});
|
|
125
|
+
|
|
126
|
+
const desc = describeCommand(args);
|
|
127
|
+
const ttl = Math.floor(CONNECT_TIMEOUT_MS / 1000);
|
|
128
|
+
// Lets the result screen turn a stream id in its summary line into a link to the
|
|
129
|
+
// explorer's own stream page — only for a real deployment id (testnet/mainnet); --chain
|
|
130
|
+
// local has no explorer route to link to, so the page just falls back to plain text.
|
|
131
|
+
const network = args.chain === 'testnet' || args.chain === 'mainnet' ? `&network=${args.chain}` : '';
|
|
132
|
+
const authUrl = `${args.authOrigin.replace(/\/$/, '')}/cli-auth?port=${port}&token=${token}&desc=${encodeURIComponent(desc)}&ttl=${ttl}${network}`;
|
|
133
|
+
console.log(' opening', authUrl, 'to sign in with a passkey...');
|
|
134
|
+
try {
|
|
135
|
+
await open(authUrl);
|
|
136
|
+
} catch {
|
|
137
|
+
console.log(' could not open a browser automatically — open this URL yourself:', authUrl);
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
const address = await connected;
|
|
141
|
+
console.log(' connected:', address);
|
|
142
|
+
|
|
143
|
+
function request(type, payload) {
|
|
144
|
+
if (!socket) {
|
|
145
|
+
return Promise.reject(userError('Browser tab disconnected. Re-run with --passkey to reconnect.'));
|
|
146
|
+
}
|
|
147
|
+
return new Promise((resolve, reject) => {
|
|
148
|
+
const id = nextId++;
|
|
149
|
+
pending.set(id, { resolve, reject });
|
|
150
|
+
socket.send(JSON.stringify({ type, id, ...payload }));
|
|
151
|
+
});
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
const account = toAccount({
|
|
155
|
+
address,
|
|
156
|
+
async signMessage({ message }) {
|
|
157
|
+
return request('sign-message', { message });
|
|
158
|
+
},
|
|
159
|
+
async signTransaction(transaction) {
|
|
160
|
+
return request('sign-transaction', { transaction: encodeForBridge(transaction) });
|
|
161
|
+
},
|
|
162
|
+
async signTypedData() {
|
|
163
|
+
// Not used anywhere in this codebase (personal_sign only — see the "Wallet-signature
|
|
164
|
+
// encryption keys" plan's locked-in decision) — a real relay implementation would
|
|
165
|
+
// need the same BigInt-safe encoding sign-transaction already has, not worth building
|
|
166
|
+
// for a path nothing calls.
|
|
167
|
+
throw userError('Signing typed data is not supported over --passkey.');
|
|
168
|
+
},
|
|
169
|
+
});
|
|
170
|
+
|
|
171
|
+
active = { wss, httpServer, account };
|
|
172
|
+
return account;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/**
|
|
176
|
+
* Called once, at the very end of the CLI's command dispatch (bin/mnemonad.js), whether the
|
|
177
|
+
* command succeeded or failed — a no-op when --passkey was never used. Tells the browser tab
|
|
178
|
+
* it's done so it can end its session and zero the key, mirroring the explorer's own
|
|
179
|
+
* disconnect() lifecycle instead of just abandoning the connection.
|
|
180
|
+
*
|
|
181
|
+
* `result` lets the tab show something more useful than a bare "done" before it closes —
|
|
182
|
+
* whether the command actually succeeded, and the one line of output that says what
|
|
183
|
+
* happened (bin/mnemonad.js captures its own last console.log line for this rather than
|
|
184
|
+
* every command needing to return a summary explicitly).
|
|
185
|
+
*
|
|
186
|
+
* @param {{ok: boolean, summary: string}} [result]
|
|
187
|
+
*/
|
|
188
|
+
export function closePasskeyBridge(result) {
|
|
189
|
+
if (!active) return;
|
|
190
|
+
const { wss, httpServer } = active;
|
|
191
|
+
active = null;
|
|
192
|
+
const payload = JSON.stringify({ type: 'done', ok: result?.ok ?? true, summary: result?.summary || '' });
|
|
193
|
+
for (const ws of wss.clients) {
|
|
194
|
+
try {
|
|
195
|
+
ws.send(payload);
|
|
196
|
+
} catch {
|
|
197
|
+
// Nothing to tell a socket that's already gone.
|
|
198
|
+
}
|
|
199
|
+
ws.close();
|
|
200
|
+
}
|
|
201
|
+
wss.close();
|
|
202
|
+
httpServer.close();
|
|
203
|
+
}
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
import { existsSync, rmSync } from 'node:fs';
|
|
2
|
+
import VectorIndex from './VectorIndex.js';
|
|
3
|
+
import TextWindowChunkingStrategy from './chunking/TextWindowChunkingStrategy.js';
|
|
4
|
+
|
|
5
|
+
/** First N bytes checked for a null byte — the standard, cheap heuristic (same one git and
|
|
6
|
+
* most diff tools use) for "this is almost certainly binary, don't try to embed it as text". */
|
|
7
|
+
const BINARY_SNIFF_BYTES = 8000;
|
|
8
|
+
|
|
9
|
+
function looksLikeText(bytes) {
|
|
10
|
+
const n = Math.min(bytes.length, BINARY_SNIFF_BYTES);
|
|
11
|
+
for (let i = 0; i < n; i++) {
|
|
12
|
+
if (bytes[i] === 0) return false;
|
|
13
|
+
}
|
|
14
|
+
return true;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
function toHex(bytes) {
|
|
18
|
+
return Array.from(bytes, (b) => b.toString(16).padStart(2, '0')).join('');
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Builds/updates a `VectorIndex` from a folder — anything shaped like
|
|
23
|
+
* `@fizzyflow/doublesync`'s `DoubleSyncFolder` (has an async `walk()` generator), which
|
|
24
|
+
* covers the CLI's real-disk `FSFolder` and any in-memory folder the same way. Incremental by
|
|
25
|
+
* construction: a file whose content hash already matches what's stored is skipped entirely,
|
|
26
|
+
* no re-embedding, no re-chunking, not even a read past the hash check.
|
|
27
|
+
*/
|
|
28
|
+
export default class Indexer {
|
|
29
|
+
/**
|
|
30
|
+
* @param {Object} [params]
|
|
31
|
+
* @param {?import('./embeddings/EmbeddingProvider.js').default} [params.embeddingProvider] -
|
|
32
|
+
* required by `index()`; `status()` never embeds, so it can go without one (see
|
|
33
|
+
* providers.js's createProvider for the usual way to get one).
|
|
34
|
+
* @param {import('./chunking/ChunkingStrategy.js').default} [params.chunkingStrategy]
|
|
35
|
+
*/
|
|
36
|
+
constructor({ embeddingProvider = null, chunkingStrategy } = {}) {
|
|
37
|
+
this.embeddingProvider = embeddingProvider;
|
|
38
|
+
this.chunkingStrategy = chunkingStrategy || new TextWindowChunkingStrategy();
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* What `index()` would do to `dbPath`, without doing it: no embedding, no model load, and
|
|
43
|
+
* the database opened read-only. Uses the same walk and the same rules as `index()` (see
|
|
44
|
+
* `_scan`), so it agrees exactly on which files count — a binary or empty file, which
|
|
45
|
+
* `index()` never records, isn't reported as missing from the index.
|
|
46
|
+
*
|
|
47
|
+
* @param {import('@fizzyflow/doublesync').DoubleSyncFolder} folder
|
|
48
|
+
* @param {string} dbPath
|
|
49
|
+
* @returns {Promise<?{changed: string[], added: string[], removed: string[]}>} null when
|
|
50
|
+
* there's no index at `dbPath` at all.
|
|
51
|
+
*/
|
|
52
|
+
async status(folder, dbPath) {
|
|
53
|
+
if (!existsSync(dbPath)) return null;
|
|
54
|
+
const index = new VectorIndex(dbPath, { readonly: true });
|
|
55
|
+
try {
|
|
56
|
+
const indexed = new Set(index.listFiles());
|
|
57
|
+
const changed = [];
|
|
58
|
+
const added = [];
|
|
59
|
+
const seen = new Set();
|
|
60
|
+
for await (const entry of this._scan(folder, index)) {
|
|
61
|
+
// An emptied file is left out of `seen`, so it comes out as removed below —
|
|
62
|
+
// exactly what index() would do to it.
|
|
63
|
+
if (!entry.unchanged && !entry.chunks) continue;
|
|
64
|
+
seen.add(entry.relPath);
|
|
65
|
+
if (entry.unchanged) continue;
|
|
66
|
+
(indexed.has(entry.relPath) ? changed : added).push(entry.relPath);
|
|
67
|
+
}
|
|
68
|
+
const removed = [...indexed].filter((p) => !seen.has(p));
|
|
69
|
+
return { changed, added, removed };
|
|
70
|
+
} finally {
|
|
71
|
+
index.close();
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* The walk both `index()` and `status()` run. Yields one entry per text file:
|
|
77
|
+
* `{relPath, unchanged: true}` when the index already has it at this content hash, or
|
|
78
|
+
* `{relPath, hash, chunks}` otherwise (`chunks` null for a file with no text to embed).
|
|
79
|
+
*/
|
|
80
|
+
async *_scan(folder, index) {
|
|
81
|
+
for await (const { path, file } of folder.walk()) {
|
|
82
|
+
const relPath = path.join('/');
|
|
83
|
+
const bytes = await file.getContent();
|
|
84
|
+
if (!looksLikeText(bytes)) continue;
|
|
85
|
+
|
|
86
|
+
const hash = toHex(await file.getFingerprint());
|
|
87
|
+
if (index.getFileHash(relPath) === hash) {
|
|
88
|
+
yield { relPath, unchanged: true };
|
|
89
|
+
continue;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
const text = new TextDecoder('utf-8', { fatal: false }).decode(bytes);
|
|
93
|
+
const chunks = this.chunkingStrategy.chunk(text);
|
|
94
|
+
yield { relPath, hash, chunks: chunks.length > 0 ? chunks : null };
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* @param {import('@fizzyflow/doublesync').DoubleSyncFolder} folder
|
|
100
|
+
* @param {string} dbPath
|
|
101
|
+
* @param {Object} [opts]
|
|
102
|
+
* @param {boolean} [opts.rebuild] - start from an empty index instead of updating this one:
|
|
103
|
+
* every file is embedded again. The one way to switch an index to a different model — and,
|
|
104
|
+
* since it rewrites the whole file, the next push carries the full index again.
|
|
105
|
+
* @returns {Promise<{filesChanged: number, filesSkipped: number, filesRemoved: number, chunksAdded: number}>}
|
|
106
|
+
*/
|
|
107
|
+
async index(folder, dbPath, { rebuild = false } = {}) {
|
|
108
|
+
if (!this.embeddingProvider) throw new Error('Indexer.index: no embeddingProvider configured');
|
|
109
|
+
if (rebuild) {
|
|
110
|
+
rmSync(dbPath, { force: true });
|
|
111
|
+
rmSync(`${dbPath}-journal`, { force: true });
|
|
112
|
+
} else {
|
|
113
|
+
assertSameModel(VectorIndex.readMeta(dbPath), this.embeddingProvider, dbPath);
|
|
114
|
+
}
|
|
115
|
+
const index = new VectorIndex(dbPath);
|
|
116
|
+
try {
|
|
117
|
+
const seenPaths = new Set();
|
|
118
|
+
// One entry per changed/new file, chunked but not yet embedded — embedding happens
|
|
119
|
+
// in one batched call across every changed file below, rather than one call per
|
|
120
|
+
// file, since a single call amortizes the model's own fixed per-call overhead.
|
|
121
|
+
const pending = [];
|
|
122
|
+
let filesSkipped = 0;
|
|
123
|
+
|
|
124
|
+
for await (const entry of this._scan(folder, index)) {
|
|
125
|
+
// A file with no text to embed (emptied since last time) is deliberately left out
|
|
126
|
+
// of seenPaths, so the removal pass below drops its old chunks like a deleted file's.
|
|
127
|
+
if (entry.unchanged) {
|
|
128
|
+
seenPaths.add(entry.relPath);
|
|
129
|
+
filesSkipped++;
|
|
130
|
+
} else if (entry.chunks) {
|
|
131
|
+
seenPaths.add(entry.relPath);
|
|
132
|
+
pending.push(entry);
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
let chunksAdded = 0;
|
|
137
|
+
if (pending.length > 0) {
|
|
138
|
+
const allTexts = pending.flatMap((p) => p.chunks.map((c) => c.text));
|
|
139
|
+
const allVectors = await this.embeddingProvider.embed(allTexts);
|
|
140
|
+
index.initVectorColumn(this.embeddingProvider.dimension);
|
|
141
|
+
|
|
142
|
+
let cursor = 0;
|
|
143
|
+
for (const p of pending) {
|
|
144
|
+
const vectors = allVectors.slice(cursor, cursor + p.chunks.length);
|
|
145
|
+
cursor += p.chunks.length;
|
|
146
|
+
index.replaceFileChunks(p.relPath, p.hash, p.chunks.map((c, i) => ({
|
|
147
|
+
startOffset: c.startOffset,
|
|
148
|
+
endOffset: c.endOffset,
|
|
149
|
+
embedding: vectors[i],
|
|
150
|
+
})));
|
|
151
|
+
chunksAdded += p.chunks.length;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
index.setMeta('model', this.embeddingProvider.modelId);
|
|
155
|
+
if (this.embeddingProvider.embedder) index.setMeta('embedder', this.embeddingProvider.embedder);
|
|
156
|
+
index.setMeta('dimension', String(this.embeddingProvider.dimension));
|
|
157
|
+
if (this.embeddingProvider.dtype) index.setMeta('dtype', this.embeddingProvider.dtype);
|
|
158
|
+
index.setMeta('chunk_size', String(this.chunkingStrategy.size ?? ''));
|
|
159
|
+
index.setMeta('chunk_overlap', String(this.chunkingStrategy.overlap ?? ''));
|
|
160
|
+
} else {
|
|
161
|
+
// Nothing changed this run — if the index already has content from a previous
|
|
162
|
+
// run, the vector column still needs registering on this fresh connection (see
|
|
163
|
+
// VectorIndex.initVectorColumn's own doc) before removeFile()'s bookkeeping below
|
|
164
|
+
// touches the same tables. Not required for a genuinely empty/new index — there's
|
|
165
|
+
// no dimension to know yet, and nothing to query either.
|
|
166
|
+
const existingDim = index.getMeta('dimension');
|
|
167
|
+
if (existingDim) index.initVectorColumn(Number(existingDim));
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
let filesRemoved = 0;
|
|
171
|
+
for (const indexedPath of index.listFiles()) {
|
|
172
|
+
if (!seenPaths.has(indexedPath)) {
|
|
173
|
+
index.removeFile(indexedPath);
|
|
174
|
+
filesRemoved++;
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
return { filesChanged: pending.length, filesSkipped, filesRemoved, chunksAdded };
|
|
179
|
+
} finally {
|
|
180
|
+
index.close();
|
|
181
|
+
}
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* Thrown when `index()` would add one model's vectors to an index built with another — two
|
|
187
|
+
* models' vectors aren't comparable (and usually aren't even the same length), so mixing them
|
|
188
|
+
* would quietly break every search. `rebuild` is the way through.
|
|
189
|
+
*/
|
|
190
|
+
export class ModelMismatchError extends Error {
|
|
191
|
+
constructor(dbPath, indexModel, providerModel) {
|
|
192
|
+
super(
|
|
193
|
+
`${dbPath} was built with ${indexModel}, not ${providerModel} — one index can't mix two models.`
|
|
194
|
+
);
|
|
195
|
+
this.name = 'ModelMismatchError';
|
|
196
|
+
this.indexModel = indexModel;
|
|
197
|
+
this.providerModel = providerModel;
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
function assertSameModel(meta, provider, dbPath) {
|
|
202
|
+
if (!meta?.model) return;
|
|
203
|
+
if (meta.model !== provider.modelId) throw new ModelMismatchError(dbPath, meta.model, provider.modelId);
|
|
204
|
+
}
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import VectorIndex from './VectorIndex.js';
|
|
2
|
+
import { LEGACY_DTYPE, EMBEDDER_TRANSFORMERS, embedderOfIndex } from './schema.js';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Queries an already-built `VectorIndex`. Deliberately the only piece that needs an embedding
|
|
6
|
+
* model loaded on the *searcher's* machine — and only to embed the query string itself, never
|
|
7
|
+
* the corpus, which is the entire point of pushing the index alongside the content: whoever
|
|
8
|
+
* pulls it gets every embedding for free, already paid for by whoever ran `Indexer` first.
|
|
9
|
+
*/
|
|
10
|
+
export default class Searcher {
|
|
11
|
+
/**
|
|
12
|
+
* @param {Object} params
|
|
13
|
+
* @param {(opts: {model: string, embedder: string, dtype: ?string}) => Promise<any>} [params.createProvider] -
|
|
14
|
+
* builds the provider the index's own metadata asks for (see providers.js) — the common
|
|
15
|
+
* case, opening someone else's index.
|
|
16
|
+
* @param {any} [params.embeddingProvider] - use this one instead, whatever the index recorded
|
|
17
|
+
* (tests; a custom provider). Its vectors must still match the index's dimension.
|
|
18
|
+
*/
|
|
19
|
+
constructor({ createProvider = null, embeddingProvider = null } = {}) {
|
|
20
|
+
if (!createProvider && !embeddingProvider) {
|
|
21
|
+
throw new Error('Searcher: pass createProvider or embeddingProvider');
|
|
22
|
+
}
|
|
23
|
+
this._createProvider = createProvider;
|
|
24
|
+
this._embeddingProvider = embeddingProvider;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* @param {string} dbPath
|
|
29
|
+
* @param {string} query
|
|
30
|
+
* @param {number} [k=5]
|
|
31
|
+
* @returns {Promise<{path: string, chunkIndex: number, startOffset: number, endOffset: number, distance: number}[]>}
|
|
32
|
+
*/
|
|
33
|
+
async search(dbPath, query, k = 5) {
|
|
34
|
+
// Read-only: the index is a file that gets pushed, and searching it must not change it.
|
|
35
|
+
const index = new VectorIndex(dbPath, { readonly: true });
|
|
36
|
+
try {
|
|
37
|
+
const dim = Number(index.getMeta('dimension'));
|
|
38
|
+
if (!dim) {
|
|
39
|
+
throw new Error(`${dbPath}: no index metadata found — run \`mnemonad index\` first`);
|
|
40
|
+
}
|
|
41
|
+
index.initVectorColumn(dim);
|
|
42
|
+
|
|
43
|
+
const provider = this._embeddingProvider || await this._createProvider(describeIndexModel(index));
|
|
44
|
+
const [queryVector] = await provider.embed([query]);
|
|
45
|
+
if (queryVector.length !== dim) {
|
|
46
|
+
throw new Error(
|
|
47
|
+
`Query embedding is ${queryVector.length}-dimensional but this index is ${dim}-dimensional — ` +
|
|
48
|
+
`the embedding model must match the one used to build it (recorded model: ${index.getMeta('model')})`
|
|
49
|
+
);
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
return index.search(queryVector, k);
|
|
53
|
+
} finally {
|
|
54
|
+
index.close();
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** The model an index says it was built with, in the shape createProvider takes. */
|
|
60
|
+
export function describeIndexModel(index) {
|
|
61
|
+
const meta = { model: index.getMeta('model'), embedder: index.getMeta('embedder') };
|
|
62
|
+
const embedder = embedderOfIndex(meta);
|
|
63
|
+
return {
|
|
64
|
+
model: meta.model,
|
|
65
|
+
embedder,
|
|
66
|
+
dtype: embedder === EMBEDDER_TRANSFORMERS ? (index.getMeta('dtype') || LEGACY_DTYPE) : null,
|
|
67
|
+
};
|
|
68
|
+
}
|