gitnexus 1.6.13-rc.11 → 1.6.13-rc.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,89 @@
1
+ import type { CachedEmbedding } from './types.js';
2
+ /** In-RAM vector copies stay below this row count; larger tables use the spill. */
3
+ export declare const DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT = 2048;
4
+ export interface CachedEmbeddingMeta {
5
+ nodeId: string;
6
+ chunkIndex: number;
7
+ startLine: number;
8
+ endLine: number;
9
+ contentHash?: string;
10
+ /** Row order in the spill file (and in `embeddings` when in-memory). */
11
+ vectorIndex: number;
12
+ }
13
+ export interface EmbeddingVectorSpill {
14
+ path: string;
15
+ dims: number;
16
+ rowCount: number;
17
+ }
18
+ export interface CachedEmbeddingsSnapshot {
19
+ embeddingNodeIds: Set<string>;
20
+ /** Populated only when the table is at or under the in-memory row limit. */
21
+ embeddings: CachedEmbedding[];
22
+ rows: CachedEmbeddingMeta[];
23
+ spill?: EmbeddingVectorSpill;
24
+ }
25
+ export interface LoadCachedEmbeddingsOptions {
26
+ /**
27
+ * Keep full `number[]` vectors in RAM at or below this many rows.
28
+ * `0` always spills. Default {@link DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT}
29
+ * or `GITNEXUS_EMBEDDING_CACHE_IN_MEMORY_LIMIT`.
30
+ */
31
+ inMemoryRowLimit?: number;
32
+ /** Directory for the spill file (default `os.tmpdir()`). */
33
+ spillDir?: string;
34
+ }
35
+ export interface CachedEmbeddingsBuilder {
36
+ embeddingNodeIds: Set<string>;
37
+ rows: CachedEmbeddingMeta[];
38
+ /** Float32 vectors kept in RAM until the in-memory row limit is exceeded. */
39
+ inMemory: Float32Array[] | null;
40
+ inMemoryRowLimit: number;
41
+ writer: EmbeddingSpillWriter;
42
+ }
43
+ export declare function emptyCachedEmbeddingsSnapshot(): CachedEmbeddingsSnapshot;
44
+ export declare function resolveEmbeddingCacheInMemoryRowLimit(override?: number): number;
45
+ export declare function normalizeCachedEmbeddings(raw: {
46
+ embeddingNodeIds?: Set<string>;
47
+ embeddings?: CachedEmbedding[];
48
+ rows?: CachedEmbeddingMeta[];
49
+ spill?: EmbeddingVectorSpill;
50
+ }): CachedEmbeddingsSnapshot;
51
+ export declare function cacheRowCount(snapshot: CachedEmbeddingsSnapshot): number;
52
+ export declare function snapshotEmbeddingDims(snapshot: CachedEmbeddingsSnapshot): number | undefined;
53
+ export declare function coerceEmbeddingToFloat32(embedding: unknown): Float32Array | null;
54
+ export declare function float32ToNumberArray(vec: Float32Array): number[];
55
+ /** Best-effort unlink of every tracked spill. Safe to call more than once. */
56
+ export declare function discardLiveEmbeddingSpills(): void;
57
+ /** Run `fn` so later {@link discardScopedEmbeddingSpills} only unlinks this run. */
58
+ export declare function withEmbeddingSpillScope<T>(fn: () => T): T;
59
+ /** Unlink spills created inside the current {@link withEmbeddingSpillScope}. */
60
+ export declare function discardScopedEmbeddingSpills(): void;
61
+ export declare class EmbeddingSpillWriter {
62
+ readonly path: string;
63
+ dims: number;
64
+ rowCount: number;
65
+ private fd;
66
+ private closed;
67
+ constructor(dir: string);
68
+ append(vec: Float32Array): void;
69
+ finish(): EmbeddingVectorSpill | undefined;
70
+ abort(): void;
71
+ private unlinkQuiet;
72
+ }
73
+ /** Validates the spill header once and reads vectors without reopening the file. */
74
+ export declare class EmbeddingSpillReader {
75
+ private fd;
76
+ private readonly bytesPerVec;
77
+ readonly dims: number;
78
+ readonly rowCount: number;
79
+ constructor(spill: EmbeddingVectorSpill);
80
+ read(indices: readonly number[]): Float32Array[];
81
+ close(): void;
82
+ }
83
+ export declare function readSpillVectors(spill: EmbeddingVectorSpill, indices: readonly number[]): Float32Array[];
84
+ export declare function disposeEmbeddingSpill(spill?: EmbeddingVectorSpill): void;
85
+ export declare function createCachedEmbeddingsBuilder(options?: LoadCachedEmbeddingsOptions): CachedEmbeddingsBuilder;
86
+ export declare function ingestCachedEmbeddingRow(builder: CachedEmbeddingsBuilder, row: Record<string, unknown> | unknown[], hasContentHash: boolean): void;
87
+ export declare function finalizeCachedEmbeddingsSnapshot(builder: CachedEmbeddingsBuilder): CachedEmbeddingsSnapshot;
88
+ export declare function abortCachedEmbeddingsBuilder(builder: CachedEmbeddingsBuilder): void;
89
+ export declare function materializeCachedEmbeddings(snapshot: CachedEmbeddingsSnapshot, metas: readonly CachedEmbeddingMeta[], spillReader?: EmbeddingSpillReader): CachedEmbedding[];
@@ -0,0 +1,388 @@
1
+ /**
2
+ * Disk-backed restore cache for CodeEmbedding rows (#3306).
3
+ *
4
+ * `loadCachedEmbeddings` used to `getAll()` the table and `map(Number)` every
5
+ * vector into a JS `number[]`. On a large already-indexed repo that single
6
+ * structure OOMs the V8 heap during "Caching embeddings..." even when the
7
+ * incremental diff is a handful of nodes.
8
+ *
9
+ * This module keeps metadata in RAM and writes vectors to a temp Float32
10
+ * spill. Restore materializes only the rows that Phase 3.5 will re-insert,
11
+ * in the existing 200-row batches.
12
+ */
13
+ import { closeSync, openSync, readSync, unlinkSync, writeSync } from 'node:fs';
14
+ import { AsyncLocalStorage } from 'node:async_hooks';
15
+ import { randomBytes } from 'node:crypto';
16
+ import os from 'node:os';
17
+ import path from 'node:path';
18
+ /** In-RAM vector copies stay below this row count; larger tables use the spill. */
19
+ export const DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT = 2048;
20
+ const SPILL_MAGIC = 'GNXE';
21
+ const SPILL_VERSION = 1;
22
+ const SPILL_HEADER_BYTES = 12;
23
+ export function emptyCachedEmbeddingsSnapshot() {
24
+ return { embeddingNodeIds: new Set(), embeddings: [], rows: [] };
25
+ }
26
+ export function resolveEmbeddingCacheInMemoryRowLimit(override) {
27
+ if (override !== undefined) {
28
+ if (!Number.isFinite(override) || override < 0) {
29
+ return DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT;
30
+ }
31
+ return Math.floor(override);
32
+ }
33
+ const raw = process.env.GITNEXUS_EMBEDDING_CACHE_IN_MEMORY_LIMIT;
34
+ if (raw === undefined || raw === '')
35
+ return DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT;
36
+ const parsed = parseInt(raw, 10);
37
+ return Number.isFinite(parsed) && parsed >= 0
38
+ ? parsed
39
+ : DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT;
40
+ }
41
+ export function normalizeCachedEmbeddings(raw) {
42
+ const embeddings = raw.embeddings ?? [];
43
+ const embeddingNodeIds = raw.embeddingNodeIds ?? new Set(embeddings.map((row) => row.nodeId));
44
+ const rows = raw.rows ??
45
+ embeddings.map((row, vectorIndex) => ({
46
+ nodeId: row.nodeId,
47
+ chunkIndex: row.chunkIndex,
48
+ startLine: row.startLine,
49
+ endLine: row.endLine,
50
+ contentHash: row.contentHash,
51
+ vectorIndex,
52
+ }));
53
+ return { embeddingNodeIds, embeddings, rows, spill: raw.spill };
54
+ }
55
+ export function cacheRowCount(snapshot) {
56
+ return snapshot.rows.length > 0 ? snapshot.rows.length : snapshot.embeddings.length;
57
+ }
58
+ export function snapshotEmbeddingDims(snapshot) {
59
+ if (snapshot.spill && snapshot.spill.dims > 0)
60
+ return snapshot.spill.dims;
61
+ const dims = snapshot.embeddings[0]?.embedding.length;
62
+ return dims && dims > 0 ? dims : undefined;
63
+ }
64
+ export function coerceEmbeddingToFloat32(embedding) {
65
+ if (embedding == null)
66
+ return null;
67
+ if (embedding instanceof Float32Array) {
68
+ return embedding.length > 0 ? embedding : null;
69
+ }
70
+ if (ArrayBuffer.isView(embedding) && !(embedding instanceof DataView)) {
71
+ const view = embedding;
72
+ if (view.length === 0)
73
+ return null;
74
+ return Float32Array.from({ length: view.length }, (_, i) => Number(view[i]));
75
+ }
76
+ if (typeof embedding === 'object' &&
77
+ typeof embedding[Symbol.iterator] === 'function') {
78
+ const arr = Array.isArray(embedding)
79
+ ? embedding
80
+ : Array.from(embedding);
81
+ if (arr.length === 0)
82
+ return null;
83
+ return Float32Array.from(arr, (value) => Number(value));
84
+ }
85
+ return null;
86
+ }
87
+ export function float32ToNumberArray(vec) {
88
+ const out = new Array(vec.length);
89
+ for (let i = 0; i < vec.length; i++)
90
+ out[i] = vec[i];
91
+ return out;
92
+ }
93
+ /**
94
+ * `fs.writeSync` can return a short byte count. Loop until the whole buffer
95
+ * lands, matching `sync-csv-writer.ts`, so a partial write never advances
96
+ * `rowCount` on a truncated vector.
97
+ */
98
+ function writeAllSync(fd, data) {
99
+ let offset = 0;
100
+ while (offset < data.length) {
101
+ const n = writeSync(fd, data, offset, data.length - offset);
102
+ if (n <= 0) {
103
+ throw new Error(`embedding spill short write: wrote ${n} of ${data.length - offset} bytes`);
104
+ }
105
+ offset += n;
106
+ }
107
+ }
108
+ function unlinkBestEffort(filePath) {
109
+ try {
110
+ unlinkSync(filePath);
111
+ }
112
+ catch {
113
+ /* ENOENT or already removed */
114
+ }
115
+ }
116
+ const liveSpillPaths = new Set();
117
+ const spillScope = new AsyncLocalStorage();
118
+ let spillExitHookInstalled = false;
119
+ function trackLiveSpillPath(filePath) {
120
+ liveSpillPaths.add(filePath);
121
+ spillScope.getStore()?.add(filePath);
122
+ if (!spillExitHookInstalled) {
123
+ spillExitHookInstalled = true;
124
+ process.on('exit', () => {
125
+ for (const spillPath of liveSpillPaths) {
126
+ unlinkBestEffort(spillPath);
127
+ }
128
+ });
129
+ }
130
+ }
131
+ function untrackLiveSpillPath(filePath) {
132
+ liveSpillPaths.delete(filePath);
133
+ }
134
+ /** Best-effort unlink of every tracked spill. Safe to call more than once. */
135
+ export function discardLiveEmbeddingSpills() {
136
+ for (const spillPath of [...liveSpillPaths]) {
137
+ unlinkBestEffort(spillPath);
138
+ liveSpillPaths.delete(spillPath);
139
+ }
140
+ }
141
+ /** Run `fn` so later {@link discardScopedEmbeddingSpills} only unlinks this run. */
142
+ export function withEmbeddingSpillScope(fn) {
143
+ return spillScope.run(new Set(), fn);
144
+ }
145
+ /** Unlink spills created inside the current {@link withEmbeddingSpillScope}. */
146
+ export function discardScopedEmbeddingSpills() {
147
+ const owned = spillScope.getStore();
148
+ if (!owned)
149
+ return;
150
+ for (const spillPath of [...owned]) {
151
+ unlinkBestEffort(spillPath);
152
+ liveSpillPaths.delete(spillPath);
153
+ owned.delete(spillPath);
154
+ }
155
+ }
156
+ export class EmbeddingSpillWriter {
157
+ path;
158
+ dims = 0;
159
+ rowCount = 0;
160
+ fd = null;
161
+ closed = false;
162
+ constructor(dir) {
163
+ this.path = path.join(dir, `gitnexus-embed-restore-${process.pid}-${randomBytes(8).toString('hex')}.bin`);
164
+ }
165
+ append(vec) {
166
+ if (this.closed) {
167
+ throw new Error('embedding spill writer already closed');
168
+ }
169
+ if (this.fd === null) {
170
+ this.dims = vec.length;
171
+ this.fd = openSync(this.path, 'wx', 0o600);
172
+ trackLiveSpillPath(this.path);
173
+ const header = Buffer.alloc(SPILL_HEADER_BYTES);
174
+ header.write(SPILL_MAGIC, 0, 4, 'ascii');
175
+ header.writeUInt8(SPILL_VERSION, 4);
176
+ header.writeUInt32LE(this.dims, 5);
177
+ writeAllSync(this.fd, header);
178
+ }
179
+ else if (vec.length !== this.dims) {
180
+ throw new Error(`embedding dim mismatch while spilling: got ${vec.length}, expected ${this.dims}`);
181
+ }
182
+ writeAllSync(this.fd, Buffer.from(vec.buffer, vec.byteOffset, vec.byteLength));
183
+ this.rowCount++;
184
+ }
185
+ finish() {
186
+ if (this.fd !== null) {
187
+ closeSync(this.fd);
188
+ this.fd = null;
189
+ }
190
+ this.closed = true;
191
+ if (this.rowCount === 0) {
192
+ this.unlinkQuiet();
193
+ return undefined;
194
+ }
195
+ return { path: this.path, dims: this.dims, rowCount: this.rowCount };
196
+ }
197
+ abort() {
198
+ const opened = this.fd !== null;
199
+ if (this.fd !== null) {
200
+ try {
201
+ closeSync(this.fd);
202
+ }
203
+ catch {
204
+ /* already closed */
205
+ }
206
+ this.fd = null;
207
+ }
208
+ this.closed = true;
209
+ if (opened || this.rowCount > 0) {
210
+ this.unlinkQuiet();
211
+ }
212
+ }
213
+ unlinkQuiet() {
214
+ unlinkBestEffort(this.path);
215
+ untrackLiveSpillPath(this.path);
216
+ }
217
+ }
218
+ /** Validates the spill header once and reads vectors without reopening the file. */
219
+ export class EmbeddingSpillReader {
220
+ fd = null;
221
+ bytesPerVec;
222
+ dims;
223
+ rowCount;
224
+ constructor(spill) {
225
+ this.rowCount = spill.rowCount;
226
+ const fd = openSync(spill.path, 'r');
227
+ try {
228
+ const header = Buffer.alloc(SPILL_HEADER_BYTES);
229
+ const headerRead = readSync(fd, header, 0, SPILL_HEADER_BYTES, 0);
230
+ if (headerRead !== SPILL_HEADER_BYTES || header.toString('ascii', 0, 4) !== SPILL_MAGIC) {
231
+ throw new Error(`invalid embedding spill header: ${spill.path}`);
232
+ }
233
+ if (header.readUInt8(4) !== SPILL_VERSION) {
234
+ throw new Error(`unsupported embedding spill version in ${spill.path}`);
235
+ }
236
+ const dims = header.readUInt32LE(5);
237
+ if (dims !== spill.dims) {
238
+ throw new Error(`embedding spill dim mismatch: file ${dims}, expected ${spill.dims}`);
239
+ }
240
+ this.dims = dims;
241
+ this.bytesPerVec = dims * 4;
242
+ this.fd = fd;
243
+ }
244
+ catch (err) {
245
+ closeSync(fd);
246
+ throw err;
247
+ }
248
+ }
249
+ read(indices) {
250
+ if (this.fd === null) {
251
+ throw new Error('embedding spill reader already closed');
252
+ }
253
+ const out = [];
254
+ for (const index of indices) {
255
+ if (!Number.isInteger(index) || index < 0 || index >= this.rowCount) {
256
+ throw new Error(`embedding spill index out of range: ${index}`);
257
+ }
258
+ const offset = SPILL_HEADER_BYTES + index * this.bytesPerVec;
259
+ const copy = new Float32Array(this.dims);
260
+ const bytes = new Uint8Array(copy.buffer, copy.byteOffset, this.bytesPerVec);
261
+ const n = readSync(this.fd, bytes, 0, this.bytesPerVec, offset);
262
+ if (n !== this.bytesPerVec) {
263
+ throw new Error(`short embedding spill read at index ${index}`);
264
+ }
265
+ out.push(copy);
266
+ }
267
+ return out;
268
+ }
269
+ close() {
270
+ if (this.fd === null)
271
+ return;
272
+ closeSync(this.fd);
273
+ this.fd = null;
274
+ }
275
+ }
276
+ export function readSpillVectors(spill, indices) {
277
+ const reader = new EmbeddingSpillReader(spill);
278
+ try {
279
+ return reader.read(indices);
280
+ }
281
+ finally {
282
+ reader.close();
283
+ }
284
+ }
285
+ export function disposeEmbeddingSpill(spill) {
286
+ if (!spill?.path)
287
+ return;
288
+ unlinkBestEffort(spill.path);
289
+ untrackLiveSpillPath(spill.path);
290
+ }
291
+ export function createCachedEmbeddingsBuilder(options) {
292
+ const inMemoryRowLimit = resolveEmbeddingCacheInMemoryRowLimit(options?.inMemoryRowLimit);
293
+ return {
294
+ embeddingNodeIds: new Set(),
295
+ rows: [],
296
+ inMemory: inMemoryRowLimit <= 0 ? null : [],
297
+ inMemoryRowLimit,
298
+ writer: new EmbeddingSpillWriter(options?.spillDir ?? os.tmpdir()),
299
+ };
300
+ }
301
+ export function ingestCachedEmbeddingRow(builder, row, hasContentHash) {
302
+ const rec = row;
303
+ const nodeId = String(rec.nodeId ?? rec[0] ?? '');
304
+ if (!nodeId)
305
+ return;
306
+ const embedding = rec.embedding ?? rec[4];
307
+ const f32 = coerceEmbeddingToFloat32(embedding);
308
+ if (!f32)
309
+ return;
310
+ builder.embeddingNodeIds.add(nodeId);
311
+ const meta = {
312
+ nodeId,
313
+ chunkIndex: Number(rec.chunkIndex ?? rec[1] ?? 0),
314
+ startLine: Number(rec.startLine ?? rec[2] ?? 0),
315
+ endLine: Number(rec.endLine ?? rec[3] ?? 0),
316
+ contentHash: hasContentHash
317
+ ? (rec.contentHash ?? rec[5] ?? undefined)
318
+ : undefined,
319
+ vectorIndex: builder.rows.length,
320
+ };
321
+ builder.rows.push(meta);
322
+ if (builder.inMemory && builder.rows.length <= builder.inMemoryRowLimit) {
323
+ builder.inMemory.push(f32);
324
+ return;
325
+ }
326
+ if (builder.inMemory) {
327
+ for (const prior of builder.inMemory) {
328
+ builder.writer.append(prior);
329
+ }
330
+ builder.inMemory = null;
331
+ }
332
+ builder.writer.append(f32);
333
+ }
334
+ export function finalizeCachedEmbeddingsSnapshot(builder) {
335
+ const inMemory = builder.inMemory;
336
+ if (inMemory) {
337
+ builder.writer.abort();
338
+ return {
339
+ embeddingNodeIds: builder.embeddingNodeIds,
340
+ embeddings: builder.rows.map((meta, i) => ({
341
+ nodeId: meta.nodeId,
342
+ chunkIndex: meta.chunkIndex,
343
+ startLine: meta.startLine,
344
+ endLine: meta.endLine,
345
+ contentHash: meta.contentHash,
346
+ embedding: float32ToNumberArray(inMemory[i]),
347
+ })),
348
+ rows: builder.rows,
349
+ };
350
+ }
351
+ return {
352
+ embeddingNodeIds: builder.embeddingNodeIds,
353
+ embeddings: [],
354
+ rows: builder.rows,
355
+ spill: builder.writer.finish(),
356
+ };
357
+ }
358
+ export function abortCachedEmbeddingsBuilder(builder) {
359
+ builder.writer.abort();
360
+ }
361
+ export function materializeCachedEmbeddings(snapshot, metas, spillReader) {
362
+ if (metas.length === 0)
363
+ return [];
364
+ if (snapshot.spill && snapshot.embeddings.length === 0) {
365
+ const indices = metas.map((meta) => meta.vectorIndex);
366
+ const vectors = spillReader
367
+ ? spillReader.read(indices)
368
+ : readSpillVectors(snapshot.spill, indices);
369
+ return metas.map((meta, i) => ({
370
+ nodeId: meta.nodeId,
371
+ chunkIndex: meta.chunkIndex,
372
+ startLine: meta.startLine,
373
+ endLine: meta.endLine,
374
+ contentHash: meta.contentHash,
375
+ embedding: float32ToNumberArray(vectors[i]),
376
+ }));
377
+ }
378
+ if (snapshot.embeddings.length === 0)
379
+ return [];
380
+ const byKey = new Map(snapshot.embeddings.map((row) => [`${row.nodeId}:${row.chunkIndex}`, row]));
381
+ return metas.map((meta) => {
382
+ const hit = byKey.get(`${meta.nodeId}:${meta.chunkIndex}`) ?? snapshot.embeddings[meta.vectorIndex];
383
+ if (!hit) {
384
+ throw new Error(`missing cached embedding ${meta.nodeId}:${meta.chunkIndex}`);
385
+ }
386
+ return hit;
387
+ });
388
+ }
@@ -4,7 +4,7 @@ import { type ContentRetention } from '../../storage/repo-meta.js';
4
4
  import { NodeTableName } from './schema.js';
5
5
  import type { GraphEmitManifest } from './graph-emit-sink.js';
6
6
  import type { PdgEmitManifest } from './pdg-emit-sink.js';
7
- import { type CachedEmbedding } from '../embeddings/types.js';
7
+ import { type CachedEmbeddingsSnapshot, type LoadCachedEmbeddingsOptions } from '../embeddings/embedding-restore-spill.js';
8
8
  import { type ExtensionEnsureOptions } from './extension-loader.js';
9
9
  /** Result of splitting the relationship CSV into per-label-pair files. */
10
10
  export interface RelCsvSplitResult {
@@ -194,15 +194,16 @@ export declare const getLbugStats: () => Promise<{
194
194
  }>;
195
195
  /**
196
196
  * Load cached embeddings from LadybugDB before a rebuild.
197
- * Returns all embedding vectors so they can be re-inserted after the graph is reloaded,
198
- * avoiding expensive re-embedding of unchanged nodes.
197
+ *
198
+ * Streams `CodeEmbedding` rows with `hasNext`/`getNext` under `withConnLock`
199
+ * (#2264, #3306). Vectors are spilled to a temp Float32 file once the table
200
+ * exceeds the in-memory row limit so incremental analyze cannot OOM the V8
201
+ * heap by materializing every `number[]` up front. Small tables still return
202
+ * in-RAM `embeddings` for existing callers/tests.
199
203
  *
200
204
  * Detects old schema (no chunkIndex column) and returns empty cache to trigger rebuild.
201
205
  */
202
- export declare const loadCachedEmbeddings: () => Promise<{
203
- embeddingNodeIds: Set<string>;
204
- embeddings: CachedEmbedding[];
205
- }>;
206
+ export declare const loadCachedEmbeddings: (options?: LoadCachedEmbeddingsOptions) => Promise<CachedEmbeddingsSnapshot>;
206
207
  /**
207
208
  * Fetch existing embedding hashes from CodeEmbedding table for incremental embedding.
208
209
  * Returns a Map<nodeId, contentHash> suitable for passing to `runEmbeddingPipeline`.
@@ -24,6 +24,7 @@ import { streamAllCSVsToDisk } from './csv-generator.js';
24
24
  import { PDG_EDGE_TYPES } from './pdg-emit-sink.js';
25
25
  import { getNodeLabel as deriveNodeLabel } from './rel-pair-routing.js';
26
26
  import { EMBEDDABLE_LABELS } from '../embeddings/types.js';
27
+ import { abortCachedEmbeddingsBuilder, createCachedEmbeddingsBuilder, emptyCachedEmbeddingsSnapshot, finalizeCachedEmbeddingsSnapshot, ingestCachedEmbeddingRow, } from '../embeddings/embedding-restore-spill.js';
27
28
  import { extensionManager, getFtsCapability, resolveAnalyzeInstallPolicy, } from './extension-loader.js';
28
29
  // Remedy classification for LOAD failures (#2374/#2383). Pure + node:fs only, so
29
30
  // this adds no cycle: `extension-loader.ts` already depends on it.
@@ -1785,24 +1786,28 @@ export const getLbugStats = async () => {
1785
1786
  };
1786
1787
  /**
1787
1788
  * Load cached embeddings from LadybugDB before a rebuild.
1788
- * Returns all embedding vectors so they can be re-inserted after the graph is reloaded,
1789
- * avoiding expensive re-embedding of unchanged nodes.
1789
+ *
1790
+ * Streams `CodeEmbedding` rows with `hasNext`/`getNext` under `withConnLock`
1791
+ * (#2264, #3306). Vectors are spilled to a temp Float32 file once the table
1792
+ * exceeds the in-memory row limit so incremental analyze cannot OOM the V8
1793
+ * heap by materializing every `number[]` up front. Small tables still return
1794
+ * in-RAM `embeddings` for existing callers/tests.
1790
1795
  *
1791
1796
  * Detects old schema (no chunkIndex column) and returns empty cache to trigger rebuild.
1792
1797
  */
1793
- export const loadCachedEmbeddings = async () => {
1798
+ export const loadCachedEmbeddings = async (options) => {
1794
1799
  const c = conn;
1795
1800
  if (!c) {
1796
- return { embeddingNodeIds: new Set(), embeddings: [] };
1801
+ return emptyCachedEmbeddingsSnapshot();
1797
1802
  }
1798
1803
  // The whole read runs inside the connection lock (#2264 review P2). It's safe
1799
1804
  // today only by call-ordering (loadCachedEmbeddings runs before the WAL driver
1800
1805
  // starts), but the lock makes it robust to future reordering — a concurrent
1801
1806
  // CHECKPOINT on the singleton connection is the documented corruption trigger.
1802
- // Leaf read: no nested withConnLock-wrapped helpers inside.
1807
+ // Leaf read: no nested withConnLock-wrapped helpers inside. Do NOT call
1808
+ // `streamQuery` here — that path is unlocked and would race a CHECKPOINT.
1803
1809
  return withConnLock(async () => {
1804
- const embeddingNodeIds = new Set();
1805
- const embeddings = [];
1810
+ const builder = createCachedEmbeddingsBuilder(options);
1806
1811
  try {
1807
1812
  // Schema migration detection: query with new columns to verify schema version.
1808
1813
  // Old schema only had (nodeId, embedding); new schema adds (id, chunkIndex, startLine, endLine, contentHash).
@@ -1815,49 +1820,47 @@ export const loadCachedEmbeddings = async () => {
1815
1820
  await readQueryRows(check);
1816
1821
  }
1817
1822
  catch {
1818
- return { embeddingNodeIds: new Set(), embeddings: [] };
1823
+ abortCachedEmbeddingsBuilder(builder);
1824
+ return emptyCachedEmbeddingsSnapshot();
1819
1825
  }
1820
- // Try to read contentHash alongside chunk columns
1821
- let rows;
1826
+ let queryResult;
1822
1827
  let hasContentHash = true;
1823
1828
  try {
1824
- rows = await c.query(`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.chunkIndex AS chunkIndex, e.startLine AS startLine, e.endLine AS endLine, e.embedding AS embedding, e.contentHash AS contentHash`);
1825
- }
1826
- catch (err) {
1827
- // Fallback for legacy DBs without contentHash column
1828
- const msg = err?.message ?? '';
1829
- if (isMissingColumnOrTableError(msg)) {
1830
- hasContentHash = false;
1831
- rows = await c.query(`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.chunkIndex AS chunkIndex, e.startLine AS startLine, e.endLine AS endLine, e.embedding AS embedding`);
1829
+ try {
1830
+ queryResult = await c.query(`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.chunkIndex AS chunkIndex, e.startLine AS startLine, e.endLine AS endLine, e.embedding AS embedding, e.contentHash AS contentHash`);
1832
1831
  }
1833
- else {
1834
- throw err;
1832
+ catch (err) {
1833
+ // Fallback for legacy DBs without contentHash column
1834
+ const msg = err?.message ?? '';
1835
+ if (isMissingColumnOrTableError(msg)) {
1836
+ hasContentHash = false;
1837
+ queryResult = await c.query(`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.chunkIndex AS chunkIndex, e.startLine AS startLine, e.endLine AS endLine, e.embedding AS embedding`);
1838
+ }
1839
+ else {
1840
+ throw err;
1841
+ }
1835
1842
  }
1836
- }
1837
- for (const row of await readQueryRows(rows)) {
1838
- const nodeId = String(row.nodeId ?? row[0] ?? '');
1839
- if (!nodeId)
1840
- continue;
1841
- embeddingNodeIds.add(nodeId);
1842
- const embedding = row.embedding ?? row[4];
1843
- if (embedding) {
1844
- embeddings.push({
1845
- nodeId,
1846
- chunkIndex: Number(row.chunkIndex ?? row[1] ?? 0),
1847
- startLine: Number(row.startLine ?? row[2] ?? 0),
1848
- endLine: Number(row.endLine ?? row[3] ?? 0),
1849
- embedding: Array.isArray(embedding)
1850
- ? embedding.map(Number)
1851
- : Array.from(embedding).map(Number),
1852
- contentHash: hasContentHash ? (row.contentHash ?? row[5] ?? undefined) : undefined,
1853
- });
1843
+ const results = Array.isArray(queryResult) ? queryResult : [queryResult];
1844
+ const result = results[0];
1845
+ while (await result.hasNext()) {
1846
+ const row = await result.getNext();
1847
+ ingestCachedEmbeddingRow(builder, row, hasContentHash);
1854
1848
  }
1849
+ return finalizeCachedEmbeddingsSnapshot(builder);
1850
+ }
1851
+ catch (err) {
1852
+ abortCachedEmbeddingsBuilder(builder);
1853
+ throw err;
1854
+ }
1855
+ finally {
1856
+ if (queryResult)
1857
+ await closeQueryResults(queryResult);
1855
1858
  }
1856
1859
  }
1857
- catch {
1858
- /* embedding table may not exist */
1860
+ catch (err) {
1861
+ abortCachedEmbeddingsBuilder(builder);
1862
+ throw err;
1859
1863
  }
1860
- return { embeddingNodeIds, embeddings };
1861
1864
  });
1862
1865
  };
1863
1866
  /**