@zerowidth/workbench-sdk 2.0.0-alpha.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/nodes/combine-document-chunks/combine-document-chunks.config.json +8 -0
- package/nodes/combine-document-chunks/combine-document-chunks.process.js +9 -2
- package/nodes/describe-tables/describe-tables.config.json +59 -0
- package/nodes/describe-tables/describe-tables.process.js +40 -0
- package/nodes/expand-chunk-context/expand-chunk-context.config.json +87 -0
- package/nodes/expand-chunk-context/expand-chunk-context.process.js +34 -0
- package/nodes/find-entity-path/find-entity-path.config.json +73 -0
- package/nodes/find-entity-path/find-entity-path.process.js +51 -0
- package/nodes/get-chunk-by-index/get-chunk-by-index.config.json +8 -0
- package/nodes/get-chunk-by-index/get-chunk-by-index.process.js +9 -2
- package/nodes/get-entity-neighbors/get-entity-neighbors.config.json +81 -0
- package/nodes/get-entity-neighbors/get-entity-neighbors.process.js +51 -0
- package/nodes/keyword-search/keyword-search.config.json +67 -0
- package/nodes/keyword-search/keyword-search.process.js +45 -0
- package/nodes/knowledge-base/knowledge-base.config.json +32 -0
- package/nodes/knowledge-base/knowledge-base.process.js +15 -0
- package/nodes/list-documents/list-documents.config.json +60 -0
- package/nodes/list-documents/list-documents.process.js +25 -0
- package/nodes/list-entities/list-entities.config.json +74 -0
- package/nodes/list-entities/list-entities.process.js +25 -0
- package/nodes/query-knowledge-base/query-knowledge-base.config.json +14 -7
- package/nodes/query-knowledge-base/query-knowledge-base.process.js +25 -5
- package/nodes/read-chunks/read-chunks.config.json +79 -0
- package/nodes/read-chunks/read-chunks.process.js +49 -0
- package/nodes/remote-mcp-tool/remote-mcp-tool.config.json +1 -0
- package/nodes/semantic-search/semantic-search.config.json +9 -1
- package/nodes/semantic-search/semantic-search.process.js +41 -18
- package/nodes/tool/tool.config.json +1 -0
- package/package.json +12 -7
- package/src/index.js +76 -9
- package/src/integrations/knowledge-base-interface.js +49 -0
- package/src/integrations/sqlite.js +535 -83
- package/src/types/knowledge_base.json +9 -0
- package/src/utilities/loaders.js +67 -11
- package/src/utilities/typers.js +15 -17
- package/src/utilities/validators.js +16 -1
- package/types/knowledge_base.json +9 -0
|
@@ -1,10 +1,11 @@
|
|
|
1
|
-
import
|
|
2
|
-
import { promisify } from 'util';
|
|
1
|
+
import { DatabaseSync } from 'node:sqlite';
|
|
3
2
|
import path from 'path';
|
|
4
3
|
import fs from 'fs';
|
|
5
4
|
import { KnowledgeBaseInterface } from './knowledge-base-interface.js';
|
|
6
5
|
|
|
7
|
-
//
|
|
6
|
+
// sqlite-vec ships a loadable extension (prebuilt per-platform .dylib/.so);
|
|
7
|
+
// `getLoadablePath()` returns the binary we hand to node:sqlite's
|
|
8
|
+
// `loadExtension`. No native node module / node-gyp build involved.
|
|
8
9
|
import * as sqliteVec from 'sqlite-vec';
|
|
9
10
|
|
|
10
11
|
export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
@@ -21,48 +22,46 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
21
22
|
}
|
|
22
23
|
|
|
23
24
|
/**
|
|
24
|
-
* Initialize the database connection
|
|
25
|
+
* Initialize the database connection.
|
|
26
|
+
*
|
|
27
|
+
* Uses Node's built-in `node:sqlite` (DatabaseSync) — synchronous,
|
|
28
|
+
* zero native dependency, uniform across Node 22+/24 and Linux — rather
|
|
29
|
+
* than the legacy `sqlite3` native module. The public methods stay
|
|
30
|
+
* `async` for call-site compatibility.
|
|
25
31
|
* @returns {Promise<void>}
|
|
26
32
|
*/
|
|
27
33
|
async connect() {
|
|
28
34
|
try {
|
|
29
|
-
|
|
30
|
-
|
|
31
35
|
// Check if database file exists
|
|
32
36
|
if (!fs.existsSync(this.dbPath)) {
|
|
33
37
|
throw new Error(`Database file not found: ${this.dbPath}`);
|
|
34
38
|
}
|
|
35
39
|
|
|
36
|
-
//
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
throw new Error(`Failed to connect to database: ${err.message}`);
|
|
40
|
-
}
|
|
41
|
-
});
|
|
42
|
-
|
|
43
|
-
// Promisify the database methods
|
|
44
|
-
this.db.run = promisify(this.db.run.bind(this.db));
|
|
45
|
-
this.db.get = promisify(this.db.get.bind(this.db));
|
|
46
|
-
this.db.all = promisify(this.db.all.bind(this.db));
|
|
47
|
-
this.db.close = promisify(this.db.close.bind(this.db));
|
|
48
|
-
|
|
49
|
-
// Load sqlite-vec extension
|
|
50
|
-
try {
|
|
51
|
-
sqliteVec.load(this.db);
|
|
52
|
-
} catch (error) {
|
|
53
|
-
console.warn('[WARN] Failed to load sqlite-vec extension:', error.message);
|
|
54
|
-
// Don't fail the connection, just warn
|
|
55
|
-
}
|
|
56
|
-
|
|
57
|
-
// Test the connection
|
|
58
|
-
await this.db.get("SELECT 1");
|
|
59
|
-
this.isConnected = true;
|
|
40
|
+
// Open the connection with extension loading allowed so sqlite-vec
|
|
41
|
+
// can be attached.
|
|
42
|
+
this.db = new DatabaseSync(this.dbPath, { allowExtension: true });
|
|
60
43
|
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
44
|
+
// Load sqlite-vec extension. Enable extension loading first where the
|
|
45
|
+
// runtime exposes the toggle (guarded for forward-compat).
|
|
46
|
+
try {
|
|
47
|
+
if (typeof this.db.enableLoadExtension === 'function') {
|
|
48
|
+
this.db.enableLoadExtension(true);
|
|
64
49
|
}
|
|
50
|
+
this.db.loadExtension(sqliteVec.getLoadablePath());
|
|
51
|
+
} catch (error) {
|
|
52
|
+
console.warn('[WARN] Failed to load sqlite-vec extension:', error.message);
|
|
53
|
+
// Don't fail the connection, just warn — semanticSearch falls back
|
|
54
|
+
// to text search when vec functions are unavailable.
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
// Test the connection
|
|
58
|
+
this._get('SELECT 1');
|
|
59
|
+
this.isConnected = true;
|
|
60
|
+
} catch (error) {
|
|
61
|
+
this.isConnected = false;
|
|
62
|
+
throw new Error(`SQLite connection failed: ${error.message}`);
|
|
65
63
|
}
|
|
64
|
+
}
|
|
66
65
|
|
|
67
66
|
/**
|
|
68
67
|
* Close the database connection and clean up temporary files
|
|
@@ -71,24 +70,35 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
71
70
|
async disconnect() {
|
|
72
71
|
if (this.db && this.isConnected) {
|
|
73
72
|
try {
|
|
74
|
-
|
|
73
|
+
this.db.close();
|
|
75
74
|
this.isConnected = false;
|
|
76
75
|
} catch (error) {
|
|
77
76
|
throw new Error(`Failed to close database: ${error.message}`);
|
|
78
77
|
}
|
|
79
78
|
}
|
|
80
|
-
|
|
81
|
-
//
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
79
|
+
|
|
80
|
+
// NOTE: disconnect() never deletes the database file. File lifecycle
|
|
81
|
+
// belongs to whoever created the file — the engine tracks and removes
|
|
82
|
+
// its own temp extractions (and hosts opt in via
|
|
83
|
+
// config.knowledgeBase.cleanupDbFiles); a path-substring heuristic
|
|
84
|
+
// here once deleted user-owned databases.
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
// ─── node:sqlite driver helpers ────────────────────────────────────────
|
|
88
|
+
// DatabaseSync is synchronous: prepare once, bind anonymous `?` params by
|
|
89
|
+
// spreading the params array. These wrap the three shapes the rest of the
|
|
90
|
+
// class needs.
|
|
91
|
+
|
|
92
|
+
_all(sql, params = []) {
|
|
93
|
+
return this.db.prepare(sql).all(...params);
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
_get(sql, params = []) {
|
|
97
|
+
return this.db.prepare(sql).get(...params);
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
_run(sql, params = []) {
|
|
101
|
+
return this.db.prepare(sql).run(...params);
|
|
92
102
|
}
|
|
93
103
|
|
|
94
104
|
/**
|
|
@@ -108,7 +118,7 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
108
118
|
const allowedOperations = ['SELECT', 'INSERT', 'UPDATE', 'DELETE', 'CREATE', 'DROP', 'ALTER'];
|
|
109
119
|
const queryUpper = query.trim().toUpperCase();
|
|
110
120
|
const isAllowed = allowedOperations.some(op => queryUpper.startsWith(op));
|
|
111
|
-
|
|
121
|
+
|
|
112
122
|
if (!isAllowed) {
|
|
113
123
|
throw new Error(`Operation not allowed: ${queryUpper.split(' ')[0]}`);
|
|
114
124
|
}
|
|
@@ -116,22 +126,30 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
116
126
|
// Execute query based on operation type
|
|
117
127
|
let result;
|
|
118
128
|
if (operation.toUpperCase() === 'SELECT') {
|
|
119
|
-
if (
|
|
120
|
-
result =
|
|
129
|
+
if (/\bLIMIT\s+1\b(?!\s*,)/.test(queryUpper)) {
|
|
130
|
+
result = this._get(query, params);
|
|
121
131
|
} else {
|
|
122
|
-
result =
|
|
132
|
+
result = this._all(query, params);
|
|
123
133
|
}
|
|
124
|
-
|
|
125
|
-
|
|
134
|
+
return {
|
|
135
|
+
success: true,
|
|
136
|
+
data: result,
|
|
137
|
+
operation: operation.toUpperCase(),
|
|
138
|
+
rowCount: Array.isArray(result) ? result.length : (result ? 1 : 0)
|
|
139
|
+
};
|
|
126
140
|
}
|
|
127
141
|
|
|
142
|
+
// Non-SELECT: run() returns { changes, lastInsertRowid }.
|
|
143
|
+
const runResult = this._run(query, params);
|
|
128
144
|
return {
|
|
129
145
|
success: true,
|
|
130
|
-
data:
|
|
146
|
+
data: {
|
|
147
|
+
changes: Number(runResult.changes),
|
|
148
|
+
lastID: runResult.lastInsertRowid,
|
|
149
|
+
},
|
|
131
150
|
operation: operation.toUpperCase(),
|
|
132
|
-
rowCount:
|
|
151
|
+
rowCount: Number(runResult.changes) || 0
|
|
133
152
|
};
|
|
134
|
-
|
|
135
153
|
} catch (error) {
|
|
136
154
|
throw new Error(`SQLite query failed: ${error.message}`);
|
|
137
155
|
}
|
|
@@ -197,7 +215,7 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
197
215
|
*/
|
|
198
216
|
async getSchema() {
|
|
199
217
|
const tables = await this.select(`
|
|
200
|
-
SELECT name FROM sqlite_master
|
|
218
|
+
SELECT name FROM sqlite_master
|
|
201
219
|
WHERE type='table' AND name NOT LIKE 'sqlite_%'
|
|
202
220
|
ORDER BY name
|
|
203
221
|
`);
|
|
@@ -218,10 +236,10 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
218
236
|
async validateKnowledgeBaseSchema() {
|
|
219
237
|
try {
|
|
220
238
|
const schema = await this.getSchema();
|
|
221
|
-
|
|
239
|
+
|
|
222
240
|
const hasDocuments = 'documents' in schema;
|
|
223
241
|
const hasChunks = 'chunks' in schema;
|
|
224
|
-
|
|
242
|
+
|
|
225
243
|
if (!hasDocuments || !hasChunks) {
|
|
226
244
|
return {
|
|
227
245
|
valid: false,
|
|
@@ -267,7 +285,7 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
267
285
|
const docCount = await this.select('SELECT COUNT(*) as count FROM documents');
|
|
268
286
|
const chunkCount = await this.select('SELECT COUNT(*) as count FROM chunks');
|
|
269
287
|
const totalSize = await this.select('SELECT SUM(file_size) as total_size FROM documents');
|
|
270
|
-
|
|
288
|
+
|
|
271
289
|
return {
|
|
272
290
|
documents: docCount[0]?.count || 0,
|
|
273
291
|
chunks: chunkCount[0]?.count || 0,
|
|
@@ -284,19 +302,29 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
284
302
|
*/
|
|
285
303
|
async getEmbeddingModel() {
|
|
286
304
|
try {
|
|
287
|
-
const
|
|
305
|
+
const rows = await this.select(
|
|
288
306
|
'SELECT embedding_model FROM chunks WHERE embedding_model IS NOT NULL ORDER BY created_at DESC LIMIT 1'
|
|
289
307
|
);
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
308
|
+
|
|
309
|
+
// `select()` returns a single object for LIMIT-1 queries (query()
|
|
310
|
+
// dispatches those through `_get`), or an array otherwise —
|
|
311
|
+
// normalize to the first row. Reading `.length` on the object
|
|
312
|
+
// silently missed the stored model and fell through to the default,
|
|
313
|
+
// whose NON-namespaced value ("text-embedding-3-small") then never
|
|
314
|
+
// matched the namespaced model the chunks are indexed under
|
|
315
|
+
// ("openai/text-embedding-3-small"), so `WHERE embedding_model = ?`
|
|
316
|
+
// dropped every row.
|
|
317
|
+
const row = Array.isArray(rows) ? rows[0] : rows;
|
|
318
|
+
if (row && row.embedding_model) {
|
|
319
|
+
return row.embedding_model;
|
|
293
320
|
}
|
|
294
|
-
|
|
295
|
-
// Fall back to default
|
|
296
|
-
|
|
321
|
+
|
|
322
|
+
// Fall back to the namespaced default (matches the ingestion
|
|
323
|
+
// default + OpenRouter's expected model id).
|
|
324
|
+
return 'openai/text-embedding-3-small';
|
|
297
325
|
} catch (error) {
|
|
298
326
|
console.warn('[WARN] Failed to get embedding model from database, using default:', error.message);
|
|
299
|
-
return 'text-embedding-3-small';
|
|
327
|
+
return 'openai/text-embedding-3-small';
|
|
300
328
|
}
|
|
301
329
|
}
|
|
302
330
|
|
|
@@ -336,7 +364,7 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
336
364
|
|
|
337
365
|
// Check if sqlite-vec extension is loaded
|
|
338
366
|
try {
|
|
339
|
-
|
|
367
|
+
this._get("SELECT vec_version()");
|
|
340
368
|
} catch (error) {
|
|
341
369
|
console.warn('[WARN] sqlite-vec extension not loaded properly, falling back to text search');
|
|
342
370
|
return await this._fallbackTextSearch(query, options);
|
|
@@ -351,10 +379,10 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
351
379
|
const queryEmbedding = options.query_embedding;
|
|
352
380
|
const embeddingDimensions = queryEmbedding.length;
|
|
353
381
|
const queryEmbeddingString = JSON.stringify(queryEmbedding);
|
|
354
|
-
|
|
382
|
+
|
|
355
383
|
// Build the KNN query using sqlite-vec scalar functions
|
|
356
384
|
let searchSql = `
|
|
357
|
-
SELECT
|
|
385
|
+
SELECT
|
|
358
386
|
c.id,
|
|
359
387
|
c.document_id,
|
|
360
388
|
c.chunk_index,
|
|
@@ -375,10 +403,16 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
375
403
|
WHERE c.embedding IS NOT NULL
|
|
376
404
|
AND c.embedding_model = ?
|
|
377
405
|
AND c.embedding_dimensions = ?
|
|
378
|
-
AND vec_distance_cosine(c.embedding, ?)
|
|
406
|
+
AND vec_distance_cosine(c.embedding, ?) <= ?
|
|
379
407
|
`;
|
|
380
408
|
|
|
381
|
-
|
|
409
|
+
// vec_distance_cosine is a DISTANCE (0 = identical, larger = less
|
|
410
|
+
// similar). Callers pass `similarity_threshold` in [0,1] (1 =
|
|
411
|
+
// identical), so convert: similarity >= t ⟺ distance <= (1 - t).
|
|
412
|
+
// (Previously this compared distance `>= threshold`, which dropped
|
|
413
|
+
// the closest matches and kept the farthest — inverted.)
|
|
414
|
+
const maxDistance = 1 - similarity_threshold;
|
|
415
|
+
const params = [queryEmbeddingString, modelToUse, embeddingDimensions, queryEmbeddingString, maxDistance];
|
|
382
416
|
|
|
383
417
|
// Add document filter if specified
|
|
384
418
|
if (document_id) {
|
|
@@ -394,8 +428,8 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
394
428
|
// Add the query embedding and limit to params
|
|
395
429
|
params.push(queryEmbeddingString, limit);
|
|
396
430
|
|
|
397
|
-
const results =
|
|
398
|
-
|
|
431
|
+
const results = this._all(searchSql, params);
|
|
432
|
+
|
|
399
433
|
// Parse JSON metadata and format results
|
|
400
434
|
return results.map(row => ({
|
|
401
435
|
id: row.id,
|
|
@@ -410,10 +444,12 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
410
444
|
metadata: row.metadata ? JSON.parse(row.metadata) : {},
|
|
411
445
|
embedding_model: row.embedding_model,
|
|
412
446
|
embedding_dimensions: row.embedding_dimensions,
|
|
413
|
-
|
|
447
|
+
// `row.similarity` is the aliased cosine DISTANCE; report it as
|
|
448
|
+
// an actual similarity in [-1, 1] (1 = identical).
|
|
449
|
+
similarity_score: 1 - row.similarity,
|
|
414
450
|
created_at: row.created_at
|
|
415
451
|
}));
|
|
416
|
-
|
|
452
|
+
|
|
417
453
|
} catch (error) {
|
|
418
454
|
console.warn('[WARN] Vector search failed, falling back to text search:', error.message);
|
|
419
455
|
return await this._fallbackTextSearch(query, options);
|
|
@@ -434,7 +470,7 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
434
470
|
|
|
435
471
|
try {
|
|
436
472
|
let sql = `
|
|
437
|
-
SELECT
|
|
473
|
+
SELECT
|
|
438
474
|
c.id,
|
|
439
475
|
c.document_id,
|
|
440
476
|
c.chunk_index,
|
|
@@ -450,28 +486,444 @@ export default class SQLiteIntegration extends KnowledgeBaseInterface {
|
|
|
450
486
|
LEFT JOIN documents d ON c.document_id = d.id
|
|
451
487
|
WHERE c.content LIKE ?
|
|
452
488
|
`;
|
|
453
|
-
|
|
489
|
+
|
|
454
490
|
const params = [`%${query}%`];
|
|
455
|
-
|
|
491
|
+
|
|
456
492
|
if (document_id) {
|
|
457
493
|
sql += ' AND c.document_id = ?';
|
|
458
494
|
params.push(document_id);
|
|
459
495
|
}
|
|
460
|
-
|
|
496
|
+
|
|
461
497
|
sql += ' ORDER BY c.created_at DESC LIMIT ?';
|
|
462
498
|
params.push(limit);
|
|
463
|
-
|
|
499
|
+
|
|
464
500
|
const results = await this.select(sql, params);
|
|
465
|
-
|
|
501
|
+
|
|
466
502
|
// Add mock similarity scores for text search
|
|
467
503
|
return results.map((result, index) => ({
|
|
468
504
|
...result,
|
|
469
505
|
similarity_score: 1.0 - (index * 0.1), // Mock decreasing similarity
|
|
470
506
|
match_type: 'text_search'
|
|
471
507
|
}));
|
|
472
|
-
|
|
508
|
+
|
|
473
509
|
} catch (error) {
|
|
474
510
|
throw new Error(`Fallback text search failed: ${error.message}`);
|
|
475
511
|
}
|
|
476
512
|
}
|
|
513
|
+
|
|
514
|
+
/**
|
|
515
|
+
* Keyword (substring) search over chunk content — the free, no-embedding
|
|
516
|
+
* complement to semanticSearch. Case-insensitive; ranks by number of
|
|
517
|
+
* occurrences (mirrors the in-app KB "test search" keyword mode). Returns
|
|
518
|
+
* the same row shape as semanticSearch (plus `match_count` / `match_type`).
|
|
519
|
+
* @param {string} query - Text to match within chunk content
|
|
520
|
+
* @param {Object} options - { limit = 10, document_id = null }
|
|
521
|
+
* @returns {Promise<Array>} Matching chunks, most matches first
|
|
522
|
+
*/
|
|
523
|
+
async keywordSearch(query, options = {}) {
|
|
524
|
+
const { limit = 10, document_id = null } = options;
|
|
525
|
+
if (!this.isConnected) {
|
|
526
|
+
await this.connect();
|
|
527
|
+
}
|
|
528
|
+
const q = String(query ?? '');
|
|
529
|
+
if (q.length === 0) return [];
|
|
530
|
+
|
|
531
|
+
let sql = `
|
|
532
|
+
SELECT
|
|
533
|
+
c.id,
|
|
534
|
+
c.document_id,
|
|
535
|
+
c.chunk_index,
|
|
536
|
+
c.content,
|
|
537
|
+
c.token_count,
|
|
538
|
+
c.chunk_type,
|
|
539
|
+
c.metadata,
|
|
540
|
+
c.embedding_model,
|
|
541
|
+
c.embedding_dimensions,
|
|
542
|
+
d.display_name as document_name,
|
|
543
|
+
d.file_type,
|
|
544
|
+
d.folder_path,
|
|
545
|
+
c.created_at
|
|
546
|
+
FROM chunks c
|
|
547
|
+
LEFT JOIN documents d ON c.document_id = d.id
|
|
548
|
+
WHERE c.content LIKE ? COLLATE NOCASE
|
|
549
|
+
`;
|
|
550
|
+
const params = [`%${q}%`];
|
|
551
|
+
if (document_id) {
|
|
552
|
+
sql += ' AND c.document_id = ?';
|
|
553
|
+
params.push(document_id);
|
|
554
|
+
}
|
|
555
|
+
|
|
556
|
+
// No SQL LIMIT: rank by occurrence count in JS, then slice.
|
|
557
|
+
const rows = this._all(sql, params);
|
|
558
|
+
const needle = q.toLowerCase();
|
|
559
|
+
const scored = rows.map((row) => {
|
|
560
|
+
const content = row.content ?? '';
|
|
561
|
+
const hay = content.toLowerCase();
|
|
562
|
+
let count = 0;
|
|
563
|
+
let idx = hay.indexOf(needle);
|
|
564
|
+
while (idx !== -1) {
|
|
565
|
+
count++;
|
|
566
|
+
idx = hay.indexOf(needle, idx + needle.length);
|
|
567
|
+
}
|
|
568
|
+
return {
|
|
569
|
+
...row,
|
|
570
|
+
metadata: row.metadata ? JSON.parse(row.metadata) : {},
|
|
571
|
+
match_count: count,
|
|
572
|
+
match_type: 'keyword'
|
|
573
|
+
};
|
|
574
|
+
});
|
|
575
|
+
scored.sort((a, b) => b.match_count - a.match_count || a.chunk_index - b.chunk_index);
|
|
576
|
+
return scored.slice(0, limit);
|
|
577
|
+
}
|
|
578
|
+
|
|
579
|
+
/**
|
|
580
|
+
* List documents in the knowledge base (navigation, not search).
|
|
581
|
+
* @param {Object} options - { limit = 100, offset = 0 }
|
|
582
|
+
* @returns {Promise<Array>} Document rows with a chunk_count each
|
|
583
|
+
*/
|
|
584
|
+
async listDocuments(options = {}) {
|
|
585
|
+
const { limit = 100, offset = 0 } = options;
|
|
586
|
+
if (!this.isConnected) {
|
|
587
|
+
await this.connect();
|
|
588
|
+
}
|
|
589
|
+
return this._all(
|
|
590
|
+
`SELECT
|
|
591
|
+
d.id, d.display_name, d.file_type, d.file_size, d.folder_path,
|
|
592
|
+
d.created_at, d.updated_at,
|
|
593
|
+
(SELECT COUNT(*) FROM chunks c WHERE c.document_id = d.id) AS chunk_count
|
|
594
|
+
FROM documents d
|
|
595
|
+
ORDER BY d.created_at ASC, d.display_name ASC
|
|
596
|
+
LIMIT ? OFFSET ?`,
|
|
597
|
+
[limit, offset],
|
|
598
|
+
);
|
|
599
|
+
}
|
|
600
|
+
|
|
601
|
+
/**
|
|
602
|
+
* Read chunks of a document in order, starting at a chunk index — for
|
|
603
|
+
* pagination and "read what came after chunk N" (e.g. following a
|
|
604
|
+
* similarity hit). Returns chunks with index >= startIndex, ascending.
|
|
605
|
+
* @param {string} documentId
|
|
606
|
+
* @param {Object} options - { startIndex = 0, limit = 10 }
|
|
607
|
+
* @returns {Promise<Array>} Chunk rows (metadata parsed)
|
|
608
|
+
*/
|
|
609
|
+
async getChunks(documentId, options = {}) {
|
|
610
|
+
const { startIndex = 0, limit = 10 } = options;
|
|
611
|
+
if (!this.isConnected) {
|
|
612
|
+
await this.connect();
|
|
613
|
+
}
|
|
614
|
+
const rows = this._all(
|
|
615
|
+
`SELECT id, document_id, chunk_index, content, token_count, chunk_type, metadata, created_at
|
|
616
|
+
FROM chunks
|
|
617
|
+
WHERE document_id = ? AND chunk_index >= ?
|
|
618
|
+
ORDER BY chunk_index ASC
|
|
619
|
+
LIMIT ?`,
|
|
620
|
+
[documentId, startIndex, limit],
|
|
621
|
+
);
|
|
622
|
+
return rows.map((r) => ({
|
|
623
|
+
...r,
|
|
624
|
+
metadata: r.metadata ? JSON.parse(r.metadata) : {},
|
|
625
|
+
}));
|
|
626
|
+
}
|
|
627
|
+
|
|
628
|
+
/**
|
|
629
|
+
* Read a window of chunks around a given chunk index — context expansion
|
|
630
|
+
* for a chunk found via search (grab the N before / after it).
|
|
631
|
+
* @param {string} documentId
|
|
632
|
+
* @param {number} chunkIndex - The anchor chunk's index
|
|
633
|
+
* @param {Object} options - { before = 1, after = 1 }
|
|
634
|
+
* @returns {Promise<Array>} Chunk rows in the window, ascending (metadata parsed)
|
|
635
|
+
*/
|
|
636
|
+
async getChunkWindow(documentId, chunkIndex, options = {}) {
|
|
637
|
+
const { before = 1, after = 1 } = options;
|
|
638
|
+
if (!this.isConnected) {
|
|
639
|
+
await this.connect();
|
|
640
|
+
}
|
|
641
|
+
const lo = Math.max(0, Number(chunkIndex) - before);
|
|
642
|
+
const hi = Number(chunkIndex) + after;
|
|
643
|
+
const rows = this._all(
|
|
644
|
+
`SELECT id, document_id, chunk_index, content, token_count, chunk_type, metadata, created_at
|
|
645
|
+
FROM chunks
|
|
646
|
+
WHERE document_id = ? AND chunk_index >= ? AND chunk_index <= ?
|
|
647
|
+
ORDER BY chunk_index ASC`,
|
|
648
|
+
[documentId, lo, hi],
|
|
649
|
+
);
|
|
650
|
+
return rows.map((r) => ({
|
|
651
|
+
...r,
|
|
652
|
+
metadata: r.metadata ? JSON.parse(r.metadata) : {},
|
|
653
|
+
}));
|
|
654
|
+
}
|
|
655
|
+
|
|
656
|
+
/**
|
|
657
|
+
* Describe the tabular tables in a Tabular→SQL KB, from the `_kb_tables`
|
|
658
|
+
* registry — everything an LLM needs to write correct SQL: the column
|
|
659
|
+
* schema, a `CREATE TABLE` DDL (the format models expect), and a few
|
|
660
|
+
* sample rows (which disambiguate value formats / casing far better than
|
|
661
|
+
* types alone). Returns [] for a KB with no registry (e.g. a docs KB).
|
|
662
|
+
* @param {Object} options - { sampleLimit = 5 } rows per table (0 = none)
|
|
663
|
+
* @returns {Promise<Array>} [{ table_name, source_name, row_count, columns:[{name,type}], ddl, sample_rows }]
|
|
664
|
+
*/
|
|
665
|
+
async listTables(options = {}) {
|
|
666
|
+
const { sampleLimit = 5 } = options;
|
|
667
|
+
if (!this.isConnected) {
|
|
668
|
+
await this.connect();
|
|
669
|
+
}
|
|
670
|
+
let rows;
|
|
671
|
+
try {
|
|
672
|
+
rows = this._all(
|
|
673
|
+
`SELECT table_name, source_name, row_count, columns
|
|
674
|
+
FROM _kb_tables
|
|
675
|
+
ORDER BY table_name ASC`,
|
|
676
|
+
);
|
|
677
|
+
} catch {
|
|
678
|
+
return []; // no _kb_tables registry — not a tabular KB
|
|
679
|
+
}
|
|
680
|
+
return rows.map((r) => {
|
|
681
|
+
const tableName = String(r.table_name ?? "");
|
|
682
|
+
const columns = r.columns ? JSON.parse(r.columns) : [];
|
|
683
|
+
let sample_rows = [];
|
|
684
|
+
// Identifiers come from our own sanitizer, but guard the
|
|
685
|
+
// interpolation before it reaches SQL anyway.
|
|
686
|
+
if (sampleLimit > 0 && /^[a-zA-Z0-9_]+$/.test(tableName)) {
|
|
687
|
+
try {
|
|
688
|
+
sample_rows = this._all(
|
|
689
|
+
`SELECT * FROM "${tableName}" LIMIT ?`,
|
|
690
|
+
[sampleLimit],
|
|
691
|
+
);
|
|
692
|
+
} catch {
|
|
693
|
+
sample_rows = [];
|
|
694
|
+
}
|
|
695
|
+
}
|
|
696
|
+
return {
|
|
697
|
+
table_name: r.table_name,
|
|
698
|
+
source_name: r.source_name,
|
|
699
|
+
row_count: r.row_count,
|
|
700
|
+
columns,
|
|
701
|
+
ddl: buildCreateTableDDL(tableName, columns),
|
|
702
|
+
sample_rows,
|
|
703
|
+
};
|
|
704
|
+
});
|
|
705
|
+
}
|
|
706
|
+
|
|
707
|
+
// ── Knowledge-graph traversal (Graph recipe, ADR 0023) ────────────
|
|
708
|
+
// Graph KBs carry `nodes` + `edges` tables (entities + typed
|
|
709
|
+
// relations, both with document provenance) alongside the same
|
|
710
|
+
// `_kb_tables` registry tabular uses — so `listTables` / `query`
|
|
711
|
+
// work on them unchanged; these methods add traversal sugar.
|
|
712
|
+
// Soft-fail convention matches listTables: a KB without graph
|
|
713
|
+
// tables returns empty shapes, never throws.
|
|
714
|
+
|
|
715
|
+
/**
|
|
716
|
+
* List entities, highest-degree first (degree = count of edges in
|
|
717
|
+
* either direction), so the "important" nodes surface first.
|
|
718
|
+
* @param {Object} options - { limit=100, offset=0, type=null, search=null }
|
|
719
|
+
* @returns {Promise<Array>} [{ id, name, type, description, document_id, degree, created_at }]
|
|
720
|
+
*/
|
|
721
|
+
async listEntities(options = {}) {
|
|
722
|
+
const { limit = 100, offset = 0, type = null, search = null } = options;
|
|
723
|
+
if (!this.isConnected) {
|
|
724
|
+
await this.connect();
|
|
725
|
+
}
|
|
726
|
+
try {
|
|
727
|
+
const where = [];
|
|
728
|
+
const params = [];
|
|
729
|
+
if (type) {
|
|
730
|
+
where.push("n.type = ? COLLATE NOCASE");
|
|
731
|
+
params.push(type);
|
|
732
|
+
}
|
|
733
|
+
if (search) {
|
|
734
|
+
where.push("n.name LIKE ? COLLATE NOCASE");
|
|
735
|
+
params.push(`%${search}%`);
|
|
736
|
+
}
|
|
737
|
+
return this._all(
|
|
738
|
+
`SELECT n.*, (
|
|
739
|
+
SELECT COUNT(*) FROM edges e
|
|
740
|
+
WHERE e.source_id = n.id OR e.target_id = n.id
|
|
741
|
+
) AS degree
|
|
742
|
+
FROM nodes n
|
|
743
|
+
${where.length ? `WHERE ${where.join(" AND ")}` : ""}
|
|
744
|
+
ORDER BY degree DESC, n.name COLLATE NOCASE ASC
|
|
745
|
+
LIMIT ? OFFSET ?`,
|
|
746
|
+
[...params, limit, offset],
|
|
747
|
+
);
|
|
748
|
+
} catch {
|
|
749
|
+
return []; // no nodes table — not a graph KB
|
|
750
|
+
}
|
|
751
|
+
}
|
|
752
|
+
|
|
753
|
+
/**
|
|
754
|
+
* Resolve an entity reference — exact id first, then
|
|
755
|
+
* case-insensitive name. Returns the node row or null.
|
|
756
|
+
*/
|
|
757
|
+
async getEntity(ref) {
|
|
758
|
+
if (!this.isConnected) {
|
|
759
|
+
await this.connect();
|
|
760
|
+
}
|
|
761
|
+
if (typeof ref !== "string" || ref.length === 0) return null;
|
|
762
|
+
try {
|
|
763
|
+
return (
|
|
764
|
+
this._get(`SELECT * FROM nodes WHERE id = ?`, [ref]) ||
|
|
765
|
+
this._get(
|
|
766
|
+
`SELECT * FROM nodes WHERE name = ? COLLATE NOCASE`,
|
|
767
|
+
[ref],
|
|
768
|
+
) ||
|
|
769
|
+
null
|
|
770
|
+
);
|
|
771
|
+
} catch {
|
|
772
|
+
return null;
|
|
773
|
+
}
|
|
774
|
+
}
|
|
775
|
+
|
|
776
|
+
/**
|
|
777
|
+
* Edges touching an entity, each joined with the node on the far
|
|
778
|
+
* side. `direction` filters to edges where the entity is the
|
|
779
|
+
* source ("out"), the target ("in"), or either ("both").
|
|
780
|
+
* @param {string} entityRef - entity id or (case-insensitive) name
|
|
781
|
+
* @param {Object} options - { direction="both", relation_type=null, limit=50 }
|
|
782
|
+
* @returns {Promise<Object>} { entity, neighbors: [{ edge_id, relation_type, description, direction, node }] }
|
|
783
|
+
*/
|
|
784
|
+
async getNeighbors(entityRef, options = {}) {
|
|
785
|
+
const { direction = "both", relation_type = null, limit = 50 } = options;
|
|
786
|
+
const entity = await this.getEntity(entityRef);
|
|
787
|
+
if (!entity) return { entity: null, neighbors: [] };
|
|
788
|
+
try {
|
|
789
|
+
let directionSql;
|
|
790
|
+
if (direction === "out") directionSql = "e.source_id = ?";
|
|
791
|
+
else if (direction === "in") directionSql = "e.target_id = ?";
|
|
792
|
+
else directionSql = "(e.source_id = ? OR e.target_id = ?)";
|
|
793
|
+
const directionParams =
|
|
794
|
+
direction === "out" || direction === "in"
|
|
795
|
+
? [entity.id]
|
|
796
|
+
: [entity.id, entity.id];
|
|
797
|
+
const typeSql = relation_type
|
|
798
|
+
? " AND e.type = ? COLLATE NOCASE"
|
|
799
|
+
: "";
|
|
800
|
+
const rows = this._all(
|
|
801
|
+
`SELECT e.id AS edge_id, e.type AS relation_type,
|
|
802
|
+
e.description, e.source_id, e.target_id,
|
|
803
|
+
n.id AS node_id, n.name AS node_name,
|
|
804
|
+
n.type AS node_type, n.description AS node_description
|
|
805
|
+
FROM edges e
|
|
806
|
+
JOIN nodes n ON n.id = CASE
|
|
807
|
+
WHEN e.source_id = ? THEN e.target_id ELSE e.source_id
|
|
808
|
+
END
|
|
809
|
+
WHERE ${directionSql}${typeSql}
|
|
810
|
+
LIMIT ?`,
|
|
811
|
+
[
|
|
812
|
+
entity.id,
|
|
813
|
+
...directionParams,
|
|
814
|
+
...(relation_type ? [relation_type] : []),
|
|
815
|
+
limit,
|
|
816
|
+
],
|
|
817
|
+
);
|
|
818
|
+
return {
|
|
819
|
+
entity,
|
|
820
|
+
neighbors: rows.map((r) => ({
|
|
821
|
+
edge_id: r.edge_id,
|
|
822
|
+
relation_type: r.relation_type,
|
|
823
|
+
description: r.description,
|
|
824
|
+
direction: r.source_id === entity.id ? "out" : "in",
|
|
825
|
+
node: {
|
|
826
|
+
id: r.node_id,
|
|
827
|
+
name: r.node_name,
|
|
828
|
+
type: r.node_type,
|
|
829
|
+
description: r.node_description,
|
|
830
|
+
},
|
|
831
|
+
})),
|
|
832
|
+
};
|
|
833
|
+
} catch {
|
|
834
|
+
return { entity, neighbors: [] };
|
|
835
|
+
}
|
|
836
|
+
}
|
|
837
|
+
|
|
838
|
+
/**
|
|
839
|
+
* Shortest path between two entities — BFS over edges, direction-
|
|
840
|
+
* agnostic (relations read both ways for pathfinding). Edge count
|
|
841
|
+
* is capped so a runaway artifact can't wedge a run.
|
|
842
|
+
* @param {string} fromRef - entity id or (case-insensitive) name
|
|
843
|
+
* @param {string} toRef - entity id or (case-insensitive) name
|
|
844
|
+
* @param {Object} options - { max_depth=4, max_edges=50000 }
|
|
845
|
+
* @returns {Promise<Object>} { found, from, to, hops, steps: [{ node, via_edge? }] }
|
|
846
|
+
*/
|
|
847
|
+
async findPath(fromRef, toRef, options = {}) {
|
|
848
|
+
const { max_depth = 4, max_edges = 50000 } = options;
|
|
849
|
+
const from = await this.getEntity(fromRef);
|
|
850
|
+
const to = await this.getEntity(toRef);
|
|
851
|
+
if (!from || !to) {
|
|
852
|
+
return { found: false, from, to, hops: 0, steps: [] };
|
|
853
|
+
}
|
|
854
|
+
if (from.id === to.id) {
|
|
855
|
+
return { found: true, from, to, hops: 0, steps: [{ node: from }] };
|
|
856
|
+
}
|
|
857
|
+
let edges;
|
|
858
|
+
try {
|
|
859
|
+
edges = this._all(
|
|
860
|
+
`SELECT id, source_id, target_id, type FROM edges LIMIT ?`,
|
|
861
|
+
[max_edges],
|
|
862
|
+
);
|
|
863
|
+
} catch {
|
|
864
|
+
return { found: false, from, to, hops: 0, steps: [] };
|
|
865
|
+
}
|
|
866
|
+
const adjacency = new Map();
|
|
867
|
+
for (const e of edges) {
|
|
868
|
+
if (!adjacency.has(e.source_id)) adjacency.set(e.source_id, []);
|
|
869
|
+
if (!adjacency.has(e.target_id)) adjacency.set(e.target_id, []);
|
|
870
|
+
adjacency.get(e.source_id).push({ next: e.target_id, edge: e });
|
|
871
|
+
adjacency.get(e.target_id).push({ next: e.source_id, edge: e });
|
|
872
|
+
}
|
|
873
|
+
// BFS with parent pointers.
|
|
874
|
+
const visited = new Map([[from.id, null]]);
|
|
875
|
+
let frontier = [from.id];
|
|
876
|
+
for (let depth = 0; depth < max_depth && frontier.length > 0; depth++) {
|
|
877
|
+
const next = [];
|
|
878
|
+
for (const nodeId of frontier) {
|
|
879
|
+
for (const { next: nextId, edge } of adjacency.get(nodeId) ?? []) {
|
|
880
|
+
if (visited.has(nextId)) continue;
|
|
881
|
+
visited.set(nextId, { prev: nodeId, edge });
|
|
882
|
+
if (nextId === to.id) {
|
|
883
|
+
return this._materializePath(from, to, visited);
|
|
884
|
+
}
|
|
885
|
+
next.push(nextId);
|
|
886
|
+
}
|
|
887
|
+
}
|
|
888
|
+
frontier = next;
|
|
889
|
+
}
|
|
890
|
+
return { found: false, from, to, hops: 0, steps: [] };
|
|
891
|
+
}
|
|
892
|
+
|
|
893
|
+
/** Walk parent pointers back from `to`, hydrating node rows. */
|
|
894
|
+
_materializePath(from, to, visited) {
|
|
895
|
+
const reversed = [];
|
|
896
|
+
let cursor = to.id;
|
|
897
|
+
while (cursor !== from.id) {
|
|
898
|
+
const link = visited.get(cursor);
|
|
899
|
+
reversed.push({ nodeId: cursor, edge: link.edge });
|
|
900
|
+
cursor = link.prev;
|
|
901
|
+
}
|
|
902
|
+
const steps = [{ node: from }];
|
|
903
|
+
for (const { nodeId, edge } of reversed.reverse()) {
|
|
904
|
+
const node =
|
|
905
|
+
this._get(`SELECT * FROM nodes WHERE id = ?`, [nodeId]) ?? {
|
|
906
|
+
id: nodeId,
|
|
907
|
+
};
|
|
908
|
+
steps.push({
|
|
909
|
+
node,
|
|
910
|
+
via_edge: {
|
|
911
|
+
id: edge.id,
|
|
912
|
+
type: edge.type,
|
|
913
|
+
source_id: edge.source_id,
|
|
914
|
+
target_id: edge.target_id,
|
|
915
|
+
},
|
|
916
|
+
});
|
|
917
|
+
}
|
|
918
|
+
return { found: true, from, to, hops: steps.length - 1, steps };
|
|
919
|
+
}
|
|
920
|
+
}
|
|
921
|
+
|
|
922
|
+
/** A `CREATE TABLE` statement for a KB table's columns — the schema format
|
|
923
|
+
* LLMs are trained on for text-to-SQL. */
|
|
924
|
+
function buildCreateTableDDL(name, columns) {
|
|
925
|
+
const cols = (columns || [])
|
|
926
|
+
.map((c) => ` "${c.name}" ${c.type}`)
|
|
927
|
+
.join(",\n");
|
|
928
|
+
return `CREATE TABLE "${name}" (\n${cols}\n);`;
|
|
477
929
|
}
|