@diegoaltoworks/chatter 2.0.0 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/chunking.d.ts +31 -0
- package/dist/core/chunking.d.ts.map +1 -0
- package/dist/core/retrieval.d.ts +39 -0
- package/dist/core/retrieval.d.ts.map +1 -1
- package/dist/index.d.ts +3 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +211 -25
- package/dist/index.mjs +205 -20
- package/dist/mcp-server.d.ts.map +1 -1
- package/dist/mcp-server.js +201 -19
- package/dist/mcp-server.mjs +200 -18
- package/dist/personas/index.js +4 -2
- package/dist/personas/index.mjs +4 -2
- package/dist/personas/timeContext.d.ts.map +1 -1
- package/dist/server.d.ts.map +1 -1
- package/dist/server.js +207 -25
- package/dist/server.mjs +202 -20
- package/dist/types.d.ts +7 -0
- package/dist/types.d.ts.map +1 -1
- package/package.json +10 -10
package/dist/index.mjs
CHANGED
|
@@ -72,6 +72,126 @@ var init_buildLock = __esm({
|
|
|
72
72
|
}
|
|
73
73
|
});
|
|
74
74
|
|
|
75
|
+
// src/core/chunking.ts
|
|
76
|
+
function headingOf(line) {
|
|
77
|
+
const match = HEADING.exec(line);
|
|
78
|
+
if (!match) return void 0;
|
|
79
|
+
const title = (match[2] ?? "").replace(/[ \t]+#+$/, "").replace(/^#+$/, "").trim();
|
|
80
|
+
return { level: match[1].length, title };
|
|
81
|
+
}
|
|
82
|
+
function startsTable(lines, i) {
|
|
83
|
+
return lines[i].includes("|") && i + 1 < lines.length && TABLE_DELIMITER.test(lines[i + 1]);
|
|
84
|
+
}
|
|
85
|
+
function fenceEnd(lines, start) {
|
|
86
|
+
const opener = FENCE.exec(lines[start]);
|
|
87
|
+
if (!opener) return start + 1;
|
|
88
|
+
const marker = opener[1];
|
|
89
|
+
const closer = new RegExp(`^ {0,3}${marker[0] === "`" ? "`" : "~"}{${marker.length},}[ \\t]*$`);
|
|
90
|
+
for (let i = start + 1; i < lines.length; i++) if (closer.test(lines[i])) return i + 1;
|
|
91
|
+
return lines.length;
|
|
92
|
+
}
|
|
93
|
+
function listEnd(lines, start) {
|
|
94
|
+
let end = start + 1;
|
|
95
|
+
for (let i = start + 1; i < lines.length; i++) {
|
|
96
|
+
const line = lines[i];
|
|
97
|
+
if (isBlank(line)) {
|
|
98
|
+
let next = i + 1;
|
|
99
|
+
while (next < lines.length && isBlank(lines[next])) next++;
|
|
100
|
+
if (next < lines.length && (LIST_ITEM.test(lines[next]) || isIndented(lines[next]))) {
|
|
101
|
+
i = next - 1;
|
|
102
|
+
continue;
|
|
103
|
+
}
|
|
104
|
+
break;
|
|
105
|
+
}
|
|
106
|
+
if (headingOf(line) && !isIndented(line)) break;
|
|
107
|
+
if (FENCE.test(line) && !isIndented(line)) break;
|
|
108
|
+
end = i + 1;
|
|
109
|
+
}
|
|
110
|
+
return end;
|
|
111
|
+
}
|
|
112
|
+
function tableEnd(lines, start) {
|
|
113
|
+
let end = start + 2;
|
|
114
|
+
while (end < lines.length && !isBlank(lines[end]) && lines[end].includes("|")) end++;
|
|
115
|
+
return end;
|
|
116
|
+
}
|
|
117
|
+
function paragraphEnd(lines, start) {
|
|
118
|
+
let end = start + 1;
|
|
119
|
+
while (end < lines.length && !isBlank(lines[end]) && !headingOf(lines[end]) && !FENCE.test(lines[end]) && !LIST_ITEM.test(lines[end])) {
|
|
120
|
+
end++;
|
|
121
|
+
}
|
|
122
|
+
return end;
|
|
123
|
+
}
|
|
124
|
+
function scan(text) {
|
|
125
|
+
const lines = text.replace(/\r\n?/g, "\n").split("\n");
|
|
126
|
+
const sections = [{ trail: [], blocks: [] }];
|
|
127
|
+
const stack = [];
|
|
128
|
+
let i = 0;
|
|
129
|
+
while (i < lines.length) {
|
|
130
|
+
const line = lines[i];
|
|
131
|
+
if (isBlank(line)) {
|
|
132
|
+
i++;
|
|
133
|
+
continue;
|
|
134
|
+
}
|
|
135
|
+
const heading = FENCE.test(line) ? void 0 : headingOf(line);
|
|
136
|
+
if (heading) {
|
|
137
|
+
while (stack.length && stack[stack.length - 1].level >= heading.level) stack.pop();
|
|
138
|
+
stack.push(heading);
|
|
139
|
+
sections.push({ trail: stack.map((h) => h.title), heading: line, blocks: [] });
|
|
140
|
+
i++;
|
|
141
|
+
continue;
|
|
142
|
+
}
|
|
143
|
+
let end;
|
|
144
|
+
if (FENCE.test(line)) end = fenceEnd(lines, i);
|
|
145
|
+
else if (LIST_ITEM.test(line)) end = listEnd(lines, i);
|
|
146
|
+
else if (startsTable(lines, i)) end = tableEnd(lines, i);
|
|
147
|
+
else end = paragraphEnd(lines, i);
|
|
148
|
+
sections[sections.length - 1].blocks.push({ lines: lines.slice(i, end) });
|
|
149
|
+
i = end;
|
|
150
|
+
}
|
|
151
|
+
return sections;
|
|
152
|
+
}
|
|
153
|
+
function chunkSections(text, max = 900) {
|
|
154
|
+
const cap = Math.max(1, max);
|
|
155
|
+
const out = [];
|
|
156
|
+
const emit = (parts, section) => {
|
|
157
|
+
const body = parts.join("\n\n");
|
|
158
|
+
if (body.trim() === "") return;
|
|
159
|
+
out.push({ text: body, section: [...section], position: out.length });
|
|
160
|
+
};
|
|
161
|
+
for (const section of scan(text)) {
|
|
162
|
+
let parts = section.heading ? [section.heading.trim()] : [];
|
|
163
|
+
let size = parts.join("\n\n").length;
|
|
164
|
+
let hasBlock = false;
|
|
165
|
+
for (const block of section.blocks) {
|
|
166
|
+
const piece = block.lines.join("\n");
|
|
167
|
+
const added = piece.length + (parts.length ? 2 : 0);
|
|
168
|
+
if (hasBlock && size + added > cap) {
|
|
169
|
+
emit(parts, section.trail);
|
|
170
|
+
parts = [];
|
|
171
|
+
size = 0;
|
|
172
|
+
hasBlock = false;
|
|
173
|
+
}
|
|
174
|
+
size += piece.length + (parts.length ? 2 : 0);
|
|
175
|
+
parts.push(piece);
|
|
176
|
+
hasBlock = true;
|
|
177
|
+
}
|
|
178
|
+
emit(parts, section.trail);
|
|
179
|
+
}
|
|
180
|
+
return out;
|
|
181
|
+
}
|
|
182
|
+
var FENCE, HEADING, LIST_ITEM, TABLE_DELIMITER, isBlank, isIndented;
|
|
183
|
+
var init_chunking = __esm({
|
|
184
|
+
"src/core/chunking.ts"() {
|
|
185
|
+
"use strict";
|
|
186
|
+
FENCE = /^ {0,3}(`{3,}|~{3,})/;
|
|
187
|
+
HEADING = /^ {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*$/;
|
|
188
|
+
LIST_ITEM = /^[ \t]*(?:[-*+]|\d{1,9}[.)])[ \t]+\S/;
|
|
189
|
+
TABLE_DELIMITER = /^[ \t]*\|?[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?[ \t]*$/;
|
|
190
|
+
isBlank = (line) => line.trim() === "";
|
|
191
|
+
isIndented = (line) => /^(?: {2,}|\t)/.test(line);
|
|
192
|
+
}
|
|
193
|
+
});
|
|
194
|
+
|
|
75
195
|
// src/core/loaders.ts
|
|
76
196
|
import { readdirSync, readFileSync, statSync } from "node:fs";
|
|
77
197
|
import { join } from "node:path";
|
|
@@ -139,6 +259,7 @@ __export(retrieval_exports, {
|
|
|
139
259
|
openLibsqlClient: () => openLibsqlClient,
|
|
140
260
|
wrapMissingLibsqlError: () => wrapMissingLibsqlError
|
|
141
261
|
});
|
|
262
|
+
import { relative, sep } from "node:path";
|
|
142
263
|
function createOpenAIEmbedder(client) {
|
|
143
264
|
return async (input) => {
|
|
144
265
|
const res = await client.embeddings.create({ model: EMB_MODEL, input });
|
|
@@ -181,6 +302,7 @@ var init_retrieval = __esm({
|
|
|
181
302
|
"src/core/retrieval.ts"() {
|
|
182
303
|
"use strict";
|
|
183
304
|
init_buildLock();
|
|
305
|
+
init_chunking();
|
|
184
306
|
init_loaders();
|
|
185
307
|
init_logger();
|
|
186
308
|
EMB_MODEL = "text-embedding-3-large";
|
|
@@ -191,6 +313,7 @@ var init_retrieval = __esm({
|
|
|
191
313
|
this.embed = embed;
|
|
192
314
|
this.db = options.databaseClient;
|
|
193
315
|
this.knowledgeDir = options.knowledgeDir || DEFAULT_KNOWLEDGE_DIR;
|
|
316
|
+
this.chunking = options.chunking ?? "lines";
|
|
194
317
|
this.logger = options.logger ?? createConsoleLogger();
|
|
195
318
|
this.buildLock = options.buildLock ?? createTursoBuildLock(this.db);
|
|
196
319
|
this.instanceId = options.instanceId ?? crypto.randomUUID();
|
|
@@ -204,6 +327,7 @@ var init_retrieval = __esm({
|
|
|
204
327
|
*/
|
|
205
328
|
db;
|
|
206
329
|
knowledgeDir;
|
|
330
|
+
chunking;
|
|
207
331
|
logger;
|
|
208
332
|
buildLock;
|
|
209
333
|
instanceId;
|
|
@@ -264,11 +388,41 @@ var init_retrieval = __esm({
|
|
|
264
388
|
this.logger.info("No knowledge documents and no existing chunks - nothing to build");
|
|
265
389
|
return;
|
|
266
390
|
}
|
|
391
|
+
this.logger.info(`Chunking mode: ${this.chunking}`);
|
|
392
|
+
if (this.chunking === "sections") await this.ensureSectionColumns();
|
|
267
393
|
const rows = [];
|
|
268
394
|
for (const d of docs) {
|
|
269
|
-
|
|
270
|
-
const
|
|
271
|
-
|
|
395
|
+
if (this.chunking === "sections") {
|
|
396
|
+
const source = relative(this.knowledgeDir, d.source).split(sep).join("/");
|
|
397
|
+
for (const c of chunkSections(d.text)) {
|
|
398
|
+
const section = c.section.join(" > ");
|
|
399
|
+
const id = await sha256(`${d.bucket}|${source}|${section}|${c.text}`);
|
|
400
|
+
const context = section ? `${source} > ${section}` : source;
|
|
401
|
+
rows.push({
|
|
402
|
+
id,
|
|
403
|
+
bucket: d.bucket,
|
|
404
|
+
source,
|
|
405
|
+
text: c.text,
|
|
406
|
+
embedText: `${context}
|
|
407
|
+
|
|
408
|
+
${c.text}`,
|
|
409
|
+
section,
|
|
410
|
+
position: c.position
|
|
411
|
+
});
|
|
412
|
+
}
|
|
413
|
+
} else {
|
|
414
|
+
for (const part of chunk(d.text)) {
|
|
415
|
+
const id = await sha256(`${d.bucket}|${d.source}|${part}`);
|
|
416
|
+
rows.push({
|
|
417
|
+
id,
|
|
418
|
+
bucket: d.bucket,
|
|
419
|
+
source: d.source,
|
|
420
|
+
text: part,
|
|
421
|
+
embedText: part,
|
|
422
|
+
section: null,
|
|
423
|
+
position: null
|
|
424
|
+
});
|
|
425
|
+
}
|
|
272
426
|
}
|
|
273
427
|
}
|
|
274
428
|
this.logger.info(`Created ${rows.length} chunks from knowledge documents`);
|
|
@@ -277,11 +431,17 @@ var init_retrieval = __esm({
|
|
|
277
431
|
for (let i = 0; i < rows.length; i += UPSERT_BATCH) {
|
|
278
432
|
const batch = rows.slice(i, i + UPSERT_BATCH);
|
|
279
433
|
await this.db.batch(
|
|
280
|
-
batch.map(
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
434
|
+
batch.map(
|
|
435
|
+
(r) => this.chunking === "sections" ? {
|
|
436
|
+
sql: `INSERT INTO chunks(id,bucket,source,text,section,position) VALUES(?,?,?,?,?,?)
|
|
437
|
+
ON CONFLICT(id) DO NOTHING`,
|
|
438
|
+
args: [r.id, r.bucket, r.source, r.text, r.section, r.position]
|
|
439
|
+
} : {
|
|
440
|
+
sql: `INSERT INTO chunks(id,bucket,source,text) VALUES(?,?,?,?)
|
|
441
|
+
ON CONFLICT(id) DO NOTHING`,
|
|
442
|
+
args: [r.id, r.bucket, r.source, r.text]
|
|
443
|
+
}
|
|
444
|
+
),
|
|
285
445
|
"write"
|
|
286
446
|
);
|
|
287
447
|
}
|
|
@@ -302,7 +462,7 @@ var init_retrieval = __esm({
|
|
|
302
462
|
return;
|
|
303
463
|
}
|
|
304
464
|
this.logger.info(`Embedding ${missing.length} new/updated chunks with ${EMB_MODEL}...`);
|
|
305
|
-
const textById = new Map(rows.map((r) => [r.id, r.
|
|
465
|
+
const textById = new Map(rows.map((r) => [r.id, r.embedText]));
|
|
306
466
|
const BATCH = 96;
|
|
307
467
|
for (let i = 0; i < missing.length; i += BATCH) {
|
|
308
468
|
const batchIds = missing.slice(i, i + BATCH);
|
|
@@ -323,6 +483,18 @@ var init_retrieval = __esm({
|
|
|
323
483
|
}
|
|
324
484
|
this.logger.info(`Successfully embedded ${missing.length} new chunks`);
|
|
325
485
|
}
|
|
486
|
+
/**
|
|
487
|
+
* Adds the nullable `section`/`position` columns to a `chunks` table that
|
|
488
|
+
* every released version created without them. Checked via table_info so it
|
|
489
|
+
* is a no-op once present; only `'sections'` mode ever calls it.
|
|
490
|
+
*/
|
|
491
|
+
async ensureSectionColumns() {
|
|
492
|
+
const info = await this.db.execute("PRAGMA table_info(chunks)");
|
|
493
|
+
const have = new Set(info.rows.map((r) => String(r.name)));
|
|
494
|
+
if (!have.has("section")) await this.db.execute("ALTER TABLE chunks ADD COLUMN section TEXT");
|
|
495
|
+
if (!have.has("position"))
|
|
496
|
+
await this.db.execute("ALTER TABLE chunks ADD COLUMN position INTEGER");
|
|
497
|
+
}
|
|
326
498
|
// Remove chunks from database that no longer exist in markdown files
|
|
327
499
|
async cleanupStaleChunks(currentIds) {
|
|
328
500
|
const result = await this.db.execute("SELECT id FROM chunks");
|
|
@@ -369,25 +541,34 @@ var init_retrieval = __esm({
|
|
|
369
541
|
* interpolated. An empty list retrieves nothing, and short-circuits before
|
|
370
542
|
* the embedding call.
|
|
371
543
|
*/
|
|
372
|
-
async
|
|
544
|
+
async queryChunks(q, k = 6, allowed = ["base"]) {
|
|
373
545
|
if (allowed.length === 0) return [];
|
|
374
546
|
const [qv] = await this.embed([q]);
|
|
375
547
|
const placeholders = allowed.map(() => "?").join(",");
|
|
376
548
|
const res = await this.db.execute({
|
|
377
|
-
sql: `SELECT c
|
|
549
|
+
sql: `SELECT c.*, e.embedding
|
|
378
550
|
FROM chunks c
|
|
379
551
|
JOIN embeddings e ON e.id = c.id
|
|
380
552
|
WHERE c.bucket IN (${placeholders})`,
|
|
381
553
|
args: allowed
|
|
382
554
|
});
|
|
383
|
-
const scored =
|
|
384
|
-
for (const row of res.rows) {
|
|
555
|
+
const scored = res.rows.map((row) => {
|
|
385
556
|
const emb = JSON.parse(String(row.embedding));
|
|
386
|
-
const
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
557
|
+
const trail = row.section == null ? "" : String(row.section);
|
|
558
|
+
return {
|
|
559
|
+
text: String(row.text),
|
|
560
|
+
bucket: String(row.bucket ?? ""),
|
|
561
|
+
source: String(row.source ?? ""),
|
|
562
|
+
section: trail ? trail.split(" > ") : [],
|
|
563
|
+
score: _VectorStore.cosine(qv, emb)
|
|
564
|
+
};
|
|
565
|
+
});
|
|
566
|
+
scored.sort((a, b) => b.score - a.score);
|
|
567
|
+
return scored.slice(0, k);
|
|
568
|
+
}
|
|
569
|
+
/** Same results as {@link queryChunks}, as bare text. */
|
|
570
|
+
async query(q, k = 6, allowed = ["base"]) {
|
|
571
|
+
return (await this.queryChunks(q, k, allowed)).map((c) => c.text);
|
|
391
572
|
}
|
|
392
573
|
};
|
|
393
574
|
}
|
|
@@ -790,6 +971,7 @@ async function resolveBuckets({
|
|
|
790
971
|
|
|
791
972
|
// src/index.ts
|
|
792
973
|
init_buildLock();
|
|
974
|
+
init_chunking();
|
|
793
975
|
init_loaders();
|
|
794
976
|
init_logger();
|
|
795
977
|
|
|
@@ -1278,6 +1460,7 @@ async function createMCPServer(config) {
|
|
|
1278
1460
|
store = new VectorStore2(createOpenAIEmbedder2(client), {
|
|
1279
1461
|
databaseClient: db,
|
|
1280
1462
|
knowledgeDir: config.knowledgeDir || DEFAULT_KNOWLEDGE_DIR2,
|
|
1463
|
+
chunking: config.chunking,
|
|
1281
1464
|
logger: log
|
|
1282
1465
|
});
|
|
1283
1466
|
}
|
|
@@ -2325,7 +2508,7 @@ function publicRoutes(deps) {
|
|
|
2325
2508
|
}
|
|
2326
2509
|
|
|
2327
2510
|
// src/server.ts
|
|
2328
|
-
import { relative } from "node:path";
|
|
2511
|
+
import { relative as relative2 } from "node:path";
|
|
2329
2512
|
import { Hono as Hono5 } from "hono";
|
|
2330
2513
|
import OpenAI2 from "openai";
|
|
2331
2514
|
|
|
@@ -2484,6 +2667,7 @@ async function createServer(config) {
|
|
|
2484
2667
|
store = new VectorStore2(createOpenAIEmbedder2(client), {
|
|
2485
2668
|
databaseClient: db,
|
|
2486
2669
|
knowledgeDir: config.knowledgeDir || DEFAULT_KNOWLEDGE_DIR2,
|
|
2670
|
+
chunking: config.chunking,
|
|
2487
2671
|
logger
|
|
2488
2672
|
});
|
|
2489
2673
|
}
|
|
@@ -2534,7 +2718,7 @@ async function createServer(config) {
|
|
|
2534
2718
|
}
|
|
2535
2719
|
}
|
|
2536
2720
|
if (serveStatic && staticDir) {
|
|
2537
|
-
const relativePath =
|
|
2721
|
+
const relativePath = relative2(process.cwd(), staticDir);
|
|
2538
2722
|
app.get("/chatter.js", serveStatic({ path: `${relativePath}/chatter.js` }));
|
|
2539
2723
|
app.get("/chatter.css", serveStatic({ path: `${relativePath}/chatter.css` }));
|
|
2540
2724
|
}
|
|
@@ -2695,6 +2879,7 @@ export {
|
|
|
2695
2879
|
applyTransformReply,
|
|
2696
2880
|
canAcquireBuildLock,
|
|
2697
2881
|
chatBodyLimit,
|
|
2882
|
+
chunkSections,
|
|
2698
2883
|
completeOnce,
|
|
2699
2884
|
completeStream,
|
|
2700
2885
|
cors,
|
package/dist/mcp-server.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"mcp-server.d.ts","sourceRoot":"","sources":["../src/mcp-server.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AAsBH,YAAY,EACV,WAAW,EACX,QAAQ,EACR,cAAc,EACd,gBAAgB,EAChB,aAAa,EACb,gBAAgB,GACjB,MAAM,oBAAoB,CAAC;AAE5B,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,oBAAoB,CAAC;AAE3D;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAsB,eAAe,CAAC,MAAM,EAAE,gBAAgB;
|
|
1
|
+
{"version":3,"file":"mcp-server.d.ts","sourceRoot":"","sources":["../src/mcp-server.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AAsBH,YAAY,EACV,WAAW,EACX,QAAQ,EACR,cAAc,EACd,gBAAgB,EAChB,aAAa,EACb,gBAAgB,GACjB,MAAM,oBAAoB,CAAC;AAE5B,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,oBAAoB,CAAC;AAE3D;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAsB,eAAe,CAAC,MAAM,EAAE,gBAAgB;qCA+U3B,IAAI;GAEtC"}
|
package/dist/mcp-server.js
CHANGED
|
@@ -114,6 +114,126 @@ var init_buildLock = __esm({
|
|
|
114
114
|
}
|
|
115
115
|
});
|
|
116
116
|
|
|
117
|
+
// src/core/chunking.ts
|
|
118
|
+
function headingOf(line) {
|
|
119
|
+
const match = HEADING.exec(line);
|
|
120
|
+
if (!match) return void 0;
|
|
121
|
+
const title = (match[2] ?? "").replace(/[ \t]+#+$/, "").replace(/^#+$/, "").trim();
|
|
122
|
+
return { level: match[1].length, title };
|
|
123
|
+
}
|
|
124
|
+
function startsTable(lines, i) {
|
|
125
|
+
return lines[i].includes("|") && i + 1 < lines.length && TABLE_DELIMITER.test(lines[i + 1]);
|
|
126
|
+
}
|
|
127
|
+
function fenceEnd(lines, start) {
|
|
128
|
+
const opener = FENCE.exec(lines[start]);
|
|
129
|
+
if (!opener) return start + 1;
|
|
130
|
+
const marker = opener[1];
|
|
131
|
+
const closer = new RegExp(`^ {0,3}${marker[0] === "`" ? "`" : "~"}{${marker.length},}[ \\t]*$`);
|
|
132
|
+
for (let i = start + 1; i < lines.length; i++) if (closer.test(lines[i])) return i + 1;
|
|
133
|
+
return lines.length;
|
|
134
|
+
}
|
|
135
|
+
function listEnd(lines, start) {
|
|
136
|
+
let end = start + 1;
|
|
137
|
+
for (let i = start + 1; i < lines.length; i++) {
|
|
138
|
+
const line = lines[i];
|
|
139
|
+
if (isBlank(line)) {
|
|
140
|
+
let next = i + 1;
|
|
141
|
+
while (next < lines.length && isBlank(lines[next])) next++;
|
|
142
|
+
if (next < lines.length && (LIST_ITEM.test(lines[next]) || isIndented(lines[next]))) {
|
|
143
|
+
i = next - 1;
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
break;
|
|
147
|
+
}
|
|
148
|
+
if (headingOf(line) && !isIndented(line)) break;
|
|
149
|
+
if (FENCE.test(line) && !isIndented(line)) break;
|
|
150
|
+
end = i + 1;
|
|
151
|
+
}
|
|
152
|
+
return end;
|
|
153
|
+
}
|
|
154
|
+
function tableEnd(lines, start) {
|
|
155
|
+
let end = start + 2;
|
|
156
|
+
while (end < lines.length && !isBlank(lines[end]) && lines[end].includes("|")) end++;
|
|
157
|
+
return end;
|
|
158
|
+
}
|
|
159
|
+
function paragraphEnd(lines, start) {
|
|
160
|
+
let end = start + 1;
|
|
161
|
+
while (end < lines.length && !isBlank(lines[end]) && !headingOf(lines[end]) && !FENCE.test(lines[end]) && !LIST_ITEM.test(lines[end])) {
|
|
162
|
+
end++;
|
|
163
|
+
}
|
|
164
|
+
return end;
|
|
165
|
+
}
|
|
166
|
+
function scan(text) {
|
|
167
|
+
const lines = text.replace(/\r\n?/g, "\n").split("\n");
|
|
168
|
+
const sections = [{ trail: [], blocks: [] }];
|
|
169
|
+
const stack = [];
|
|
170
|
+
let i = 0;
|
|
171
|
+
while (i < lines.length) {
|
|
172
|
+
const line = lines[i];
|
|
173
|
+
if (isBlank(line)) {
|
|
174
|
+
i++;
|
|
175
|
+
continue;
|
|
176
|
+
}
|
|
177
|
+
const heading = FENCE.test(line) ? void 0 : headingOf(line);
|
|
178
|
+
if (heading) {
|
|
179
|
+
while (stack.length && stack[stack.length - 1].level >= heading.level) stack.pop();
|
|
180
|
+
stack.push(heading);
|
|
181
|
+
sections.push({ trail: stack.map((h) => h.title), heading: line, blocks: [] });
|
|
182
|
+
i++;
|
|
183
|
+
continue;
|
|
184
|
+
}
|
|
185
|
+
let end;
|
|
186
|
+
if (FENCE.test(line)) end = fenceEnd(lines, i);
|
|
187
|
+
else if (LIST_ITEM.test(line)) end = listEnd(lines, i);
|
|
188
|
+
else if (startsTable(lines, i)) end = tableEnd(lines, i);
|
|
189
|
+
else end = paragraphEnd(lines, i);
|
|
190
|
+
sections[sections.length - 1].blocks.push({ lines: lines.slice(i, end) });
|
|
191
|
+
i = end;
|
|
192
|
+
}
|
|
193
|
+
return sections;
|
|
194
|
+
}
|
|
195
|
+
function chunkSections(text, max = 900) {
|
|
196
|
+
const cap = Math.max(1, max);
|
|
197
|
+
const out = [];
|
|
198
|
+
const emit = (parts, section) => {
|
|
199
|
+
const body = parts.join("\n\n");
|
|
200
|
+
if (body.trim() === "") return;
|
|
201
|
+
out.push({ text: body, section: [...section], position: out.length });
|
|
202
|
+
};
|
|
203
|
+
for (const section of scan(text)) {
|
|
204
|
+
let parts = section.heading ? [section.heading.trim()] : [];
|
|
205
|
+
let size = parts.join("\n\n").length;
|
|
206
|
+
let hasBlock = false;
|
|
207
|
+
for (const block of section.blocks) {
|
|
208
|
+
const piece = block.lines.join("\n");
|
|
209
|
+
const added = piece.length + (parts.length ? 2 : 0);
|
|
210
|
+
if (hasBlock && size + added > cap) {
|
|
211
|
+
emit(parts, section.trail);
|
|
212
|
+
parts = [];
|
|
213
|
+
size = 0;
|
|
214
|
+
hasBlock = false;
|
|
215
|
+
}
|
|
216
|
+
size += piece.length + (parts.length ? 2 : 0);
|
|
217
|
+
parts.push(piece);
|
|
218
|
+
hasBlock = true;
|
|
219
|
+
}
|
|
220
|
+
emit(parts, section.trail);
|
|
221
|
+
}
|
|
222
|
+
return out;
|
|
223
|
+
}
|
|
224
|
+
var FENCE, HEADING, LIST_ITEM, TABLE_DELIMITER, isBlank, isIndented;
|
|
225
|
+
var init_chunking = __esm({
|
|
226
|
+
"src/core/chunking.ts"() {
|
|
227
|
+
"use strict";
|
|
228
|
+
FENCE = /^ {0,3}(`{3,}|~{3,})/;
|
|
229
|
+
HEADING = /^ {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*$/;
|
|
230
|
+
LIST_ITEM = /^[ \t]*(?:[-*+]|\d{1,9}[.)])[ \t]+\S/;
|
|
231
|
+
TABLE_DELIMITER = /^[ \t]*\|?[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?[ \t]*$/;
|
|
232
|
+
isBlank = (line) => line.trim() === "";
|
|
233
|
+
isIndented = (line) => /^(?: {2,}|\t)/.test(line);
|
|
234
|
+
}
|
|
235
|
+
});
|
|
236
|
+
|
|
117
237
|
// src/core/loaders.ts
|
|
118
238
|
function walk(dir) {
|
|
119
239
|
const items = [];
|
|
@@ -194,11 +314,13 @@ async function sha256(input) {
|
|
|
194
314
|
const hash = await crypto.subtle.digest("SHA-256", data);
|
|
195
315
|
return Array.from(new Uint8Array(hash)).map((b) => b.toString(16).padStart(2, "0")).join("");
|
|
196
316
|
}
|
|
197
|
-
var EMB_MODEL, DEFAULT_KNOWLEDGE_DIR, DATABASE_CONFIG_REQUIRED_MESSAGE, VectorStore;
|
|
317
|
+
var import_node_path3, EMB_MODEL, DEFAULT_KNOWLEDGE_DIR, DATABASE_CONFIG_REQUIRED_MESSAGE, VectorStore;
|
|
198
318
|
var init_retrieval = __esm({
|
|
199
319
|
"src/core/retrieval.ts"() {
|
|
200
320
|
"use strict";
|
|
321
|
+
import_node_path3 = require("node:path");
|
|
201
322
|
init_buildLock();
|
|
323
|
+
init_chunking();
|
|
202
324
|
init_loaders();
|
|
203
325
|
init_logger();
|
|
204
326
|
EMB_MODEL = "text-embedding-3-large";
|
|
@@ -209,6 +331,7 @@ var init_retrieval = __esm({
|
|
|
209
331
|
this.embed = embed;
|
|
210
332
|
this.db = options.databaseClient;
|
|
211
333
|
this.knowledgeDir = options.knowledgeDir || DEFAULT_KNOWLEDGE_DIR;
|
|
334
|
+
this.chunking = options.chunking ?? "lines";
|
|
212
335
|
this.logger = options.logger ?? createConsoleLogger();
|
|
213
336
|
this.buildLock = options.buildLock ?? createTursoBuildLock(this.db);
|
|
214
337
|
this.instanceId = options.instanceId ?? crypto.randomUUID();
|
|
@@ -222,6 +345,7 @@ var init_retrieval = __esm({
|
|
|
222
345
|
*/
|
|
223
346
|
db;
|
|
224
347
|
knowledgeDir;
|
|
348
|
+
chunking;
|
|
225
349
|
logger;
|
|
226
350
|
buildLock;
|
|
227
351
|
instanceId;
|
|
@@ -282,11 +406,41 @@ var init_retrieval = __esm({
|
|
|
282
406
|
this.logger.info("No knowledge documents and no existing chunks - nothing to build");
|
|
283
407
|
return;
|
|
284
408
|
}
|
|
409
|
+
this.logger.info(`Chunking mode: ${this.chunking}`);
|
|
410
|
+
if (this.chunking === "sections") await this.ensureSectionColumns();
|
|
285
411
|
const rows = [];
|
|
286
412
|
for (const d of docs) {
|
|
287
|
-
|
|
288
|
-
const
|
|
289
|
-
|
|
413
|
+
if (this.chunking === "sections") {
|
|
414
|
+
const source = (0, import_node_path3.relative)(this.knowledgeDir, d.source).split(import_node_path3.sep).join("/");
|
|
415
|
+
for (const c of chunkSections(d.text)) {
|
|
416
|
+
const section = c.section.join(" > ");
|
|
417
|
+
const id = await sha256(`${d.bucket}|${source}|${section}|${c.text}`);
|
|
418
|
+
const context = section ? `${source} > ${section}` : source;
|
|
419
|
+
rows.push({
|
|
420
|
+
id,
|
|
421
|
+
bucket: d.bucket,
|
|
422
|
+
source,
|
|
423
|
+
text: c.text,
|
|
424
|
+
embedText: `${context}
|
|
425
|
+
|
|
426
|
+
${c.text}`,
|
|
427
|
+
section,
|
|
428
|
+
position: c.position
|
|
429
|
+
});
|
|
430
|
+
}
|
|
431
|
+
} else {
|
|
432
|
+
for (const part of chunk(d.text)) {
|
|
433
|
+
const id = await sha256(`${d.bucket}|${d.source}|${part}`);
|
|
434
|
+
rows.push({
|
|
435
|
+
id,
|
|
436
|
+
bucket: d.bucket,
|
|
437
|
+
source: d.source,
|
|
438
|
+
text: part,
|
|
439
|
+
embedText: part,
|
|
440
|
+
section: null,
|
|
441
|
+
position: null
|
|
442
|
+
});
|
|
443
|
+
}
|
|
290
444
|
}
|
|
291
445
|
}
|
|
292
446
|
this.logger.info(`Created ${rows.length} chunks from knowledge documents`);
|
|
@@ -295,11 +449,17 @@ var init_retrieval = __esm({
|
|
|
295
449
|
for (let i = 0; i < rows.length; i += UPSERT_BATCH) {
|
|
296
450
|
const batch = rows.slice(i, i + UPSERT_BATCH);
|
|
297
451
|
await this.db.batch(
|
|
298
|
-
batch.map(
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
452
|
+
batch.map(
|
|
453
|
+
(r) => this.chunking === "sections" ? {
|
|
454
|
+
sql: `INSERT INTO chunks(id,bucket,source,text,section,position) VALUES(?,?,?,?,?,?)
|
|
455
|
+
ON CONFLICT(id) DO NOTHING`,
|
|
456
|
+
args: [r.id, r.bucket, r.source, r.text, r.section, r.position]
|
|
457
|
+
} : {
|
|
458
|
+
sql: `INSERT INTO chunks(id,bucket,source,text) VALUES(?,?,?,?)
|
|
459
|
+
ON CONFLICT(id) DO NOTHING`,
|
|
460
|
+
args: [r.id, r.bucket, r.source, r.text]
|
|
461
|
+
}
|
|
462
|
+
),
|
|
303
463
|
"write"
|
|
304
464
|
);
|
|
305
465
|
}
|
|
@@ -320,7 +480,7 @@ var init_retrieval = __esm({
|
|
|
320
480
|
return;
|
|
321
481
|
}
|
|
322
482
|
this.logger.info(`Embedding ${missing.length} new/updated chunks with ${EMB_MODEL}...`);
|
|
323
|
-
const textById = new Map(rows.map((r) => [r.id, r.
|
|
483
|
+
const textById = new Map(rows.map((r) => [r.id, r.embedText]));
|
|
324
484
|
const BATCH = 96;
|
|
325
485
|
for (let i = 0; i < missing.length; i += BATCH) {
|
|
326
486
|
const batchIds = missing.slice(i, i + BATCH);
|
|
@@ -341,6 +501,18 @@ var init_retrieval = __esm({
|
|
|
341
501
|
}
|
|
342
502
|
this.logger.info(`Successfully embedded ${missing.length} new chunks`);
|
|
343
503
|
}
|
|
504
|
+
/**
|
|
505
|
+
* Adds the nullable `section`/`position` columns to a `chunks` table that
|
|
506
|
+
* every released version created without them. Checked via table_info so it
|
|
507
|
+
* is a no-op once present; only `'sections'` mode ever calls it.
|
|
508
|
+
*/
|
|
509
|
+
async ensureSectionColumns() {
|
|
510
|
+
const info = await this.db.execute("PRAGMA table_info(chunks)");
|
|
511
|
+
const have = new Set(info.rows.map((r) => String(r.name)));
|
|
512
|
+
if (!have.has("section")) await this.db.execute("ALTER TABLE chunks ADD COLUMN section TEXT");
|
|
513
|
+
if (!have.has("position"))
|
|
514
|
+
await this.db.execute("ALTER TABLE chunks ADD COLUMN position INTEGER");
|
|
515
|
+
}
|
|
344
516
|
// Remove chunks from database that no longer exist in markdown files
|
|
345
517
|
async cleanupStaleChunks(currentIds) {
|
|
346
518
|
const result = await this.db.execute("SELECT id FROM chunks");
|
|
@@ -387,25 +559,34 @@ var init_retrieval = __esm({
|
|
|
387
559
|
* interpolated. An empty list retrieves nothing, and short-circuits before
|
|
388
560
|
* the embedding call.
|
|
389
561
|
*/
|
|
390
|
-
async
|
|
562
|
+
async queryChunks(q, k = 6, allowed = ["base"]) {
|
|
391
563
|
if (allowed.length === 0) return [];
|
|
392
564
|
const [qv] = await this.embed([q]);
|
|
393
565
|
const placeholders = allowed.map(() => "?").join(",");
|
|
394
566
|
const res = await this.db.execute({
|
|
395
|
-
sql: `SELECT c
|
|
567
|
+
sql: `SELECT c.*, e.embedding
|
|
396
568
|
FROM chunks c
|
|
397
569
|
JOIN embeddings e ON e.id = c.id
|
|
398
570
|
WHERE c.bucket IN (${placeholders})`,
|
|
399
571
|
args: allowed
|
|
400
572
|
});
|
|
401
|
-
const scored =
|
|
402
|
-
for (const row of res.rows) {
|
|
573
|
+
const scored = res.rows.map((row) => {
|
|
403
574
|
const emb = JSON.parse(String(row.embedding));
|
|
404
|
-
const
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
575
|
+
const trail = row.section == null ? "" : String(row.section);
|
|
576
|
+
return {
|
|
577
|
+
text: String(row.text),
|
|
578
|
+
bucket: String(row.bucket ?? ""),
|
|
579
|
+
source: String(row.source ?? ""),
|
|
580
|
+
section: trail ? trail.split(" > ") : [],
|
|
581
|
+
score: _VectorStore.cosine(qv, emb)
|
|
582
|
+
};
|
|
583
|
+
});
|
|
584
|
+
scored.sort((a, b) => b.score - a.score);
|
|
585
|
+
return scored.slice(0, k);
|
|
586
|
+
}
|
|
587
|
+
/** Same results as {@link queryChunks}, as bare text. */
|
|
588
|
+
async query(q, k = 6, allowed = ["base"]) {
|
|
589
|
+
return (await this.queryChunks(q, k, allowed)).map((c) => c.text);
|
|
409
590
|
}
|
|
410
591
|
};
|
|
411
592
|
}
|
|
@@ -1032,6 +1213,7 @@ async function createMCPServer(config) {
|
|
|
1032
1213
|
store = new VectorStore2(createOpenAIEmbedder2(client), {
|
|
1033
1214
|
databaseClient: db,
|
|
1034
1215
|
knowledgeDir: config.knowledgeDir || DEFAULT_KNOWLEDGE_DIR2,
|
|
1216
|
+
chunking: config.chunking,
|
|
1035
1217
|
logger: log
|
|
1036
1218
|
});
|
|
1037
1219
|
}
|