@diegoaltoworks/chatter 2.0.0 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/chunking.d.ts +31 -0
- package/dist/core/chunking.d.ts.map +1 -0
- package/dist/core/retrieval.d.ts +39 -0
- package/dist/core/retrieval.d.ts.map +1 -1
- package/dist/index.d.ts +3 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +211 -25
- package/dist/index.mjs +205 -20
- package/dist/mcp-server.d.ts.map +1 -1
- package/dist/mcp-server.js +201 -19
- package/dist/mcp-server.mjs +200 -18
- package/dist/personas/index.js +4 -2
- package/dist/personas/index.mjs +4 -2
- package/dist/personas/timeContext.d.ts.map +1 -1
- package/dist/server.d.ts.map +1 -1
- package/dist/server.js +207 -25
- package/dist/server.mjs +202 -20
- package/dist/types.d.ts +7 -0
- package/dist/types.d.ts.map +1 -1
- package/package.json +10 -10
package/dist/server.js
CHANGED
|
@@ -114,6 +114,126 @@ var init_buildLock = __esm({
|
|
|
114
114
|
}
|
|
115
115
|
});
|
|
116
116
|
|
|
117
|
+
// src/core/chunking.ts
|
|
118
|
+
function headingOf(line) {
|
|
119
|
+
const match = HEADING.exec(line);
|
|
120
|
+
if (!match) return void 0;
|
|
121
|
+
const title = (match[2] ?? "").replace(/[ \t]+#+$/, "").replace(/^#+$/, "").trim();
|
|
122
|
+
return { level: match[1].length, title };
|
|
123
|
+
}
|
|
124
|
+
function startsTable(lines, i) {
|
|
125
|
+
return lines[i].includes("|") && i + 1 < lines.length && TABLE_DELIMITER.test(lines[i + 1]);
|
|
126
|
+
}
|
|
127
|
+
function fenceEnd(lines, start) {
|
|
128
|
+
const opener = FENCE.exec(lines[start]);
|
|
129
|
+
if (!opener) return start + 1;
|
|
130
|
+
const marker = opener[1];
|
|
131
|
+
const closer = new RegExp(`^ {0,3}${marker[0] === "`" ? "`" : "~"}{${marker.length},}[ \\t]*$`);
|
|
132
|
+
for (let i = start + 1; i < lines.length; i++) if (closer.test(lines[i])) return i + 1;
|
|
133
|
+
return lines.length;
|
|
134
|
+
}
|
|
135
|
+
function listEnd(lines, start) {
|
|
136
|
+
let end = start + 1;
|
|
137
|
+
for (let i = start + 1; i < lines.length; i++) {
|
|
138
|
+
const line = lines[i];
|
|
139
|
+
if (isBlank(line)) {
|
|
140
|
+
let next = i + 1;
|
|
141
|
+
while (next < lines.length && isBlank(lines[next])) next++;
|
|
142
|
+
if (next < lines.length && (LIST_ITEM.test(lines[next]) || isIndented(lines[next]))) {
|
|
143
|
+
i = next - 1;
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
break;
|
|
147
|
+
}
|
|
148
|
+
if (headingOf(line) && !isIndented(line)) break;
|
|
149
|
+
if (FENCE.test(line) && !isIndented(line)) break;
|
|
150
|
+
end = i + 1;
|
|
151
|
+
}
|
|
152
|
+
return end;
|
|
153
|
+
}
|
|
154
|
+
function tableEnd(lines, start) {
|
|
155
|
+
let end = start + 2;
|
|
156
|
+
while (end < lines.length && !isBlank(lines[end]) && lines[end].includes("|")) end++;
|
|
157
|
+
return end;
|
|
158
|
+
}
|
|
159
|
+
function paragraphEnd(lines, start) {
|
|
160
|
+
let end = start + 1;
|
|
161
|
+
while (end < lines.length && !isBlank(lines[end]) && !headingOf(lines[end]) && !FENCE.test(lines[end]) && !LIST_ITEM.test(lines[end])) {
|
|
162
|
+
end++;
|
|
163
|
+
}
|
|
164
|
+
return end;
|
|
165
|
+
}
|
|
166
|
+
function scan(text) {
|
|
167
|
+
const lines = text.replace(/\r\n?/g, "\n").split("\n");
|
|
168
|
+
const sections = [{ trail: [], blocks: [] }];
|
|
169
|
+
const stack = [];
|
|
170
|
+
let i = 0;
|
|
171
|
+
while (i < lines.length) {
|
|
172
|
+
const line = lines[i];
|
|
173
|
+
if (isBlank(line)) {
|
|
174
|
+
i++;
|
|
175
|
+
continue;
|
|
176
|
+
}
|
|
177
|
+
const heading = FENCE.test(line) ? void 0 : headingOf(line);
|
|
178
|
+
if (heading) {
|
|
179
|
+
while (stack.length && stack[stack.length - 1].level >= heading.level) stack.pop();
|
|
180
|
+
stack.push(heading);
|
|
181
|
+
sections.push({ trail: stack.map((h) => h.title), heading: line, blocks: [] });
|
|
182
|
+
i++;
|
|
183
|
+
continue;
|
|
184
|
+
}
|
|
185
|
+
let end;
|
|
186
|
+
if (FENCE.test(line)) end = fenceEnd(lines, i);
|
|
187
|
+
else if (LIST_ITEM.test(line)) end = listEnd(lines, i);
|
|
188
|
+
else if (startsTable(lines, i)) end = tableEnd(lines, i);
|
|
189
|
+
else end = paragraphEnd(lines, i);
|
|
190
|
+
sections[sections.length - 1].blocks.push({ lines: lines.slice(i, end) });
|
|
191
|
+
i = end;
|
|
192
|
+
}
|
|
193
|
+
return sections;
|
|
194
|
+
}
|
|
195
|
+
function chunkSections(text, max = 900) {
|
|
196
|
+
const cap = Math.max(1, max);
|
|
197
|
+
const out = [];
|
|
198
|
+
const emit = (parts, section) => {
|
|
199
|
+
const body = parts.join("\n\n");
|
|
200
|
+
if (body.trim() === "") return;
|
|
201
|
+
out.push({ text: body, section: [...section], position: out.length });
|
|
202
|
+
};
|
|
203
|
+
for (const section of scan(text)) {
|
|
204
|
+
let parts = section.heading ? [section.heading.trim()] : [];
|
|
205
|
+
let size = parts.join("\n\n").length;
|
|
206
|
+
let hasBlock = false;
|
|
207
|
+
for (const block of section.blocks) {
|
|
208
|
+
const piece = block.lines.join("\n");
|
|
209
|
+
const added = piece.length + (parts.length ? 2 : 0);
|
|
210
|
+
if (hasBlock && size + added > cap) {
|
|
211
|
+
emit(parts, section.trail);
|
|
212
|
+
parts = [];
|
|
213
|
+
size = 0;
|
|
214
|
+
hasBlock = false;
|
|
215
|
+
}
|
|
216
|
+
size += piece.length + (parts.length ? 2 : 0);
|
|
217
|
+
parts.push(piece);
|
|
218
|
+
hasBlock = true;
|
|
219
|
+
}
|
|
220
|
+
emit(parts, section.trail);
|
|
221
|
+
}
|
|
222
|
+
return out;
|
|
223
|
+
}
|
|
224
|
+
var FENCE, HEADING, LIST_ITEM, TABLE_DELIMITER, isBlank, isIndented;
|
|
225
|
+
var init_chunking = __esm({
|
|
226
|
+
"src/core/chunking.ts"() {
|
|
227
|
+
"use strict";
|
|
228
|
+
FENCE = /^ {0,3}(`{3,}|~{3,})/;
|
|
229
|
+
HEADING = /^ {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*$/;
|
|
230
|
+
LIST_ITEM = /^[ \t]*(?:[-*+]|\d{1,9}[.)])[ \t]+\S/;
|
|
231
|
+
TABLE_DELIMITER = /^[ \t]*\|?[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?[ \t]*$/;
|
|
232
|
+
isBlank = (line) => line.trim() === "";
|
|
233
|
+
isIndented = (line) => /^(?: {2,}|\t)/.test(line);
|
|
234
|
+
}
|
|
235
|
+
});
|
|
236
|
+
|
|
117
237
|
// src/core/loaders.ts
|
|
118
238
|
function walk(dir) {
|
|
119
239
|
const items = [];
|
|
@@ -194,11 +314,13 @@ async function sha256(input) {
|
|
|
194
314
|
const hash = await crypto.subtle.digest("SHA-256", data);
|
|
195
315
|
return Array.from(new Uint8Array(hash)).map((b) => b.toString(16).padStart(2, "0")).join("");
|
|
196
316
|
}
|
|
197
|
-
var EMB_MODEL, DEFAULT_KNOWLEDGE_DIR, DATABASE_CONFIG_REQUIRED_MESSAGE, VectorStore;
|
|
317
|
+
var import_node_path3, EMB_MODEL, DEFAULT_KNOWLEDGE_DIR, DATABASE_CONFIG_REQUIRED_MESSAGE, VectorStore;
|
|
198
318
|
var init_retrieval = __esm({
|
|
199
319
|
"src/core/retrieval.ts"() {
|
|
200
320
|
"use strict";
|
|
321
|
+
import_node_path3 = require("node:path");
|
|
201
322
|
init_buildLock();
|
|
323
|
+
init_chunking();
|
|
202
324
|
init_loaders();
|
|
203
325
|
init_logger();
|
|
204
326
|
EMB_MODEL = "text-embedding-3-large";
|
|
@@ -209,6 +331,7 @@ var init_retrieval = __esm({
|
|
|
209
331
|
this.embed = embed;
|
|
210
332
|
this.db = options.databaseClient;
|
|
211
333
|
this.knowledgeDir = options.knowledgeDir || DEFAULT_KNOWLEDGE_DIR;
|
|
334
|
+
this.chunking = options.chunking ?? "lines";
|
|
212
335
|
this.logger = options.logger ?? createConsoleLogger();
|
|
213
336
|
this.buildLock = options.buildLock ?? createTursoBuildLock(this.db);
|
|
214
337
|
this.instanceId = options.instanceId ?? crypto.randomUUID();
|
|
@@ -222,6 +345,7 @@ var init_retrieval = __esm({
|
|
|
222
345
|
*/
|
|
223
346
|
db;
|
|
224
347
|
knowledgeDir;
|
|
348
|
+
chunking;
|
|
225
349
|
logger;
|
|
226
350
|
buildLock;
|
|
227
351
|
instanceId;
|
|
@@ -282,11 +406,41 @@ var init_retrieval = __esm({
|
|
|
282
406
|
this.logger.info("No knowledge documents and no existing chunks - nothing to build");
|
|
283
407
|
return;
|
|
284
408
|
}
|
|
409
|
+
this.logger.info(`Chunking mode: ${this.chunking}`);
|
|
410
|
+
if (this.chunking === "sections") await this.ensureSectionColumns();
|
|
285
411
|
const rows = [];
|
|
286
412
|
for (const d of docs) {
|
|
287
|
-
|
|
288
|
-
const
|
|
289
|
-
|
|
413
|
+
if (this.chunking === "sections") {
|
|
414
|
+
const source = (0, import_node_path3.relative)(this.knowledgeDir, d.source).split(import_node_path3.sep).join("/");
|
|
415
|
+
for (const c of chunkSections(d.text)) {
|
|
416
|
+
const section = c.section.join(" > ");
|
|
417
|
+
const id = await sha256(`${d.bucket}|${source}|${section}|${c.text}`);
|
|
418
|
+
const context = section ? `${source} > ${section}` : source;
|
|
419
|
+
rows.push({
|
|
420
|
+
id,
|
|
421
|
+
bucket: d.bucket,
|
|
422
|
+
source,
|
|
423
|
+
text: c.text,
|
|
424
|
+
embedText: `${context}
|
|
425
|
+
|
|
426
|
+
${c.text}`,
|
|
427
|
+
section,
|
|
428
|
+
position: c.position
|
|
429
|
+
});
|
|
430
|
+
}
|
|
431
|
+
} else {
|
|
432
|
+
for (const part of chunk(d.text)) {
|
|
433
|
+
const id = await sha256(`${d.bucket}|${d.source}|${part}`);
|
|
434
|
+
rows.push({
|
|
435
|
+
id,
|
|
436
|
+
bucket: d.bucket,
|
|
437
|
+
source: d.source,
|
|
438
|
+
text: part,
|
|
439
|
+
embedText: part,
|
|
440
|
+
section: null,
|
|
441
|
+
position: null
|
|
442
|
+
});
|
|
443
|
+
}
|
|
290
444
|
}
|
|
291
445
|
}
|
|
292
446
|
this.logger.info(`Created ${rows.length} chunks from knowledge documents`);
|
|
@@ -295,11 +449,17 @@ var init_retrieval = __esm({
|
|
|
295
449
|
for (let i = 0; i < rows.length; i += UPSERT_BATCH) {
|
|
296
450
|
const batch = rows.slice(i, i + UPSERT_BATCH);
|
|
297
451
|
await this.db.batch(
|
|
298
|
-
batch.map(
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
452
|
+
batch.map(
|
|
453
|
+
(r) => this.chunking === "sections" ? {
|
|
454
|
+
sql: `INSERT INTO chunks(id,bucket,source,text,section,position) VALUES(?,?,?,?,?,?)
|
|
455
|
+
ON CONFLICT(id) DO NOTHING`,
|
|
456
|
+
args: [r.id, r.bucket, r.source, r.text, r.section, r.position]
|
|
457
|
+
} : {
|
|
458
|
+
sql: `INSERT INTO chunks(id,bucket,source,text) VALUES(?,?,?,?)
|
|
459
|
+
ON CONFLICT(id) DO NOTHING`,
|
|
460
|
+
args: [r.id, r.bucket, r.source, r.text]
|
|
461
|
+
}
|
|
462
|
+
),
|
|
303
463
|
"write"
|
|
304
464
|
);
|
|
305
465
|
}
|
|
@@ -320,7 +480,7 @@ var init_retrieval = __esm({
|
|
|
320
480
|
return;
|
|
321
481
|
}
|
|
322
482
|
this.logger.info(`Embedding ${missing.length} new/updated chunks with ${EMB_MODEL}...`);
|
|
323
|
-
const textById = new Map(rows.map((r) => [r.id, r.
|
|
483
|
+
const textById = new Map(rows.map((r) => [r.id, r.embedText]));
|
|
324
484
|
const BATCH = 96;
|
|
325
485
|
for (let i = 0; i < missing.length; i += BATCH) {
|
|
326
486
|
const batchIds = missing.slice(i, i + BATCH);
|
|
@@ -341,6 +501,18 @@ var init_retrieval = __esm({
|
|
|
341
501
|
}
|
|
342
502
|
this.logger.info(`Successfully embedded ${missing.length} new chunks`);
|
|
343
503
|
}
|
|
504
|
+
/**
|
|
505
|
+
* Adds the nullable `section`/`position` columns to a `chunks` table that
|
|
506
|
+
* every released version created without them. Checked via table_info so it
|
|
507
|
+
* is a no-op once present; only `'sections'` mode ever calls it.
|
|
508
|
+
*/
|
|
509
|
+
async ensureSectionColumns() {
|
|
510
|
+
const info = await this.db.execute("PRAGMA table_info(chunks)");
|
|
511
|
+
const have = new Set(info.rows.map((r) => String(r.name)));
|
|
512
|
+
if (!have.has("section")) await this.db.execute("ALTER TABLE chunks ADD COLUMN section TEXT");
|
|
513
|
+
if (!have.has("position"))
|
|
514
|
+
await this.db.execute("ALTER TABLE chunks ADD COLUMN position INTEGER");
|
|
515
|
+
}
|
|
344
516
|
// Remove chunks from database that no longer exist in markdown files
|
|
345
517
|
async cleanupStaleChunks(currentIds) {
|
|
346
518
|
const result = await this.db.execute("SELECT id FROM chunks");
|
|
@@ -387,25 +559,34 @@ var init_retrieval = __esm({
|
|
|
387
559
|
* interpolated. An empty list retrieves nothing, and short-circuits before
|
|
388
560
|
* the embedding call.
|
|
389
561
|
*/
|
|
390
|
-
async
|
|
562
|
+
async queryChunks(q, k = 6, allowed = ["base"]) {
|
|
391
563
|
if (allowed.length === 0) return [];
|
|
392
564
|
const [qv] = await this.embed([q]);
|
|
393
565
|
const placeholders = allowed.map(() => "?").join(",");
|
|
394
566
|
const res = await this.db.execute({
|
|
395
|
-
sql: `SELECT c
|
|
567
|
+
sql: `SELECT c.*, e.embedding
|
|
396
568
|
FROM chunks c
|
|
397
569
|
JOIN embeddings e ON e.id = c.id
|
|
398
570
|
WHERE c.bucket IN (${placeholders})`,
|
|
399
571
|
args: allowed
|
|
400
572
|
});
|
|
401
|
-
const scored =
|
|
402
|
-
for (const row of res.rows) {
|
|
573
|
+
const scored = res.rows.map((row) => {
|
|
403
574
|
const emb = JSON.parse(String(row.embedding));
|
|
404
|
-
const
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
575
|
+
const trail = row.section == null ? "" : String(row.section);
|
|
576
|
+
return {
|
|
577
|
+
text: String(row.text),
|
|
578
|
+
bucket: String(row.bucket ?? ""),
|
|
579
|
+
source: String(row.source ?? ""),
|
|
580
|
+
section: trail ? trail.split(" > ") : [],
|
|
581
|
+
score: _VectorStore.cosine(qv, emb)
|
|
582
|
+
};
|
|
583
|
+
});
|
|
584
|
+
scored.sort((a, b) => b.score - a.score);
|
|
585
|
+
return scored.slice(0, k);
|
|
586
|
+
}
|
|
587
|
+
/** Same results as {@link queryChunks}, as bare text. */
|
|
588
|
+
async query(q, k = 6, allowed = ["base"]) {
|
|
589
|
+
return (await this.queryChunks(q, k, allowed)).map((c) => c.text);
|
|
409
590
|
}
|
|
410
591
|
};
|
|
411
592
|
}
|
|
@@ -530,7 +711,7 @@ __export(server_exports, {
|
|
|
530
711
|
createServer: () => createServer
|
|
531
712
|
});
|
|
532
713
|
module.exports = __toCommonJS(server_exports);
|
|
533
|
-
var
|
|
714
|
+
var import_node_path5 = require("node:path");
|
|
534
715
|
var import_hono5 = require("hono");
|
|
535
716
|
var import_openai = __toESM(require("openai"));
|
|
536
717
|
|
|
@@ -1078,7 +1259,7 @@ function loadServeStatic(runtime = detectRuntime()) {
|
|
|
1078
1259
|
|
|
1079
1260
|
// src/core/widgets.ts
|
|
1080
1261
|
var import_node_fs3 = require("node:fs");
|
|
1081
|
-
var
|
|
1262
|
+
var import_node_path4 = require("node:path");
|
|
1082
1263
|
var import_node_url = require("node:url");
|
|
1083
1264
|
var import_meta = {};
|
|
1084
1265
|
function resolveStatic(configStaticDir) {
|
|
@@ -1087,15 +1268,15 @@ function resolveStatic(configStaticDir) {
|
|
|
1087
1268
|
}
|
|
1088
1269
|
try {
|
|
1089
1270
|
if (import_meta.url) {
|
|
1090
|
-
const moduleDir = (0,
|
|
1091
|
-
const staticDir = (0,
|
|
1271
|
+
const moduleDir = (0, import_node_path4.dirname)((0, import_node_url.fileURLToPath)(import_meta.url));
|
|
1272
|
+
const staticDir = (0, import_node_path4.join)(moduleDir, "static");
|
|
1092
1273
|
if ((0, import_node_fs3.existsSync)(staticDir)) return { staticDir };
|
|
1093
1274
|
}
|
|
1094
1275
|
} catch {
|
|
1095
1276
|
}
|
|
1096
1277
|
try {
|
|
1097
1278
|
if (typeof __dirname !== "undefined") {
|
|
1098
|
-
const staticDir = (0,
|
|
1279
|
+
const staticDir = (0, import_node_path4.join)(__dirname, "static");
|
|
1099
1280
|
if ((0, import_node_fs3.existsSync)(staticDir)) return { staticDir };
|
|
1100
1281
|
}
|
|
1101
1282
|
} catch {
|
|
@@ -2021,6 +2202,7 @@ async function createServer(config) {
|
|
|
2021
2202
|
store = new VectorStore2(createOpenAIEmbedder2(client), {
|
|
2022
2203
|
databaseClient: db,
|
|
2023
2204
|
knowledgeDir: config.knowledgeDir || DEFAULT_KNOWLEDGE_DIR2,
|
|
2205
|
+
chunking: config.chunking,
|
|
2024
2206
|
logger
|
|
2025
2207
|
});
|
|
2026
2208
|
}
|
|
@@ -2071,7 +2253,7 @@ async function createServer(config) {
|
|
|
2071
2253
|
}
|
|
2072
2254
|
}
|
|
2073
2255
|
if (serveStatic && staticDir) {
|
|
2074
|
-
const relativePath = (0,
|
|
2256
|
+
const relativePath = (0, import_node_path5.relative)(process.cwd(), staticDir);
|
|
2075
2257
|
app.get("/chatter.js", serveStatic({ path: `${relativePath}/chatter.js` }));
|
|
2076
2258
|
app.get("/chatter.css", serveStatic({ path: `${relativePath}/chatter.css` }));
|
|
2077
2259
|
}
|
package/dist/server.mjs
CHANGED
|
@@ -92,6 +92,126 @@ var init_buildLock = __esm({
|
|
|
92
92
|
}
|
|
93
93
|
});
|
|
94
94
|
|
|
95
|
+
// src/core/chunking.ts
|
|
96
|
+
function headingOf(line) {
|
|
97
|
+
const match = HEADING.exec(line);
|
|
98
|
+
if (!match) return void 0;
|
|
99
|
+
const title = (match[2] ?? "").replace(/[ \t]+#+$/, "").replace(/^#+$/, "").trim();
|
|
100
|
+
return { level: match[1].length, title };
|
|
101
|
+
}
|
|
102
|
+
function startsTable(lines, i) {
|
|
103
|
+
return lines[i].includes("|") && i + 1 < lines.length && TABLE_DELIMITER.test(lines[i + 1]);
|
|
104
|
+
}
|
|
105
|
+
function fenceEnd(lines, start) {
|
|
106
|
+
const opener = FENCE.exec(lines[start]);
|
|
107
|
+
if (!opener) return start + 1;
|
|
108
|
+
const marker = opener[1];
|
|
109
|
+
const closer = new RegExp(`^ {0,3}${marker[0] === "`" ? "`" : "~"}{${marker.length},}[ \\t]*$`);
|
|
110
|
+
for (let i = start + 1; i < lines.length; i++) if (closer.test(lines[i])) return i + 1;
|
|
111
|
+
return lines.length;
|
|
112
|
+
}
|
|
113
|
+
function listEnd(lines, start) {
|
|
114
|
+
let end = start + 1;
|
|
115
|
+
for (let i = start + 1; i < lines.length; i++) {
|
|
116
|
+
const line = lines[i];
|
|
117
|
+
if (isBlank(line)) {
|
|
118
|
+
let next = i + 1;
|
|
119
|
+
while (next < lines.length && isBlank(lines[next])) next++;
|
|
120
|
+
if (next < lines.length && (LIST_ITEM.test(lines[next]) || isIndented(lines[next]))) {
|
|
121
|
+
i = next - 1;
|
|
122
|
+
continue;
|
|
123
|
+
}
|
|
124
|
+
break;
|
|
125
|
+
}
|
|
126
|
+
if (headingOf(line) && !isIndented(line)) break;
|
|
127
|
+
if (FENCE.test(line) && !isIndented(line)) break;
|
|
128
|
+
end = i + 1;
|
|
129
|
+
}
|
|
130
|
+
return end;
|
|
131
|
+
}
|
|
132
|
+
function tableEnd(lines, start) {
|
|
133
|
+
let end = start + 2;
|
|
134
|
+
while (end < lines.length && !isBlank(lines[end]) && lines[end].includes("|")) end++;
|
|
135
|
+
return end;
|
|
136
|
+
}
|
|
137
|
+
function paragraphEnd(lines, start) {
|
|
138
|
+
let end = start + 1;
|
|
139
|
+
while (end < lines.length && !isBlank(lines[end]) && !headingOf(lines[end]) && !FENCE.test(lines[end]) && !LIST_ITEM.test(lines[end])) {
|
|
140
|
+
end++;
|
|
141
|
+
}
|
|
142
|
+
return end;
|
|
143
|
+
}
|
|
144
|
+
function scan(text) {
|
|
145
|
+
const lines = text.replace(/\r\n?/g, "\n").split("\n");
|
|
146
|
+
const sections = [{ trail: [], blocks: [] }];
|
|
147
|
+
const stack = [];
|
|
148
|
+
let i = 0;
|
|
149
|
+
while (i < lines.length) {
|
|
150
|
+
const line = lines[i];
|
|
151
|
+
if (isBlank(line)) {
|
|
152
|
+
i++;
|
|
153
|
+
continue;
|
|
154
|
+
}
|
|
155
|
+
const heading = FENCE.test(line) ? void 0 : headingOf(line);
|
|
156
|
+
if (heading) {
|
|
157
|
+
while (stack.length && stack[stack.length - 1].level >= heading.level) stack.pop();
|
|
158
|
+
stack.push(heading);
|
|
159
|
+
sections.push({ trail: stack.map((h) => h.title), heading: line, blocks: [] });
|
|
160
|
+
i++;
|
|
161
|
+
continue;
|
|
162
|
+
}
|
|
163
|
+
let end;
|
|
164
|
+
if (FENCE.test(line)) end = fenceEnd(lines, i);
|
|
165
|
+
else if (LIST_ITEM.test(line)) end = listEnd(lines, i);
|
|
166
|
+
else if (startsTable(lines, i)) end = tableEnd(lines, i);
|
|
167
|
+
else end = paragraphEnd(lines, i);
|
|
168
|
+
sections[sections.length - 1].blocks.push({ lines: lines.slice(i, end) });
|
|
169
|
+
i = end;
|
|
170
|
+
}
|
|
171
|
+
return sections;
|
|
172
|
+
}
|
|
173
|
+
function chunkSections(text, max = 900) {
|
|
174
|
+
const cap = Math.max(1, max);
|
|
175
|
+
const out = [];
|
|
176
|
+
const emit = (parts, section) => {
|
|
177
|
+
const body = parts.join("\n\n");
|
|
178
|
+
if (body.trim() === "") return;
|
|
179
|
+
out.push({ text: body, section: [...section], position: out.length });
|
|
180
|
+
};
|
|
181
|
+
for (const section of scan(text)) {
|
|
182
|
+
let parts = section.heading ? [section.heading.trim()] : [];
|
|
183
|
+
let size = parts.join("\n\n").length;
|
|
184
|
+
let hasBlock = false;
|
|
185
|
+
for (const block of section.blocks) {
|
|
186
|
+
const piece = block.lines.join("\n");
|
|
187
|
+
const added = piece.length + (parts.length ? 2 : 0);
|
|
188
|
+
if (hasBlock && size + added > cap) {
|
|
189
|
+
emit(parts, section.trail);
|
|
190
|
+
parts = [];
|
|
191
|
+
size = 0;
|
|
192
|
+
hasBlock = false;
|
|
193
|
+
}
|
|
194
|
+
size += piece.length + (parts.length ? 2 : 0);
|
|
195
|
+
parts.push(piece);
|
|
196
|
+
hasBlock = true;
|
|
197
|
+
}
|
|
198
|
+
emit(parts, section.trail);
|
|
199
|
+
}
|
|
200
|
+
return out;
|
|
201
|
+
}
|
|
202
|
+
var FENCE, HEADING, LIST_ITEM, TABLE_DELIMITER, isBlank, isIndented;
|
|
203
|
+
var init_chunking = __esm({
|
|
204
|
+
"src/core/chunking.ts"() {
|
|
205
|
+
"use strict";
|
|
206
|
+
FENCE = /^ {0,3}(`{3,}|~{3,})/;
|
|
207
|
+
HEADING = /^ {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*$/;
|
|
208
|
+
LIST_ITEM = /^[ \t]*(?:[-*+]|\d{1,9}[.)])[ \t]+\S/;
|
|
209
|
+
TABLE_DELIMITER = /^[ \t]*\|?[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?[ \t]*$/;
|
|
210
|
+
isBlank = (line) => line.trim() === "";
|
|
211
|
+
isIndented = (line) => /^(?: {2,}|\t)/.test(line);
|
|
212
|
+
}
|
|
213
|
+
});
|
|
214
|
+
|
|
95
215
|
// src/core/loaders.ts
|
|
96
216
|
import { readdirSync, readFileSync as readFileSync2, statSync } from "node:fs";
|
|
97
217
|
import { join as join2 } from "node:path";
|
|
@@ -134,6 +254,7 @@ __export(retrieval_exports, {
|
|
|
134
254
|
openLibsqlClient: () => openLibsqlClient,
|
|
135
255
|
wrapMissingLibsqlError: () => wrapMissingLibsqlError
|
|
136
256
|
});
|
|
257
|
+
import { relative, sep } from "node:path";
|
|
137
258
|
function createOpenAIEmbedder(client) {
|
|
138
259
|
return async (input) => {
|
|
139
260
|
const res = await client.embeddings.create({ model: EMB_MODEL, input });
|
|
@@ -176,6 +297,7 @@ var init_retrieval = __esm({
|
|
|
176
297
|
"src/core/retrieval.ts"() {
|
|
177
298
|
"use strict";
|
|
178
299
|
init_buildLock();
|
|
300
|
+
init_chunking();
|
|
179
301
|
init_loaders();
|
|
180
302
|
init_logger();
|
|
181
303
|
EMB_MODEL = "text-embedding-3-large";
|
|
@@ -186,6 +308,7 @@ var init_retrieval = __esm({
|
|
|
186
308
|
this.embed = embed;
|
|
187
309
|
this.db = options.databaseClient;
|
|
188
310
|
this.knowledgeDir = options.knowledgeDir || DEFAULT_KNOWLEDGE_DIR;
|
|
311
|
+
this.chunking = options.chunking ?? "lines";
|
|
189
312
|
this.logger = options.logger ?? createConsoleLogger();
|
|
190
313
|
this.buildLock = options.buildLock ?? createTursoBuildLock(this.db);
|
|
191
314
|
this.instanceId = options.instanceId ?? crypto.randomUUID();
|
|
@@ -199,6 +322,7 @@ var init_retrieval = __esm({
|
|
|
199
322
|
*/
|
|
200
323
|
db;
|
|
201
324
|
knowledgeDir;
|
|
325
|
+
chunking;
|
|
202
326
|
logger;
|
|
203
327
|
buildLock;
|
|
204
328
|
instanceId;
|
|
@@ -259,11 +383,41 @@ var init_retrieval = __esm({
|
|
|
259
383
|
this.logger.info("No knowledge documents and no existing chunks - nothing to build");
|
|
260
384
|
return;
|
|
261
385
|
}
|
|
386
|
+
this.logger.info(`Chunking mode: ${this.chunking}`);
|
|
387
|
+
if (this.chunking === "sections") await this.ensureSectionColumns();
|
|
262
388
|
const rows = [];
|
|
263
389
|
for (const d of docs) {
|
|
264
|
-
|
|
265
|
-
const
|
|
266
|
-
|
|
390
|
+
if (this.chunking === "sections") {
|
|
391
|
+
const source = relative(this.knowledgeDir, d.source).split(sep).join("/");
|
|
392
|
+
for (const c of chunkSections(d.text)) {
|
|
393
|
+
const section = c.section.join(" > ");
|
|
394
|
+
const id = await sha256(`${d.bucket}|${source}|${section}|${c.text}`);
|
|
395
|
+
const context = section ? `${source} > ${section}` : source;
|
|
396
|
+
rows.push({
|
|
397
|
+
id,
|
|
398
|
+
bucket: d.bucket,
|
|
399
|
+
source,
|
|
400
|
+
text: c.text,
|
|
401
|
+
embedText: `${context}
|
|
402
|
+
|
|
403
|
+
${c.text}`,
|
|
404
|
+
section,
|
|
405
|
+
position: c.position
|
|
406
|
+
});
|
|
407
|
+
}
|
|
408
|
+
} else {
|
|
409
|
+
for (const part of chunk(d.text)) {
|
|
410
|
+
const id = await sha256(`${d.bucket}|${d.source}|${part}`);
|
|
411
|
+
rows.push({
|
|
412
|
+
id,
|
|
413
|
+
bucket: d.bucket,
|
|
414
|
+
source: d.source,
|
|
415
|
+
text: part,
|
|
416
|
+
embedText: part,
|
|
417
|
+
section: null,
|
|
418
|
+
position: null
|
|
419
|
+
});
|
|
420
|
+
}
|
|
267
421
|
}
|
|
268
422
|
}
|
|
269
423
|
this.logger.info(`Created ${rows.length} chunks from knowledge documents`);
|
|
@@ -272,11 +426,17 @@ var init_retrieval = __esm({
|
|
|
272
426
|
for (let i = 0; i < rows.length; i += UPSERT_BATCH) {
|
|
273
427
|
const batch = rows.slice(i, i + UPSERT_BATCH);
|
|
274
428
|
await this.db.batch(
|
|
275
|
-
batch.map(
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
429
|
+
batch.map(
|
|
430
|
+
(r) => this.chunking === "sections" ? {
|
|
431
|
+
sql: `INSERT INTO chunks(id,bucket,source,text,section,position) VALUES(?,?,?,?,?,?)
|
|
432
|
+
ON CONFLICT(id) DO NOTHING`,
|
|
433
|
+
args: [r.id, r.bucket, r.source, r.text, r.section, r.position]
|
|
434
|
+
} : {
|
|
435
|
+
sql: `INSERT INTO chunks(id,bucket,source,text) VALUES(?,?,?,?)
|
|
436
|
+
ON CONFLICT(id) DO NOTHING`,
|
|
437
|
+
args: [r.id, r.bucket, r.source, r.text]
|
|
438
|
+
}
|
|
439
|
+
),
|
|
280
440
|
"write"
|
|
281
441
|
);
|
|
282
442
|
}
|
|
@@ -297,7 +457,7 @@ var init_retrieval = __esm({
|
|
|
297
457
|
return;
|
|
298
458
|
}
|
|
299
459
|
this.logger.info(`Embedding ${missing.length} new/updated chunks with ${EMB_MODEL}...`);
|
|
300
|
-
const textById = new Map(rows.map((r) => [r.id, r.
|
|
460
|
+
const textById = new Map(rows.map((r) => [r.id, r.embedText]));
|
|
301
461
|
const BATCH = 96;
|
|
302
462
|
for (let i = 0; i < missing.length; i += BATCH) {
|
|
303
463
|
const batchIds = missing.slice(i, i + BATCH);
|
|
@@ -318,6 +478,18 @@ var init_retrieval = __esm({
|
|
|
318
478
|
}
|
|
319
479
|
this.logger.info(`Successfully embedded ${missing.length} new chunks`);
|
|
320
480
|
}
|
|
481
|
+
/**
|
|
482
|
+
* Adds the nullable `section`/`position` columns to a `chunks` table that
|
|
483
|
+
* every released version created without them. Checked via table_info so it
|
|
484
|
+
* is a no-op once present; only `'sections'` mode ever calls it.
|
|
485
|
+
*/
|
|
486
|
+
async ensureSectionColumns() {
|
|
487
|
+
const info = await this.db.execute("PRAGMA table_info(chunks)");
|
|
488
|
+
const have = new Set(info.rows.map((r) => String(r.name)));
|
|
489
|
+
if (!have.has("section")) await this.db.execute("ALTER TABLE chunks ADD COLUMN section TEXT");
|
|
490
|
+
if (!have.has("position"))
|
|
491
|
+
await this.db.execute("ALTER TABLE chunks ADD COLUMN position INTEGER");
|
|
492
|
+
}
|
|
321
493
|
// Remove chunks from database that no longer exist in markdown files
|
|
322
494
|
async cleanupStaleChunks(currentIds) {
|
|
323
495
|
const result = await this.db.execute("SELECT id FROM chunks");
|
|
@@ -364,25 +536,34 @@ var init_retrieval = __esm({
|
|
|
364
536
|
* interpolated. An empty list retrieves nothing, and short-circuits before
|
|
365
537
|
* the embedding call.
|
|
366
538
|
*/
|
|
367
|
-
async
|
|
539
|
+
async queryChunks(q, k = 6, allowed = ["base"]) {
|
|
368
540
|
if (allowed.length === 0) return [];
|
|
369
541
|
const [qv] = await this.embed([q]);
|
|
370
542
|
const placeholders = allowed.map(() => "?").join(",");
|
|
371
543
|
const res = await this.db.execute({
|
|
372
|
-
sql: `SELECT c
|
|
544
|
+
sql: `SELECT c.*, e.embedding
|
|
373
545
|
FROM chunks c
|
|
374
546
|
JOIN embeddings e ON e.id = c.id
|
|
375
547
|
WHERE c.bucket IN (${placeholders})`,
|
|
376
548
|
args: allowed
|
|
377
549
|
});
|
|
378
|
-
const scored =
|
|
379
|
-
for (const row of res.rows) {
|
|
550
|
+
const scored = res.rows.map((row) => {
|
|
380
551
|
const emb = JSON.parse(String(row.embedding));
|
|
381
|
-
const
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
552
|
+
const trail = row.section == null ? "" : String(row.section);
|
|
553
|
+
return {
|
|
554
|
+
text: String(row.text),
|
|
555
|
+
bucket: String(row.bucket ?? ""),
|
|
556
|
+
source: String(row.source ?? ""),
|
|
557
|
+
section: trail ? trail.split(" > ") : [],
|
|
558
|
+
score: _VectorStore.cosine(qv, emb)
|
|
559
|
+
};
|
|
560
|
+
});
|
|
561
|
+
scored.sort((a, b) => b.score - a.score);
|
|
562
|
+
return scored.slice(0, k);
|
|
563
|
+
}
|
|
564
|
+
/** Same results as {@link queryChunks}, as bare text. */
|
|
565
|
+
async query(q, k = 6, allowed = ["base"]) {
|
|
566
|
+
return (await this.queryChunks(q, k, allowed)).map((c) => c.text);
|
|
386
567
|
}
|
|
387
568
|
};
|
|
388
569
|
}
|
|
@@ -502,7 +683,7 @@ var init_knowledgeHealth = __esm({
|
|
|
502
683
|
});
|
|
503
684
|
|
|
504
685
|
// src/server.ts
|
|
505
|
-
import { relative } from "node:path";
|
|
686
|
+
import { relative as relative2 } from "node:path";
|
|
506
687
|
import { Hono as Hono5 } from "hono";
|
|
507
688
|
import OpenAI from "openai";
|
|
508
689
|
|
|
@@ -1992,6 +2173,7 @@ async function createServer(config) {
|
|
|
1992
2173
|
store = new VectorStore2(createOpenAIEmbedder2(client), {
|
|
1993
2174
|
databaseClient: db,
|
|
1994
2175
|
knowledgeDir: config.knowledgeDir || DEFAULT_KNOWLEDGE_DIR2,
|
|
2176
|
+
chunking: config.chunking,
|
|
1995
2177
|
logger
|
|
1996
2178
|
});
|
|
1997
2179
|
}
|
|
@@ -2042,7 +2224,7 @@ async function createServer(config) {
|
|
|
2042
2224
|
}
|
|
2043
2225
|
}
|
|
2044
2226
|
if (serveStatic && staticDir) {
|
|
2045
|
-
const relativePath =
|
|
2227
|
+
const relativePath = relative2(process.cwd(), staticDir);
|
|
2046
2228
|
app.get("/chatter.js", serveStatic({ path: `${relativePath}/chatter.js` }));
|
|
2047
2229
|
app.get("/chatter.css", serveStatic({ path: `${relativePath}/chatter.css` }));
|
|
2048
2230
|
}
|
package/dist/types.d.ts
CHANGED
|
@@ -187,6 +187,13 @@ export interface ChatterConfig extends BrainHooks {
|
|
|
187
187
|
configDir?: string;
|
|
188
188
|
/** Knowledge base directory. Default: ./config/knowledge */
|
|
189
189
|
knowledgeDir?: string;
|
|
190
|
+
/**
|
|
191
|
+
* How knowledge files are chunked by the default `VectorStore`: `'lines'`
|
|
192
|
+
* (default, unchanged) or `'sections'` (heading-aware, with section context
|
|
193
|
+
* in the embedding). Switching re-embeds the knowledge base once. Ignored
|
|
194
|
+
* when `retriever` is set.
|
|
195
|
+
*/
|
|
196
|
+
chunking?: "lines" | "sections";
|
|
190
197
|
/** Prompts directory. Default: ./config/prompts */
|
|
191
198
|
promptsDir?: string;
|
|
192
199
|
/** Public static files directory. Default: ./public */
|