@diegoaltoworks/chatter 2.0.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -72,6 +72,126 @@ var init_buildLock = __esm({
72
72
  }
73
73
  });
74
74
 
75
+ // src/core/chunking.ts
76
+ function headingOf(line) {
77
+ const match = HEADING.exec(line);
78
+ if (!match) return void 0;
79
+ const title = (match[2] ?? "").replace(/[ \t]+#+$/, "").replace(/^#+$/, "").trim();
80
+ return { level: match[1].length, title };
81
+ }
82
+ function startsTable(lines, i) {
83
+ return lines[i].includes("|") && i + 1 < lines.length && TABLE_DELIMITER.test(lines[i + 1]);
84
+ }
85
+ function fenceEnd(lines, start) {
86
+ const opener = FENCE.exec(lines[start]);
87
+ if (!opener) return start + 1;
88
+ const marker = opener[1];
89
+ const closer = new RegExp(`^ {0,3}${marker[0] === "`" ? "`" : "~"}{${marker.length},}[ \\t]*$`);
90
+ for (let i = start + 1; i < lines.length; i++) if (closer.test(lines[i])) return i + 1;
91
+ return lines.length;
92
+ }
93
+ function listEnd(lines, start) {
94
+ let end = start + 1;
95
+ for (let i = start + 1; i < lines.length; i++) {
96
+ const line = lines[i];
97
+ if (isBlank(line)) {
98
+ let next = i + 1;
99
+ while (next < lines.length && isBlank(lines[next])) next++;
100
+ if (next < lines.length && (LIST_ITEM.test(lines[next]) || isIndented(lines[next]))) {
101
+ i = next - 1;
102
+ continue;
103
+ }
104
+ break;
105
+ }
106
+ if (headingOf(line) && !isIndented(line)) break;
107
+ if (FENCE.test(line) && !isIndented(line)) break;
108
+ end = i + 1;
109
+ }
110
+ return end;
111
+ }
112
+ function tableEnd(lines, start) {
113
+ let end = start + 2;
114
+ while (end < lines.length && !isBlank(lines[end]) && lines[end].includes("|")) end++;
115
+ return end;
116
+ }
117
+ function paragraphEnd(lines, start) {
118
+ let end = start + 1;
119
+ while (end < lines.length && !isBlank(lines[end]) && !headingOf(lines[end]) && !FENCE.test(lines[end]) && !LIST_ITEM.test(lines[end])) {
120
+ end++;
121
+ }
122
+ return end;
123
+ }
124
+ function scan(text) {
125
+ const lines = text.replace(/\r\n?/g, "\n").split("\n");
126
+ const sections = [{ trail: [], blocks: [] }];
127
+ const stack = [];
128
+ let i = 0;
129
+ while (i < lines.length) {
130
+ const line = lines[i];
131
+ if (isBlank(line)) {
132
+ i++;
133
+ continue;
134
+ }
135
+ const heading = FENCE.test(line) ? void 0 : headingOf(line);
136
+ if (heading) {
137
+ while (stack.length && stack[stack.length - 1].level >= heading.level) stack.pop();
138
+ stack.push(heading);
139
+ sections.push({ trail: stack.map((h) => h.title), heading: line, blocks: [] });
140
+ i++;
141
+ continue;
142
+ }
143
+ let end;
144
+ if (FENCE.test(line)) end = fenceEnd(lines, i);
145
+ else if (LIST_ITEM.test(line)) end = listEnd(lines, i);
146
+ else if (startsTable(lines, i)) end = tableEnd(lines, i);
147
+ else end = paragraphEnd(lines, i);
148
+ sections[sections.length - 1].blocks.push({ lines: lines.slice(i, end) });
149
+ i = end;
150
+ }
151
+ return sections;
152
+ }
153
+ function chunkSections(text, max = 900) {
154
+ const cap = Math.max(1, max);
155
+ const out = [];
156
+ const emit = (parts, section) => {
157
+ const body = parts.join("\n\n");
158
+ if (body.trim() === "") return;
159
+ out.push({ text: body, section: [...section], position: out.length });
160
+ };
161
+ for (const section of scan(text)) {
162
+ let parts = section.heading ? [section.heading.trim()] : [];
163
+ let size = parts.join("\n\n").length;
164
+ let hasBlock = false;
165
+ for (const block of section.blocks) {
166
+ const piece = block.lines.join("\n");
167
+ const added = piece.length + (parts.length ? 2 : 0);
168
+ if (hasBlock && size + added > cap) {
169
+ emit(parts, section.trail);
170
+ parts = [];
171
+ size = 0;
172
+ hasBlock = false;
173
+ }
174
+ size += piece.length + (parts.length ? 2 : 0);
175
+ parts.push(piece);
176
+ hasBlock = true;
177
+ }
178
+ emit(parts, section.trail);
179
+ }
180
+ return out;
181
+ }
182
+ var FENCE, HEADING, LIST_ITEM, TABLE_DELIMITER, isBlank, isIndented;
183
+ var init_chunking = __esm({
184
+ "src/core/chunking.ts"() {
185
+ "use strict";
186
+ FENCE = /^ {0,3}(`{3,}|~{3,})/;
187
+ HEADING = /^ {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*$/;
188
+ LIST_ITEM = /^[ \t]*(?:[-*+]|\d{1,9}[.)])[ \t]+\S/;
189
+ TABLE_DELIMITER = /^[ \t]*\|?[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?[ \t]*$/;
190
+ isBlank = (line) => line.trim() === "";
191
+ isIndented = (line) => /^(?: {2,}|\t)/.test(line);
192
+ }
193
+ });
194
+
75
195
  // src/core/loaders.ts
76
196
  import { readdirSync, readFileSync, statSync } from "node:fs";
77
197
  import { join } from "node:path";
@@ -139,6 +259,7 @@ __export(retrieval_exports, {
139
259
  openLibsqlClient: () => openLibsqlClient,
140
260
  wrapMissingLibsqlError: () => wrapMissingLibsqlError
141
261
  });
262
+ import { relative, sep } from "node:path";
142
263
  function createOpenAIEmbedder(client) {
143
264
  return async (input) => {
144
265
  const res = await client.embeddings.create({ model: EMB_MODEL, input });
@@ -181,6 +302,7 @@ var init_retrieval = __esm({
181
302
  "src/core/retrieval.ts"() {
182
303
  "use strict";
183
304
  init_buildLock();
305
+ init_chunking();
184
306
  init_loaders();
185
307
  init_logger();
186
308
  EMB_MODEL = "text-embedding-3-large";
@@ -191,6 +313,7 @@ var init_retrieval = __esm({
191
313
  this.embed = embed;
192
314
  this.db = options.databaseClient;
193
315
  this.knowledgeDir = options.knowledgeDir || DEFAULT_KNOWLEDGE_DIR;
316
+ this.chunking = options.chunking ?? "lines";
194
317
  this.logger = options.logger ?? createConsoleLogger();
195
318
  this.buildLock = options.buildLock ?? createTursoBuildLock(this.db);
196
319
  this.instanceId = options.instanceId ?? crypto.randomUUID();
@@ -204,6 +327,7 @@ var init_retrieval = __esm({
204
327
  */
205
328
  db;
206
329
  knowledgeDir;
330
+ chunking;
207
331
  logger;
208
332
  buildLock;
209
333
  instanceId;
@@ -264,11 +388,41 @@ var init_retrieval = __esm({
264
388
  this.logger.info("No knowledge documents and no existing chunks - nothing to build");
265
389
  return;
266
390
  }
391
+ this.logger.info(`Chunking mode: ${this.chunking}`);
392
+ if (this.chunking === "sections") await this.ensureSectionColumns();
267
393
  const rows = [];
268
394
  for (const d of docs) {
269
- for (const part of chunk(d.text)) {
270
- const id = await sha256(`${d.bucket}|${d.source}|${part}`);
271
- rows.push({ id, bucket: d.bucket, source: d.source, text: part });
395
+ if (this.chunking === "sections") {
396
+ const source = relative(this.knowledgeDir, d.source).split(sep).join("/");
397
+ for (const c of chunkSections(d.text)) {
398
+ const section = c.section.join(" > ");
399
+ const id = await sha256(`${d.bucket}|${source}|${section}|${c.text}`);
400
+ const context = section ? `${source} > ${section}` : source;
401
+ rows.push({
402
+ id,
403
+ bucket: d.bucket,
404
+ source,
405
+ text: c.text,
406
+ embedText: `${context}
407
+
408
+ ${c.text}`,
409
+ section,
410
+ position: c.position
411
+ });
412
+ }
413
+ } else {
414
+ for (const part of chunk(d.text)) {
415
+ const id = await sha256(`${d.bucket}|${d.source}|${part}`);
416
+ rows.push({
417
+ id,
418
+ bucket: d.bucket,
419
+ source: d.source,
420
+ text: part,
421
+ embedText: part,
422
+ section: null,
423
+ position: null
424
+ });
425
+ }
272
426
  }
273
427
  }
274
428
  this.logger.info(`Created ${rows.length} chunks from knowledge documents`);
@@ -277,11 +431,17 @@ var init_retrieval = __esm({
277
431
  for (let i = 0; i < rows.length; i += UPSERT_BATCH) {
278
432
  const batch = rows.slice(i, i + UPSERT_BATCH);
279
433
  await this.db.batch(
280
- batch.map((r) => ({
281
- sql: `INSERT INTO chunks(id,bucket,source,text) VALUES(?,?,?,?)
282
- ON CONFLICT(id) DO NOTHING`,
283
- args: [r.id, r.bucket, r.source, r.text]
284
- })),
434
+ batch.map(
435
+ (r) => this.chunking === "sections" ? {
436
+ sql: `INSERT INTO chunks(id,bucket,source,text,section,position) VALUES(?,?,?,?,?,?)
437
+ ON CONFLICT(id) DO NOTHING`,
438
+ args: [r.id, r.bucket, r.source, r.text, r.section, r.position]
439
+ } : {
440
+ sql: `INSERT INTO chunks(id,bucket,source,text) VALUES(?,?,?,?)
441
+ ON CONFLICT(id) DO NOTHING`,
442
+ args: [r.id, r.bucket, r.source, r.text]
443
+ }
444
+ ),
285
445
  "write"
286
446
  );
287
447
  }
@@ -302,7 +462,7 @@ var init_retrieval = __esm({
302
462
  return;
303
463
  }
304
464
  this.logger.info(`Embedding ${missing.length} new/updated chunks with ${EMB_MODEL}...`);
305
- const textById = new Map(rows.map((r) => [r.id, r.text]));
465
+ const textById = new Map(rows.map((r) => [r.id, r.embedText]));
306
466
  const BATCH = 96;
307
467
  for (let i = 0; i < missing.length; i += BATCH) {
308
468
  const batchIds = missing.slice(i, i + BATCH);
@@ -323,6 +483,18 @@ var init_retrieval = __esm({
323
483
  }
324
484
  this.logger.info(`Successfully embedded ${missing.length} new chunks`);
325
485
  }
486
+ /**
487
+ * Adds the nullable `section`/`position` columns to a `chunks` table that
488
+ * every released version created without them. Checked via table_info so it
489
+ * is a no-op once present; only `'sections'` mode ever calls it.
490
+ */
491
+ async ensureSectionColumns() {
492
+ const info = await this.db.execute("PRAGMA table_info(chunks)");
493
+ const have = new Set(info.rows.map((r) => String(r.name)));
494
+ if (!have.has("section")) await this.db.execute("ALTER TABLE chunks ADD COLUMN section TEXT");
495
+ if (!have.has("position"))
496
+ await this.db.execute("ALTER TABLE chunks ADD COLUMN position INTEGER");
497
+ }
326
498
  // Remove chunks from database that no longer exist in markdown files
327
499
  async cleanupStaleChunks(currentIds) {
328
500
  const result = await this.db.execute("SELECT id FROM chunks");
@@ -369,25 +541,34 @@ var init_retrieval = __esm({
369
541
  * interpolated. An empty list retrieves nothing, and short-circuits before
370
542
  * the embedding call.
371
543
  */
372
- async query(q, k = 6, allowed = ["base"]) {
544
+ async queryChunks(q, k = 6, allowed = ["base"]) {
373
545
  if (allowed.length === 0) return [];
374
546
  const [qv] = await this.embed([q]);
375
547
  const placeholders = allowed.map(() => "?").join(",");
376
548
  const res = await this.db.execute({
377
- sql: `SELECT c.id, c.text, e.embedding
549
+ sql: `SELECT c.*, e.embedding
378
550
  FROM chunks c
379
551
  JOIN embeddings e ON e.id = c.id
380
552
  WHERE c.bucket IN (${placeholders})`,
381
553
  args: allowed
382
554
  });
383
- const scored = [];
384
- for (const row of res.rows) {
555
+ const scored = res.rows.map((row) => {
385
556
  const emb = JSON.parse(String(row.embedding));
386
- const s = _VectorStore.cosine(qv, emb);
387
- scored.push({ s, text: String(row.text) });
388
- }
389
- scored.sort((a, b) => b.s - a.s);
390
- return scored.slice(0, k).map((r) => r.text);
557
+ const trail = row.section == null ? "" : String(row.section);
558
+ return {
559
+ text: String(row.text),
560
+ bucket: String(row.bucket ?? ""),
561
+ source: String(row.source ?? ""),
562
+ section: trail ? trail.split(" > ") : [],
563
+ score: _VectorStore.cosine(qv, emb)
564
+ };
565
+ });
566
+ scored.sort((a, b) => b.score - a.score);
567
+ return scored.slice(0, k);
568
+ }
569
+ /** Same results as {@link queryChunks}, as bare text. */
570
+ async query(q, k = 6, allowed = ["base"]) {
571
+ return (await this.queryChunks(q, k, allowed)).map((c) => c.text);
391
572
  }
392
573
  };
393
574
  }
@@ -790,6 +971,7 @@ async function resolveBuckets({
790
971
 
791
972
  // src/index.ts
792
973
  init_buildLock();
974
+ init_chunking();
793
975
  init_loaders();
794
976
  init_logger();
795
977
 
@@ -1278,6 +1460,7 @@ async function createMCPServer(config) {
1278
1460
  store = new VectorStore2(createOpenAIEmbedder2(client), {
1279
1461
  databaseClient: db,
1280
1462
  knowledgeDir: config.knowledgeDir || DEFAULT_KNOWLEDGE_DIR2,
1463
+ chunking: config.chunking,
1281
1464
  logger: log
1282
1465
  });
1283
1466
  }
@@ -2325,7 +2508,7 @@ function publicRoutes(deps) {
2325
2508
  }
2326
2509
 
2327
2510
  // src/server.ts
2328
- import { relative } from "node:path";
2511
+ import { relative as relative2 } from "node:path";
2329
2512
  import { Hono as Hono5 } from "hono";
2330
2513
  import OpenAI2 from "openai";
2331
2514
 
@@ -2484,6 +2667,7 @@ async function createServer(config) {
2484
2667
  store = new VectorStore2(createOpenAIEmbedder2(client), {
2485
2668
  databaseClient: db,
2486
2669
  knowledgeDir: config.knowledgeDir || DEFAULT_KNOWLEDGE_DIR2,
2670
+ chunking: config.chunking,
2487
2671
  logger
2488
2672
  });
2489
2673
  }
@@ -2534,7 +2718,7 @@ async function createServer(config) {
2534
2718
  }
2535
2719
  }
2536
2720
  if (serveStatic && staticDir) {
2537
- const relativePath = relative(process.cwd(), staticDir);
2721
+ const relativePath = relative2(process.cwd(), staticDir);
2538
2722
  app.get("/chatter.js", serveStatic({ path: `${relativePath}/chatter.js` }));
2539
2723
  app.get("/chatter.css", serveStatic({ path: `${relativePath}/chatter.css` }));
2540
2724
  }
@@ -2695,6 +2879,7 @@ export {
2695
2879
  applyTransformReply,
2696
2880
  canAcquireBuildLock,
2697
2881
  chatBodyLimit,
2882
+ chunkSections,
2698
2883
  completeOnce,
2699
2884
  completeStream,
2700
2885
  cors,
@@ -1 +1 @@
1
- {"version":3,"file":"mcp-server.d.ts","sourceRoot":"","sources":["../src/mcp-server.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AAsBH,YAAY,EACV,WAAW,EACX,QAAQ,EACR,cAAc,EACd,gBAAgB,EAChB,aAAa,EACb,gBAAgB,GACjB,MAAM,oBAAoB,CAAC;AAE5B,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,oBAAoB,CAAC;AAE3D;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAsB,eAAe,CAAC,MAAM,EAAE,gBAAgB;qCA8U3B,IAAI;GAEtC"}
1
+ {"version":3,"file":"mcp-server.d.ts","sourceRoot":"","sources":["../src/mcp-server.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AAsBH,YAAY,EACV,WAAW,EACX,QAAQ,EACR,cAAc,EACd,gBAAgB,EAChB,aAAa,EACb,gBAAgB,GACjB,MAAM,oBAAoB,CAAC;AAE5B,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,oBAAoB,CAAC;AAE3D;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAsB,eAAe,CAAC,MAAM,EAAE,gBAAgB;qCA+U3B,IAAI;GAEtC"}
@@ -114,6 +114,126 @@ var init_buildLock = __esm({
114
114
  }
115
115
  });
116
116
 
117
+ // src/core/chunking.ts
118
+ function headingOf(line) {
119
+ const match = HEADING.exec(line);
120
+ if (!match) return void 0;
121
+ const title = (match[2] ?? "").replace(/[ \t]+#+$/, "").replace(/^#+$/, "").trim();
122
+ return { level: match[1].length, title };
123
+ }
124
+ function startsTable(lines, i) {
125
+ return lines[i].includes("|") && i + 1 < lines.length && TABLE_DELIMITER.test(lines[i + 1]);
126
+ }
127
+ function fenceEnd(lines, start) {
128
+ const opener = FENCE.exec(lines[start]);
129
+ if (!opener) return start + 1;
130
+ const marker = opener[1];
131
+ const closer = new RegExp(`^ {0,3}${marker[0] === "`" ? "`" : "~"}{${marker.length},}[ \\t]*$`);
132
+ for (let i = start + 1; i < lines.length; i++) if (closer.test(lines[i])) return i + 1;
133
+ return lines.length;
134
+ }
135
+ function listEnd(lines, start) {
136
+ let end = start + 1;
137
+ for (let i = start + 1; i < lines.length; i++) {
138
+ const line = lines[i];
139
+ if (isBlank(line)) {
140
+ let next = i + 1;
141
+ while (next < lines.length && isBlank(lines[next])) next++;
142
+ if (next < lines.length && (LIST_ITEM.test(lines[next]) || isIndented(lines[next]))) {
143
+ i = next - 1;
144
+ continue;
145
+ }
146
+ break;
147
+ }
148
+ if (headingOf(line) && !isIndented(line)) break;
149
+ if (FENCE.test(line) && !isIndented(line)) break;
150
+ end = i + 1;
151
+ }
152
+ return end;
153
+ }
154
+ function tableEnd(lines, start) {
155
+ let end = start + 2;
156
+ while (end < lines.length && !isBlank(lines[end]) && lines[end].includes("|")) end++;
157
+ return end;
158
+ }
159
+ function paragraphEnd(lines, start) {
160
+ let end = start + 1;
161
+ while (end < lines.length && !isBlank(lines[end]) && !headingOf(lines[end]) && !FENCE.test(lines[end]) && !LIST_ITEM.test(lines[end])) {
162
+ end++;
163
+ }
164
+ return end;
165
+ }
166
+ function scan(text) {
167
+ const lines = text.replace(/\r\n?/g, "\n").split("\n");
168
+ const sections = [{ trail: [], blocks: [] }];
169
+ const stack = [];
170
+ let i = 0;
171
+ while (i < lines.length) {
172
+ const line = lines[i];
173
+ if (isBlank(line)) {
174
+ i++;
175
+ continue;
176
+ }
177
+ const heading = FENCE.test(line) ? void 0 : headingOf(line);
178
+ if (heading) {
179
+ while (stack.length && stack[stack.length - 1].level >= heading.level) stack.pop();
180
+ stack.push(heading);
181
+ sections.push({ trail: stack.map((h) => h.title), heading: line, blocks: [] });
182
+ i++;
183
+ continue;
184
+ }
185
+ let end;
186
+ if (FENCE.test(line)) end = fenceEnd(lines, i);
187
+ else if (LIST_ITEM.test(line)) end = listEnd(lines, i);
188
+ else if (startsTable(lines, i)) end = tableEnd(lines, i);
189
+ else end = paragraphEnd(lines, i);
190
+ sections[sections.length - 1].blocks.push({ lines: lines.slice(i, end) });
191
+ i = end;
192
+ }
193
+ return sections;
194
+ }
195
+ function chunkSections(text, max = 900) {
196
+ const cap = Math.max(1, max);
197
+ const out = [];
198
+ const emit = (parts, section) => {
199
+ const body = parts.join("\n\n");
200
+ if (body.trim() === "") return;
201
+ out.push({ text: body, section: [...section], position: out.length });
202
+ };
203
+ for (const section of scan(text)) {
204
+ let parts = section.heading ? [section.heading.trim()] : [];
205
+ let size = parts.join("\n\n").length;
206
+ let hasBlock = false;
207
+ for (const block of section.blocks) {
208
+ const piece = block.lines.join("\n");
209
+ const added = piece.length + (parts.length ? 2 : 0);
210
+ if (hasBlock && size + added > cap) {
211
+ emit(parts, section.trail);
212
+ parts = [];
213
+ size = 0;
214
+ hasBlock = false;
215
+ }
216
+ size += piece.length + (parts.length ? 2 : 0);
217
+ parts.push(piece);
218
+ hasBlock = true;
219
+ }
220
+ emit(parts, section.trail);
221
+ }
222
+ return out;
223
+ }
224
+ var FENCE, HEADING, LIST_ITEM, TABLE_DELIMITER, isBlank, isIndented;
225
+ var init_chunking = __esm({
226
+ "src/core/chunking.ts"() {
227
+ "use strict";
228
+ FENCE = /^ {0,3}(`{3,}|~{3,})/;
229
+ HEADING = /^ {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*$/;
230
+ LIST_ITEM = /^[ \t]*(?:[-*+]|\d{1,9}[.)])[ \t]+\S/;
231
+ TABLE_DELIMITER = /^[ \t]*\|?[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?[ \t]*$/;
232
+ isBlank = (line) => line.trim() === "";
233
+ isIndented = (line) => /^(?: {2,}|\t)/.test(line);
234
+ }
235
+ });
236
+
117
237
  // src/core/loaders.ts
118
238
  function walk(dir) {
119
239
  const items = [];
@@ -194,11 +314,13 @@ async function sha256(input) {
194
314
  const hash = await crypto.subtle.digest("SHA-256", data);
195
315
  return Array.from(new Uint8Array(hash)).map((b) => b.toString(16).padStart(2, "0")).join("");
196
316
  }
197
- var EMB_MODEL, DEFAULT_KNOWLEDGE_DIR, DATABASE_CONFIG_REQUIRED_MESSAGE, VectorStore;
317
+ var import_node_path3, EMB_MODEL, DEFAULT_KNOWLEDGE_DIR, DATABASE_CONFIG_REQUIRED_MESSAGE, VectorStore;
198
318
  var init_retrieval = __esm({
199
319
  "src/core/retrieval.ts"() {
200
320
  "use strict";
321
+ import_node_path3 = require("node:path");
201
322
  init_buildLock();
323
+ init_chunking();
202
324
  init_loaders();
203
325
  init_logger();
204
326
  EMB_MODEL = "text-embedding-3-large";
@@ -209,6 +331,7 @@ var init_retrieval = __esm({
209
331
  this.embed = embed;
210
332
  this.db = options.databaseClient;
211
333
  this.knowledgeDir = options.knowledgeDir || DEFAULT_KNOWLEDGE_DIR;
334
+ this.chunking = options.chunking ?? "lines";
212
335
  this.logger = options.logger ?? createConsoleLogger();
213
336
  this.buildLock = options.buildLock ?? createTursoBuildLock(this.db);
214
337
  this.instanceId = options.instanceId ?? crypto.randomUUID();
@@ -222,6 +345,7 @@ var init_retrieval = __esm({
222
345
  */
223
346
  db;
224
347
  knowledgeDir;
348
+ chunking;
225
349
  logger;
226
350
  buildLock;
227
351
  instanceId;
@@ -282,11 +406,41 @@ var init_retrieval = __esm({
282
406
  this.logger.info("No knowledge documents and no existing chunks - nothing to build");
283
407
  return;
284
408
  }
409
+ this.logger.info(`Chunking mode: ${this.chunking}`);
410
+ if (this.chunking === "sections") await this.ensureSectionColumns();
285
411
  const rows = [];
286
412
  for (const d of docs) {
287
- for (const part of chunk(d.text)) {
288
- const id = await sha256(`${d.bucket}|${d.source}|${part}`);
289
- rows.push({ id, bucket: d.bucket, source: d.source, text: part });
413
+ if (this.chunking === "sections") {
414
+ const source = (0, import_node_path3.relative)(this.knowledgeDir, d.source).split(import_node_path3.sep).join("/");
415
+ for (const c of chunkSections(d.text)) {
416
+ const section = c.section.join(" > ");
417
+ const id = await sha256(`${d.bucket}|${source}|${section}|${c.text}`);
418
+ const context = section ? `${source} > ${section}` : source;
419
+ rows.push({
420
+ id,
421
+ bucket: d.bucket,
422
+ source,
423
+ text: c.text,
424
+ embedText: `${context}
425
+
426
+ ${c.text}`,
427
+ section,
428
+ position: c.position
429
+ });
430
+ }
431
+ } else {
432
+ for (const part of chunk(d.text)) {
433
+ const id = await sha256(`${d.bucket}|${d.source}|${part}`);
434
+ rows.push({
435
+ id,
436
+ bucket: d.bucket,
437
+ source: d.source,
438
+ text: part,
439
+ embedText: part,
440
+ section: null,
441
+ position: null
442
+ });
443
+ }
290
444
  }
291
445
  }
292
446
  this.logger.info(`Created ${rows.length} chunks from knowledge documents`);
@@ -295,11 +449,17 @@ var init_retrieval = __esm({
295
449
  for (let i = 0; i < rows.length; i += UPSERT_BATCH) {
296
450
  const batch = rows.slice(i, i + UPSERT_BATCH);
297
451
  await this.db.batch(
298
- batch.map((r) => ({
299
- sql: `INSERT INTO chunks(id,bucket,source,text) VALUES(?,?,?,?)
300
- ON CONFLICT(id) DO NOTHING`,
301
- args: [r.id, r.bucket, r.source, r.text]
302
- })),
452
+ batch.map(
453
+ (r) => this.chunking === "sections" ? {
454
+ sql: `INSERT INTO chunks(id,bucket,source,text,section,position) VALUES(?,?,?,?,?,?)
455
+ ON CONFLICT(id) DO NOTHING`,
456
+ args: [r.id, r.bucket, r.source, r.text, r.section, r.position]
457
+ } : {
458
+ sql: `INSERT INTO chunks(id,bucket,source,text) VALUES(?,?,?,?)
459
+ ON CONFLICT(id) DO NOTHING`,
460
+ args: [r.id, r.bucket, r.source, r.text]
461
+ }
462
+ ),
303
463
  "write"
304
464
  );
305
465
  }
@@ -320,7 +480,7 @@ var init_retrieval = __esm({
320
480
  return;
321
481
  }
322
482
  this.logger.info(`Embedding ${missing.length} new/updated chunks with ${EMB_MODEL}...`);
323
- const textById = new Map(rows.map((r) => [r.id, r.text]));
483
+ const textById = new Map(rows.map((r) => [r.id, r.embedText]));
324
484
  const BATCH = 96;
325
485
  for (let i = 0; i < missing.length; i += BATCH) {
326
486
  const batchIds = missing.slice(i, i + BATCH);
@@ -341,6 +501,18 @@ var init_retrieval = __esm({
341
501
  }
342
502
  this.logger.info(`Successfully embedded ${missing.length} new chunks`);
343
503
  }
504
+ /**
505
+ * Adds the nullable `section`/`position` columns to a `chunks` table that
506
+ * every released version created without them. Checked via table_info so it
507
+ * is a no-op once present; only `'sections'` mode ever calls it.
508
+ */
509
+ async ensureSectionColumns() {
510
+ const info = await this.db.execute("PRAGMA table_info(chunks)");
511
+ const have = new Set(info.rows.map((r) => String(r.name)));
512
+ if (!have.has("section")) await this.db.execute("ALTER TABLE chunks ADD COLUMN section TEXT");
513
+ if (!have.has("position"))
514
+ await this.db.execute("ALTER TABLE chunks ADD COLUMN position INTEGER");
515
+ }
344
516
  // Remove chunks from database that no longer exist in markdown files
345
517
  async cleanupStaleChunks(currentIds) {
346
518
  const result = await this.db.execute("SELECT id FROM chunks");
@@ -387,25 +559,34 @@ var init_retrieval = __esm({
387
559
  * interpolated. An empty list retrieves nothing, and short-circuits before
388
560
  * the embedding call.
389
561
  */
390
- async query(q, k = 6, allowed = ["base"]) {
562
+ async queryChunks(q, k = 6, allowed = ["base"]) {
391
563
  if (allowed.length === 0) return [];
392
564
  const [qv] = await this.embed([q]);
393
565
  const placeholders = allowed.map(() => "?").join(",");
394
566
  const res = await this.db.execute({
395
- sql: `SELECT c.id, c.text, e.embedding
567
+ sql: `SELECT c.*, e.embedding
396
568
  FROM chunks c
397
569
  JOIN embeddings e ON e.id = c.id
398
570
  WHERE c.bucket IN (${placeholders})`,
399
571
  args: allowed
400
572
  });
401
- const scored = [];
402
- for (const row of res.rows) {
573
+ const scored = res.rows.map((row) => {
403
574
  const emb = JSON.parse(String(row.embedding));
404
- const s = _VectorStore.cosine(qv, emb);
405
- scored.push({ s, text: String(row.text) });
406
- }
407
- scored.sort((a, b) => b.s - a.s);
408
- return scored.slice(0, k).map((r) => r.text);
575
+ const trail = row.section == null ? "" : String(row.section);
576
+ return {
577
+ text: String(row.text),
578
+ bucket: String(row.bucket ?? ""),
579
+ source: String(row.source ?? ""),
580
+ section: trail ? trail.split(" > ") : [],
581
+ score: _VectorStore.cosine(qv, emb)
582
+ };
583
+ });
584
+ scored.sort((a, b) => b.score - a.score);
585
+ return scored.slice(0, k);
586
+ }
587
+ /** Same results as {@link queryChunks}, as bare text. */
588
+ async query(q, k = 6, allowed = ["base"]) {
589
+ return (await this.queryChunks(q, k, allowed)).map((c) => c.text);
409
590
  }
410
591
  };
411
592
  }
@@ -1032,6 +1213,7 @@ async function createMCPServer(config) {
1032
1213
  store = new VectorStore2(createOpenAIEmbedder2(client), {
1033
1214
  databaseClient: db,
1034
1215
  knowledgeDir: config.knowledgeDir || DEFAULT_KNOWLEDGE_DIR2,
1216
+ chunking: config.chunking,
1035
1217
  logger: log
1036
1218
  });
1037
1219
  }