@diegoaltoworks/chatter 2.0.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/server.js CHANGED
@@ -114,6 +114,126 @@ var init_buildLock = __esm({
114
114
  }
115
115
  });
116
116
 
117
+ // src/core/chunking.ts
118
+ function headingOf(line) {
119
+ const match = HEADING.exec(line);
120
+ if (!match) return void 0;
121
+ const title = (match[2] ?? "").replace(/[ \t]+#+$/, "").replace(/^#+$/, "").trim();
122
+ return { level: match[1].length, title };
123
+ }
124
+ function startsTable(lines, i) {
125
+ return lines[i].includes("|") && i + 1 < lines.length && TABLE_DELIMITER.test(lines[i + 1]);
126
+ }
127
+ function fenceEnd(lines, start) {
128
+ const opener = FENCE.exec(lines[start]);
129
+ if (!opener) return start + 1;
130
+ const marker = opener[1];
131
+ const closer = new RegExp(`^ {0,3}${marker[0] === "`" ? "`" : "~"}{${marker.length},}[ \\t]*$`);
132
+ for (let i = start + 1; i < lines.length; i++) if (closer.test(lines[i])) return i + 1;
133
+ return lines.length;
134
+ }
135
+ function listEnd(lines, start) {
136
+ let end = start + 1;
137
+ for (let i = start + 1; i < lines.length; i++) {
138
+ const line = lines[i];
139
+ if (isBlank(line)) {
140
+ let next = i + 1;
141
+ while (next < lines.length && isBlank(lines[next])) next++;
142
+ if (next < lines.length && (LIST_ITEM.test(lines[next]) || isIndented(lines[next]))) {
143
+ i = next - 1;
144
+ continue;
145
+ }
146
+ break;
147
+ }
148
+ if (headingOf(line) && !isIndented(line)) break;
149
+ if (FENCE.test(line) && !isIndented(line)) break;
150
+ end = i + 1;
151
+ }
152
+ return end;
153
+ }
154
+ function tableEnd(lines, start) {
155
+ let end = start + 2;
156
+ while (end < lines.length && !isBlank(lines[end]) && lines[end].includes("|")) end++;
157
+ return end;
158
+ }
159
+ function paragraphEnd(lines, start) {
160
+ let end = start + 1;
161
+ while (end < lines.length && !isBlank(lines[end]) && !headingOf(lines[end]) && !FENCE.test(lines[end]) && !LIST_ITEM.test(lines[end])) {
162
+ end++;
163
+ }
164
+ return end;
165
+ }
166
+ function scan(text) {
167
+ const lines = text.replace(/\r\n?/g, "\n").split("\n");
168
+ const sections = [{ trail: [], blocks: [] }];
169
+ const stack = [];
170
+ let i = 0;
171
+ while (i < lines.length) {
172
+ const line = lines[i];
173
+ if (isBlank(line)) {
174
+ i++;
175
+ continue;
176
+ }
177
+ const heading = FENCE.test(line) ? void 0 : headingOf(line);
178
+ if (heading) {
179
+ while (stack.length && stack[stack.length - 1].level >= heading.level) stack.pop();
180
+ stack.push(heading);
181
+ sections.push({ trail: stack.map((h) => h.title), heading: line, blocks: [] });
182
+ i++;
183
+ continue;
184
+ }
185
+ let end;
186
+ if (FENCE.test(line)) end = fenceEnd(lines, i);
187
+ else if (LIST_ITEM.test(line)) end = listEnd(lines, i);
188
+ else if (startsTable(lines, i)) end = tableEnd(lines, i);
189
+ else end = paragraphEnd(lines, i);
190
+ sections[sections.length - 1].blocks.push({ lines: lines.slice(i, end) });
191
+ i = end;
192
+ }
193
+ return sections;
194
+ }
195
+ function chunkSections(text, max = 900) {
196
+ const cap = Math.max(1, max);
197
+ const out = [];
198
+ const emit = (parts, section) => {
199
+ const body = parts.join("\n\n");
200
+ if (body.trim() === "") return;
201
+ out.push({ text: body, section: [...section], position: out.length });
202
+ };
203
+ for (const section of scan(text)) {
204
+ let parts = section.heading ? [section.heading.trim()] : [];
205
+ let size = parts.join("\n\n").length;
206
+ let hasBlock = false;
207
+ for (const block of section.blocks) {
208
+ const piece = block.lines.join("\n");
209
+ const added = piece.length + (parts.length ? 2 : 0);
210
+ if (hasBlock && size + added > cap) {
211
+ emit(parts, section.trail);
212
+ parts = [];
213
+ size = 0;
214
+ hasBlock = false;
215
+ }
216
+ size += piece.length + (parts.length ? 2 : 0);
217
+ parts.push(piece);
218
+ hasBlock = true;
219
+ }
220
+ emit(parts, section.trail);
221
+ }
222
+ return out;
223
+ }
224
+ var FENCE, HEADING, LIST_ITEM, TABLE_DELIMITER, isBlank, isIndented;
225
+ var init_chunking = __esm({
226
+ "src/core/chunking.ts"() {
227
+ "use strict";
228
+ FENCE = /^ {0,3}(`{3,}|~{3,})/;
229
+ HEADING = /^ {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*$/;
230
+ LIST_ITEM = /^[ \t]*(?:[-*+]|\d{1,9}[.)])[ \t]+\S/;
231
+ TABLE_DELIMITER = /^[ \t]*\|?[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?[ \t]*$/;
232
+ isBlank = (line) => line.trim() === "";
233
+ isIndented = (line) => /^(?: {2,}|\t)/.test(line);
234
+ }
235
+ });
236
+
117
237
  // src/core/loaders.ts
118
238
  function walk(dir) {
119
239
  const items = [];
@@ -194,11 +314,13 @@ async function sha256(input) {
194
314
  const hash = await crypto.subtle.digest("SHA-256", data);
195
315
  return Array.from(new Uint8Array(hash)).map((b) => b.toString(16).padStart(2, "0")).join("");
196
316
  }
197
- var EMB_MODEL, DEFAULT_KNOWLEDGE_DIR, DATABASE_CONFIG_REQUIRED_MESSAGE, VectorStore;
317
+ var import_node_path3, EMB_MODEL, DEFAULT_KNOWLEDGE_DIR, DATABASE_CONFIG_REQUIRED_MESSAGE, VectorStore;
198
318
  var init_retrieval = __esm({
199
319
  "src/core/retrieval.ts"() {
200
320
  "use strict";
321
+ import_node_path3 = require("node:path");
201
322
  init_buildLock();
323
+ init_chunking();
202
324
  init_loaders();
203
325
  init_logger();
204
326
  EMB_MODEL = "text-embedding-3-large";
@@ -209,6 +331,7 @@ var init_retrieval = __esm({
209
331
  this.embed = embed;
210
332
  this.db = options.databaseClient;
211
333
  this.knowledgeDir = options.knowledgeDir || DEFAULT_KNOWLEDGE_DIR;
334
+ this.chunking = options.chunking ?? "lines";
212
335
  this.logger = options.logger ?? createConsoleLogger();
213
336
  this.buildLock = options.buildLock ?? createTursoBuildLock(this.db);
214
337
  this.instanceId = options.instanceId ?? crypto.randomUUID();
@@ -222,6 +345,7 @@ var init_retrieval = __esm({
222
345
  */
223
346
  db;
224
347
  knowledgeDir;
348
+ chunking;
225
349
  logger;
226
350
  buildLock;
227
351
  instanceId;
@@ -282,11 +406,41 @@ var init_retrieval = __esm({
282
406
  this.logger.info("No knowledge documents and no existing chunks - nothing to build");
283
407
  return;
284
408
  }
409
+ this.logger.info(`Chunking mode: ${this.chunking}`);
410
+ if (this.chunking === "sections") await this.ensureSectionColumns();
285
411
  const rows = [];
286
412
  for (const d of docs) {
287
- for (const part of chunk(d.text)) {
288
- const id = await sha256(`${d.bucket}|${d.source}|${part}`);
289
- rows.push({ id, bucket: d.bucket, source: d.source, text: part });
413
+ if (this.chunking === "sections") {
414
+ const source = (0, import_node_path3.relative)(this.knowledgeDir, d.source).split(import_node_path3.sep).join("/");
415
+ for (const c of chunkSections(d.text)) {
416
+ const section = c.section.join(" > ");
417
+ const id = await sha256(`${d.bucket}|${source}|${section}|${c.text}`);
418
+ const context = section ? `${source} > ${section}` : source;
419
+ rows.push({
420
+ id,
421
+ bucket: d.bucket,
422
+ source,
423
+ text: c.text,
424
+ embedText: `${context}
425
+
426
+ ${c.text}`,
427
+ section,
428
+ position: c.position
429
+ });
430
+ }
431
+ } else {
432
+ for (const part of chunk(d.text)) {
433
+ const id = await sha256(`${d.bucket}|${d.source}|${part}`);
434
+ rows.push({
435
+ id,
436
+ bucket: d.bucket,
437
+ source: d.source,
438
+ text: part,
439
+ embedText: part,
440
+ section: null,
441
+ position: null
442
+ });
443
+ }
290
444
  }
291
445
  }
292
446
  this.logger.info(`Created ${rows.length} chunks from knowledge documents`);
@@ -295,11 +449,17 @@ var init_retrieval = __esm({
295
449
  for (let i = 0; i < rows.length; i += UPSERT_BATCH) {
296
450
  const batch = rows.slice(i, i + UPSERT_BATCH);
297
451
  await this.db.batch(
298
- batch.map((r) => ({
299
- sql: `INSERT INTO chunks(id,bucket,source,text) VALUES(?,?,?,?)
300
- ON CONFLICT(id) DO NOTHING`,
301
- args: [r.id, r.bucket, r.source, r.text]
302
- })),
452
+ batch.map(
453
+ (r) => this.chunking === "sections" ? {
454
+ sql: `INSERT INTO chunks(id,bucket,source,text,section,position) VALUES(?,?,?,?,?,?)
455
+ ON CONFLICT(id) DO NOTHING`,
456
+ args: [r.id, r.bucket, r.source, r.text, r.section, r.position]
457
+ } : {
458
+ sql: `INSERT INTO chunks(id,bucket,source,text) VALUES(?,?,?,?)
459
+ ON CONFLICT(id) DO NOTHING`,
460
+ args: [r.id, r.bucket, r.source, r.text]
461
+ }
462
+ ),
303
463
  "write"
304
464
  );
305
465
  }
@@ -320,7 +480,7 @@ var init_retrieval = __esm({
320
480
  return;
321
481
  }
322
482
  this.logger.info(`Embedding ${missing.length} new/updated chunks with ${EMB_MODEL}...`);
323
- const textById = new Map(rows.map((r) => [r.id, r.text]));
483
+ const textById = new Map(rows.map((r) => [r.id, r.embedText]));
324
484
  const BATCH = 96;
325
485
  for (let i = 0; i < missing.length; i += BATCH) {
326
486
  const batchIds = missing.slice(i, i + BATCH);
@@ -341,6 +501,18 @@ var init_retrieval = __esm({
341
501
  }
342
502
  this.logger.info(`Successfully embedded ${missing.length} new chunks`);
343
503
  }
504
+ /**
505
+ * Adds the nullable `section`/`position` columns to a `chunks` table that
506
+ * every released version created without them. Checked via table_info so it
507
+ * is a no-op once present; only `'sections'` mode ever calls it.
508
+ */
509
+ async ensureSectionColumns() {
510
+ const info = await this.db.execute("PRAGMA table_info(chunks)");
511
+ const have = new Set(info.rows.map((r) => String(r.name)));
512
+ if (!have.has("section")) await this.db.execute("ALTER TABLE chunks ADD COLUMN section TEXT");
513
+ if (!have.has("position"))
514
+ await this.db.execute("ALTER TABLE chunks ADD COLUMN position INTEGER");
515
+ }
344
516
  // Remove chunks from database that no longer exist in markdown files
345
517
  async cleanupStaleChunks(currentIds) {
346
518
  const result = await this.db.execute("SELECT id FROM chunks");
@@ -387,25 +559,34 @@ var init_retrieval = __esm({
387
559
  * interpolated. An empty list retrieves nothing, and short-circuits before
388
560
  * the embedding call.
389
561
  */
390
- async query(q, k = 6, allowed = ["base"]) {
562
+ async queryChunks(q, k = 6, allowed = ["base"]) {
391
563
  if (allowed.length === 0) return [];
392
564
  const [qv] = await this.embed([q]);
393
565
  const placeholders = allowed.map(() => "?").join(",");
394
566
  const res = await this.db.execute({
395
- sql: `SELECT c.id, c.text, e.embedding
567
+ sql: `SELECT c.*, e.embedding
396
568
  FROM chunks c
397
569
  JOIN embeddings e ON e.id = c.id
398
570
  WHERE c.bucket IN (${placeholders})`,
399
571
  args: allowed
400
572
  });
401
- const scored = [];
402
- for (const row of res.rows) {
573
+ const scored = res.rows.map((row) => {
403
574
  const emb = JSON.parse(String(row.embedding));
404
- const s = _VectorStore.cosine(qv, emb);
405
- scored.push({ s, text: String(row.text) });
406
- }
407
- scored.sort((a, b) => b.s - a.s);
408
- return scored.slice(0, k).map((r) => r.text);
575
+ const trail = row.section == null ? "" : String(row.section);
576
+ return {
577
+ text: String(row.text),
578
+ bucket: String(row.bucket ?? ""),
579
+ source: String(row.source ?? ""),
580
+ section: trail ? trail.split(" > ") : [],
581
+ score: _VectorStore.cosine(qv, emb)
582
+ };
583
+ });
584
+ scored.sort((a, b) => b.score - a.score);
585
+ return scored.slice(0, k);
586
+ }
587
+ /** Same results as {@link queryChunks}, as bare text. */
588
+ async query(q, k = 6, allowed = ["base"]) {
589
+ return (await this.queryChunks(q, k, allowed)).map((c) => c.text);
409
590
  }
410
591
  };
411
592
  }
@@ -530,7 +711,7 @@ __export(server_exports, {
530
711
  createServer: () => createServer
531
712
  });
532
713
  module.exports = __toCommonJS(server_exports);
533
- var import_node_path4 = require("node:path");
714
+ var import_node_path5 = require("node:path");
534
715
  var import_hono5 = require("hono");
535
716
  var import_openai = __toESM(require("openai"));
536
717
 
@@ -1078,7 +1259,7 @@ function loadServeStatic(runtime = detectRuntime()) {
1078
1259
 
1079
1260
  // src/core/widgets.ts
1080
1261
  var import_node_fs3 = require("node:fs");
1081
- var import_node_path3 = require("node:path");
1262
+ var import_node_path4 = require("node:path");
1082
1263
  var import_node_url = require("node:url");
1083
1264
  var import_meta = {};
1084
1265
  function resolveStatic(configStaticDir) {
@@ -1087,15 +1268,15 @@ function resolveStatic(configStaticDir) {
1087
1268
  }
1088
1269
  try {
1089
1270
  if (import_meta.url) {
1090
- const moduleDir = (0, import_node_path3.dirname)((0, import_node_url.fileURLToPath)(import_meta.url));
1091
- const staticDir = (0, import_node_path3.join)(moduleDir, "static");
1271
+ const moduleDir = (0, import_node_path4.dirname)((0, import_node_url.fileURLToPath)(import_meta.url));
1272
+ const staticDir = (0, import_node_path4.join)(moduleDir, "static");
1092
1273
  if ((0, import_node_fs3.existsSync)(staticDir)) return { staticDir };
1093
1274
  }
1094
1275
  } catch {
1095
1276
  }
1096
1277
  try {
1097
1278
  if (typeof __dirname !== "undefined") {
1098
- const staticDir = (0, import_node_path3.join)(__dirname, "static");
1279
+ const staticDir = (0, import_node_path4.join)(__dirname, "static");
1099
1280
  if ((0, import_node_fs3.existsSync)(staticDir)) return { staticDir };
1100
1281
  }
1101
1282
  } catch {
@@ -2021,6 +2202,7 @@ async function createServer(config) {
2021
2202
  store = new VectorStore2(createOpenAIEmbedder2(client), {
2022
2203
  databaseClient: db,
2023
2204
  knowledgeDir: config.knowledgeDir || DEFAULT_KNOWLEDGE_DIR2,
2205
+ chunking: config.chunking,
2024
2206
  logger
2025
2207
  });
2026
2208
  }
@@ -2071,7 +2253,7 @@ async function createServer(config) {
2071
2253
  }
2072
2254
  }
2073
2255
  if (serveStatic && staticDir) {
2074
- const relativePath = (0, import_node_path4.relative)(process.cwd(), staticDir);
2256
+ const relativePath = (0, import_node_path5.relative)(process.cwd(), staticDir);
2075
2257
  app.get("/chatter.js", serveStatic({ path: `${relativePath}/chatter.js` }));
2076
2258
  app.get("/chatter.css", serveStatic({ path: `${relativePath}/chatter.css` }));
2077
2259
  }
package/dist/server.mjs CHANGED
@@ -92,6 +92,126 @@ var init_buildLock = __esm({
92
92
  }
93
93
  });
94
94
 
95
+ // src/core/chunking.ts
96
+ function headingOf(line) {
97
+ const match = HEADING.exec(line);
98
+ if (!match) return void 0;
99
+ const title = (match[2] ?? "").replace(/[ \t]+#+$/, "").replace(/^#+$/, "").trim();
100
+ return { level: match[1].length, title };
101
+ }
102
+ function startsTable(lines, i) {
103
+ return lines[i].includes("|") && i + 1 < lines.length && TABLE_DELIMITER.test(lines[i + 1]);
104
+ }
105
+ function fenceEnd(lines, start) {
106
+ const opener = FENCE.exec(lines[start]);
107
+ if (!opener) return start + 1;
108
+ const marker = opener[1];
109
+ const closer = new RegExp(`^ {0,3}${marker[0] === "`" ? "`" : "~"}{${marker.length},}[ \\t]*$`);
110
+ for (let i = start + 1; i < lines.length; i++) if (closer.test(lines[i])) return i + 1;
111
+ return lines.length;
112
+ }
113
+ function listEnd(lines, start) {
114
+ let end = start + 1;
115
+ for (let i = start + 1; i < lines.length; i++) {
116
+ const line = lines[i];
117
+ if (isBlank(line)) {
118
+ let next = i + 1;
119
+ while (next < lines.length && isBlank(lines[next])) next++;
120
+ if (next < lines.length && (LIST_ITEM.test(lines[next]) || isIndented(lines[next]))) {
121
+ i = next - 1;
122
+ continue;
123
+ }
124
+ break;
125
+ }
126
+ if (headingOf(line) && !isIndented(line)) break;
127
+ if (FENCE.test(line) && !isIndented(line)) break;
128
+ end = i + 1;
129
+ }
130
+ return end;
131
+ }
132
+ function tableEnd(lines, start) {
133
+ let end = start + 2;
134
+ while (end < lines.length && !isBlank(lines[end]) && lines[end].includes("|")) end++;
135
+ return end;
136
+ }
137
+ function paragraphEnd(lines, start) {
138
+ let end = start + 1;
139
+ while (end < lines.length && !isBlank(lines[end]) && !headingOf(lines[end]) && !FENCE.test(lines[end]) && !LIST_ITEM.test(lines[end])) {
140
+ end++;
141
+ }
142
+ return end;
143
+ }
144
+ function scan(text) {
145
+ const lines = text.replace(/\r\n?/g, "\n").split("\n");
146
+ const sections = [{ trail: [], blocks: [] }];
147
+ const stack = [];
148
+ let i = 0;
149
+ while (i < lines.length) {
150
+ const line = lines[i];
151
+ if (isBlank(line)) {
152
+ i++;
153
+ continue;
154
+ }
155
+ const heading = FENCE.test(line) ? void 0 : headingOf(line);
156
+ if (heading) {
157
+ while (stack.length && stack[stack.length - 1].level >= heading.level) stack.pop();
158
+ stack.push(heading);
159
+ sections.push({ trail: stack.map((h) => h.title), heading: line, blocks: [] });
160
+ i++;
161
+ continue;
162
+ }
163
+ let end;
164
+ if (FENCE.test(line)) end = fenceEnd(lines, i);
165
+ else if (LIST_ITEM.test(line)) end = listEnd(lines, i);
166
+ else if (startsTable(lines, i)) end = tableEnd(lines, i);
167
+ else end = paragraphEnd(lines, i);
168
+ sections[sections.length - 1].blocks.push({ lines: lines.slice(i, end) });
169
+ i = end;
170
+ }
171
+ return sections;
172
+ }
173
+ function chunkSections(text, max = 900) {
174
+ const cap = Math.max(1, max);
175
+ const out = [];
176
+ const emit = (parts, section) => {
177
+ const body = parts.join("\n\n");
178
+ if (body.trim() === "") return;
179
+ out.push({ text: body, section: [...section], position: out.length });
180
+ };
181
+ for (const section of scan(text)) {
182
+ let parts = section.heading ? [section.heading.trim()] : [];
183
+ let size = parts.join("\n\n").length;
184
+ let hasBlock = false;
185
+ for (const block of section.blocks) {
186
+ const piece = block.lines.join("\n");
187
+ const added = piece.length + (parts.length ? 2 : 0);
188
+ if (hasBlock && size + added > cap) {
189
+ emit(parts, section.trail);
190
+ parts = [];
191
+ size = 0;
192
+ hasBlock = false;
193
+ }
194
+ size += piece.length + (parts.length ? 2 : 0);
195
+ parts.push(piece);
196
+ hasBlock = true;
197
+ }
198
+ emit(parts, section.trail);
199
+ }
200
+ return out;
201
+ }
202
+ var FENCE, HEADING, LIST_ITEM, TABLE_DELIMITER, isBlank, isIndented;
203
+ var init_chunking = __esm({
204
+ "src/core/chunking.ts"() {
205
+ "use strict";
206
+ FENCE = /^ {0,3}(`{3,}|~{3,})/;
207
+ HEADING = /^ {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*$/;
208
+ LIST_ITEM = /^[ \t]*(?:[-*+]|\d{1,9}[.)])[ \t]+\S/;
209
+ TABLE_DELIMITER = /^[ \t]*\|?[ \t]*:?-+:?[ \t]*(?:\|[ \t]*:?-+:?[ \t]*)*\|?[ \t]*$/;
210
+ isBlank = (line) => line.trim() === "";
211
+ isIndented = (line) => /^(?: {2,}|\t)/.test(line);
212
+ }
213
+ });
214
+
95
215
  // src/core/loaders.ts
96
216
  import { readdirSync, readFileSync as readFileSync2, statSync } from "node:fs";
97
217
  import { join as join2 } from "node:path";
@@ -134,6 +254,7 @@ __export(retrieval_exports, {
134
254
  openLibsqlClient: () => openLibsqlClient,
135
255
  wrapMissingLibsqlError: () => wrapMissingLibsqlError
136
256
  });
257
+ import { relative, sep } from "node:path";
137
258
  function createOpenAIEmbedder(client) {
138
259
  return async (input) => {
139
260
  const res = await client.embeddings.create({ model: EMB_MODEL, input });
@@ -176,6 +297,7 @@ var init_retrieval = __esm({
176
297
  "src/core/retrieval.ts"() {
177
298
  "use strict";
178
299
  init_buildLock();
300
+ init_chunking();
179
301
  init_loaders();
180
302
  init_logger();
181
303
  EMB_MODEL = "text-embedding-3-large";
@@ -186,6 +308,7 @@ var init_retrieval = __esm({
186
308
  this.embed = embed;
187
309
  this.db = options.databaseClient;
188
310
  this.knowledgeDir = options.knowledgeDir || DEFAULT_KNOWLEDGE_DIR;
311
+ this.chunking = options.chunking ?? "lines";
189
312
  this.logger = options.logger ?? createConsoleLogger();
190
313
  this.buildLock = options.buildLock ?? createTursoBuildLock(this.db);
191
314
  this.instanceId = options.instanceId ?? crypto.randomUUID();
@@ -199,6 +322,7 @@ var init_retrieval = __esm({
199
322
  */
200
323
  db;
201
324
  knowledgeDir;
325
+ chunking;
202
326
  logger;
203
327
  buildLock;
204
328
  instanceId;
@@ -259,11 +383,41 @@ var init_retrieval = __esm({
259
383
  this.logger.info("No knowledge documents and no existing chunks - nothing to build");
260
384
  return;
261
385
  }
386
+ this.logger.info(`Chunking mode: ${this.chunking}`);
387
+ if (this.chunking === "sections") await this.ensureSectionColumns();
262
388
  const rows = [];
263
389
  for (const d of docs) {
264
- for (const part of chunk(d.text)) {
265
- const id = await sha256(`${d.bucket}|${d.source}|${part}`);
266
- rows.push({ id, bucket: d.bucket, source: d.source, text: part });
390
+ if (this.chunking === "sections") {
391
+ const source = relative(this.knowledgeDir, d.source).split(sep).join("/");
392
+ for (const c of chunkSections(d.text)) {
393
+ const section = c.section.join(" > ");
394
+ const id = await sha256(`${d.bucket}|${source}|${section}|${c.text}`);
395
+ const context = section ? `${source} > ${section}` : source;
396
+ rows.push({
397
+ id,
398
+ bucket: d.bucket,
399
+ source,
400
+ text: c.text,
401
+ embedText: `${context}
402
+
403
+ ${c.text}`,
404
+ section,
405
+ position: c.position
406
+ });
407
+ }
408
+ } else {
409
+ for (const part of chunk(d.text)) {
410
+ const id = await sha256(`${d.bucket}|${d.source}|${part}`);
411
+ rows.push({
412
+ id,
413
+ bucket: d.bucket,
414
+ source: d.source,
415
+ text: part,
416
+ embedText: part,
417
+ section: null,
418
+ position: null
419
+ });
420
+ }
267
421
  }
268
422
  }
269
423
  this.logger.info(`Created ${rows.length} chunks from knowledge documents`);
@@ -272,11 +426,17 @@ var init_retrieval = __esm({
272
426
  for (let i = 0; i < rows.length; i += UPSERT_BATCH) {
273
427
  const batch = rows.slice(i, i + UPSERT_BATCH);
274
428
  await this.db.batch(
275
- batch.map((r) => ({
276
- sql: `INSERT INTO chunks(id,bucket,source,text) VALUES(?,?,?,?)
277
- ON CONFLICT(id) DO NOTHING`,
278
- args: [r.id, r.bucket, r.source, r.text]
279
- })),
429
+ batch.map(
430
+ (r) => this.chunking === "sections" ? {
431
+ sql: `INSERT INTO chunks(id,bucket,source,text,section,position) VALUES(?,?,?,?,?,?)
432
+ ON CONFLICT(id) DO NOTHING`,
433
+ args: [r.id, r.bucket, r.source, r.text, r.section, r.position]
434
+ } : {
435
+ sql: `INSERT INTO chunks(id,bucket,source,text) VALUES(?,?,?,?)
436
+ ON CONFLICT(id) DO NOTHING`,
437
+ args: [r.id, r.bucket, r.source, r.text]
438
+ }
439
+ ),
280
440
  "write"
281
441
  );
282
442
  }
@@ -297,7 +457,7 @@ var init_retrieval = __esm({
297
457
  return;
298
458
  }
299
459
  this.logger.info(`Embedding ${missing.length} new/updated chunks with ${EMB_MODEL}...`);
300
- const textById = new Map(rows.map((r) => [r.id, r.text]));
460
+ const textById = new Map(rows.map((r) => [r.id, r.embedText]));
301
461
  const BATCH = 96;
302
462
  for (let i = 0; i < missing.length; i += BATCH) {
303
463
  const batchIds = missing.slice(i, i + BATCH);
@@ -318,6 +478,18 @@ var init_retrieval = __esm({
318
478
  }
319
479
  this.logger.info(`Successfully embedded ${missing.length} new chunks`);
320
480
  }
481
+ /**
482
+ * Adds the nullable `section`/`position` columns to a `chunks` table that
483
+ * every released version created without them. Checked via table_info so it
484
+ * is a no-op once present; only `'sections'` mode ever calls it.
485
+ */
486
+ async ensureSectionColumns() {
487
+ const info = await this.db.execute("PRAGMA table_info(chunks)");
488
+ const have = new Set(info.rows.map((r) => String(r.name)));
489
+ if (!have.has("section")) await this.db.execute("ALTER TABLE chunks ADD COLUMN section TEXT");
490
+ if (!have.has("position"))
491
+ await this.db.execute("ALTER TABLE chunks ADD COLUMN position INTEGER");
492
+ }
321
493
  // Remove chunks from database that no longer exist in markdown files
322
494
  async cleanupStaleChunks(currentIds) {
323
495
  const result = await this.db.execute("SELECT id FROM chunks");
@@ -364,25 +536,34 @@ var init_retrieval = __esm({
364
536
  * interpolated. An empty list retrieves nothing, and short-circuits before
365
537
  * the embedding call.
366
538
  */
367
- async query(q, k = 6, allowed = ["base"]) {
539
+ async queryChunks(q, k = 6, allowed = ["base"]) {
368
540
  if (allowed.length === 0) return [];
369
541
  const [qv] = await this.embed([q]);
370
542
  const placeholders = allowed.map(() => "?").join(",");
371
543
  const res = await this.db.execute({
372
- sql: `SELECT c.id, c.text, e.embedding
544
+ sql: `SELECT c.*, e.embedding
373
545
  FROM chunks c
374
546
  JOIN embeddings e ON e.id = c.id
375
547
  WHERE c.bucket IN (${placeholders})`,
376
548
  args: allowed
377
549
  });
378
- const scored = [];
379
- for (const row of res.rows) {
550
+ const scored = res.rows.map((row) => {
380
551
  const emb = JSON.parse(String(row.embedding));
381
- const s = _VectorStore.cosine(qv, emb);
382
- scored.push({ s, text: String(row.text) });
383
- }
384
- scored.sort((a, b) => b.s - a.s);
385
- return scored.slice(0, k).map((r) => r.text);
552
+ const trail = row.section == null ? "" : String(row.section);
553
+ return {
554
+ text: String(row.text),
555
+ bucket: String(row.bucket ?? ""),
556
+ source: String(row.source ?? ""),
557
+ section: trail ? trail.split(" > ") : [],
558
+ score: _VectorStore.cosine(qv, emb)
559
+ };
560
+ });
561
+ scored.sort((a, b) => b.score - a.score);
562
+ return scored.slice(0, k);
563
+ }
564
+ /** Same results as {@link queryChunks}, as bare text. */
565
+ async query(q, k = 6, allowed = ["base"]) {
566
+ return (await this.queryChunks(q, k, allowed)).map((c) => c.text);
386
567
  }
387
568
  };
388
569
  }
@@ -502,7 +683,7 @@ var init_knowledgeHealth = __esm({
502
683
  });
503
684
 
504
685
  // src/server.ts
505
- import { relative } from "node:path";
686
+ import { relative as relative2 } from "node:path";
506
687
  import { Hono as Hono5 } from "hono";
507
688
  import OpenAI from "openai";
508
689
 
@@ -1992,6 +2173,7 @@ async function createServer(config) {
1992
2173
  store = new VectorStore2(createOpenAIEmbedder2(client), {
1993
2174
  databaseClient: db,
1994
2175
  knowledgeDir: config.knowledgeDir || DEFAULT_KNOWLEDGE_DIR2,
2176
+ chunking: config.chunking,
1995
2177
  logger
1996
2178
  });
1997
2179
  }
@@ -2042,7 +2224,7 @@ async function createServer(config) {
2042
2224
  }
2043
2225
  }
2044
2226
  if (serveStatic && staticDir) {
2045
- const relativePath = relative(process.cwd(), staticDir);
2227
+ const relativePath = relative2(process.cwd(), staticDir);
2046
2228
  app.get("/chatter.js", serveStatic({ path: `${relativePath}/chatter.js` }));
2047
2229
  app.get("/chatter.css", serveStatic({ path: `${relativePath}/chatter.css` }));
2048
2230
  }
package/dist/types.d.ts CHANGED
@@ -187,6 +187,13 @@ export interface ChatterConfig extends BrainHooks {
187
187
  configDir?: string;
188
188
  /** Knowledge base directory. Default: ./config/knowledge */
189
189
  knowledgeDir?: string;
190
+ /**
191
+ * How knowledge files are chunked by the default `VectorStore`: `'lines'`
192
+ * (default, unchanged) or `'sections'` (heading-aware, with section context
193
+ * in the embedding). Switching re-embeds the knowledge base once. Ignored
194
+ * when `retriever` is set.
195
+ */
196
+ chunking?: "lines" | "sections";
190
197
  /** Prompts directory. Default: ./config/prompts */
191
198
  promptsDir?: string;
192
199
  /** Public static files directory. Default: ./public */