@puwenhui/dsh-rag-kb 0.2.0 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +30 -2
  2. package/lib/plugin.mjs +90 -25
  3. package/package.json +1 -1
package/README.md CHANGED
@@ -13,7 +13,8 @@ DSH RAG 知识库插件:**文档导入 → 智能切块 → 本地向量化
13
13
  - **会话语义管理**:所有操作(添加/删除/保存/检索)直接在对话里用大白话完成
14
14
  - **多格式支持**:pdf / docx / md / txt / csv / json
15
15
  - **完全本地**:模型本地运行,数据不出机器
16
- - **目录自动索引**:配置目录后文件放入/变更自动入库
16
+ - **目录自动索引**:配置目录后,放入/修改的文件自动入库;**DSH 未运行期间放入的也会在下次启动时补扫**,不用管当时开没开
17
+ - **文档原地更新**:同一文件改了内容会**覆盖更新**旧索引(以文件路径为身份),不会在库里留下旧版本
17
18
  - **对话内容保存**:把对话中的结论/总结一键存入知识库
18
19
 
19
20
  ## 版本要求
@@ -50,7 +51,7 @@ dsh plugin --profile web add @puwenhui/dsh-rag-kb
50
51
  ## 检索架构
51
52
 
52
53
  ```
53
- 文档 → 解析(pdf/docx/md) → 标题感知切块(400 token + 10% 重叠)
54
+ 文档 → 解析(pdf/docx/md) → 标题感知切块(节标题强制断块 + 400 token + 10% 重叠)
54
55
  → 向量化(transformers.js bge-small-zh-v1.5, 512维, GPU/CPU)
55
56
  → FTS5 trigram 关键词索引(与向量索引同步写入)
56
57
 
@@ -59,6 +60,33 @@ dsh plugin --profile web add @puwenhui/dsh-rag-kb
59
60
  → 引用溯源(citation + position 字段)
60
61
  ```
61
62
 
63
+ ### 分块策略(v0.3.0 起)
64
+
65
+ 面向 Markdown 手册/知识库文档优化,保证**一节 = 一块**,引用溯源可精确到节:
66
+
67
+ | 规则 | 说明 |
68
+ |---|---|
69
+ | 节标题强制断块 | 遇 `##` ~ `######` 立即断块,短节不再被合并;刻意不含单级 `#`——配置/代码注释普遍用它开头,误判会把命令序列切碎 |
70
+ | 续块补回节标题 | 长节切成多块时,每块都带节标题,检索命中时能看清出处 |
71
+ | 软下限 + 硬上限 | 块过短(如仅「节标题 + 来源行」)宁可略微超长也不单独成块;合并超过 1.2 倍上限则强制断开 |
72
+ | 滑窗按换行对齐 | 超长无结构段落按换行位置切分,不把命令行从中间截断 |
73
+ | 兜底宽度 | 块长控制在约 600 字符(≈400 token),实测上限 ~920 字符,避免超出 embedding 序列长度被截断 |
74
+
75
+ 实测效果(13 篇技术/管理文档,114 节):切成 605 块,**96% 的块带节标题**。
76
+
77
+ ### 索引行为(v0.3.1 起)
78
+
79
+ | 行为 | 说明 |
80
+ |---|---|
81
+ | 身份标识 | 文档以**文件绝对路径**为 doc_id(非内容),同一文件改了内容走覆盖更新,库里不留旧版本 |
82
+ | 变更检测 | 单独存 `content_hash` 判断内容是否真的变了——未变则跳过,连向量化都不跑 |
83
+ | 启动补扫 | 插件加载时扫描 `watchDir`,把已存在但未入库的文件补索引(`watch` 只捕获建立之后的变动) |
84
+ | 实时监视 | 文件放入/修改后等 2 秒再索引,避免读到半写文件 |
85
+ | 范围限制 | 监视与扫描均为**单层**,不递归子目录;扩展名限 pdf/docx/md/txt/csv/json/log |
86
+ | 旧库兼容 | 自动补 `content_hash` 列;索引时清理同名旧格式记录(`content_hash` 为空的条目) |
87
+
88
+ > 一次索引 = 一次 `documents` 行 + N 条 `chunks` 行;重复触发不会重复插入,由 doc_id 主键与「先删后插」双重保证。
89
+
62
90
  ## 技术栈
63
91
 
64
92
  | 层 | 选型 | 说明 |
package/lib/plugin.mjs CHANGED
@@ -66,7 +66,7 @@ function resolveKbConfig(c = {}) {
66
66
  */
67
67
  /** 库归属标识(PRAGMA application_id,抄宿主 session-query-sqlite 惯例) */
68
68
  const APP_ID = 1380009794;
69
- const SCHEMA_VERSION = 1;
69
+ const SCHEMA_VERSION = 2;
70
70
  var KbStore = class {
71
71
  db;
72
72
  vectors = [];
@@ -83,6 +83,7 @@ var KbStore = class {
83
83
  this.db.exec(`
84
84
  CREATE TABLE IF NOT EXISTS documents(
85
85
  doc_id TEXT PRIMARY KEY,
86
+ content_hash TEXT NOT NULL DEFAULT '',
86
87
  name TEXT NOT NULL,
87
88
  source TEXT NOT NULL DEFAULT 'upload',
88
89
  bytes INTEGER NOT NULL DEFAULT 0,
@@ -106,21 +107,32 @@ var KbStore = class {
106
107
  tokenize='trigram'
107
108
  );
108
109
  `);
110
+ try {
111
+ this.db.exec("ALTER TABLE documents ADD COLUMN content_hash TEXT NOT NULL DEFAULT ''");
112
+ } catch {}
109
113
  }
110
- /** 文档内容 hash 作 doc_id(同名不同内容=不同文档) */
111
- static docId(content) {
114
+ /**
115
+ * 文件绝对路径 → doc_id:同一路径的文件内容变化时 doc_id 保持不变,
116
+ * 由 upsertDoc + insertChunks 覆盖更新,不会在库里留下旧版本。
117
+ * 路径转小写以适配 Windows 大小写不敏感。
118
+ */
119
+ static docIdFromPath(filePath) {
120
+ return createHash("sha256").update(resolve(filePath).toLowerCase()).digest("hex").slice(0, 16);
121
+ }
122
+ /** 内容 hash:save 文档以它为 doc_id;文件文档用它判断内容是否变化 */
123
+ static hashContent(content) {
112
124
  return createHash("sha256").update(content).digest("hex").slice(0, 16);
113
125
  }
114
126
  hasDoc(docId) {
115
127
  return this.db.prepare("SELECT 1 FROM documents WHERE doc_id=?").get(docId) !== void 0;
116
128
  }
117
- upsertDoc(docId, name, source, bytes, status, error = "") {
129
+ upsertDoc(docId, name, source, bytes, contentHash, status, error = "") {
118
130
  const now = (/* @__PURE__ */ new Date()).toISOString();
119
131
  this.db.prepare(`
120
- INSERT INTO documents(doc_id, name, source, bytes, chunk_count, status, error, created_at, updated_at)
121
- VALUES(?,?,?,?,0,?,?,?,?)
122
- ON CONFLICT(doc_id) DO UPDATE SET name=excluded.name, bytes=excluded.bytes, status=excluded.status, error=excluded.error, updated_at=excluded.updated_at
123
- `).run(docId, name, source, bytes, status, error, now, now);
132
+ INSERT INTO documents(doc_id, content_hash, name, source, bytes, chunk_count, status, error, created_at, updated_at)
133
+ VALUES(?,?,?,?,?,0,?,?,?,?)
134
+ ON CONFLICT(doc_id) DO UPDATE SET content_hash=excluded.content_hash, name=excluded.name, bytes=excluded.bytes, status=excluded.status, error=excluded.error, updated_at=excluded.updated_at
135
+ `).run(docId, contentHash, name, source, bytes, status, error, now, now);
124
136
  this.loaded = false;
125
137
  }
126
138
  setDocStatus(docId, status, chunkCount, error = "") {
@@ -266,7 +278,12 @@ var indexer_exports = /* @__PURE__ */ __exportAll({
266
278
  });
267
279
  /** 中文友好的切块参数 */
268
280
  const CHUNK_TOKENS = 400;
269
- const OVERLAP_RATIO = .1;
281
+ /**
282
+ * 节标题行:`##` ~ `######` 视为节标题,遇之强制断块(保证「一节 = 一块」)。
283
+ * 刻意不含单级 `#`——nginx/shell/mysql 等配置与代码的注释普遍用单 `#` 开头,
284
+ * 误判会把命令序列切碎(实测语音平台文档 27 节被切成 231 块)。
285
+ */
286
+ const HEADING_RE = /^#{2,6} /;
270
287
  async function extractText(filePath) {
271
288
  const ext = extname(filePath).toLowerCase();
272
289
  const buf = await readFile(filePath);
@@ -285,28 +302,47 @@ async function extractText(filePath) {
285
302
  if (ext === ".md" || ext === ".txt" || ext === ".log" || ext === ".csv" || ext === ".json") return { text: buf.toString("utf8") };
286
303
  throw new Error(`不支持的文件类型 ${ext}(支持 pdf/docx/md/txt/csv/json)`);
287
304
  }
288
- /** 标题感知切块:按标题/空行分段,过长段再固定长度切 */
305
+ /** 标题感知切块:节标题强制断块(一节 = 一块)+ 空行分段 + 过长滑窗兜底 */
289
306
  function chunkText(text) {
290
307
  const normalized = text.replace(/\r\n/g, "\n").replace(/\n{3,}/g, "\n\n").trim();
291
308
  if (normalized === "") return [];
292
- const paragraphs = normalized.split(/\n(?=#{1,4} )|\n\n+/).map((p) => p.trim()).filter((p) => p !== "");
309
+ const paragraphs = normalized.split(/\n(?=#{2,6} )|\n\n+/).map((p) => p.trim()).filter((p) => p !== "");
293
310
  const chunks = [];
294
311
  let current = "";
312
+ let sectionHeading = "";
295
313
  const maxChars = CHUNK_TOKENS * 1.5;
296
- const overlap = Math.floor(maxChars * OVERLAP_RATIO);
297
314
  for (const para of paragraphs) {
298
- if (current !== "" && (current + "\n" + para).length > maxChars) {
315
+ const isHeading = HEADING_RE.test(para);
316
+ if (isHeading) sectionHeading = para.split("\n")[0].trim();
317
+ if (isHeading && current !== "") {
299
318
  chunks.push(current);
300
- current = current.length > overlap ? current.slice(-60) : "";
319
+ current = "";
301
320
  }
302
- if (para.length > maxChars * 2) {
303
- if (current !== "") {
304
- chunks.push(current);
305
- current = "";
321
+ if (para.length > maxChars * 1.2) {
322
+ const prefix = current;
323
+ current = "";
324
+ const mergePrefix = prefix !== "" && prefix.length <= maxChars * .6;
325
+ if (prefix !== "" && !mergePrefix) chunks.push(prefix);
326
+ for (let i = 0; i < para.length;) {
327
+ let end = Math.min(i + maxChars, para.length);
328
+ if (end < para.length) {
329
+ const nl = para.lastIndexOf("\n", end);
330
+ if (nl > i + maxChars * .5) end = nl + 1;
331
+ }
332
+ const piece = para.slice(i, end);
333
+ const isFirst = i === 0;
334
+ i = end;
335
+ chunks.push(isFirst ? mergePrefix ? `${prefix}\n\n${piece}` : piece : sectionHeading !== "" ? `${sectionHeading}\n\n${piece}` : piece);
306
336
  }
307
- for (let i = 0; i < para.length; i += 540) chunks.push(para.slice(i, i + maxChars));
337
+ current = sectionHeading;
308
338
  continue;
309
339
  }
340
+ const merged = current + "\n" + para;
341
+ if (current !== "" && merged.length > maxChars && (current.length >= maxChars * .5 || merged.length > maxChars * 1.2)) {
342
+ chunks.push(current);
343
+ const tail = current.slice(-60);
344
+ current = sectionHeading !== "" ? `${sectionHeading}\n\n${tail}` : tail;
345
+ }
310
346
  current = current === "" ? para : current + "\n" + para;
311
347
  }
312
348
  if (current !== "") chunks.push(current);
@@ -412,6 +448,10 @@ const inject = [
412
448
  let store;
413
449
  let cfg;
414
450
  const progressListeners = /* @__PURE__ */ new Set();
451
+ /** 目录监视与启动扫描支持的文件类型(两处必须用同一份,否则会出现「监视到了却扫不到」的错位) */
452
+ const WATCH_EXT = /\.(pdf|docx|md|txt|csv|json|log)$/i;
453
+ /** 去掉扩展名:save 文档的 name 是标题(无后缀),文件文档的 name 带后缀,比对前需归一 */
454
+ const stripExt = (n) => n.replace(WATCH_EXT, "");
415
455
  function broadcastProgress(msg) {
416
456
  for (const write of progressListeners) try {
417
457
  write(`event: progress\ndata: ${JSON.stringify(msg)}\n\n`);
@@ -429,12 +469,15 @@ async function indexOne(ctx, filePath, source) {
429
469
  const name = basename(filePath);
430
470
  const bytes = statSync(filePath).size;
431
471
  const content = await (await import("node:fs/promises")).readFile(filePath);
432
- const docId = KbStore.docId(content);
433
- if (s.hasDoc(docId) && s.getDoc(docId)?.status === "ready") return {
472
+ const docId = KbStore.docIdFromPath(filePath);
473
+ const hash = KbStore.hashContent(content);
474
+ for (const d of s.listDocs()) if (d.content_hash === "" && stripExt(d.name) === stripExt(name)) s.deleteDoc(d.doc_id);
475
+ const prev = s.getDoc(docId);
476
+ if (prev?.status === "ready" && prev.content_hash === hash) return {
434
477
  ok: true,
435
478
  message: `${name} 已索引(内容未变化,跳过)`
436
479
  };
437
- s.upsertDoc(docId, name, source, bytes, "indexing");
480
+ s.upsertDoc(docId, name, source, bytes, hash, "indexing");
438
481
  broadcastProgress(`正在索引 ${name}`);
439
482
  try {
440
483
  const chunks = await indexFile(filePath, await getEmbedder(cfg?.embeddingModel ?? "Xenova/bge-small-zh-v1.5"), cfg?.batchSize ?? 32);
@@ -459,6 +502,27 @@ async function indexOne(ctx, filePath, source) {
459
502
  };
460
503
  }
461
504
  }
505
+ /**
506
+ * 启动时补索引:扫描 watchDir 里已存在但尚未入库的文件。
507
+ * watch() 只捕获监视建立之后的变动,DSH 未运行期间放进目录的文件它看不到,故需补扫。
508
+ * 内容未变化的已入库文件会被 indexOne 直接跳过,不会重复向量化。
509
+ */
510
+ async function scanWatchDir(ctx, dir) {
511
+ let names;
512
+ try {
513
+ const { readdir } = await import("node:fs/promises");
514
+ names = (await readdir(dir)).filter((f) => WATCH_EXT.test(f));
515
+ } catch (e) {
516
+ console.warn(`[rag-kb] 扫描目录失败 ${dir}: ${String(e.message)}`);
517
+ return;
518
+ }
519
+ let indexed = 0;
520
+ for (const f of names) {
521
+ const r = await indexOne(ctx, join(dir, f), "watch");
522
+ if (r.ok && !r.message.includes("跳过")) indexed++;
523
+ }
524
+ if (indexed > 0) console.log(`[rag-kb] 启动扫描补索引 ${indexed} 个文件(${dir})`);
525
+ }
462
526
  /** 混合语义检索(向量+关键词)+ 可选 reranker 精排 + 引用元数据 */
463
527
  async function search(query, topK) {
464
528
  const s = await ensureStore();
@@ -536,7 +600,7 @@ function apply(ctx, config = {}) {
536
600
  const watcher = watch(dir, { persistent: false }, (_ev, filename) => {
537
601
  if (filename === null || filename === void 0) return;
538
602
  const fname = String(filename);
539
- if (!/\.(pdf|docx|md|txt|csv|json|log)$/i.test(fname)) return;
603
+ if (!WATCH_EXT.test(fname)) return;
540
604
  const f = join(dir, fname);
541
605
  setTimeout(() => {
542
606
  if (existsSync(f)) indexOne(ctx, f, "watch");
@@ -545,6 +609,7 @@ function apply(ctx, config = {}) {
545
609
  ctx.effect(() => {
546
610
  watcher.close();
547
611
  }, "rag-kb: dir watcher");
612
+ scanWatchDir(ctx, dir);
548
613
  } catch (e) {
549
614
  console.warn(`[rag-kb] 目录监视失败 ${dir}: ${String(e.message)}`);
550
615
  }
@@ -667,12 +732,12 @@ function apply(ctx, config = {}) {
667
732
  if (a.action === "save" && a.text !== void 0 && a.text !== "") {
668
733
  const title = a.title !== void 0 && a.title !== "" ? a.title : `对话保存 ${(/* @__PURE__ */ new Date()).toISOString().slice(0, 16)}`;
669
734
  const content = `# ${title}\n\n${a.text}`;
670
- const docId = KbStore.docId(content);
735
+ const docId = KbStore.hashContent(content);
671
736
  if (s.hasDoc(docId) && s.getDoc(docId)?.status === "ready") return {
672
737
  ok: true,
673
738
  message: `${title} 已在知识库中(内容相同,跳过)`
674
739
  };
675
- s.upsertDoc(docId, title, "save", content.length, "indexing");
740
+ s.upsertDoc(docId, title, "save", content.length, docId, "indexing");
676
741
  broadcastProgress(`正在索引对话内容「${title}」`);
677
742
  try {
678
743
  const embedder = await getEmbedder(cfg?.embeddingModel ?? "Xenova/bge-small-zh-v1.5");
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@puwenhui/dsh-rag-kb",
3
- "version": "0.2.0",
3
+ "version": "0.3.1",
4
4
  "description": "DSH RAG 知识库插件:混合检索(向量+FTS5)+Reranker 重排+GPU 加速+引用溯源+会话语义管理",
5
5
  "publishConfig": {
6
6
  "registry": "https://registry.npmjs.org/",