@puwenhui/dsh-rag-kb 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +15 -1
  2. package/lib/plugin.mjs +35 -11
  3. package/package.json +1 -1
package/README.md CHANGED
@@ -50,7 +50,7 @@ dsh plugin --profile web add @puwenhui/dsh-rag-kb
50
50
  ## 检索架构
51
51
 
52
52
  ```
53
- 文档 → 解析(pdf/docx/md) → 标题感知切块(400 token + 10% 重叠)
53
+ 文档 → 解析(pdf/docx/md) → 标题感知切块(节标题强制断块 + 400 token + 10% 重叠)
54
54
  → 向量化(transformers.js bge-small-zh-v1.5, 512维, GPU/CPU)
55
55
  → FTS5 trigram 关键词索引(与向量索引同步写入)
56
56
 
@@ -59,6 +59,20 @@ dsh plugin --profile web add @puwenhui/dsh-rag-kb
59
59
  → 引用溯源(citation + position 字段)
60
60
  ```
61
61
 
62
+ ### 分块策略(v0.3.0 起)
63
+
64
+ 面向 Markdown 手册/知识库文档优化,保证**一节 = 一块**,引用溯源可精确到节:
65
+
66
+ | 规则 | 说明 |
67
+ |---|---|
68
+ | 节标题强制断块 | 遇 `##` ~ `######` 立即断块,短节不再被合并;刻意不含单级 `#`——配置/代码注释普遍用它开头,误判会把命令序列切碎 |
69
+ | 续块补回节标题 | 长节切成多块时,每块都带节标题,检索命中时能看清出处 |
70
+ | 软下限 + 硬上限 | 块过短(如仅「节标题 + 来源行」)宁可略微超长也不单独成块;合并超过 1.2 倍上限则强制断开 |
71
+ | 滑窗按换行对齐 | 超长无结构段落按换行位置切分,不把命令行从中间截断 |
72
+ | 兜底宽度 | 块长控制在约 600 字符(≈400 token),实测上限 ~920 字符,避免超出 embedding 序列长度被截断 |
73
+
74
+ 实测效果(13 篇技术/管理文档,114 节):切成 605 块,**96% 的块带节标题**。
75
+
62
76
  ## 技术栈
63
77
 
64
78
  | 层 | 选型 | 说明 |
package/lib/plugin.mjs CHANGED
@@ -266,7 +266,12 @@ var indexer_exports = /* @__PURE__ */ __exportAll({
266
266
  });
267
267
  /** 中文友好的切块参数 */
268
268
  const CHUNK_TOKENS = 400;
269
- const OVERLAP_RATIO = .1;
269
+ /**
270
+ * 节标题行:`##` ~ `######` 视为节标题,遇之强制断块(保证「一节 = 一块」)。
271
+ * 刻意不含单级 `#`——nginx/shell/mysql 等配置与代码的注释普遍用单 `#` 开头,
272
+ * 误判会把命令序列切碎(实测语音平台文档 27 节被切成 231 块)。
273
+ */
274
+ const HEADING_RE = /^#{2,6} /;
270
275
  async function extractText(filePath) {
271
276
  const ext = extname(filePath).toLowerCase();
272
277
  const buf = await readFile(filePath);
@@ -285,28 +290,47 @@ async function extractText(filePath) {
285
290
  if (ext === ".md" || ext === ".txt" || ext === ".log" || ext === ".csv" || ext === ".json") return { text: buf.toString("utf8") };
286
291
  throw new Error(`不支持的文件类型 ${ext}(支持 pdf/docx/md/txt/csv/json)`);
287
292
  }
288
- /** 标题感知切块:按标题/空行分段,过长段再固定长度切 */
293
+ /** 标题感知切块:节标题强制断块(一节 = 一块)+ 空行分段 + 过长滑窗兜底 */
289
294
  function chunkText(text) {
290
295
  const normalized = text.replace(/\r\n/g, "\n").replace(/\n{3,}/g, "\n\n").trim();
291
296
  if (normalized === "") return [];
292
- const paragraphs = normalized.split(/\n(?=#{1,4} )|\n\n+/).map((p) => p.trim()).filter((p) => p !== "");
297
+ const paragraphs = normalized.split(/\n(?=#{2,6} )|\n\n+/).map((p) => p.trim()).filter((p) => p !== "");
293
298
  const chunks = [];
294
299
  let current = "";
300
+ let sectionHeading = "";
295
301
  const maxChars = CHUNK_TOKENS * 1.5;
296
- const overlap = Math.floor(maxChars * OVERLAP_RATIO);
297
302
  for (const para of paragraphs) {
298
- if (current !== "" && (current + "\n" + para).length > maxChars) {
303
+ const isHeading = HEADING_RE.test(para);
304
+ if (isHeading) sectionHeading = para.split("\n")[0].trim();
305
+ if (isHeading && current !== "") {
299
306
  chunks.push(current);
300
- current = current.length > overlap ? current.slice(-60) : "";
307
+ current = "";
301
308
  }
302
- if (para.length > maxChars * 2) {
303
- if (current !== "") {
304
- chunks.push(current);
305
- current = "";
309
+ if (para.length > maxChars * 1.2) {
310
+ const prefix = current;
311
+ current = "";
312
+ const mergePrefix = prefix !== "" && prefix.length <= maxChars * .6;
313
+ if (prefix !== "" && !mergePrefix) chunks.push(prefix);
314
+ for (let i = 0; i < para.length;) {
315
+ let end = Math.min(i + maxChars, para.length);
316
+ if (end < para.length) {
317
+ const nl = para.lastIndexOf("\n", end);
318
+ if (nl > i + maxChars * .5) end = nl + 1;
319
+ }
320
+ const piece = para.slice(i, end);
321
+ const isFirst = i === 0;
322
+ i = end;
323
+ chunks.push(isFirst ? mergePrefix ? `${prefix}\n\n${piece}` : piece : sectionHeading !== "" ? `${sectionHeading}\n\n${piece}` : piece);
306
324
  }
307
- for (let i = 0; i < para.length; i += 540) chunks.push(para.slice(i, i + maxChars));
325
+ current = sectionHeading;
308
326
  continue;
309
327
  }
328
+ const merged = current + "\n" + para;
329
+ if (current !== "" && merged.length > maxChars && (current.length >= maxChars * .5 || merged.length > maxChars * 1.2)) {
330
+ chunks.push(current);
331
+ const tail = current.slice(-60);
332
+ current = sectionHeading !== "" ? `${sectionHeading}\n\n${tail}` : tail;
333
+ }
310
334
  current = current === "" ? para : current + "\n" + para;
311
335
  }
312
336
  if (current !== "") chunks.push(current);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@puwenhui/dsh-rag-kb",
3
- "version": "0.2.0",
3
+ "version": "0.3.0",
4
4
  "description": "DSH RAG 知识库插件:混合检索(向量+FTS5)+Reranker 重排+GPU 加速+引用溯源+会话语义管理",
5
5
  "publishConfig": {
6
6
  "registry": "https://registry.npmjs.org/",