@puwenhui/dsh-rag-kb 0.3.0 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -1
- package/lib/plugin.mjs +55 -14
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -13,7 +13,8 @@ DSH RAG 知识库插件:**文档导入 → 智能切块 → 本地向量化
|
|
|
13
13
|
- **会话语义管理**:所有操作(添加/删除/保存/检索)直接在对话里用大白话完成
|
|
14
14
|
- **多格式支持**:pdf / docx / md / txt / csv / json
|
|
15
15
|
- **完全本地**:模型本地运行,数据不出机器
|
|
16
|
-
-
|
|
16
|
+
- **目录自动索引**:配置目录后,放入/修改的文件自动入库;**DSH 未运行期间放入的也会在下次启动时补扫**,不用管当时开没开
|
|
17
|
+
- **文档原地更新**:同一文件改了内容会**覆盖更新**旧索引(以文件路径为身份),不会在库里留下旧版本
|
|
17
18
|
- **对话内容保存**:把对话中的结论/总结一键存入知识库
|
|
18
19
|
|
|
19
20
|
## 版本要求
|
|
@@ -73,6 +74,19 @@ dsh plugin --profile web add @puwenhui/dsh-rag-kb
|
|
|
73
74
|
|
|
74
75
|
实测效果(13 篇技术/管理文档,114 节):切成 605 块,**96% 的块带节标题**。
|
|
75
76
|
|
|
77
|
+
### 索引行为(v0.3.1 起)
|
|
78
|
+
|
|
79
|
+
| 行为 | 说明 |
|
|
80
|
+
|---|---|
|
|
81
|
+
| 身份标识 | 文档以**文件绝对路径**为 doc_id(非内容),同一文件改了内容走覆盖更新,库里不留旧版本 |
|
|
82
|
+
| 变更检测 | 单独存 `content_hash` 判断内容是否真的变了——未变则跳过,连向量化都不跑 |
|
|
83
|
+
| 启动补扫 | 插件加载时扫描 `watchDir`,把已存在但未入库的文件补索引(`watch` 只捕获建立之后的变动) |
|
|
84
|
+
| 实时监视 | 文件放入/修改后等 2 秒再索引,避免读到半写文件 |
|
|
85
|
+
| 范围限制 | 监视与扫描均为**单层**,不递归子目录;扩展名限 pdf/docx/md/txt/csv/json/log |
|
|
86
|
+
| 旧库兼容 | 自动补 `content_hash` 列;索引时清理同名旧格式记录(`content_hash` 为空的条目) |
|
|
87
|
+
|
|
88
|
+
> 一次索引 = 一次 `documents` 行 + N 条 `chunks` 行;重复触发不会重复插入,由 doc_id 主键与「先删后插」双重保证。
|
|
89
|
+
|
|
76
90
|
## 技术栈
|
|
77
91
|
|
|
78
92
|
| 层 | 选型 | 说明 |
|
package/lib/plugin.mjs
CHANGED
|
@@ -66,7 +66,7 @@ function resolveKbConfig(c = {}) {
|
|
|
66
66
|
*/
|
|
67
67
|
/** 库归属标识(PRAGMA application_id,抄宿主 session-query-sqlite 惯例) */
|
|
68
68
|
const APP_ID = 1380009794;
|
|
69
|
-
const SCHEMA_VERSION =
|
|
69
|
+
const SCHEMA_VERSION = 2;
|
|
70
70
|
var KbStore = class {
|
|
71
71
|
db;
|
|
72
72
|
vectors = [];
|
|
@@ -83,6 +83,7 @@ var KbStore = class {
|
|
|
83
83
|
this.db.exec(`
|
|
84
84
|
CREATE TABLE IF NOT EXISTS documents(
|
|
85
85
|
doc_id TEXT PRIMARY KEY,
|
|
86
|
+
content_hash TEXT NOT NULL DEFAULT '',
|
|
86
87
|
name TEXT NOT NULL,
|
|
87
88
|
source TEXT NOT NULL DEFAULT 'upload',
|
|
88
89
|
bytes INTEGER NOT NULL DEFAULT 0,
|
|
@@ -106,21 +107,32 @@ var KbStore = class {
|
|
|
106
107
|
tokenize='trigram'
|
|
107
108
|
);
|
|
108
109
|
`);
|
|
110
|
+
try {
|
|
111
|
+
this.db.exec("ALTER TABLE documents ADD COLUMN content_hash TEXT NOT NULL DEFAULT ''");
|
|
112
|
+
} catch {}
|
|
113
|
+
}
|
|
114
|
+
/**
|
|
115
|
+
* 文件绝对路径 → doc_id:同一路径的文件内容变化时 doc_id 保持不变,
|
|
116
|
+
* 由 upsertDoc + insertChunks 覆盖更新,不会在库里留下旧版本。
|
|
117
|
+
* 路径转小写以适配 Windows 大小写不敏感。
|
|
118
|
+
*/
|
|
119
|
+
static docIdFromPath(filePath) {
|
|
120
|
+
return createHash("sha256").update(resolve(filePath).toLowerCase()).digest("hex").slice(0, 16);
|
|
109
121
|
}
|
|
110
|
-
/**
|
|
111
|
-
static
|
|
122
|
+
/** 内容 hash:save 文档以它为 doc_id;文件文档用它判断内容是否变化 */
|
|
123
|
+
static hashContent(content) {
|
|
112
124
|
return createHash("sha256").update(content).digest("hex").slice(0, 16);
|
|
113
125
|
}
|
|
114
126
|
hasDoc(docId) {
|
|
115
127
|
return this.db.prepare("SELECT 1 FROM documents WHERE doc_id=?").get(docId) !== void 0;
|
|
116
128
|
}
|
|
117
|
-
upsertDoc(docId, name, source, bytes, status, error = "") {
|
|
129
|
+
upsertDoc(docId, name, source, bytes, contentHash, status, error = "") {
|
|
118
130
|
const now = (/* @__PURE__ */ new Date()).toISOString();
|
|
119
131
|
this.db.prepare(`
|
|
120
|
-
INSERT INTO documents(doc_id, name, source, bytes, chunk_count, status, error, created_at, updated_at)
|
|
121
|
-
VALUES(
|
|
122
|
-
ON CONFLICT(doc_id) DO UPDATE SET name=excluded.name, bytes=excluded.bytes, status=excluded.status, error=excluded.error, updated_at=excluded.updated_at
|
|
123
|
-
`).run(docId, name, source, bytes, status, error, now, now);
|
|
132
|
+
INSERT INTO documents(doc_id, content_hash, name, source, bytes, chunk_count, status, error, created_at, updated_at)
|
|
133
|
+
VALUES(?,?,?,?,?,0,?,?,?,?)
|
|
134
|
+
ON CONFLICT(doc_id) DO UPDATE SET content_hash=excluded.content_hash, name=excluded.name, bytes=excluded.bytes, status=excluded.status, error=excluded.error, updated_at=excluded.updated_at
|
|
135
|
+
`).run(docId, contentHash, name, source, bytes, status, error, now, now);
|
|
124
136
|
this.loaded = false;
|
|
125
137
|
}
|
|
126
138
|
setDocStatus(docId, status, chunkCount, error = "") {
|
|
@@ -436,6 +448,10 @@ const inject = [
|
|
|
436
448
|
let store;
|
|
437
449
|
let cfg;
|
|
438
450
|
const progressListeners = /* @__PURE__ */ new Set();
|
|
451
|
+
/** 目录监视与启动扫描支持的文件类型(两处必须用同一份,否则会出现「监视到了却扫不到」的错位) */
|
|
452
|
+
const WATCH_EXT = /\.(pdf|docx|md|txt|csv|json|log)$/i;
|
|
453
|
+
/** 去掉扩展名:save 文档的 name 是标题(无后缀),文件文档的 name 带后缀,比对前需归一 */
|
|
454
|
+
const stripExt = (n) => n.replace(WATCH_EXT, "");
|
|
439
455
|
function broadcastProgress(msg) {
|
|
440
456
|
for (const write of progressListeners) try {
|
|
441
457
|
write(`event: progress\ndata: ${JSON.stringify(msg)}\n\n`);
|
|
@@ -453,12 +469,15 @@ async function indexOne(ctx, filePath, source) {
|
|
|
453
469
|
const name = basename(filePath);
|
|
454
470
|
const bytes = statSync(filePath).size;
|
|
455
471
|
const content = await (await import("node:fs/promises")).readFile(filePath);
|
|
456
|
-
const docId = KbStore.
|
|
457
|
-
|
|
472
|
+
const docId = KbStore.docIdFromPath(filePath);
|
|
473
|
+
const hash = KbStore.hashContent(content);
|
|
474
|
+
for (const d of s.listDocs()) if (d.content_hash === "" && stripExt(d.name) === stripExt(name)) s.deleteDoc(d.doc_id);
|
|
475
|
+
const prev = s.getDoc(docId);
|
|
476
|
+
if (prev?.status === "ready" && prev.content_hash === hash) return {
|
|
458
477
|
ok: true,
|
|
459
478
|
message: `${name} 已索引(内容未变化,跳过)`
|
|
460
479
|
};
|
|
461
|
-
s.upsertDoc(docId, name, source, bytes, "indexing");
|
|
480
|
+
s.upsertDoc(docId, name, source, bytes, hash, "indexing");
|
|
462
481
|
broadcastProgress(`正在索引 ${name}`);
|
|
463
482
|
try {
|
|
464
483
|
const chunks = await indexFile(filePath, await getEmbedder(cfg?.embeddingModel ?? "Xenova/bge-small-zh-v1.5"), cfg?.batchSize ?? 32);
|
|
@@ -483,6 +502,27 @@ async function indexOne(ctx, filePath, source) {
|
|
|
483
502
|
};
|
|
484
503
|
}
|
|
485
504
|
}
|
|
505
|
+
/**
|
|
506
|
+
* 启动时补索引:扫描 watchDir 里已存在但尚未入库的文件。
|
|
507
|
+
* watch() 只捕获监视建立之后的变动,DSH 未运行期间放进目录的文件它看不到,故需补扫。
|
|
508
|
+
* 内容未变化的已入库文件会被 indexOne 直接跳过,不会重复向量化。
|
|
509
|
+
*/
|
|
510
|
+
async function scanWatchDir(ctx, dir) {
|
|
511
|
+
let names;
|
|
512
|
+
try {
|
|
513
|
+
const { readdir } = await import("node:fs/promises");
|
|
514
|
+
names = (await readdir(dir)).filter((f) => WATCH_EXT.test(f));
|
|
515
|
+
} catch (e) {
|
|
516
|
+
console.warn(`[rag-kb] 扫描目录失败 ${dir}: ${String(e.message)}`);
|
|
517
|
+
return;
|
|
518
|
+
}
|
|
519
|
+
let indexed = 0;
|
|
520
|
+
for (const f of names) {
|
|
521
|
+
const r = await indexOne(ctx, join(dir, f), "watch");
|
|
522
|
+
if (r.ok && !r.message.includes("跳过")) indexed++;
|
|
523
|
+
}
|
|
524
|
+
if (indexed > 0) console.log(`[rag-kb] 启动扫描补索引 ${indexed} 个文件(${dir})`);
|
|
525
|
+
}
|
|
486
526
|
/** 混合语义检索(向量+关键词)+ 可选 reranker 精排 + 引用元数据 */
|
|
487
527
|
async function search(query, topK) {
|
|
488
528
|
const s = await ensureStore();
|
|
@@ -560,7 +600,7 @@ function apply(ctx, config = {}) {
|
|
|
560
600
|
const watcher = watch(dir, { persistent: false }, (_ev, filename) => {
|
|
561
601
|
if (filename === null || filename === void 0) return;
|
|
562
602
|
const fname = String(filename);
|
|
563
|
-
if (
|
|
603
|
+
if (!WATCH_EXT.test(fname)) return;
|
|
564
604
|
const f = join(dir, fname);
|
|
565
605
|
setTimeout(() => {
|
|
566
606
|
if (existsSync(f)) indexOne(ctx, f, "watch");
|
|
@@ -569,6 +609,7 @@ function apply(ctx, config = {}) {
|
|
|
569
609
|
ctx.effect(() => {
|
|
570
610
|
watcher.close();
|
|
571
611
|
}, "rag-kb: dir watcher");
|
|
612
|
+
scanWatchDir(ctx, dir);
|
|
572
613
|
} catch (e) {
|
|
573
614
|
console.warn(`[rag-kb] 目录监视失败 ${dir}: ${String(e.message)}`);
|
|
574
615
|
}
|
|
@@ -691,12 +732,12 @@ function apply(ctx, config = {}) {
|
|
|
691
732
|
if (a.action === "save" && a.text !== void 0 && a.text !== "") {
|
|
692
733
|
const title = a.title !== void 0 && a.title !== "" ? a.title : `对话保存 ${(/* @__PURE__ */ new Date()).toISOString().slice(0, 16)}`;
|
|
693
734
|
const content = `# ${title}\n\n${a.text}`;
|
|
694
|
-
const docId = KbStore.
|
|
735
|
+
const docId = KbStore.hashContent(content);
|
|
695
736
|
if (s.hasDoc(docId) && s.getDoc(docId)?.status === "ready") return {
|
|
696
737
|
ok: true,
|
|
697
738
|
message: `${title} 已在知识库中(内容相同,跳过)`
|
|
698
739
|
};
|
|
699
|
-
s.upsertDoc(docId, title, "save", content.length, "indexing");
|
|
740
|
+
s.upsertDoc(docId, title, "save", content.length, docId, "indexing");
|
|
700
741
|
broadcastProgress(`正在索引对话内容「${title}」`);
|
|
701
742
|
try {
|
|
702
743
|
const embedder = await getEmbedder(cfg?.embeddingModel ?? "Xenova/bge-small-zh-v1.5");
|