ppxans-harness 2.4.0 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -201
- package/README.md +218 -265
- package/bin/ppx-channels.js +2 -2
- package/bin/ppx-serve.js +5 -5
- package/bin/ppx-setup.js +124 -0
- package/bin/ppx-web.js +140 -0
- package/bin/ppx.js +2 -2
- package/config/identity.md +6 -6
- package/config/ishiki.md +16 -16
- package/config/ppx.json +15 -151
- package/config/ppx.json.example +143 -0
- package/package.json +17 -10
- package/skills/.usage.json +6 -0
- package/skills/agent-professional-training/SKILL.md +94 -0
- package/skills/brainstorm/SKILL.md +24 -0
- package/skills/cupid-lover-comms/SKILL.md +37 -0
- package/skills/debug/SKILL.md +26 -0
- package/skills/plan/SKILL.md +25 -0
- package/skills/ponytail/SKILL.md +25 -0
- package/skills/ppx-memory/SKILL.md +91 -0
- package/skills/ppx-memory/scripts/cli.js +192 -0
- package/skills/ppx-memory/scripts/experience.js +133 -0
- package/skills/ppx-memory/scripts/fact-store.js +842 -0
- package/skills/ppx-memory/scripts/l0.js +52 -0
- package/skills/ppx-memory/scripts/l2.js +146 -0
- package/skills/ppx-memory/scripts/l3.js +112 -0
- package/skills/ppx-memory/scripts/memory-ticker.js +238 -0
- package/skills/ppx-memory/scripts/pii.js +42 -0
- package/skills/ppx-memory/scripts/schema.js +80 -0
- package/skills/ppx-memory/scripts/session.js +398 -0
- package/skills/ppx-memory/scripts/similarity.js +43 -0
- package/skills/ppx-memory/scripts/store.js +116 -0
- package/skills/ppx-memory/scripts/wal.js +38 -0
- package/skills/ppx-selfheal/SKILL.md +24 -0
- package/skills/ppx-selfheal/scripts/cli.js +80 -0
- package/skills/ppx-selfheal/scripts/healer.js +184 -0
- package/skills/ppx-selfheal/scripts/logger.js +17 -0
- package/skills/ppx-selfheal/scripts/store.js +116 -0
- package/skills/prompt-depth-kit/SKILL.md +28 -0
- package/skills/session-naming/SKILL.md +36 -0
- package/skills/verify/SKILL.md +25 -0
- package/src/agent/index.js +1340 -717
- package/src/agent/prompts.js +47 -3
- package/src/aml-server.js +197 -151
- package/src/ans/eviction.js +123 -143
- package/src/ans/guard.js +159 -120
- package/src/ans/lifecycle.js +96 -93
- package/src/ans/proactive.js +112 -129
- package/src/ans/reward.js +95 -111
- package/src/ans/values.js +15 -15
- package/src/audit/audit-chain.js +43 -7
- package/src/audit/verifier.js +157 -120
- package/src/bus/circuit-breaker.js +9 -1
- package/src/bus/runtime-bus.js +107 -93
- package/src/channels/base.js +57 -34
- package/src/channels/feishu.js +118 -126
- package/src/channels/http.js +1090 -592
- package/src/channels/index.js +111 -110
- package/src/channels/log.js +29 -29
- package/src/channels/wechat-crypto.js +73 -74
- package/src/channels/wechat.js +191 -197
- package/src/channels/workspace.js +94 -0
- package/src/channels-cli.js +126 -124
- package/src/cli.js +129 -120
- package/src/commands/index.js +142 -0
- package/src/config/channels.js +137 -170
- package/src/config/index.js +277 -224
- package/src/config/placeholder.js +31 -0
- package/src/config/providers.js +148 -188
- package/src/config/settings.js +156 -182
- package/src/core/policy.js +69 -18
- package/src/core/trace.js +8 -6
- package/src/edit/editblock.js +266 -0
- package/src/edit/snapshot.js +67 -0
- package/src/evidence/index.js +153 -0
- package/src/evolve/playbook.js +9 -10
- package/src/hooks/index.js +113 -0
- package/src/llm/client.js +188 -446
- package/src/llm/dsml.js +74 -74
- package/src/llm/embedder.js +41 -35
- package/src/llm/fence.js +52 -105
- package/src/llm/index.js +4 -4
- package/src/llm/local-embedder.js +94 -0
- package/src/llm/presets.js +113 -0
- package/src/llm/pricing.js +93 -0
- package/src/llm/retry.js +73 -73
- package/src/llm/router.js +92 -97
- package/src/mcp/admin.js +326 -0
- package/src/mcp/client.js +487 -375
- package/src/mcp/http.js +203 -0
- package/src/mcp/index.js +116 -116
- package/src/mcp/server.js +392 -0
- package/src/mcp/tasks.js +133 -0
- package/src/memory/asset-hub.js +6 -11
- package/src/memory/canvas.js +2 -5
- package/src/memory/compaction.js +28 -28
- package/src/memory/experience.js +133 -122
- package/src/memory/fact-store.js +914 -698
- package/src/memory/failure-episode.js +20 -11
- package/src/memory/fork.js +17 -8
- package/src/memory/index.js +8 -6
- package/src/memory/l0.js +53 -52
- package/src/memory/l2.js +145 -130
- package/src/memory/l3.js +111 -111
- package/src/memory/legion-board.js +71 -0
- package/src/memory/memory-ticker.js +239 -240
- package/src/memory/session.js +398 -397
- package/src/memory/sqlite-store.js +581 -0
- package/src/mode/blackboard.js +49 -49
- package/src/mode/graph.js +42 -41
- package/src/mode/index.js +64 -64
- package/src/mode/legion.js +54 -51
- package/src/mode/plan-exec.js +50 -50
- package/src/mode/router.js +28 -40
- package/src/orchestrator/agent-worker.js +69 -69
- package/src/orchestrator/dag.js +90 -83
- package/src/orchestrator/experts.js +76 -0
- package/src/orchestrator/index.js +1 -1
- package/src/orchestrator/legion.js +179 -187
- package/src/orchestrator/supervisor.js +6 -8
- package/src/permissions/index.js +378 -0
- package/src/persona/index.js +28 -29
- package/src/plugin/builtin.js +313 -212
- package/src/plugin/context.js +80 -79
- package/src/plugin/index.js +62 -62
- package/src/plugin/v3.js +73 -0
- package/src/protocol/index.js +148 -0
- package/src/repomap/index.js +309 -0
- package/src/review/index.js +393 -0
- package/src/seam/registry.js +3 -0
- package/src/seam/shell.js +55 -55
- package/src/security/injection.js +79 -0
- package/src/selfheal/evolve.js +67 -67
- package/src/selfheal/healer.js +184 -167
- package/src/selfheal/run.js +9 -9
- package/src/server.js +63 -60
- package/src/services/diagnose.js +180 -0
- package/src/services/learning-service.js +9 -0
- package/src/services/memory-health.js +34 -6
- package/src/services/memory-service.js +42 -9
- package/src/services/triage.js +138 -0
- package/src/session/parts.js +76 -0
- package/src/session/projection.js +73 -0
- package/src/session/rollout.js +54 -0
- package/src/session/turn.js +137 -0
- package/src/skills/lint.js +72 -0
- package/src/skills/loader.js +231 -150
- package/src/skills/search.js +58 -0
- package/src/skills/verify.js +95 -100
- package/src/tools/advanced.js +388 -352
- package/src/tools/builtin.js +384 -297
- package/src/tools/catalog.js +283 -159
- package/src/tools/command-guard.js +112 -112
- package/src/tools/custom.js +47 -47
- package/src/tools/delegate.js +383 -297
- package/src/tools/document.js +254 -253
- package/src/tools/git.js +151 -0
- package/src/tools/governance.js +47 -20
- package/src/tools/index.js +16 -11
- package/src/tools/methods.js +178 -178
- package/src/tools/ocr.js +59 -59
- package/src/tools/sandbox-worker.js +40 -0
- package/src/tools/sandbox.js +92 -0
- package/src/tools/seam.js +162 -125
- package/src/tools/selfmod.js +196 -176
- package/src/tools/v3.js +225 -0
- package/src/tools/vad.js +176 -0
- package/src/tools/voice.js +238 -0
- package/src/utils/async.js +14 -0
- package/src/utils/config-file.js +53 -0
- package/src/utils/crashguard.js +88 -0
- package/src/utils/http.js +53 -0
- package/src/utils/id.js +8 -0
- package/src/utils/json-state.js +33 -0
- package/src/utils/logger.js +17 -17
- package/src/utils/ndjson.js +25 -0
- package/src/utils/pii.js +42 -42
- package/src/utils/rate-limit.js +50 -0
- package/src/utils/schema.js +80 -0
- package/src/utils/similarity.js +43 -0
- package/src/utils/store.js +170 -108
- package/src/utils/text.js +15 -15
- package/src/utils/trace.js +153 -153
- package/src/utils/wal.js +39 -0
- package/src/utils/winutf8.js +16 -15
- package/src/wiki/index.js +170 -0
package/src/tools/document.js
CHANGED
|
@@ -1,253 +1,254 @@
|
|
|
1
|
-
// src/tools/document.js - 文档加载器 (对标 LangChain Document Loaders, 零依赖)
|
|
2
|
-
// 支持: txt / md / json / csv / html / pdf (文字型, 扫描件需 OCR)
|
|
3
|
-
// PDF 零依赖提取: 解压 FlateDecode 流 (zlib) + 提取 Tj/TJ 文本操作符
|
|
4
|
-
import fs from "node:fs";
|
|
5
|
-
import os from "node:os";
|
|
6
|
-
import path from "node:path";
|
|
7
|
-
import zlib from "node:zlib";
|
|
8
|
-
import { ocrImage } from "./ocr.js";
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
const
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
const
|
|
47
|
-
const
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
const
|
|
76
|
-
const
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
.replace(/<
|
|
91
|
-
.replace(/<
|
|
92
|
-
.replace(/<[
|
|
93
|
-
.replace(
|
|
94
|
-
.replace(
|
|
95
|
-
.
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
}
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
const
|
|
119
|
-
const
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
cur
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
}
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
const
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
const
|
|
181
|
-
const
|
|
182
|
-
const
|
|
183
|
-
const
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
const
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
const
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
const
|
|
238
|
-
const
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
1
|
+
// src/tools/document.js - 文档加载器 (对标 LangChain Document Loaders, 零依赖)
|
|
2
|
+
// 支持: txt / md / json / csv / html / pdf (文字型, 扫描件需 OCR)
|
|
3
|
+
// PDF 零依赖提取: 解压 FlateDecode 流 (zlib) + 提取 Tj/TJ 文本操作符
|
|
4
|
+
import fs from "node:fs";
|
|
5
|
+
import os from "node:os";
|
|
6
|
+
import path from "node:path";
|
|
7
|
+
import zlib from "node:zlib";
|
|
8
|
+
import { ocrImage } from "./ocr.js";
|
|
9
|
+
import { debug } from "../utils/logger.js";
|
|
10
|
+
|
|
11
|
+
const MAX_CHARS = 20000; // 单文档返回上限
|
|
12
|
+
|
|
13
|
+
// 安全路径: 阻止逃出工作目录 (防路径穿越, 与 builtin.js 同策略)
|
|
14
|
+
function safePath(root, p) {
|
|
15
|
+
const resolved = path.resolve(root, p);
|
|
16
|
+
if (resolved !== root && !resolved.startsWith(root + path.sep)) {
|
|
17
|
+
throw new Error(`路径越界拒绝: ${p}`);
|
|
18
|
+
}
|
|
19
|
+
return resolved;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
// 解码 PDF 文本字符串: 处理 UTF-16BE (FE FF BOM) 与 UTF-8 与转义字符
|
|
23
|
+
function decodePdfString(latin1Str) {
|
|
24
|
+
let s = String(latin1Str || "").replace(/\\(\r?\n)/g, ""); // 续行
|
|
25
|
+
const buf = Buffer.from(s, "latin1");
|
|
26
|
+
// UTF-16BE BOM
|
|
27
|
+
if (buf.length >= 2 && buf[0] === 0xfe && buf[1] === 0xff) {
|
|
28
|
+
return buf.slice(2).toString("utf16le");
|
|
29
|
+
}
|
|
30
|
+
// 字节级反转义 \( \) \\
|
|
31
|
+
const out = [];
|
|
32
|
+
for (let i = 0; i < buf.length; i++) {
|
|
33
|
+
if (buf[i] === 0x5c && i + 1 < buf.length) {
|
|
34
|
+
const n = buf[i + 1];
|
|
35
|
+
if (n === 0x28 || n === 0x29 || n === 0x5c) { out.push(n); i++; continue; }
|
|
36
|
+
}
|
|
37
|
+
out.push(buf[i]);
|
|
38
|
+
}
|
|
39
|
+
const clean = Buffer.from(out);
|
|
40
|
+
const utf8 = clean.toString("utf8");
|
|
41
|
+
return utf8.includes("\uFFFD") ? clean.toString("latin1") : utf8;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
// 零依赖 PDF 文本提取: 遍历 stream, FlateDecode 解压, 提取 (text) Tj 与 [(a)(b)] TJ
|
|
45
|
+
export function extractPdfText(buf) {
|
|
46
|
+
const raw = buf.toString("latin1");
|
|
47
|
+
const texts = [];
|
|
48
|
+
const streamRe = /stream\r?\n([\s\S]*?)endstream/g;
|
|
49
|
+
let m;
|
|
50
|
+
while ((m = streamRe.exec(raw)) !== null) {
|
|
51
|
+
let data = m[1];
|
|
52
|
+
// 尝试 FlateDecode 解压 (内容流通常是压缩的)
|
|
53
|
+
try {
|
|
54
|
+
const inf = zlib.inflateSync(Buffer.from(data, "latin1")).toString("latin1");
|
|
55
|
+
if (inf.length > 0) data = inf;
|
|
56
|
+
} catch { /* 未压缩则用原文 */ }
|
|
57
|
+
// (text) Tj
|
|
58
|
+
for (const tm of data.matchAll(/\(([^)]*)\)\s*Tj/g)) {
|
|
59
|
+
const t = decodePdfString(tm[1]).trim();
|
|
60
|
+
if (t) texts.push(t);
|
|
61
|
+
}
|
|
62
|
+
// [(a)(b)] TJ (数组形式)
|
|
63
|
+
for (const tm of data.matchAll(/\[((?:\([^)]*\)[\s<>0-9.-]*)+)\]\s*TJ/g)) {
|
|
64
|
+
for (const pm of tm[1].matchAll(/\(([^)]*)\)/g)) {
|
|
65
|
+
const t = decodePdfString(pm[1]).trim();
|
|
66
|
+
if (t) texts.push(t);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
return texts.join(" ");
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// 提取 PDF 内嵌 JPEG 图片 (DCTDecode 流, 扫描件 PDF 的页面图), 供 OCR
|
|
74
|
+
export function extractPdfJpegs(buf) {
|
|
75
|
+
const raw = buf.toString("latin1");
|
|
76
|
+
const jpegs = [];
|
|
77
|
+
const re = /\/DCTDecode[\s\S]{0,200}?stream\r?\n([\s\S]*?)endstream/g;
|
|
78
|
+
let m;
|
|
79
|
+
while ((m = re.exec(raw)) !== null) {
|
|
80
|
+
const data = Buffer.from(m[1], "latin1");
|
|
81
|
+
// JPEG 魔数 FFD8
|
|
82
|
+
if (data.length > 2 && data[0] === 0xff && data[1] === 0xd8) jpegs.push(data);
|
|
83
|
+
}
|
|
84
|
+
return jpegs;
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
// HTML 转纯文本 (基础 strip, 与 advanced.js fetch_page 同思路)
|
|
88
|
+
function htmlToText(html) {
|
|
89
|
+
return String(html || "")
|
|
90
|
+
.replace(/<script[\s\S]*?<\/script>/gi, " ")
|
|
91
|
+
.replace(/<style[\s\S]*?<\/style>/gi, " ")
|
|
92
|
+
.replace(/<br\s*\/?>|<\/p>|<\/div>|<\/li>|<\/h[1-6]>|<\/tr>/gi, "\n")
|
|
93
|
+
.replace(/<[^>]+>/g, " ")
|
|
94
|
+
.replace(/ /gi, " ").replace(/&/gi, "&").replace(/</gi, "<").replace(/>/gi, ">")
|
|
95
|
+
.replace(/\s+/g, " ")
|
|
96
|
+
.trim();
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
// 按扩展名提取文档文本 (纯函数, 供 read_document 与 ingest_document 复用)
|
|
100
|
+
export function extractDocumentText(filePath) {
|
|
101
|
+
const ext = path.extname(filePath).toLowerCase();
|
|
102
|
+
if (!fs.existsSync(filePath)) throw new Error(`文件不存在: ${filePath}`);
|
|
103
|
+
const buf = fs.readFileSync(filePath);
|
|
104
|
+
switch (ext) {
|
|
105
|
+
case ".txt": case ".md": case ".csv": case ".json": case ".log":
|
|
106
|
+
return buf.toString("utf8");
|
|
107
|
+
case ".html": case ".htm":
|
|
108
|
+
return htmlToText(buf.toString("utf8"));
|
|
109
|
+
case ".pdf":
|
|
110
|
+
return extractPdfText(buf);
|
|
111
|
+
default:
|
|
112
|
+
throw new Error(`不支持的文档类型: ${ext} (支持 txt/md/csv/json/html/pdf)`);
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
// 分块: 按段落切, 每块约 chunkSize 字 (供 ingest 向量化)
|
|
117
|
+
export function splitChunks(text, chunkSize = 500) {
|
|
118
|
+
const clean = String(text || "").replace(/\r/g, "");
|
|
119
|
+
const paras = clean.split(/\n{2,}/).map((p) => p.trim()).filter(Boolean);
|
|
120
|
+
const chunks = [];
|
|
121
|
+
let cur = "";
|
|
122
|
+
for (const p of paras) {
|
|
123
|
+
if ((cur + p).length > chunkSize && cur) {
|
|
124
|
+
chunks.push(cur.trim());
|
|
125
|
+
cur = p;
|
|
126
|
+
} else {
|
|
127
|
+
cur = cur ? cur + "\n" + p : p;
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
if (cur.trim()) chunks.push(cur.trim());
|
|
131
|
+
return chunks.filter((c) => c.length >= 10);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// 从 config 构造 OCR 选项 (未显式关闭时默认启用本地 tesseract 自动 OCR)
|
|
135
|
+
function ocrOptsFromConfig(cfg) {
|
|
136
|
+
const c = (cfg && cfg.ocr) || {};
|
|
137
|
+
if (c.auto === false) return null; // 显式关闭自动 OCR
|
|
138
|
+
return { tesseract: c.tesseract || "tesseract", lang: c.lang || "chi_sim", cloud: c.cloud || null };
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
// 读文档文本, PDF 无文本层(扫描件)时自动提取内嵌图片 OCR (_ocrFn 供测试注入)
|
|
142
|
+
export async function readDocumentText(filePath, ocrOpts, _ocrFn = ocrImage) {
|
|
143
|
+
let text = extractDocumentText(filePath);
|
|
144
|
+
if (path.extname(filePath).toLowerCase() === ".pdf" && !text.trim() && ocrOpts) {
|
|
145
|
+
const jpegs = extractPdfJpegs(fs.readFileSync(filePath));
|
|
146
|
+
if (jpegs.length) {
|
|
147
|
+
const parts = [];
|
|
148
|
+
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), "ppx-ocr-pdf-"));
|
|
149
|
+
try {
|
|
150
|
+
for (let i = 0; i < jpegs.length; i++) {
|
|
151
|
+
const tmp = path.join(tmpDir, `page-${i + 1}.jpg`);
|
|
152
|
+
fs.writeFileSync(tmp, jpegs[i]);
|
|
153
|
+
try { parts.push(await _ocrFn(tmp, ocrOpts)); } catch (e) { debug(`[tools/document] 已忽略异常: ${e && e.message ? e.message : e}`); }
|
|
154
|
+
}
|
|
155
|
+
} finally {
|
|
156
|
+
fs.rmSync(tmpDir, { recursive: true, force: true });
|
|
157
|
+
}
|
|
158
|
+
text = parts.filter(Boolean).join("\n").trim();
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
return text;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
// 注册文档工具
|
|
165
|
+
export function registerDocumentTools(catalog, { rootDir }) {
|
|
166
|
+
// 1. 读文档 (加载器, PDF 扫描件自动 OCR)
|
|
167
|
+
catalog.register({
|
|
168
|
+
name: "read_document",
|
|
169
|
+
description: "读取本地文档并转纯文本。支持 .txt/.md/.json/.csv/.html/.pdf (文字型 PDF 直接提取, 扫描件 PDF 自动 OCR)。用于读文档/报告/数据文件后回答问题。",
|
|
170
|
+
parameters: {
|
|
171
|
+
type: "object",
|
|
172
|
+
properties: {
|
|
173
|
+
path: { type: "string", description: "文档路径 (相对工作目录)" },
|
|
174
|
+
maxChars: { type: "number", description: "返回最大字符数, 默认 20000" },
|
|
175
|
+
},
|
|
176
|
+
required: ["path"],
|
|
177
|
+
},
|
|
178
|
+
execute: async (args, ctx) => {
|
|
179
|
+
try {
|
|
180
|
+
const p = safePath(rootDir, args.path);
|
|
181
|
+
const cfg = (ctx && ctx.agent && ctx.agent.config) || {};
|
|
182
|
+
const text = await readDocumentText(p, ocrOptsFromConfig(cfg));
|
|
183
|
+
const max = Math.min(args.maxChars || MAX_CHARS, 40000);
|
|
184
|
+
const truncated = text.length > max ? text.slice(0, max) + `\n...[已截断, 共 ${text.length} 字符]` : text;
|
|
185
|
+
return truncated || "(文档无文本内容, 且未识别出扫描件文字)";
|
|
186
|
+
} catch (e) {
|
|
187
|
+
return JSON.stringify({ error: "read_document 失败: " + e.message });
|
|
188
|
+
}
|
|
189
|
+
},
|
|
190
|
+
});
|
|
191
|
+
|
|
192
|
+
// 2. OCR 识别图片/扫描件文字
|
|
193
|
+
catalog.register({
|
|
194
|
+
name: "ocr_image",
|
|
195
|
+
description: "识别图片或扫描件里的文字 (OCR)。需系统安装 tesseract (含中文语言包) 或配置 config.ocr 云 key。用于 read_image/read_document 读到图片却无法理解文字时。",
|
|
196
|
+
parameters: {
|
|
197
|
+
type: "object",
|
|
198
|
+
properties: {
|
|
199
|
+
path: { type: "string", description: "图片或扫描件路径 (相对工作目录)" },
|
|
200
|
+
lang: { type: "string", description: "识别语言, 默认 chi_sim (中文)" },
|
|
201
|
+
},
|
|
202
|
+
required: ["path"],
|
|
203
|
+
},
|
|
204
|
+
execute: async (args, ctx) => {
|
|
205
|
+
const agent = ctx && ctx.agent;
|
|
206
|
+
const cfg = (agent && agent.config && agent.config.ocr) || {};
|
|
207
|
+
try {
|
|
208
|
+
const p = safePath(rootDir, args.path);
|
|
209
|
+
const text = await ocrImage(p, {
|
|
210
|
+
tesseract: cfg.tesseract || "tesseract",
|
|
211
|
+
lang: args.lang || cfg.lang || "chi_sim",
|
|
212
|
+
cloud: cfg.cloud || null,
|
|
213
|
+
});
|
|
214
|
+
return text || "(未识别出文字)";
|
|
215
|
+
} catch (e) {
|
|
216
|
+
return JSON.stringify({ error: "ocr_image 失败: " + e.message });
|
|
217
|
+
}
|
|
218
|
+
},
|
|
219
|
+
});
|
|
220
|
+
|
|
221
|
+
// 3. 文档入库 (RAG): 读文档 → 分块 → 存记忆 (带 scope 隔离)
|
|
222
|
+
catalog.register({
|
|
223
|
+
name: "ingest_document",
|
|
224
|
+
description: "读取文档, 分块后写入长期记忆 (RAG 入库), 之后可被语义检索命中。scope 用于隔离文档来源 (如 '公司制度'/'项目文档'), 避免与其他记忆混淆。",
|
|
225
|
+
parameters: {
|
|
226
|
+
type: "object",
|
|
227
|
+
properties: {
|
|
228
|
+
path: { type: "string", description: "文档路径 (相对工作目录)" },
|
|
229
|
+
scope: { type: "string", description: "文档来源标签 (可选, 便于按来源检索)" },
|
|
230
|
+
},
|
|
231
|
+
required: ["path"],
|
|
232
|
+
},
|
|
233
|
+
execute: async (args, ctx) => {
|
|
234
|
+
const agent = ctx && ctx.agent;
|
|
235
|
+
if (!agent || !agent.facts) return JSON.stringify({ error: "ingest_document: 缺少 agent 上下文" });
|
|
236
|
+
try {
|
|
237
|
+
const p = safePath(rootDir, args.path);
|
|
238
|
+
const text = await readDocumentText(p, ocrOptsFromConfig(agent.config));
|
|
239
|
+
const chunks = splitChunks(text, 500);
|
|
240
|
+
if (!chunks.length) return JSON.stringify({ error: "文档无可入库的文本 (可能是扫描件 PDF, 且 OCR 不可用)" });
|
|
241
|
+
let added = 0;
|
|
242
|
+
for (const c of chunks) {
|
|
243
|
+
const f = agent.facts.add(c, { source: "document", scope: args.scope || null, dedupe: false });
|
|
244
|
+
if (f) added++;
|
|
245
|
+
}
|
|
246
|
+
return JSON.stringify({ ok: true, chunks: chunks.length, added, scope: args.scope || null });
|
|
247
|
+
} catch (e) {
|
|
248
|
+
return JSON.stringify({ error: "ingest_document 失败: " + e.message });
|
|
249
|
+
}
|
|
250
|
+
},
|
|
251
|
+
});
|
|
252
|
+
|
|
253
|
+
return catalog;
|
|
254
|
+
}
|