dsh-plugin-office-markdown 1.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js ADDED
@@ -0,0 +1,1443 @@
1
+ /*!
2
+ * dsh-plugin-office-markdown — host half
3
+ *
4
+ * Registers:
5
+ * - the `read_office_as_markdown` tool (Office/PDF -> Markdown spill),
6
+ * - a bundled runtime skill that tells the model to use it,
7
+ * - an optional guard that redirects a raw `read` of a binary Office/PDF
8
+ * file to that tool.
9
+ *
10
+ * Everything is registered through `ctx.effect(...)`, and `config.enabled ===
11
+ * false` short-circuits `apply()` before any registration happens: disabling
12
+ * the plugin (config or DSH plugin manager) leaves the process exactly as if
13
+ * the package were never installed.
14
+ */
15
+ import fs from 'node:fs'
16
+ import path from 'node:path'
17
+ import { fileURLToPath } from 'node:url'
18
+
19
+ import {
20
+ analyzeMarkdown,
21
+ classify,
22
+ converterLabelFor,
23
+ convertFile,
24
+ ensureDirSync,
25
+ findLatestArtifact,
26
+ listArtifacts,
27
+ OFFICE_EXTS,
28
+ probeConverters,
29
+ pythonSourceLabel,
30
+ readFidelityMarker,
31
+ safeBaseName,
32
+ sha8,
33
+ sourceContentHash
34
+ } from './convert.js'
35
+ import {
36
+ currentProfileDir,
37
+ profilesRoot,
38
+ removalLogPath,
39
+ setProfileDir,
40
+ watchdogLockPath
41
+ } from './paths.js'
42
+ import { spawnRemovalWatchdog } from './removal-watchdog.js'
43
+ import { registerSettingsApi } from './settings-api.js'
44
+ import {
45
+ adoptMarkitdown,
46
+ readSnapshot,
47
+ snapshotPath,
48
+ snapshotSummary,
49
+ uninstallMarkitdown
50
+ } from './env.js'
51
+
52
+ /* `defineTool` is the same helper `dsh-plugin-save-token` imports. A static
53
+ * import is used on purpose: dynamic import + top-level await would make this
54
+ * module's namespace asynchronous, and dsh's loader would then evaluate the
55
+ * plugin object too late (the entry silently fails to activate). */
56
+ import { defineTool } from '@deepseek-ai/dsh-tools'
57
+
58
+ export const name = 'office-markdown'
59
+
60
+ export const inject = ['tools', 'skills']
61
+
62
+ const TOOL_NAME = 'read_office_as_markdown'
63
+ const SKILL_NAME = 'office-pdf-to-markdown'
64
+ const PROVIDER = 'dsh-plugin-office-markdown'
65
+
66
+ const DEFAULTS = Object.freeze({
67
+ enabled: true,
68
+ tmpDir: '',
69
+ converter: 'auto',
70
+ pythonPath: '',
71
+ pythonPrefer: 'auto',
72
+ allowUvxDownload: true,
73
+ uvxExtras: 'markitdown[all]',
74
+ fallbackEnabled: true,
75
+ guardReadTool: true,
76
+ probeTtlMs: 600000,
77
+ timeoutMs: 300000,
78
+ reuseFresh: true,
79
+ maxPreviewChars: 4000,
80
+ maxRowsPerSheet: 400,
81
+ maxTableCols: 24,
82
+ maxCellsPerSheet: 20000,
83
+ pruneStaleArtifacts: false,
84
+ registerSkill: true,
85
+ registerSettings: true,
86
+ autoAdoptEnv: true,
87
+ removeEnvOnUninstall: true
88
+ })
89
+
90
+ /** Absolute directory of this module and of the plugin package itself. */
91
+ const HERE = path.dirname(fileURLToPath(import.meta.url))
92
+ const PACKAGE_ROOT = path.dirname(HERE)
93
+
94
+ /** Read our own package version for the settings page; never fatal. */
95
+ function pluginVersion() {
96
+ try {
97
+ const pkg = JSON.parse(fs.readFileSync(path.join(PACKAGE_ROOT, 'package.json'), 'utf8'))
98
+ return typeof pkg.version === 'string' ? pkg.version : '0.0.0'
99
+ } catch {
100
+ return '0.0.0'
101
+ }
102
+ }
103
+
104
+ const VERSION = pluginVersion()
105
+
106
+ const SKILL_CONTENT = `# Office / PDF 文件先转 Markdown 再读取
107
+
108
+ 遇到 .docx、.xlsx、.pptx、.pdf 等 Office / PDF 文件时,不要直接读原文件,必须先调用 read_office_as_markdown 工具转换成 Markdown,再读取转换后的 .md 文件。转换结果保存在工作区里(就在源文件旁边),按需读取,避免全文注入上下文。
109
+
110
+ ## 为什么
111
+
112
+ Office / PDF 是二进制压缩容器。直接 read 只会得到乱码或超长的 XML 片段,既浪费 token 又拿不到有用信息。先转成 Markdown 再按需读取,可以显著降低上下文占用。
113
+
114
+ ## 标准流程
115
+
116
+ 1. \`read_office_as_markdown({ path: "报表.xlsx" })\`
117
+ - 返回转换后的 .md 路径(**不会**把全文塞进上下文),例如 \`报表-3f9a2c1d.md\`(就生成在源文件旁边)。
118
+ - 返回值里同时给出体积、行数、token 估算,以及一份**结构索引**(章节 / 工作表 / 幻灯片的标题与大致行号)。
119
+ - 一次要处理多个文件或整个文件夹时,用 \`paths: ["a.xlsx", "b.docx"]\` 或直接传目录,不要一个文件来回一次。
120
+ 2. 用 read 工具读取该 .md 文件,按需分段读取:
121
+ - 先读开头(例如 \`read({ file_path: "...", limit: 200 })\`)确认结构,再决定要不要继续读。
122
+ - **不要一次把整个 .md 读进上下文**:产物不会被裁剪,读多少完全由你控制。
123
+ 3. 如果只需要某一部分(某个工作表 / 某张幻灯片 / 某个章节):
124
+ - 先 \`read_office_as_markdown({ path: "...", action: "outline" })\` 拿一份标题索引(**只读索引,绝不触发转换**),
125
+ - 再用 grep 在该 .md 里定位,最后定点 read。
126
+
127
+ ## 参数说明
128
+
129
+ - \`path\`:工作区内的文件**或目录**路径,相对或绝对均可(\`action: "status"\` 时可省略)。传目录时按扩展名找出其中的 Office / PDF 文件。
130
+ - \`paths\`:多个文件 / 目录的数组,一次调用批量转换(逐文件串行,结果按“一行一个文件”汇总)。
131
+ - \`action\`:\`auto\`(默认,自动判断)/ \`convert\`(强制转换)/ \`read\`(读取已转换结果的开头)/ \`outline\`(只给结构索引)/ \`clean\`(清理该源文件的陈旧产物)/ \`status\`(查看当前可用的转换器,以及每个 Python 环境有没有 markitdown)。
132
+ - \`force\`:\`true\` 时忽略缓存重新转换。
133
+ - \`preview\`:可选,返回转换后 Markdown 开头 N 个字符(默认 0,即只给路径,最省 token)。
134
+ - \`recursive\`:\`path\` 是目录时是否进子目录(默认 false,一次最多 200 个文件)。
135
+ - \`dryRun\`:只对 \`action: "clean"\` 有效,默认 \`true\`,只报告会删哪些陈旧产物。
136
+
137
+ ## 注意
138
+
139
+ - 纯文本 / Markdown / CSV 等文件本来就能直接读取,工具会直接告知,不需要转换(CSV 转 Markdown 表格通常会**更费** token)。
140
+ - 转换在本地完成,不修改原文件,不消耗任何 API 额度。
141
+ - 转换结果 \`.md\` 就写在**源文件旁边**(同目录),除此之外不产生任何临时目录、登记表或缓存元数据。
142
+ - 同一个源文件反复改动时可能留下多份历史产物。要清理就用 \`action: "clean"\`(默认先干跑,确认后再传 \`dryRun: false\`),或请用户在配置里打开 \`pruneStaleArtifacts\`。**不要**自己用命令行删文件。
143
+ - 用户想提升保真度时,让他们打开 DSH 设置里的「Office 转换」页面:那里能查看本机 Python 环境、一键配置 MarkItDown、**试转一个文件确认转换链真的可用**、以及卸载插件配置的环境。**不要**自己执行 pip 安装。
144
+ - 卸载这个插件时,它会自动把设置页登记过的 MarkItDown 与依赖一并卸载;但**禁用 / 关闭 / 重启插件都不会卸载任何 Python 包**。
145
+ - 如果转换失败,工具会给出明确错误;此时可以回退到直接读取,或提示用户打开设置页面配置 MarkItDown。
146
+ `
147
+
148
+ function loggerFor(ctx) {
149
+ try {
150
+ if (ctx && ctx.logger && typeof ctx.logger.info === 'function') return ctx.logger
151
+ } catch { /* ignore */ }
152
+ return null
153
+ }
154
+
155
+ function humanBytes(n) {
156
+ if (!Number.isFinite(n)) return '未知'
157
+ if (n < 1024) return n + ' B'
158
+ if (n < 1024 * 1024) return (n / 1024).toFixed(1) + ' KB'
159
+ return (n / 1024 / 1024).toFixed(2) + ' MB'
160
+ }
161
+
162
+ function workspaceRootOf(ctx, exec) {
163
+ try {
164
+ const sessions = ctx.get('sessions')
165
+ const id = exec && exec.agent ? exec.agent.id : undefined
166
+ const session = id && sessions ? sessions.get(id) : undefined
167
+ const cwd = session && session.header ? session.header.cwd : undefined
168
+ if (typeof cwd === 'string' && cwd) return cwd
169
+ } catch { /* ignore */ }
170
+ return undefined
171
+ }
172
+
173
+ function resolveInputPath(rawPath, root) {
174
+ const cwd = root || process.cwd()
175
+ return path.isAbsolute(rawPath) ? path.normalize(rawPath) : path.resolve(cwd, rawPath)
176
+ }
177
+
178
+ function displayPath(abs, root) {
179
+ if (!root) return abs
180
+ try {
181
+ const rel = path.relative(root, abs)
182
+ if (rel && !rel.startsWith('..') && !path.isAbsolute(rel)) return rel.split(path.sep).join('/')
183
+ } catch { /* ignore */ }
184
+ return abs
185
+ }
186
+
187
+ function configSignature(cfg) {
188
+ return JSON.stringify([
189
+ cfg.converter,
190
+ cfg.uvxExtras,
191
+ cfg.fallbackEnabled,
192
+ cfg.maxRowsPerSheet,
193
+ cfg.maxTableCols,
194
+ cfg.maxCellsPerSheet
195
+ ])
196
+ }
197
+
198
+ /* ------------------------------------------------------------------ */
199
+ /* artifacts: volume, outline, reuse, cleanup */
200
+ /* ------------------------------------------------------------------ */
201
+
202
+ const OUTLINE_MAX_BYTES = 262144
203
+ const OUTLINE_MAX_ENTRIES = 40
204
+ const READ_LINE_HINT = 200
205
+ const MAX_EXPAND_FILES = 200
206
+ const LOUD_ARTIFACT_LINES = 4000
207
+
208
+ function fidelityWord(fidelity) {
209
+ if (fidelity === 'limited') return '有限'
210
+ if (fidelity === 'high') return '高'
211
+ return '未知'
212
+ }
213
+
214
+ /**
215
+ * A cheap table of contents for an artifact.
216
+ *
217
+ * MarkItDown emits a heading per workbook sheet, per slide and per Word
218
+ * heading, so `## Sheet: 汇总` really is navigation: it lets the model ask for
219
+ * one section instead of reading a whole workbook to find it. Only the head of
220
+ * the file is scanned, so the cost stays bounded no matter how large the
221
+ * artifact is; `truncated` says whether headings past that point were missed.
222
+ */
223
+ function buildOutline(mdPath, options = {}) {
224
+ const maxBytes = Math.max(4096, Number(options.maxBytes) || OUTLINE_MAX_BYTES)
225
+ const maxEntries = Math.max(1, Number(options.maxEntries) || OUTLINE_MAX_ENTRIES)
226
+ const entries = []
227
+ let truncated = false
228
+ let fd = null
229
+ try {
230
+ fd = fs.openSync(mdPath, 'r')
231
+ const buf = Buffer.alloc(maxBytes)
232
+ const n = fs.readSync(fd, buf, 0, buf.length, 0)
233
+ truncated = safeSize(mdPath) > n
234
+ const text = buf.subarray(0, n).toString('utf8').replace(/\r\n?/g, '\n')
235
+ let line = 1
236
+ for (const raw of text.split('\n')) {
237
+ if (entries.length >= maxEntries) break
238
+ const m = raw.match(/^(#{1,4})\s+(.+?)\s*$/)
239
+ if (m) entries.push({ level: m[1].length, line, title: m[2].slice(0, 120) })
240
+ line++
241
+ }
242
+ } catch {
243
+ /* an unreadable artifact simply has no outline */
244
+ } finally {
245
+ if (fd !== null) { try { fs.closeSync(fd) } catch { /* ignore */ } }
246
+ }
247
+ return { entries, truncated }
248
+ }
249
+
250
+ /**
251
+ * What the model should actually do with an artifact of this size.
252
+ *
253
+ * Nothing is ever trimmed — the point of the whole plugin is to hand over the
254
+ * complete document — so the honest alternative is to say how big it is and how
255
+ * to read it without swallowing it whole.
256
+ */
257
+ function readingStrategy(shownDst, bytes, tokens, lines) {
258
+ const parts = [
259
+ '请用 read 工具读取 ' + shownDst + ':先读前 ' + READ_LINE_HINT +
260
+ ' 行看清结构(read 本身也只返回前 2000 行),再用 grep 在同一个文件里定位需要的片段,不要一次性把整份读进上下文。'
261
+ ]
262
+ if (lines > LOUD_ARTIFACT_LINES) {
263
+ parts.push(
264
+ '该产物约 ' + lines.toLocaleString('en-US') + ' 行、约 ' + tokens.toLocaleString('en-US') +
265
+ ' tokens(' + humanBytes(bytes) + '):整份读入会明显挤占上下文,建议先用 action:"outline" 拿章节 / 工作表索引,再定点读取。'
266
+ )
267
+ }
268
+ return parts.join(' ')
269
+ }
270
+
271
+ function pushOutline(lines, value) {
272
+ if (!value.outline || !value.outline.length) return
273
+ lines.push('• 结构索引(' + (value.outlineTruncated ? '只看了解文件开头,' : '') +
274
+ '共 ' + value.outline.length + ' 个标题,行号为约值):')
275
+ for (const h of value.outline) {
276
+ lines.push(' ' + ' '.repeat(Math.max(0, h.level - 1)) + h.title + '(约第 ' + h.line + ' 行)')
277
+ }
278
+ }
279
+
280
+ function pushPreview(lines, value) {
281
+ if (!value.preview) return
282
+ lines.push('')
283
+ lines.push('--- 预览(前 ' + value.preview.length + ' 字符)---')
284
+ lines.push(value.preview)
285
+ if (value.previewTruncated) lines.push('…(预览已截断)')
286
+ }
287
+
288
+ /**
289
+ * The artifact the current cache key points at, or — when the source has moved
290
+ * on — the newest artifact that still exists, flagged `stale`.
291
+ *
292
+ * `action:"read"` on a file whose mtime changed used to be a dead end: the key
293
+ * missed, the tool answered "还没有转换结果,请先执行转换", and the already
294
+ * converted `.md` sitting right next to it was ignored. Handing back the newest
295
+ * artifact with an honest `stale` flag turns that into a useful answer.
296
+ */
297
+ function findExistingArtifact(cfg, absSrc, st, root) {
298
+ const target = conversionTarget(cfg, absSrc, st, root)
299
+ if (fs.existsSync(target.absDst) && safeSize(target.absDst) > 0) return { ...target, stale: false }
300
+ const latest = findLatestArtifact(absSrc, tmpDirFor(cfg, absSrc, root))
301
+ if (!latest) return null
302
+ return { absDst: latest.path, shownDst: displayPath(latest.path, root), stale: true }
303
+ }
304
+
305
+ /**
306
+ * Reuse an artifact that was really produced from *this* source content.
307
+ *
308
+ * The cache key ends in the source's mtime, which `git checkout`, a copy or a
309
+ * restore changes without touching a single byte — so a naive lookup writes a
310
+ * second, identical `.md` next to the first. Every artifact records the size
311
+ * and a content hash of the source it came from, so a moved mtime is checked
312
+ * against the real content before paying for a re-conversion.
313
+ *
314
+ * On a hit the artifact is renamed onto the current key, so the fast path hits
315
+ * again next time and one source keeps exactly one artifact.
316
+ */
317
+ function reuseByContent(cfg, absSrc, st, root) {
318
+ const candidates = listArtifacts(absSrc, tmpDirFor(cfg, absSrc, root))
319
+ if (!candidates.length) return null
320
+ const comparable = []
321
+ for (const art of candidates) {
322
+ const stamp = readFidelityMarker(art.path)
323
+ if (!stamp || !stamp.srchash || stamp.srchash === 'unknown') continue
324
+ if (String(stamp.srcbytes) !== String(st.size)) continue
325
+ /* Never trade a MarkItDown artifact for a fallback one: if the only match
326
+ * is limited fidelity, converting fresh is the better answer. */
327
+ if (stamp.fidelity === 'limited') continue
328
+ comparable.push(art)
329
+ }
330
+ if (!comparable.length) return null
331
+ const hash = sourceContentHash(absSrc)
332
+ if (!hash) return null
333
+ for (const art of comparable) {
334
+ const stamp = readFidelityMarker(art.path)
335
+ if (!stamp || stamp.srchash !== hash) continue
336
+ const target = conversionTarget(cfg, absSrc, st, root)
337
+ let absDst = art.path
338
+ try {
339
+ fs.renameSync(art.path, target.absDst)
340
+ absDst = target.absDst
341
+ } catch { /* keep the old name if it cannot be renamed */ }
342
+ return {
343
+ absDst,
344
+ shownDst: displayPath(absDst, root),
345
+ converter: stamp.converter,
346
+ fidelity: stamp.fidelity
347
+ }
348
+ }
349
+ return null
350
+ }
351
+
352
+ /** Every artifact of this source except the one the current key points at. */
353
+ function staleArtifactsFor(cfg, absSrc, st, root) {
354
+ const dir = tmpDirFor(cfg, absSrc, root)
355
+ const current = path.resolve(conversionTarget(cfg, absSrc, st, root).absDst)
356
+ const stale = listArtifacts(absSrc, dir).filter((art) => path.resolve(art.path) !== current)
357
+ return { dir, current, stale }
358
+ }
359
+
360
+ /**
361
+ * Delete artifacts of one source the current key no longer points at.
362
+ *
363
+ * The blast radius is deliberately tiny: one directory, names matching this
364
+ * plugin's own `<stem>-<8hex>.md` pattern, regular files only, never recursive,
365
+ * and never the artifact the current key resolves to. Anything else in the
366
+ * folder — including a `.md` the user wrote by hand — is untouched.
367
+ */
368
+ function pruneArtifactsFor(cfg, absSrc, st, root, keepAbs) {
369
+ const keep = keepAbs ? path.resolve(keepAbs) : ''
370
+ const removed = []
371
+ const failed = []
372
+ for (const art of staleArtifactsFor(cfg, absSrc, st, root).stale) {
373
+ if (keep && path.resolve(art.path) === keep) continue
374
+ try {
375
+ fs.rmSync(art.path, { force: true })
376
+ removed.push(art)
377
+ } catch (error) {
378
+ failed.push({ path: art.path, error: String((error && error.message) || error) })
379
+ }
380
+ }
381
+ return { removed, failed }
382
+ }
383
+
384
+ /** `path` / `paths` normalised into the raw inputs the caller asked for. */
385
+ function requestedPaths(args) {
386
+ const out = []
387
+ const push = (value) => {
388
+ if (typeof value !== 'string') return
389
+ for (const part of value.split(/\r?\n/)) {
390
+ const trimmed = part.trim()
391
+ if (trimmed) out.push(trimmed)
392
+ }
393
+ }
394
+ push(args.path)
395
+ const list = Array.isArray(args.paths) ? args.paths : (Array.isArray(args.path) ? args.path : [])
396
+ for (const item of list) push(item)
397
+ return out
398
+ }
399
+
400
+ /**
401
+ * Turn one raw input into the concrete files to process.
402
+ *
403
+ * A directory expands to the Office/PDF files inside it (`recursive` walks
404
+ * deeper), because "convert every workbook in this folder" is one intent, not N
405
+ * round trips. The cap keeps a stray `path:"."` from turning into an unbounded
406
+ * walk over a whole repository.
407
+ */
408
+ function expandInput(rawPath, root, recursive) {
409
+ const abs = resolveInputPath(rawPath, root)
410
+ let st = null
411
+ try { st = fs.statSync(abs) } catch { st = null }
412
+ if (!st) {
413
+ return {
414
+ error: fail('文件不存在:' + abs, {
415
+ sourcePath: rawPath,
416
+ fallbackHint: '请确认路径是否正确(相对路径以工作区 ' + (root || process.cwd()) + ' 为基准)。'
417
+ })
418
+ }
419
+ }
420
+ if (st.isFile()) return { files: [{ abs, st }] }
421
+ if (!st.isDirectory()) return { error: fail('不是文件也不是目录:' + abs, { sourcePath: rawPath }) }
422
+
423
+ const files = []
424
+ /* Counted so the answer can say what the walk actually did: a directory that
425
+ * yields a single workbook must not read as "this folder holds one file",
426
+ * and a silently skipped .txt must not look like it was never there. */
427
+ let scanned = 0
428
+ let skipped = 0
429
+ let subdirs = 0
430
+ const walk = (dir, depth) => {
431
+ let entries = []
432
+ try { entries = fs.readdirSync(dir, { withFileTypes: true }) } catch { return }
433
+ for (const entry of entries) {
434
+ if (files.length >= MAX_EXPAND_FILES) return
435
+ const child = path.join(dir, entry.name)
436
+ if (entry.isDirectory()) {
437
+ if (recursive && depth < 8) walk(child, depth + 1)
438
+ else subdirs++ // left unvisited: not recursive, or too deep
439
+ continue
440
+ }
441
+ if (!entry.isFile()) continue
442
+ scanned++
443
+ if (classify(child).kind !== 'office') { skipped++; continue }
444
+ try { files.push({ abs: child, st: fs.statSync(child) }) } catch { /* ignore */ }
445
+ }
446
+ }
447
+ walk(abs, 0)
448
+ return {
449
+ files,
450
+ directory: abs,
451
+ capped: files.length >= MAX_EXPAND_FILES,
452
+ scanned,
453
+ skipped,
454
+ subdirs
455
+ }
456
+ }
457
+
458
+ function buildDefinition(ctx, cfg) {
459
+ return defineTool({
460
+ name: TOOL_NAME,
461
+ description:
462
+ '把工作区里的 Office / PDF 文件(.docx、.xlsx、.pptx、.pdf、.xls、.odt、.epub 等)先用 MarkItDown 转成 Markdown,' +
463
+ '把结果写成源文件旁边的 .md,只返回该 .md 的路径、体积、行数、token 估算和标题结构索引,**不把全文塞进上下文**;随后用 read 工具按需读取该 .md。' +
464
+ '产物不会被裁剪,读取方式建议:先读前 200 行看结构,再用 grep 在同一个 .md 里定位需要的片段。' +
465
+ '一次可以处理多个文件或整个目录(用 paths,或直接把目录路径交给 path),也可以用 action:"outline" 只看已有结果的结构索引(不触发转换)。' +
466
+ '纯文本 / Markdown / CSV 文件本来就能直接读取,本工具会直接告知,不需要转换。' +
467
+ '转换在本地完成,不改动原文件。当模型需要读取 .docx/.xlsx/.pptx/.pdf 等文件时必须先调用本工具。',
468
+ parameters: {
469
+ path: {
470
+ type: 'string',
471
+ description: '要处理的文件或目录路径(相对工作区或绝对路径)。传目录时按扩展名找出其中的 Office/PDF 文件。action 为 "status" 时可省略'
472
+ },
473
+ paths: {
474
+ type: 'array',
475
+ items: { type: 'string' },
476
+ description: '批量处理:多个文件或目录路径(等价于把数组交给 path)。一次调用处理多个文件,省掉逐文件来回'
477
+ },
478
+ action: {
479
+ type: 'string',
480
+ description:
481
+ "动作:'auto'(默认,自动判断类型并转换或提示直接读取)、'convert'(强制转换)、'read'(读取已转换结果的开头)、'outline'(只给已转换结果的结构索引,绝不触发转换)、'clean'(清理同一源文件的陈旧产物)、'status'(查看当前可用的转换器与 Python 环境)"
482
+ },
483
+ force: {
484
+ type: 'boolean',
485
+ description: 'true 时忽略已有转换结果,重新转换'
486
+ },
487
+ preview: {
488
+ type: 'number',
489
+ description: '可选:返回转换后 Markdown 开头的字符数(默认 0 = 只返回路径,最省 token;上限为配置的 maxPreviewChars)'
490
+ },
491
+ recursive: {
492
+ type: 'boolean',
493
+ description: 'path 是目录时是否递归子目录(默认 false);无论是否递归,一次最多展开 200 个文件'
494
+ },
495
+ dryRun: {
496
+ type: 'boolean',
497
+ description: "仅对 action:'clean' 有效:true(默认)只报告将删除哪些陈旧产物,传 false 才真的删除"
498
+ }
499
+ },
500
+ output: {
501
+ schema: { type: 'object', additionalProperties: true },
502
+ render(args, value) {
503
+ return [{ type: 'text', text: renderResult(value) }]
504
+ }
505
+ },
506
+ execute: async (args, exec) => { const v = await executeTool(ctx, cfg, args || {}, exec); return JSON.parse(JSON.stringify(v)) }
507
+ })
508
+ }
509
+
510
+ function renderResult(value) {
511
+ if (!value || typeof value !== 'object') return '(无结果)'
512
+ if (value.status === 'probe') {
513
+ const lines = ['🔍 转换器探测结果']
514
+ for (const item of value.converters || []) lines.push(' • ' + item)
515
+ for (const note of value.notices || []) lines.push(' ! ' + note)
516
+ const pythons = value.pythons || []
517
+ if (pythons.length) {
518
+ lines.push('')
519
+ lines.push('🐍 Python 环境(pythonPrefer:' + (value.pythonPrefer || 'auto') + ';按此顺序探测,第一个装了 markitdown 的胜出)')
520
+ const firstOk = pythons.find((p) => p.ok)
521
+ for (const p of pythons) {
522
+ const used = firstOk && p === firstOk
523
+ const mark = p.ok ? (used ? '✅ 使用中' : '✅ 可用 ') : '❌ 未安装'
524
+ const why = p.ok ? '' : ' ← ' + (p.reason || '无 markitdown')
525
+ lines.push(' ' + mark + ' ' + p.cmd + ' [' + pythonSourceLabel(p.source) + ']' + why)
526
+ }
527
+ lines.push(' 提示:想固定用某一个,把它的路径填进配置 pythonPath;想换探测顺序,改 pythonPrefer(auto / bundled / system / config)。')
528
+ }
529
+ const snap = value.snapshot || {}
530
+ lines.push('')
531
+ if (snap.present) {
532
+ lines.push('📦 环境记录:' + (snap.adopted ? '已接管 ' : '已安装 ') + (snap.python || '未知解释器') +
533
+ '(' + (snap.installedAt || '未知时间') + '):' + (snap.added || []).length + ' 个包由本插件负责'
534
+ + (snap.keep && snap.keep.length ? ',另有 ' + snap.keep.length + ' 个包被运行时共用、保留' : '') + '。')
535
+ lines.push(' 在 DSH 里卸载本插件时,这些包会被自动 pip uninstall。')
536
+ } else {
537
+ lines.push('📦 环境记录:无 —— markitdown 不是由本插件安装或登记的,卸载插件时不会卸载任何 Python 包。')
538
+ lines.push(' 想让它也一并清理,就在设置页点一次「一键配置」完成登记。')
539
+ }
540
+ lines.push(' 打开 DSH 设置里的「Office 转换」页面,可以图形化地检查环境 / 一键配置 MarkItDown。')
541
+ return lines.join('\n')
542
+ }
543
+ if (!value.ok) {
544
+ const lines = ['❌ 处理失败', value.error || '未知错误']
545
+ if (value.sourcePath) lines.push('源文件:' + value.sourcePath)
546
+ if (value.fallbackHint) lines.push('建议:' + value.fallbackHint)
547
+ return lines.join('\n')
548
+ }
549
+ if (value.status === 'clean') {
550
+ const lines = [(value.dryRun ? '🧹 陈旧产物检查(没有删除任何文件)' : '🧹 已清理陈旧产物') + ':' + value.sourcePath]
551
+ lines.push('• 目录:' + value.directory)
552
+ if (value.markdownPath) lines.push('• 保留(当前源文件对应的产物):' + value.markdownPath)
553
+ if (!value.staleCount) {
554
+ lines.push('• 没有需要清理的陈旧产物。')
555
+ } else {
556
+ lines.push('• 陈旧产物 ' + value.staleCount + ' 个,共 ' + humanBytes(value.staleBytes) + ':')
557
+ for (const item of value.stale) lines.push(' ' + item.path + '(' + humanBytes(item.size) + ')')
558
+ }
559
+ for (const item of value.failed || []) lines.push('• 删除失败:' + item.path + '(' + item.error + ')')
560
+ for (const note of value.notices || []) lines.push('• 提示:' + note)
561
+ lines.push('下一步:' + value.guidance)
562
+ return lines.join('\n')
563
+ }
564
+ if (value.status === 'outline') {
565
+ const lines = ['🗂 结构索引:' + value.sourcePath]
566
+ lines.push('• Markdown:' + value.markdownPath + '(' + humanBytes(value.markdownBytes) + ',约 ' +
567
+ (value.lineCount || 0).toLocaleString('en-US') + ' 行,约 ' + (value.estimatedTokens || 0).toLocaleString('en-US') + ' tokens)')
568
+ lines.push('• 转换器:' + (value.converterLabel || '未知') + '(保真度:' + fidelityWord(value.fidelity) + ')')
569
+ for (const note of value.notices || []) lines.push('• 提示:' + note)
570
+ if (value.outline && value.outline.length) pushOutline(lines, value)
571
+ else lines.push('• 没有可用的标题索引。')
572
+ lines.push('下一步:' + value.guidance)
573
+ return lines.join('\n')
574
+ }
575
+ if (value.status === 'read') {
576
+ const lines = ['📖 已有转换结果:' + value.sourcePath]
577
+ lines.push('• Markdown:' + value.markdownPath + '(' + humanBytes(value.markdownBytes) + ',约 ' +
578
+ (value.lineCount || 0).toLocaleString('en-US') + ' 行,约 ' + (value.estimatedTokens || 0).toLocaleString('en-US') + ' tokens)')
579
+ lines.push('• 转换器:' + (value.converterLabel || '未知') + '(保真度:' + fidelityWord(value.fidelity) + ')')
580
+ for (const note of value.notices || []) lines.push('• 提示:' + note)
581
+ pushOutline(lines, value)
582
+ lines.push('下一步:' + value.guidance)
583
+ pushPreview(lines, value)
584
+ return lines.join('\n')
585
+ }
586
+ if (value.status === 'batch') {
587
+ const lines = ['📚 批量处理 ' + value.total + ' 个文件(成功 ' + value.succeeded + ',失败 ' + value.failed + '):' + value.sourcePath]
588
+ for (const item of value.results || []) {
589
+ if (!item.ok) {
590
+ lines.push(' ✗ ' + item.sourcePath + ' —— ' + item.error)
591
+ } else if (item.direct) {
592
+ lines.push(' • ' + item.sourcePath + '(' + (item.note || '纯文本') + ',直接用 read 读取即可,无需转换)')
593
+ } else {
594
+ lines.push(' ✓ ' + item.sourcePath + ' → ' + item.markdownPath + '(' + humanBytes(item.markdownBytes) +
595
+ ',约 ' + (item.estimatedTokens || 0).toLocaleString('en-US') + ' tokens,' + (item.converterLabel || item.converter || '') + ')')
596
+ }
597
+ }
598
+ for (const note of value.notices || []) lines.push('• 提示:' + note)
599
+ lines.push('下一步:' + value.guidance)
600
+ return lines.join('\n')
601
+ }
602
+ if (value.status === 'direct-read') {
603
+ const lines = ['📄 ' + value.sourcePath + ' 是 ' + (value.sourceKind || '纯文本') + ',可以直接读取,无需转换。']
604
+ for (const note of value.notices || []) lines.push('• 提示:' + note)
605
+ lines.push(value.guidance)
606
+ return lines.filter(Boolean).join('\n')
607
+ }
608
+ const lines = []
609
+ lines.push((value.cached ? '♻️ 已有转换结果' : '✅ 已转换为 Markdown') + ':' + value.sourcePath)
610
+ lines.push('• 源文件:' + value.sourcePath + '(' + humanBytes(value.sourceBytes) + ',未修改)')
611
+ lines.push(
612
+ '• Markdown:' + value.markdownPath + '(' + humanBytes(value.markdownBytes) +
613
+ (value.lineCount ? ',约 ' + value.lineCount.toLocaleString('en-US') + ' 行' : '') +
614
+ ',约 ' + (value.estimatedTokens || 0).toLocaleString('en-US') + ' tokens)'
615
+ )
616
+ lines.push('• 转换器:' + (value.converterLabel || value.converter) + '(保真度:' + fidelityWord(value.fidelity) + ')')
617
+ if (value.converterNote) lines.push('• 说明:' + value.converterNote)
618
+ for (const note of value.notices || []) lines.push('• 提示:' + note)
619
+ pushOutline(lines, value)
620
+ lines.push('下一步:' + value.guidance)
621
+ pushPreview(lines, value)
622
+ return lines.join('\n')
623
+ }
624
+
625
+ function fail(message, extra = {}) {
626
+ return { ok: false, error: message, ...extra }
627
+ }
628
+
629
+ /**
630
+ * Convert / read / outline ONE source file.
631
+ *
632
+ * Split out of `executeTool` so a directory or a `paths` list can run the same
633
+ * logic per file. Files are handled one at a time on purpose: conversion is CPU-
634
+ * and disk-heavy, and a serial loop keeps a 50-file folder from spawning 50
635
+ * interpreters at once.
636
+ */
637
+ async function executeOne(ctx, cfg, args, action, absSrc, st, root, exec, logger) {
638
+ const signal = exec && exec.signal ? exec.signal : undefined
639
+ const cls = classify(absSrc)
640
+ const shownSrc = displayPath(absSrc, root)
641
+
642
+ // Plain text / Markdown / CSV: reading directly is cheaper than converting.
643
+ if (action === 'auto' && cls.kind === 'plain') {
644
+ const csvLike = cls.ext === '.csv' || cls.ext === '.tsv'
645
+ return {
646
+ ok: true,
647
+ status: 'direct-read',
648
+ sourcePath: shownSrc,
649
+ sourceAbsolutePath: absSrc,
650
+ sourceBytes: st.size,
651
+ sourceKind: cls.label,
652
+ guidance: csvLike
653
+ ? '这是分隔符文本,直接用 read 工具读取最省 token(转成 Markdown 表格通常更费)。若确实需要结构化 Markdown 表格,再调用本工具并传 action:"convert"。'
654
+ : '请直接用 read 工具读取 ' + shownSrc + ',不要转换。'
655
+ }
656
+ }
657
+
658
+ if (action === 'outline') {
659
+ const existing = findExistingArtifact(cfg, absSrc, st, root)
660
+ if (!existing) {
661
+ return fail('还没有转换结果,无法给出结构索引。请先转换(action:"convert" 或不传 action)。', { sourcePath: shownSrc })
662
+ }
663
+ return outlineValue(existing, shownSrc)
664
+ }
665
+
666
+ if (action === 'read') {
667
+ const existing = findExistingArtifact(cfg, absSrc, st, root)
668
+ if (!existing) {
669
+ return fail('还没有转换结果,请先执行转换(action:"convert" 或不传 action)。', { sourcePath: shownSrc })
670
+ }
671
+ return readExisting(cfg, existing, shownSrc, args)
672
+ }
673
+
674
+ // convert / auto on Office or unknown-binary files
675
+ ensureTmpDir(ctx, cfg, absSrc, root)
676
+ const target = conversionTarget(cfg, absSrc, st, root)
677
+ const hasTarget = cfg.reuseFresh !== false && fs.existsSync(target.absDst) && safeSize(target.absDst) > 0
678
+ const startedAt = Date.now()
679
+ const notices = []
680
+ let activeTarget = target
681
+ let result = null
682
+
683
+ if (hasTarget && !args.force) {
684
+ const stamp = readFidelityMarker(target.absDst)
685
+ if (!stamp) {
686
+ /* Written by 1.1.x, before artifacts recorded how they were made. Reusing
687
+ * it is still right; claiming "high" fidelity for it is not. */
688
+ result = {
689
+ ok: true,
690
+ converter: 'cache',
691
+ converterLabel: '已有转换结果(旧版本写入,未记录转换器)',
692
+ fidelity: 'unknown',
693
+ converterNote: null,
694
+ probeNotes: []
695
+ }
696
+ } else if (stamp.fidelity === 'limited') {
697
+ /* A fallback converter produced this. MarkItDown may have been installed
698
+ * since — exactly the case where reusing the cache silently downgrades
699
+ * the answer — so consult the TTL-cached probe and re-convert only when a
700
+ * high-fidelity converter has actually appeared. */
701
+ let best = null
702
+ try {
703
+ best = ((await probeConverters(cfg, { signal })).chain || [])[0] || null
704
+ } catch { /* probing must never block a reuse */ }
705
+ if (best && best.fidelity === 'high') {
706
+ notices.push('检测到可用的 MarkItDown(' + best.label + '),已重新转换以提升保真度。')
707
+ } else {
708
+ result = {
709
+ ok: true,
710
+ converter: 'cache',
711
+ converterLabel: converterLabelFor(stamp.converter, '已有转换结果'),
712
+ fidelity: stamp.fidelity,
713
+ converterNote: null,
714
+ probeNotes: []
715
+ }
716
+ }
717
+ } else {
718
+ result = {
719
+ ok: true,
720
+ converter: 'cache',
721
+ converterLabel: converterLabelFor(stamp.converter, '已有转换结果'),
722
+ fidelity: stamp.fidelity,
723
+ converterNote: null,
724
+ probeNotes: []
725
+ }
726
+ }
727
+ }
728
+
729
+ if (!result && !hasTarget && !args.force) {
730
+ /* The mtime moved, but the bytes may not have (git checkout, a copy, a
731
+ * restore). Reuse the existing artifact and re-key it rather than writing a
732
+ * second identical .md next to the first. */
733
+ const reused = reuseByContent(cfg, absSrc, st, root)
734
+ if (reused) {
735
+ activeTarget = { absDst: reused.absDst, shownDst: reused.shownDst }
736
+ result = {
737
+ ok: true,
738
+ converter: 'cache',
739
+ converterLabel: converterLabelFor(reused.converter, '已有转换结果'),
740
+ fidelity: reused.fidelity || 'high',
741
+ converterNote: null,
742
+ probeNotes: []
743
+ }
744
+ notices.push('源文件的修改时间变了,但内容没有变(已比对内容哈希),直接复用已有 .md,没有生成重复文件。')
745
+ }
746
+ }
747
+
748
+ if (!result) {
749
+ result = await convertFile({ srcPath: absSrc, dstPath: target.absDst, cfg, signal, force: !!args.force })
750
+ if (result.ok && cfg.pruneStaleArtifacts === true) {
751
+ const pruned = pruneArtifactsFor(cfg, absSrc, st, root, target.absDst)
752
+ if (pruned.removed.length) {
753
+ notices.push('已按 pruneStaleArtifacts 清理同一源文件的 ' + pruned.removed.length + ' 个陈旧产物。')
754
+ }
755
+ }
756
+ }
757
+
758
+ if (!result.ok) {
759
+ safeLog(logger, 'warn', '转换失败 ' + shownSrc + ':' + String(result.error || '未知错误').split('\n')[0].slice(0, 200))
760
+ return fail(result.error || '转换失败', {
761
+ sourcePath: shownSrc,
762
+ sourceAbsolutePath: absSrc,
763
+ converterAttempts: result.attempts || [],
764
+ notices: result.probeNotes || [],
765
+ fallbackHint:
766
+ '可回退为直接用 read 工具读取 ' + shownSrc + '(可能得到乱码),或请用户手动处理:' +
767
+ '安装 uv 后重试(uvx markitdown 无需永久安装),或执行 pip install "markitdown[all]"。' +
768
+ '也可以传 action:"status" 查看当前可用的转换器。'
769
+ })
770
+ }
771
+
772
+ /* The artifact is never pulled into one string any more: `preview:0` used to
773
+ * read the whole Markdown into memory just to report its size and token
774
+ * count, on the default path, for a value it then declined to return. */
775
+ const previewLimit = Math.max(0, Math.min(Number(args.preview) || 0, cfg.maxPreviewChars || 4000))
776
+ const info = analyzeMarkdown(activeTarget.absDst, { headChars: previewLimit + 1 })
777
+ const previewTruncated = previewLimit > 0 && info.head.length > previewLimit
778
+ const preview = previewLimit > 0 ? info.head.slice(0, previewLimit) : ''
779
+ const outline = buildOutline(activeTarget.absDst)
780
+
781
+ if (result.converter === 'cache' && notices.length === 0) {
782
+ notices.push('命中缓存:源文件未变化,直接复用已有 .md(如需强制重转请传 force:true)。')
783
+ }
784
+ if (cls.kind === 'unknown') notices.push('无法从扩展名判断类型,已按二进制处理。')
785
+ if (result.fidelity === 'limited') notices.push('未使用 MarkItDown,保真度有限(见上方说明)。')
786
+ if (result.fidelity === 'unknown') notices.push('这个 .md 由旧版本插件生成,没有记录转换器与保真度;需要确认时传 force:true 重新转换。')
787
+
788
+ safeLog(
789
+ logger,
790
+ 'info',
791
+ (result.converter === 'cache' ? '复用已有转换结果 ' : '转换成功 ') + shownSrc + ' → ' + activeTarget.shownDst +
792
+ '(' + (result.converter === 'cache'
793
+ ? '未重新转换'
794
+ : String(result.converter) + ',' + ((Date.now() - startedAt) / 1000).toFixed(1) + 's') +
795
+ ',' + humanBytes(info.bytes) + ',约 ' + info.lines.toLocaleString('en-US') + ' 行)'
796
+ )
797
+
798
+ return {
799
+ ok: true,
800
+ status: result.converter === 'cache' ? 'cached' : 'converted',
801
+ cached: result.converter === 'cache',
802
+ sourcePath: shownSrc,
803
+ sourceAbsolutePath: absSrc,
804
+ sourceBytes: st.size,
805
+ sourceKind: cls.label,
806
+ markdownPath: activeTarget.shownDst,
807
+ markdownAbsolutePath: activeTarget.absDst,
808
+ markdownBytes: info.bytes,
809
+ lineCount: info.lines,
810
+ estimatedTokens: info.tokens,
811
+ converter: result.converter,
812
+ converterLabel: result.converterLabel,
813
+ fidelity: result.fidelity,
814
+ converterNote: result.converterNote,
815
+ outline: outline.entries,
816
+ outlineTruncated: outline.truncated,
817
+ preview: preview || undefined,
818
+ previewTruncated,
819
+ notices,
820
+ guidance: readingStrategy(activeTarget.shownDst, info.bytes, info.tokens, info.lines)
821
+ }
822
+ }
823
+
824
+ async function executeTool(ctx, cfg, args, exec) {
825
+ const logger = loggerFor(ctx)
826
+ const signal = exec && exec.signal ? exec.signal : undefined
827
+ const root = workspaceRootOf(ctx, exec)
828
+ const action = String(args.action || 'auto').toLowerCase()
829
+
830
+ if (action === 'status') {
831
+ const probe = await probeConverters(cfg, { signal, force: true })
832
+ return {
833
+ ok: true,
834
+ status: 'probe',
835
+ converters: probe.chain.map((c) => c.label + ' [' + c.id + ']'),
836
+ pythons: probe.pythonProbes || [],
837
+ pythonPrefer: probe.pythonPrefer || String(cfg.pythonPrefer || 'auto'),
838
+ notices: probe.notes || [],
839
+ snapshot: snapshotSummary()
840
+ }
841
+ }
842
+
843
+ const raws = requestedPaths(args)
844
+ if (!raws.length) {
845
+ return fail('缺少 path 参数。用法:read_office_as_markdown({ path: "报表.xlsx" });批量时用 paths: ["a.xlsx", "b.docx"],或直接传一个目录。')
846
+ }
847
+
848
+ if (action === 'clean') {
849
+ if (raws.length > 1) return fail('action:"clean" 一次只处理一个源文件,请分开调用。', { sourcePath: raws[0] })
850
+ const raw = raws[0]
851
+ const abs = resolveInputPath(raw, root)
852
+ let st = null
853
+ try { st = fs.statSync(abs) } catch { st = null }
854
+ if (!st) {
855
+ return fail('文件不存在:' + abs, {
856
+ sourcePath: raw,
857
+ fallbackHint: '请确认路径是否正确(相对路径以工作区 ' + (root || process.cwd()) + ' 为基准)。'
858
+ })
859
+ }
860
+ if (!st.isFile()) return fail('不是文件(可能是目录):' + abs, { sourcePath: raw })
861
+
862
+ const shown = displayPath(abs, root)
863
+ const dryRun = args.dryRun !== false
864
+ const { dir, current, stale } = staleArtifactsFor(cfg, abs, st, root)
865
+ const kept = fs.existsSync(current) && safeSize(current) > 0
866
+ const outcome = dryRun ? { removed: [], failed: [] } : pruneArtifactsFor(cfg, abs, st, root, '')
867
+ const staleBytes = stale.reduce((total, art) => total + art.size, 0)
868
+ if (!dryRun && outcome.removed.length) {
869
+ safeLog(logger, 'info', '清理陈旧产物 ' + shown + ':删除 ' + outcome.removed.length + ' 个,共 ' + humanBytes(staleBytes) + '。')
870
+ }
871
+ return {
872
+ ok: true,
873
+ status: 'clean',
874
+ dryRun,
875
+ sourcePath: shown,
876
+ sourceAbsolutePath: abs,
877
+ directory: displayPath(dir, root),
878
+ markdownPath: kept ? displayPath(current, root) : '',
879
+ staleCount: stale.length,
880
+ staleBytes,
881
+ stale: stale.map((art) => ({ path: displayPath(art.path, root), size: art.size })),
882
+ removed: outcome.removed.map((art) => displayPath(art.path, root)),
883
+ failed: outcome.failed.map((item) => ({ path: displayPath(item.path, root), error: item.error })),
884
+ notices: kept ? [] : ['当前源文件对应的产物还不存在,下面列出的都是源文件变化前留下的旧版本。'],
885
+ guidance: stale.length
886
+ ? (dryRun ? '确认无误后,再调用一次并传 dryRun:false 才会真正删除。' : '已删除上面列出的文件。')
887
+ : '没有需要清理的产物。'
888
+ }
889
+ }
890
+
891
+ const recursive = args.recursive === true
892
+ const files = []
893
+ let directory = ''
894
+ let capped = false
895
+ const dirStats = { dirs: 0, picked: 0, scanned: 0, skipped: 0, subdirs: 0 }
896
+ for (const raw of raws) {
897
+ const expanded = expandInput(raw, root, recursive)
898
+ if (expanded.error) {
899
+ if (raws.length === 1) return expanded.error
900
+ files.push({ error: expanded.error, raw })
901
+ continue
902
+ }
903
+ if (expanded.directory) {
904
+ directory = expanded.directory
905
+ dirStats.dirs++
906
+ dirStats.picked += expanded.files.length
907
+ dirStats.scanned += expanded.scanned || 0
908
+ dirStats.skipped += expanded.skipped || 0
909
+ dirStats.subdirs += expanded.subdirs || 0
910
+ }
911
+ if (expanded.capped) capped = true
912
+ for (const item of expanded.files) files.push(item)
913
+ }
914
+
915
+ /* What the directory walk decided, stated out loud. Without this the caller
916
+ * only sees "1 file" and cannot tell a folder with one workbook from a folder
917
+ * with one workbook and nine spreadsheets it declined to touch. */
918
+ const dirNotices = []
919
+ if (dirStats.dirs) {
920
+ const parts = ['目录展开:' + (recursive ? '已递归子目录' : '默认不递归子目录')]
921
+ if (!recursive && dirStats.subdirs) {
922
+ parts.push('跳过 ' + dirStats.subdirs + ' 个子目录(需要就传 recursive:true)')
923
+ }
924
+ parts.push('扫到 ' + dirStats.scanned + ' 个文件,挑出 ' + dirStats.picked + ' 个 Office / PDF')
925
+ if (dirStats.skipped) parts.push('另有 ' + dirStats.skipped + ' 个非 Office / PDF 文件已跳过')
926
+ parts.push('上限 ' + MAX_EXPAND_FILES + ' 个')
927
+ dirNotices.push(parts.join(';') + '。')
928
+ }
929
+
930
+ if (!files.length) {
931
+ if (directory) {
932
+ return fail(
933
+ '这个目录里没有可转换的 Office / PDF 文件:' + displayPath(directory, root) +
934
+ (recursive ? '' : '(默认不递归子目录,需要的话传 recursive:true)'),
935
+ { sourcePath: displayPath(directory, root) }
936
+ )
937
+ }
938
+ return fail('没有可处理的文件。', { sourcePath: raws.join(', ') })
939
+ }
940
+
941
+ /* One file keeps the full, detailed answer. A batch gets one compact line per
942
+ * file, because twelve weekly reports should not cost twelve turns. */
943
+ if (files.length === 1) {
944
+ const value = await executeOne(ctx, cfg, args, action, files[0].abs, files[0].st, root, exec, logger)
945
+ if (!dirNotices.length || !value || typeof value !== 'object') return value
946
+ return { ...value, notices: [...(value.notices || []), ...dirNotices] }
947
+ }
948
+
949
+ const results = []
950
+ for (const item of files) {
951
+ if (item.error) {
952
+ results.push({ ok: false, sourcePath: item.raw, error: item.error.error })
953
+ continue
954
+ }
955
+ const value = await executeOne(ctx, cfg, args, action, item.abs, item.st, root, exec, logger)
956
+ if (!value.ok) {
957
+ results.push({ ok: false, sourcePath: value.sourcePath || item.abs, error: value.error })
958
+ continue
959
+ }
960
+ if (value.status === 'direct-read') {
961
+ results.push({ ok: true, direct: true, sourcePath: value.sourcePath, note: value.sourceKind || '纯文本' })
962
+ continue
963
+ }
964
+ results.push({
965
+ ok: true,
966
+ sourcePath: value.sourcePath,
967
+ markdownPath: value.markdownPath,
968
+ markdownBytes: value.markdownBytes,
969
+ estimatedTokens: value.estimatedTokens,
970
+ converter: value.converter,
971
+ converterLabel: value.converterLabel,
972
+ fidelity: value.fidelity,
973
+ stale: value.stale === true
974
+ })
975
+ }
976
+
977
+ const succeeded = results.filter((item) => item.ok).length
978
+ const notices = []
979
+ if (capped) {
980
+ notices.push('目录展开达到上限(' + MAX_EXPAND_FILES + ' 个文件),只处理了前 ' + MAX_EXPAND_FILES + ' 个;其余请再调用一次或缩小范围。')
981
+ }
982
+ notices.push(...dirNotices)
983
+ return {
984
+ ok: true,
985
+ status: 'batch',
986
+ sourcePath: directory ? displayPath(directory, root) : raws.join(', '),
987
+ total: results.length,
988
+ succeeded,
989
+ failed: results.length - succeeded,
990
+ results,
991
+ notices,
992
+ guidance: '每个文件都已在源文件旁边生成 .md,按需用 read / grep 逐个读取,不要一次性全部读入。'
993
+ }
994
+ }
995
+
996
+ function safeSize(p) {
997
+ try { return fs.statSync(p).size } catch { return 0 }
998
+ }
999
+
1000
+ function ensureTmpDir(ctx, cfg, absSrc, root) {
1001
+ const dir = tmpDirFor(cfg, absSrc, root)
1002
+ ensureDirSync(dir)
1003
+ return dir
1004
+ }
1005
+
1006
+ function tmpDirFor(cfg, absSrc, root) {
1007
+ const d = cfg.tmpDir === undefined || cfg.tmpDir === null ? '' : String(cfg.tmpDir)
1008
+ // 默认(tmpDir 为空):与源文件同目录,转换产物就躺在原文件旁边。
1009
+ if (!d) return path.dirname(absSrc)
1010
+ const base = root || path.dirname(absSrc)
1011
+ return path.isAbsolute(d) ? d : path.join(base, d)
1012
+ }
1013
+
1014
+ function conversionTarget(cfg, absSrc, st, root) {
1015
+ const dir = tmpDirFor(cfg, absSrc, root)
1016
+ const hash = sha8(absSrc + '|' + st.size + '|' + st.mtimeMs + '|' + configSignature(cfg))
1017
+ const absDst = path.join(dir, safeBaseName(absSrc) + '-' + hash + '.md')
1018
+ return { absDst, shownDst: displayPath(absDst, root) }
1019
+ }
1020
+
1021
+ /** Outline-only answer: never converts, so it is safe on any turn. */
1022
+ function outlineValue(existing, shownSrc) {
1023
+ const info = analyzeMarkdown(existing.absDst, { headChars: 0 })
1024
+ const outline = buildOutline(existing.absDst)
1025
+ const stamp = readFidelityMarker(existing.absDst)
1026
+ return {
1027
+ ok: true,
1028
+ status: 'outline',
1029
+ sourcePath: shownSrc,
1030
+ markdownPath: existing.shownDst,
1031
+ markdownAbsolutePath: existing.absDst,
1032
+ markdownBytes: info.bytes,
1033
+ lineCount: info.lines,
1034
+ estimatedTokens: info.tokens,
1035
+ converter: stamp ? stamp.converter : '',
1036
+ converterLabel: stamp ? converterLabelFor(stamp.converter, '已有转换结果') : '已有转换结果',
1037
+ fidelity: stamp && stamp.fidelity ? stamp.fidelity : 'unknown',
1038
+ stale: !!existing.stale,
1039
+ outline: outline.entries,
1040
+ outlineTruncated: outline.truncated,
1041
+ notices: existing.stale ? ['源文件已变化,这是较早的产物;要基于最新源文件重转,请传 force:true。'] : [],
1042
+ guidance: outline.entries.length
1043
+ ? '按上面的标题挑需要的章节,用 grep 在该 .md 里定位后定点 read,不要整份读入。'
1044
+ : '这个产物没有 Markdown 标题可作索引;请用 grep 在该 .md 里搜索关键词,再定点 read。'
1045
+ }
1046
+ }
1047
+
1048
+ /**
1049
+ * Read the head of an existing artifact.
1050
+ *
1051
+ * The preview cap is `cfg.maxPreviewChars` — the same knob the conversion path
1052
+ * honours — instead of a second, hard-coded 4000 that quietly ignored config.
1053
+ */
1054
+ function readExisting(cfg, target, shownSrc, args) {
1055
+ const limit = Math.max(0, Math.min(Number(args.preview) || 2000, cfg.maxPreviewChars || 4000))
1056
+ const info = analyzeMarkdown(target.absDst, { headChars: limit + 1 })
1057
+ const truncated = info.head.length > limit
1058
+ const head = truncated ? info.head.slice(0, limit) : info.head
1059
+ const stamp = readFidelityMarker(target.absDst)
1060
+ const outline = buildOutline(target.absDst)
1061
+ const notices = []
1062
+ if (target.stale) notices.push('源文件已变化,这是较早的产物;要基于最新源文件重转,请传 force:true。')
1063
+ if (!stamp) notices.push('这个 .md 由旧版本插件生成,没有记录转换器与保真度。')
1064
+ return {
1065
+ ok: true,
1066
+ status: 'read',
1067
+ cached: true,
1068
+ sourcePath: shownSrc,
1069
+ markdownPath: target.shownDst,
1070
+ markdownAbsolutePath: target.absDst,
1071
+ markdownBytes: info.bytes,
1072
+ lineCount: info.lines,
1073
+ estimatedTokens: info.tokens,
1074
+ converter: stamp ? stamp.converter : '',
1075
+ converterLabel: stamp ? converterLabelFor(stamp.converter, '已有转换结果') : '已有转换结果',
1076
+ fidelity: stamp && stamp.fidelity ? stamp.fidelity : 'unknown',
1077
+ stale: !!target.stale,
1078
+ outline: outline.entries,
1079
+ outlineTruncated: outline.truncated,
1080
+ notices,
1081
+ preview: head || undefined,
1082
+ previewTruncated: truncated,
1083
+ guidance: '如需完整内容,请用 read 工具分段读取 ' + target.shownDst + '(先读前 200 行看结构,再 grep 定位)。'
1084
+ }
1085
+ }
1086
+
1087
+ /* ------------------------------------------------------------------ */
1088
+ /* uninstalling the plugin also uninstalls the environment it recorded */
1089
+ /* ------------------------------------------------------------------ */
1090
+
1091
+ /** Logging must never break a disposal path. */
1092
+ function safeLog(logger, level, message) {
1093
+ try {
1094
+ if (logger && typeof logger[level] === 'function') logger[level]('[' + PROVIDER + '] ' + message)
1095
+ } catch { /* ignore */ }
1096
+ }
1097
+
1098
+ /**
1099
+ * True only when this package has really been **uninstalled** from the profile.
1100
+ *
1101
+ * Being disposed is not enough: Cordis disposes a plugin when the app shuts
1102
+ * down, when the user merely disables it, and when the profile is reloaded.
1103
+ * A real removal is the only case where BOTH the package directory is gone AND
1104
+ * no profile config still references us — so both are required before anything
1105
+ * is uninstalled. A false negative only leaves packages behind; a false
1106
+ * positive would silently break the user's Python environment, so the check
1107
+ * errs toward doing nothing.
1108
+ */
1109
+ function removalConfirmed() {
1110
+ try {
1111
+ if (fs.existsSync(path.join(PACKAGE_ROOT, 'lib', 'index.js'))) return false
1112
+ } catch {
1113
+ return false
1114
+ }
1115
+ const root = profilesRoot()
1116
+ let profiles = []
1117
+ try {
1118
+ profiles = fs.readdirSync(root)
1119
+ } catch {
1120
+ return false
1121
+ }
1122
+ for (const profile of profiles) {
1123
+ for (const file of ['cordis.patch.yml', 'cordis.patch.yaml', 'cordis.yml', 'package.json']) {
1124
+ const candidate = path.join(root, profile, file)
1125
+ try {
1126
+ if (!fs.existsSync(candidate)) continue
1127
+ if (fs.readFileSync(candidate, 'utf8').includes('office-markdown')) return false
1128
+ } catch {
1129
+ return false
1130
+ }
1131
+ }
1132
+ }
1133
+ return true
1134
+ }
1135
+
1136
+ async function runEnvRemoval(cfg, logger) {
1137
+ try {
1138
+ if (cfg && cfg.removeEnvOnUninstall === false) return
1139
+ const snap = readSnapshot()
1140
+ if (!snap) return
1141
+ const count = Array.isArray(snap.added) ? snap.added.length : 0
1142
+ safeLog(logger, 'info', '插件已从 profile 移除,正在一并卸载登记过的 ' + count + ' 个 Python 包…')
1143
+ const result = await uninstallMarkitdown({})
1144
+ if (result && result.error) safeLog(logger, 'warn', '环境清理失败:' + result.error)
1145
+ } catch (error) {
1146
+ safeLog(logger, 'warn', '环境清理异常:' + String((error && error.message) || error))
1147
+ }
1148
+ }
1149
+
1150
+ /**
1151
+ * 卸载后清环境:这里**不能**用宿主内的定时器轮询。
1152
+ *
1153
+ * `dsh-plugin-manager` 的 `removeBundle()` 是 `selectBundle(false)` →
1154
+ * `reload()` → `pnpm remove`(见 @deepseek-ai/dsh-plugin-manager/lib/index.js
1155
+ * :1844-1863):本插件的 fiber 先被销毁,几秒后包目录才被删掉。宿主内的定时
1156
+ * 器一旦遇到用户顺手重启一次 harness 就全部消失,环境便永远清不掉了。
1157
+ *
1158
+ * 因此这里派一个脱离宿主的看门狗进程(`lib/removal-watchdog.js`)去做:它自己
1159
+ * 轮询,确认插件真的被移除后才卸载登记的 Python 包。
1160
+ *
1161
+ * 但派发本身也要克制:这个函数在**每一次 dispose** 时都会被调用,而关闭插件、
1162
+ * 重载 profile、退出 DSH 同样会 dispose。所以下面先用 `stillSelectedInProfile()`
1163
+ * 判断「profile 是不是已经不要本插件了」——只有真正的卸载会命中,平时连一个
1164
+ * 后台进程都不会起。
1165
+ */
1166
+ /**
1167
+ * True while some profile still lists this plugin among the bundles it wants.
1168
+ *
1169
+ * `dsh-plugin-manager.removeBundle()` runs `selectBundle(name, false)` *before*
1170
+ * `reload()` (index.js:1844-1863), so at dispose time a real removal has
1171
+ * already dropped us from `dsh.profile.bundles` while an ordinary shutdown, a
1172
+ * profile reload or a plain disable still has us there. That makes this a cheap
1173
+ * "is somebody actually removing me?" test — and it is what keeps the watchdog
1174
+ * from ever being spawned during normal use.
1175
+ */
1176
+ function stillSelectedInProfile() {
1177
+ const candidates = []
1178
+ const current = currentProfileDir()
1179
+ if (current !== '') {
1180
+ candidates.push(path.join(current, 'package.json'))
1181
+ } else {
1182
+ let names = []
1183
+ try {
1184
+ names = fs.readdirSync(profilesRoot())
1185
+ } catch {
1186
+ return true // 读不到 profiles 就当作「仍在启用」,宁可什么都不做
1187
+ }
1188
+ for (const profile of names) candidates.push(path.join(profilesRoot(), profile, 'package.json'))
1189
+ }
1190
+ for (const candidate of candidates) {
1191
+ try {
1192
+ if (!fs.existsSync(candidate)) continue
1193
+ const manifest = JSON.parse(fs.readFileSync(candidate, 'utf8'))
1194
+ const profileSection = manifest && manifest.dsh ? manifest.dsh.profile : null
1195
+ const bundles = profileSection ? profileSection.bundles : null
1196
+ if (Array.isArray(bundles) && bundles.includes(PROVIDER)) return true
1197
+ } catch { /* 单个 profile 解析失败不影响其它 profile */ }
1198
+ }
1199
+ return false
1200
+ }
1201
+
1202
+ function scheduleRemovalCleanup(cfg, logger) {
1203
+ if (cfg && cfg.removeEnvOnUninstall === false) return
1204
+ const snap = readSnapshot()
1205
+ if (!snap) return
1206
+ const packages = Array.isArray(snap.added)
1207
+ ? snap.added.filter((item) => typeof item === 'string' && item !== '')
1208
+ : []
1209
+ if (packages.length === 0) return
1210
+
1211
+ let gone = false
1212
+ try {
1213
+ gone = removalConfirmed()
1214
+ } catch {
1215
+ gone = false
1216
+ }
1217
+ if (gone) {
1218
+ void runEnvRemoval(cfg, logger)
1219
+ return
1220
+ }
1221
+
1222
+ // Normal use must stay free of background processes: shutting DSH down,
1223
+ // reloading the profile or merely disabling the plugin all dispose us while
1224
+ // the profile still selects the bundle. Only a real removal has already
1225
+ // dropped us from `dsh.profile.bundles` by this point.
1226
+ let stillSelected = true
1227
+ try {
1228
+ stillSelected = stillSelectedInProfile()
1229
+ } catch {
1230
+ stillSelected = true
1231
+ }
1232
+ if (stillSelected) return
1233
+
1234
+ if (!snap.python) {
1235
+ safeLog(
1236
+ logger,
1237
+ 'warn',
1238
+ '环境记录里没有解释器路径,插件被卸载后不会自动清理;可在 profile 目录里跑 install.ps1 -Uninstall。'
1239
+ )
1240
+ return
1241
+ }
1242
+
1243
+ try {
1244
+ const info = spawnRemovalWatchdog({
1245
+ packageRoot: PACKAGE_ROOT,
1246
+ profilesRoot: profilesRoot(),
1247
+ snapshotPath: snapshotPath(),
1248
+ lockPath: watchdogLockPath(),
1249
+ python: snap.python,
1250
+ packages,
1251
+ logPath: removalLogPath(),
1252
+ timeoutSec: 120
1253
+ })
1254
+ safeLog(
1255
+ logger,
1256
+ 'info',
1257
+ '检测到本插件正被移除,已派发卸载看门狗(pid ' +
1258
+ info.pid +
1259
+ '):确认插件真的被移除后,它会卸载登记的 ' +
1260
+ packages.length +
1261
+ ' 个 Python 包。'
1262
+ )
1263
+ } catch (error) {
1264
+ safeLog(logger, 'warn', '无法派发卸载看门狗:' + String((error && error.message) || error))
1265
+ }
1266
+ }
1267
+
1268
+ /**
1269
+ * The DSH-owned runtime Python is the environment this plugin may clean up
1270
+ * later, so record it once if markitdown is already there and nothing has been
1271
+ * recorded yet. An interpreter the user installed themselves is deliberately
1272
+ * left alone — the plugin only ever removes what it can prove it brought in.
1273
+ */
1274
+ function scheduleAutoAdoptEnv(cfg) {
1275
+ if (cfg.autoAdoptEnv === false) return
1276
+ const timer = setTimeout(() => { void autoAdoptEnv(cfg) }, 5000)
1277
+ if (timer && typeof timer.unref === 'function') timer.unref()
1278
+ }
1279
+
1280
+ async function autoAdoptEnv(cfg) {
1281
+ try {
1282
+ if (readSnapshot()) return
1283
+ const probe = await probeConverters(cfg, { force: false })
1284
+ const entry = (probe.chain || []).find((c) => c && c.id === 'python-module' && c.python)
1285
+ if (!entry) return
1286
+ if (entry.python.source !== 'bundled') return
1287
+ await adoptMarkitdown(entry.python, {})
1288
+ } catch { /* best effort only */ }
1289
+ }
1290
+
1291
+ export function apply(ctx, config) {
1292
+ const cfg = { ...DEFAULTS, ...(config || {}) }
1293
+
1294
+ // Learn where this host actually keeps its data before anything looks up a
1295
+ // path. `profileContext.dir` is the profile directory DSH itself reports, and
1296
+ // `<dsh home>/profiles/<name>` gives the DSH home back; `lib/paths.js` falls
1297
+ // back to `DSH_HOME` and then `~/.dsh` when the host says nothing.
1298
+ try {
1299
+ const profileContext = ctx.get('profileContext')
1300
+ const dir = profileContext && typeof profileContext.dir === 'string' ? profileContext.dir : ''
1301
+ setProfileDir(dir)
1302
+ } catch { /* keep the environment-variable and homedir fallbacks */ }
1303
+
1304
+ if (cfg.enabled === false) {
1305
+ const logger = loggerFor(ctx)
1306
+ if (logger) logger.info('[' + PROVIDER + '] enabled=false:未注册工具、技能或守卫。')
1307
+ return
1308
+ }
1309
+
1310
+ const tools = ctx.get('tools')
1311
+ if (!tools || typeof tools.register !== 'function') {
1312
+ const logger = loggerFor(ctx)
1313
+ if (logger) logger.warn('[' + PROVIDER + '] 未找到 tools 服务,插件未注册。')
1314
+ return
1315
+ }
1316
+
1317
+ ctx.effect(() => tools.register(buildDefinition(ctx, cfg)), PROVIDER + ': tool ' + TOOL_NAME)
1318
+
1319
+ if (cfg.registerSkill !== false) {
1320
+ const skills = ctx.get('skills')
1321
+ if (skills && typeof skills.register === 'function') {
1322
+ ctx.effect(
1323
+ () =>
1324
+ skills.register({
1325
+ name: SKILL_NAME,
1326
+ description:
1327
+ '遇到 .docx、.xlsx、.pptx、.pdf 等 Office / PDF 文件时,必须先调用 read_office_as_markdown 转成 Markdown 再读取:结果写在源文件旁边(工作区里),只把 .md 路径、体积与标题结构索引返回,避免全文注入上下文;产物完整保留、绝不裁剪,一次调用还能批量处理多个文件或整个目录。',
1328
+ whenToUse:
1329
+ '当需要读取工作区里的 Office(Word/Excel/PowerPoint)或 PDF 文件,或用户要求总结/分析这类文件内容时。',
1330
+ source: 'runtime',
1331
+ provider: PROVIDER,
1332
+ content: SKILL_CONTENT,
1333
+ invocation: { modelInvocable: true, userInvocable: true }
1334
+ }),
1335
+ PROVIDER + ': skill ' + SKILL_NAME
1336
+ )
1337
+ }
1338
+ }
1339
+
1340
+ if (cfg.guardReadTool !== false && typeof tools.guard === 'function') {
1341
+ ctx.effect(
1342
+ () => tools.guard((execution) => guardRead(ctx, cfg, execution)),
1343
+ PROVIDER + ': read guard'
1344
+ )
1345
+ }
1346
+
1347
+ // Settings page backend. `webServer` is deliberately NOT in `inject`: a
1348
+ // profile without a web server (a plain CLI session) must keep the tool and
1349
+ // skill working. `ctx.inject` starts a nested fiber that waits for the
1350
+ // service instead, and that fiber is owned by ours, so it is disposed with
1351
+ // the plugin.
1352
+ if (cfg.registerSettings !== false) {
1353
+ try {
1354
+ ctx.inject(['webServer'], (child) => {
1355
+ child.effect(
1356
+ () =>
1357
+ registerSettingsApi(child, cfg, {
1358
+ name: PROVIDER,
1359
+ version: VERSION,
1360
+ provider: PROVIDER,
1361
+ toolName: TOOL_NAME
1362
+ }),
1363
+ PROVIDER + ': settings API'
1364
+ )
1365
+ })
1366
+ } catch (error) {
1367
+ const logger = loggerFor(ctx)
1368
+ if (logger) logger.warn('[' + PROVIDER + '] 设置页面接口注册失败:' + String((error && error.message) || error))
1369
+ }
1370
+ }
1371
+
1372
+ // Conversion itself leaves NOTHING behind: a single `.md` next to the source
1373
+ // file, owned by the user — no temp directory, no registry, no sidecar
1374
+ // metadata. The one thing that *can* survive is the environment the plugin
1375
+ // recorded when it was configured, so that snapshot is released when — and
1376
+ // only when — the plugin is really uninstalled from the profile.
1377
+ scheduleAutoAdoptEnv(cfg)
1378
+
1379
+ try {
1380
+ const logger = loggerFor(ctx)
1381
+ ctx.effect(
1382
+ () => () => scheduleRemovalCleanup(cfg, logger),
1383
+ PROVIDER + ': uninstall cleanup'
1384
+ )
1385
+ } catch (error) {
1386
+ safeLog(loggerFor(ctx), 'warn', '卸载清理钩子注册失败:' + String((error && error.message) || error))
1387
+ }
1388
+ }
1389
+
1390
+ /**
1391
+ * Best-effort artifact lookup for the read guard.
1392
+ *
1393
+ * The guard runs outside any tool call, so there is no session workspace to
1394
+ * resolve a relative path against. It is tried against the process working
1395
+ * directory and simply yields no hint when it does not land on a real file —
1396
+ * the guard message itself stays useful either way.
1397
+ */
1398
+ function existingArtifactFor(cfg, raw) {
1399
+ try {
1400
+ const abs = path.isAbsolute(raw) ? path.normalize(raw) : path.resolve(process.cwd(), raw)
1401
+ const st = fs.statSync(abs)
1402
+ if (!st.isFile()) return null
1403
+ return findExistingArtifact(cfg, abs, st, undefined)
1404
+ } catch {
1405
+ return null
1406
+ }
1407
+ }
1408
+
1409
+ function guardRead(ctx, cfg, execution) {
1410
+ try {
1411
+ if (!execution || execution.name !== 'read') return undefined
1412
+ const args = execution.arguments || {}
1413
+ const raw = args.file_path || args.path || args.filePath || args.filename
1414
+ if (typeof raw !== 'string' || !raw) return undefined
1415
+ const cls = classify(raw)
1416
+ if (cls.kind !== 'office') return undefined
1417
+ const shown = raw.replace(/\\/g, '/')
1418
+
1419
+ /* If the file already HAS an artifact, answering "convert it first" is
1420
+ * simply wrong — the work is done, and the model is being told to redo it.
1421
+ * Hand over the .md path so the very next read succeeds. */
1422
+ const existing = existingArtifactFor(cfg, raw)
1423
+ if (existing) {
1424
+ return (
1425
+ '「' + shown + '」是 ' + cls.label + '(' + cls.ext + '),不能直接读取(二进制容器,读出来是乱码)。' +
1426
+ '但它已经转换好了' + (existing.stale ? '(较早的产物,源文件此后有过改动)' : '') + ':\n' +
1427
+ existing.absDst + '\n' +
1428
+ '请直接 read 上面这个 .md 文件。要基于最新源文件重新转换,调用 ' + TOOL_NAME +
1429
+ '({ path: "' + shown + '", force: true })。'
1430
+ )
1431
+ }
1432
+
1433
+ return (
1434
+ '「' + shown + '」是 ' + cls.label + '(' + cls.ext + '),属于二进制压缩容器,直接读取会得到乱码并浪费 token。' +
1435
+ '请先调用 ' + TOOL_NAME + '({ path: "' + shown + '" }) 转换成 Markdown,再 read 生成的 .md 文件。' +
1436
+ '(这条拦截来自 office-markdown 插件,为的是不让二进制内容进上下文;换成别的路径读同一个文件并不能绕过它。确实需要直接读取原始字节时,请让用户把配置项 guardReadTool 设为 false。)'
1437
+ )
1438
+ } catch {
1439
+ return undefined
1440
+ }
1441
+ }
1442
+
1443
+ export { SKILL_CONTENT, TOOL_NAME, SKILL_NAME, DEFAULTS }