dsh-plugin-office-markdown 1.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.en.md +156 -0
- package/README.md +144 -0
- package/cordis.patch.yml +35 -0
- package/lib/client.js +589 -0
- package/lib/convert.js +796 -0
- package/lib/env.js +602 -0
- package/lib/fallback-node.js +334 -0
- package/lib/fallback.py +579 -0
- package/lib/index.js +1443 -0
- package/lib/paths.js +76 -0
- package/lib/removal-watchdog.js +359 -0
- package/lib/settings-api.js +381 -0
- package/package.json +55 -0
package/lib/convert.js
ADDED
|
@@ -0,0 +1,796 @@
|
|
|
1
|
+
/*!
|
|
2
|
+
* dsh-plugin-office-markdown — converter core (host half)
|
|
3
|
+
*
|
|
4
|
+
* Responsibilities:
|
|
5
|
+
* - classify a workspace path as Office/PDF (needs conversion) or plain text;
|
|
6
|
+
* - probe the converter chain in the user's required priority order:
|
|
7
|
+
* 1. `uvx markitdown` (no permanent install)
|
|
8
|
+
* 2. local `markitdown` CLI
|
|
9
|
+
* 3. `python -m markitdown`
|
|
10
|
+
* 4. bundled fallback (fallback.py, needs a local Python, limited fidelity)
|
|
11
|
+
* 5. bundled pure-Node fallback (fallback-node.js, NO Python, NO network)
|
|
12
|
+
* - run the conversion into a workspace temp directory, never touching the
|
|
13
|
+
* source file, and report honestly which converter produced the output.
|
|
14
|
+
*
|
|
15
|
+
* Pure Node standard library: no @deepseek-ai imports, no npm dependencies.
|
|
16
|
+
*/
|
|
17
|
+
import { spawn } from 'node:child_process'
|
|
18
|
+
import fs from 'node:fs'
|
|
19
|
+
import path from 'node:path'
|
|
20
|
+
import { createHash } from 'node:crypto'
|
|
21
|
+
import { StringDecoder } from 'node:string_decoder'
|
|
22
|
+
import { fileURLToPath } from 'node:url'
|
|
23
|
+
|
|
24
|
+
import { convertFileNode, NODE_DEFAULTS } from './fallback-node.js'
|
|
25
|
+
import { runtimesRoot } from './paths.js'
|
|
26
|
+
|
|
27
|
+
const HERE = path.dirname(fileURLToPath(import.meta.url))
|
|
28
|
+
const FALLBACK_PY = path.join(HERE, 'fallback.py')
|
|
29
|
+
const FALLBACK_NODE = path.join(HERE, 'fallback-node.js')
|
|
30
|
+
|
|
31
|
+
/** Binary Office / PDF containers: converting them is what saves tokens. */
|
|
32
|
+
export const OFFICE_EXTS = Object.freeze({
|
|
33
|
+
'.docx': 'Word 文档',
|
|
34
|
+
'.doc': 'Word 97-2003 文档',
|
|
35
|
+
'.docm': 'Word 宏文档',
|
|
36
|
+
'.xlsx': 'Excel 工作簿',
|
|
37
|
+
'.xlsm': 'Excel 宏工作簿',
|
|
38
|
+
'.xls': 'Excel 97-2003 工作簿',
|
|
39
|
+
'.pptx': 'PowerPoint 演示文稿',
|
|
40
|
+
'.pptm': 'PowerPoint 宏演示文稿',
|
|
41
|
+
'.ppt': 'PowerPoint 97-2003 演示文稿',
|
|
42
|
+
'.pdf': 'PDF 文档',
|
|
43
|
+
'.odt': 'OpenDocument 文本',
|
|
44
|
+
'.ods': 'OpenDocument 表格',
|
|
45
|
+
'.odp': 'OpenDocument 演示',
|
|
46
|
+
'.rtf': 'RTF 文档',
|
|
47
|
+
'.epub': 'EPUB 电子书',
|
|
48
|
+
'.msg': 'Outlook 邮件',
|
|
49
|
+
'.ipynb': 'Jupyter Notebook'
|
|
50
|
+
})
|
|
51
|
+
|
|
52
|
+
/** Text-ish formats the model may read directly (no conversion needed). */
|
|
53
|
+
export const PLAIN_EXTS = new Set([
|
|
54
|
+
'.txt', '.md', '.markdown', '.mdx', '.csv', '.tsv', '.json', '.jsonl', '.ndjson',
|
|
55
|
+
'.yaml', '.yml', '.xml', '.html', '.htm', '.log', '.ini', '.cfg', '.toml',
|
|
56
|
+
'.py', '.js', '.mjs', '.cjs', '.ts', '.tsx', '.jsx', '.sql', '.rst', '.tex',
|
|
57
|
+
'.srt', '.vtt', '.gitignore', '.env', '.properties'
|
|
58
|
+
])
|
|
59
|
+
|
|
60
|
+
const CONVERTIBLE = new Set(Object.keys(OFFICE_EXTS))
|
|
61
|
+
|
|
62
|
+
export function extOf(filePath) {
|
|
63
|
+
const base = path.basename(String(filePath || ''))
|
|
64
|
+
const i = base.lastIndexOf('.')
|
|
65
|
+
return i <= 0 ? '' : base.slice(i).toLowerCase()
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* @returns {{kind:'office'|'plain'|'unknown', ext:string, label:string, convertible:boolean}}
|
|
70
|
+
*/
|
|
71
|
+
export function classify(filePath) {
|
|
72
|
+
const ext = extOf(filePath)
|
|
73
|
+
if (CONVERTIBLE.has(ext)) {
|
|
74
|
+
return { kind: 'office', ext, label: OFFICE_EXTS[ext], convertible: true }
|
|
75
|
+
}
|
|
76
|
+
if (PLAIN_EXTS.has(ext)) return { kind: 'plain', ext, label: '纯文本', convertible: false }
|
|
77
|
+
return { kind: 'unknown', ext, label: ext ? ext + ' 文件' : '未知类型', convertible: false }
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/** Rough token estimate (CJK counts ~0.75 token/char, latin ~1 token/3.8 chars). */
|
|
81
|
+
export function estimateTokens(text) {
|
|
82
|
+
if (!text) return 0
|
|
83
|
+
let cjk = 0
|
|
84
|
+
let other = 0
|
|
85
|
+
for (let i = 0; i < text.length; i++) {
|
|
86
|
+
const c = text.charCodeAt(i)
|
|
87
|
+
if (c >= 0x2e80 && c <= 0x9fff) cjk++
|
|
88
|
+
else if (c >= 0xd800 && c <= 0xdfff) {
|
|
89
|
+
cjk++
|
|
90
|
+
i++
|
|
91
|
+
} else if (c === 10 || c === 13 || c === 32 || c === 9) other++
|
|
92
|
+
else other++
|
|
93
|
+
}
|
|
94
|
+
return Math.ceil(other / 3.8 + cjk * 0.75)
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
export function sha8(input) {
|
|
98
|
+
return createHash('sha1').update(String(input)).digest('hex').slice(0, 8)
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
export function safeBaseName(filePath) {
|
|
102
|
+
const base = path.basename(String(filePath || 'file'))
|
|
103
|
+
const i = base.lastIndexOf('.')
|
|
104
|
+
const stem = i > 0 ? base.slice(0, i) : base
|
|
105
|
+
const cleaned = stem.replace(/[^0-9A-Za-z\u4e00-\u9fff._-]+/g, '_').replace(/^_+|_+$/g, '')
|
|
106
|
+
return (cleaned || 'file').slice(0, 80)
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/* ------------------------------------------------------------------ */
|
|
110
|
+
/* artifact provenance */
|
|
111
|
+
/* ------------------------------------------------------------------ */
|
|
112
|
+
|
|
113
|
+
/** Opening of the comment every artifact this plugin writes starts with. */
|
|
114
|
+
export const MARKER_PREFIX = '<!-- dsh-office-markdown '
|
|
115
|
+
|
|
116
|
+
const MARKER_RE = /^<!--\s*dsh-office-markdown\s+([^>]*?)-->\s*\n?/
|
|
117
|
+
|
|
118
|
+
/** Human labels for every converter, usable without probing the machine. */
|
|
119
|
+
export const CONVERTER_LABELS = Object.freeze({
|
|
120
|
+
uvx: 'uvx markitdown(临时运行,无需永久安装)',
|
|
121
|
+
'markitdown-cli': 'markitdown 命令行',
|
|
122
|
+
cli: 'markitdown 命令行',
|
|
123
|
+
'python-module': 'python -m markitdown',
|
|
124
|
+
python: 'python -m markitdown',
|
|
125
|
+
builtin: '插件内置 Python 兜底转换器(保真度有限)',
|
|
126
|
+
'node-builtin': '插件内置 Node 兜底转换器(纯 Node,无需 Python 与网络,保真度有限)'
|
|
127
|
+
})
|
|
128
|
+
|
|
129
|
+
/** Label for a converter id recorded in an artifact. */
|
|
130
|
+
export function converterLabelFor(id, fallback) {
|
|
131
|
+
const key = String(id || '')
|
|
132
|
+
return CONVERTER_LABELS[key] || fallback || key || '未知转换器'
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/**
|
|
136
|
+
* Stamp the artifact with what actually produced it.
|
|
137
|
+
*
|
|
138
|
+
* Without this a reuse can only guess: `index.js` used to hard-code
|
|
139
|
+
* `fidelity:"high"` for every cache hit, so a file really produced by the
|
|
140
|
+
* limited-fidelity Node fallback reported the same confidence as a real
|
|
141
|
+
* MarkItDown run. The stamp is a Markdown comment, so it never renders.
|
|
142
|
+
*
|
|
143
|
+
* @returns {number} the artifact's new size in bytes
|
|
144
|
+
*/
|
|
145
|
+
export function writeFidelityMarker(dstPath, converterId, fidelity, extra) {
|
|
146
|
+
const body = fs.readFileSync(dstPath, 'utf8')
|
|
147
|
+
const attrs = {
|
|
148
|
+
converter: String(converterId || 'unknown'),
|
|
149
|
+
fidelity: String(fidelity || 'unknown'),
|
|
150
|
+
at: new Date().toISOString()
|
|
151
|
+
}
|
|
152
|
+
/* `srcbytes` / `srchash` describe the SOURCE file this artifact came out of.
|
|
153
|
+
* They are what lets a moved mtime be checked against the real content
|
|
154
|
+
* instead of paying for a re-conversion. Values must stay whitespace-free:
|
|
155
|
+
* the marker is parsed by splitting on spaces. */
|
|
156
|
+
for (const key of Object.keys(extra || {})) {
|
|
157
|
+
const value = String(extra[key] == null ? '' : extra[key]).replace(/\s+/g, '')
|
|
158
|
+
if (value) attrs[key] = value
|
|
159
|
+
}
|
|
160
|
+
const marker = MARKER_PREFIX + Object.keys(attrs).map((k) => k + '=' + attrs[k]).join(' ') + ' -->\n'
|
|
161
|
+
fs.writeFileSync(dstPath, marker + body, 'utf8')
|
|
162
|
+
return Buffer.byteLength(marker, 'utf8') + Buffer.byteLength(body, 'utf8')
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/** Read the stamp back. `null` for an artifact written by an earlier version. */
|
|
166
|
+
export function readFidelityMarker(absPath) {
|
|
167
|
+
let fd = null
|
|
168
|
+
try {
|
|
169
|
+
fd = fs.openSync(absPath, 'r')
|
|
170
|
+
const buf = Buffer.alloc(512)
|
|
171
|
+
const n = fs.readSync(fd, buf, 0, buf.length, 0)
|
|
172
|
+
const m = buf.subarray(0, n).toString('utf8').match(MARKER_RE)
|
|
173
|
+
if (!m) return null
|
|
174
|
+
const attrs = {}
|
|
175
|
+
for (const part of String(m[1]).trim().split(/\s+/)) {
|
|
176
|
+
const i = part.indexOf('=')
|
|
177
|
+
if (i > 0) attrs[part.slice(0, i)] = part.slice(i + 1)
|
|
178
|
+
}
|
|
179
|
+
return {
|
|
180
|
+
...attrs,
|
|
181
|
+
converter: attrs.converter || '',
|
|
182
|
+
fidelity: attrs.fidelity || '',
|
|
183
|
+
at: attrs.at || ''
|
|
184
|
+
}
|
|
185
|
+
} catch {
|
|
186
|
+
return null
|
|
187
|
+
} finally {
|
|
188
|
+
if (fd !== null) { try { fs.closeSync(fd) } catch { /* ignore */ } }
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/* ------------------------------------------------------------------ */
|
|
193
|
+
/* artifact bookkeeping */
|
|
194
|
+
/* ------------------------------------------------------------------ */
|
|
195
|
+
|
|
196
|
+
/** `^<safeBaseName(src)>-[0-9a-f]{8}\.md$` — the only names this plugin writes. */
|
|
197
|
+
export function artifactNamePattern(srcAbs) {
|
|
198
|
+
const stem = safeBaseName(srcAbs).replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
|
|
199
|
+
return new RegExp('^' + stem + '-[0-9a-f]{8}\\.md$')
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* Every artifact this plugin ever wrote for `srcAbs` inside `tmpDir`.
|
|
204
|
+
*
|
|
205
|
+
* Scoped on purpose: only that one directory, only files (never directories),
|
|
206
|
+
* only names matching the plugin's own pattern, never recursive. The same rules
|
|
207
|
+
* back `action:"clean"` and the optional auto-prune, so there is exactly one
|
|
208
|
+
* definition of "an artifact of this source file" to audit.
|
|
209
|
+
*
|
|
210
|
+
* @returns {Array<{path:string, name:string, size:number, mtimeMs:number}>} newest first
|
|
211
|
+
*/
|
|
212
|
+
export function listArtifacts(srcAbs, tmpDir) {
|
|
213
|
+
const out = []
|
|
214
|
+
let names = []
|
|
215
|
+
try { names = fs.readdirSync(tmpDir) } catch { return out }
|
|
216
|
+
const re = artifactNamePattern(srcAbs)
|
|
217
|
+
for (const file of names) {
|
|
218
|
+
if (!re.test(file)) continue
|
|
219
|
+
const abs = path.join(tmpDir, file)
|
|
220
|
+
let st = null
|
|
221
|
+
try { st = fs.statSync(abs) } catch { continue }
|
|
222
|
+
if (!st || !st.isFile()) continue
|
|
223
|
+
out.push({ path: abs, name: file, size: st.size, mtimeMs: st.mtimeMs })
|
|
224
|
+
}
|
|
225
|
+
out.sort((a, b) => b.mtimeMs - a.mtimeMs)
|
|
226
|
+
return out
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
/** The most recently written artifact for `srcAbs`, or `null`. */
|
|
230
|
+
export function findLatestArtifact(srcAbs, tmpDir) {
|
|
231
|
+
const list = listArtifacts(srcAbs, tmpDir)
|
|
232
|
+
return list.length ? list[0] : null
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
/* ------------------------------------------------------------------ */
|
|
236
|
+
/* artifact analysis */
|
|
237
|
+
/* ------------------------------------------------------------------ */
|
|
238
|
+
|
|
239
|
+
const ANALYZE_CHUNK = 262144
|
|
240
|
+
|
|
241
|
+
/**
|
|
242
|
+
* Walk an artifact once, reporting its real byte size, line count and a token
|
|
243
|
+
* estimate — without ever materialising it as one string.
|
|
244
|
+
*
|
|
245
|
+
* `preview:0` (the default) used to `readFileSync` the whole Markdown anyway and
|
|
246
|
+
* then run a per-character loop over it: for a 50 MB workbook that is tens of
|
|
247
|
+
* megabytes of transient string on top of the file's own cost, on the default
|
|
248
|
+
* path, for a value the caller had just decided not to return. The estimate is
|
|
249
|
+
* the same one {@link estimateTokens} makes; only the retention changed.
|
|
250
|
+
*
|
|
251
|
+
* @returns {{bytes:number, lines:number, tokens:number, head:string, error:string}}
|
|
252
|
+
*/
|
|
253
|
+
export function analyzeMarkdown(absPath, options = {}) {
|
|
254
|
+
const headChars = Math.max(0, Number(options.headChars) || 0)
|
|
255
|
+
const out = { bytes: 0, lines: 0, tokens: 0, head: '', error: '' }
|
|
256
|
+
let cjk = 0
|
|
257
|
+
let other = 0
|
|
258
|
+
let lines = 1
|
|
259
|
+
const count = (s) => {
|
|
260
|
+
for (let i = 0; i < s.length; i++) {
|
|
261
|
+
const c = s.charCodeAt(i)
|
|
262
|
+
if (c === 10) lines++
|
|
263
|
+
if (c >= 0x2e80 && c <= 0x9fff) cjk++
|
|
264
|
+
else if (c >= 0xd800 && c <= 0xdfff) { cjk++; i++ }
|
|
265
|
+
else other++
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
const take = (s) => {
|
|
269
|
+
if (headChars > 0 && out.head.length < headChars) out.head += s.slice(0, headChars - out.head.length)
|
|
270
|
+
}
|
|
271
|
+
let fd = null
|
|
272
|
+
try {
|
|
273
|
+
fd = fs.openSync(absPath, 'r')
|
|
274
|
+
const buf = Buffer.alloc(ANALYZE_CHUNK)
|
|
275
|
+
const dec = new StringDecoder('utf8')
|
|
276
|
+
for (;;) {
|
|
277
|
+
const n = fs.readSync(fd, buf, 0, buf.length, null)
|
|
278
|
+
if (n <= 0) break
|
|
279
|
+
out.bytes += n
|
|
280
|
+
const s = dec.write(buf.subarray(0, n))
|
|
281
|
+
take(s)
|
|
282
|
+
count(s)
|
|
283
|
+
}
|
|
284
|
+
const tail = dec.end()
|
|
285
|
+
if (tail) { take(tail); count(tail) }
|
|
286
|
+
out.lines = lines
|
|
287
|
+
out.tokens = Math.ceil(other / 3.8 + cjk * 0.75)
|
|
288
|
+
} catch (error) {
|
|
289
|
+
out.error = String((error && error.message) || error)
|
|
290
|
+
} finally {
|
|
291
|
+
if (fd !== null) { try { fs.closeSync(fd) } catch { /* ignore */ } }
|
|
292
|
+
}
|
|
293
|
+
return out
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
/**
|
|
297
|
+
* Streaming content hash of a source file.
|
|
298
|
+
*
|
|
299
|
+
* The cache key ends in the source's mtime, which `git checkout`, a copy or a
|
|
300
|
+
* restore changes without touching the content. Comparing this hash against the
|
|
301
|
+
* one recorded in an artifact is what tells "really edited" apart from "same
|
|
302
|
+
* bytes, new timestamp". Returns `''` when the file cannot be read.
|
|
303
|
+
*/
|
|
304
|
+
export function sourceContentHash(absPath) {
|
|
305
|
+
let fd = null
|
|
306
|
+
try {
|
|
307
|
+
fd = fs.openSync(absPath, 'r')
|
|
308
|
+
const buf = Buffer.alloc(ANALYZE_CHUNK)
|
|
309
|
+
const hash = createHash('sha1')
|
|
310
|
+
for (;;) {
|
|
311
|
+
const n = fs.readSync(fd, buf, 0, buf.length, null)
|
|
312
|
+
if (n <= 0) break
|
|
313
|
+
hash.update(buf.subarray(0, n))
|
|
314
|
+
}
|
|
315
|
+
return hash.digest('hex').slice(0, 16)
|
|
316
|
+
} catch {
|
|
317
|
+
return ''
|
|
318
|
+
} finally {
|
|
319
|
+
if (fd !== null) { try { fs.closeSync(fd) } catch { /* ignore */ } }
|
|
320
|
+
}
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
/* ------------------------------------------------------------------ */
|
|
324
|
+
/* subprocess */
|
|
325
|
+
/* ------------------------------------------------------------------ */
|
|
326
|
+
|
|
327
|
+
const MAX_STDOUT = 262144
|
|
328
|
+
const MAX_STDERR = 65536
|
|
329
|
+
|
|
330
|
+
/**
|
|
331
|
+
* Run one child process. Never rejects: every failure is reported as a value
|
|
332
|
+
* so the caller can fall through the converter chain.
|
|
333
|
+
*/
|
|
334
|
+
/**
|
|
335
|
+
* Kill a child process **and everything it spawned**.
|
|
336
|
+
*
|
|
337
|
+
* `child.kill()` on Windows is `TerminateProcess`, which stops only the process
|
|
338
|
+
* it is called on: `uvx` and the `python` it starts keep running, keep the
|
|
339
|
+
* output file open, and can even write into it after the caller has already
|
|
340
|
+
* declared the conversion dead. `taskkill /T /F` walks the whole tree. On POSIX
|
|
341
|
+
* the child is spawned into its own process group (`detached`) so the group can
|
|
342
|
+
* be signalled at once.
|
|
343
|
+
*
|
|
344
|
+
* Resolves once the tree is gone, or after `graceMs` at the latest, so the
|
|
345
|
+
* caller can safely delete and re-create the output file afterwards.
|
|
346
|
+
*/
|
|
347
|
+
function killTree(child, graceMs = 2000) {
|
|
348
|
+
return new Promise((resolve) => {
|
|
349
|
+
let done = false
|
|
350
|
+
const settle = () => { if (!done) { done = true; clearTimeout(timer); resolve() } }
|
|
351
|
+
if (!child || typeof child.pid !== 'number') { resolve(); return }
|
|
352
|
+
const timer = setTimeout(settle, graceMs)
|
|
353
|
+
if (timer && typeof timer.unref === 'function') timer.unref()
|
|
354
|
+
if (process.platform === 'win32') {
|
|
355
|
+
try {
|
|
356
|
+
const killer = spawn('taskkill', ['/pid', String(child.pid), '/T', '/F'], { windowsHide: true, stdio: 'ignore' })
|
|
357
|
+
killer.on('error', () => { try { child.kill() } catch { /* ignore */ } ; settle() })
|
|
358
|
+
killer.on('close', settle)
|
|
359
|
+
} catch {
|
|
360
|
+
try { child.kill() } catch { /* ignore */ }
|
|
361
|
+
settle()
|
|
362
|
+
}
|
|
363
|
+
return
|
|
364
|
+
}
|
|
365
|
+
try { process.kill(-child.pid, 'SIGKILL') } catch { try { child.kill('SIGKILL') } catch { /* ignore */ } }
|
|
366
|
+
settle()
|
|
367
|
+
})
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
export function run(cmd, args, options = {}) {
|
|
371
|
+
const timeoutMs = Number.isFinite(options.timeoutMs) ? options.timeoutMs : 300000
|
|
372
|
+
const signal = options.signal
|
|
373
|
+
return new Promise((resolve) => {
|
|
374
|
+
if (signal && signal.aborted) {
|
|
375
|
+
resolve({ ok: false, code: -1, stdout: '', stderr: 'aborted before start', aborted: true })
|
|
376
|
+
return
|
|
377
|
+
}
|
|
378
|
+
let child
|
|
379
|
+
try {
|
|
380
|
+
child = spawn(cmd, args, {
|
|
381
|
+
cwd: options.cwd,
|
|
382
|
+
windowsHide: true,
|
|
383
|
+
shell: false,
|
|
384
|
+
/* Windows never gets `detached`: there the tree is taken down with
|
|
385
|
+
* `taskkill /T /F`, and a detached console child would only be harder
|
|
386
|
+
* to reach. */
|
|
387
|
+
detached: process.platform !== 'win32',
|
|
388
|
+
env: options.env ? { ...process.env, ...options.env } : process.env,
|
|
389
|
+
stdio: ['ignore', 'pipe', 'pipe']
|
|
390
|
+
})
|
|
391
|
+
} catch (error) {
|
|
392
|
+
resolve({ ok: false, code: -1, stdout: '', stderr: String((error && error.message) || error), spawnError: true })
|
|
393
|
+
return
|
|
394
|
+
}
|
|
395
|
+
let stdout = ''
|
|
396
|
+
let stderr = ''
|
|
397
|
+
let timedOut = false
|
|
398
|
+
let settled = false
|
|
399
|
+
let killPromise = null
|
|
400
|
+
const treeKill = () => {
|
|
401
|
+
if (!killPromise) killPromise = killTree(child)
|
|
402
|
+
return killPromise
|
|
403
|
+
}
|
|
404
|
+
const timer = setTimeout(() => {
|
|
405
|
+
timedOut = true
|
|
406
|
+
void treeKill()
|
|
407
|
+
}, timeoutMs)
|
|
408
|
+
const onAbort = () => { void treeKill() }
|
|
409
|
+
if (signal && typeof signal.addEventListener === 'function') signal.addEventListener('abort', onAbort, { once: true })
|
|
410
|
+
|
|
411
|
+
const finish = (result) => {
|
|
412
|
+
if (settled) return
|
|
413
|
+
settled = true
|
|
414
|
+
clearTimeout(timer)
|
|
415
|
+
if (signal && typeof signal.removeEventListener === 'function') signal.removeEventListener('abort', onAbort)
|
|
416
|
+
/* A timed-out or cancelled run must not report back before the tree is
|
|
417
|
+
* really gone: the caller deletes and re-creates the output file next. */
|
|
418
|
+
if (killPromise) killPromise.then(() => resolve(result))
|
|
419
|
+
else resolve(result)
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
/* `StringDecoder` holds the bytes of a multi-byte character together when
|
|
423
|
+
* one arrives split across two chunks. The old per-chunk `toString('utf8')`
|
|
424
|
+
* turned every such character into U+FFFD, and that text is what the failure
|
|
425
|
+
* message handed to the model is built from. Every chunk must be written
|
|
426
|
+
* through the decoder, even after the capture cap is reached, or the decoder
|
|
427
|
+
* loses track of where it was. */
|
|
428
|
+
const outDec = new StringDecoder('utf8')
|
|
429
|
+
const errDec = new StringDecoder('utf8')
|
|
430
|
+
if (child.stdout) child.stdout.on('data', (d) => {
|
|
431
|
+
const s = outDec.write(d)
|
|
432
|
+
if (stdout.length < MAX_STDOUT) stdout += s
|
|
433
|
+
})
|
|
434
|
+
if (child.stderr) child.stderr.on('data', (d) => {
|
|
435
|
+
const s = errDec.write(d)
|
|
436
|
+
if (stderr.length < MAX_STDERR) stderr += s
|
|
437
|
+
})
|
|
438
|
+
child.on('error', (error) => finish({ ok: false, code: -1, stdout, stderr: stderr + '\n' + String((error && error.message) || error), spawnError: true }))
|
|
439
|
+
child.on('close', (code) => finish({ ok: !timedOut && code === 0, code, stdout, stderr, timedOut, aborted: !!(signal && signal.aborted) }))
|
|
440
|
+
})
|
|
441
|
+
}
|
|
442
|
+
|
|
443
|
+
/* ------------------------------------------------------------------ */
|
|
444
|
+
/* converter probing */
|
|
445
|
+
/* ------------------------------------------------------------------ */
|
|
446
|
+
|
|
447
|
+
function discoverBundledPythons() {
|
|
448
|
+
const out = []
|
|
449
|
+
try {
|
|
450
|
+
const root = runtimesRoot()
|
|
451
|
+
for (const name of fs.readdirSync(root)) {
|
|
452
|
+
const p = path.join(root, name, 'dependencies', 'python', 'python.exe')
|
|
453
|
+
if (fs.existsSync(p)) out.push(p)
|
|
454
|
+
const p2 = path.join(root, name, 'dependencies', 'python', 'bin', 'python3')
|
|
455
|
+
if (fs.existsSync(p2)) out.push(p2)
|
|
456
|
+
}
|
|
457
|
+
} catch { /* no bundled runtime */ }
|
|
458
|
+
return out
|
|
459
|
+
}
|
|
460
|
+
|
|
461
|
+
/** Chinese label for where a Python candidate came from. */
|
|
462
|
+
export function pythonSourceLabel(source) {
|
|
463
|
+
if (source === 'config') return '配置 pythonPath'
|
|
464
|
+
if (source === 'bundled') return 'DSH 自带运行时'
|
|
465
|
+
return '系统 PATH'
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
export const PYTHON_PREFERS = ['auto', 'bundled', 'system', 'config']
|
|
469
|
+
|
|
470
|
+
/**
|
|
471
|
+
* Ordered list of Python interpreters to try, honouring `cfg.pythonPrefer`.
|
|
472
|
+
* Each entry carries its `source` so `action:"status"` can explain the order.
|
|
473
|
+
* @returns {Array<{cmd: string, prefix: string[], source: string}>}
|
|
474
|
+
*/
|
|
475
|
+
export function pythonCandidates(cfg) {
|
|
476
|
+
const out = []
|
|
477
|
+
const seen = new Set()
|
|
478
|
+
const push = (cmd, prefix, source) => {
|
|
479
|
+
if (!cmd) return
|
|
480
|
+
const key = cmd + ' ' + prefix.join(' ')
|
|
481
|
+
if (seen.has(key)) return
|
|
482
|
+
seen.add(key)
|
|
483
|
+
out.push({ cmd, prefix, source: source || 'system' })
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
const configured = cfg && cfg.pythonPath ? String(cfg.pythonPath) : ''
|
|
487
|
+
const bundled = discoverBundledPythons()
|
|
488
|
+
const system = [['python', []], ['python3', []], ['py', ['-3']]]
|
|
489
|
+
|
|
490
|
+
let prefer = String((cfg && cfg.pythonPrefer) || 'auto').toLowerCase()
|
|
491
|
+
if (!PYTHON_PREFERS.includes(prefer)) prefer = 'auto'
|
|
492
|
+
/* A mode that cannot be satisfied degrades to the full order rather than
|
|
493
|
+
* leaving the plugin with no interpreter at all. */
|
|
494
|
+
if (prefer === 'config' && !configured) prefer = 'auto'
|
|
495
|
+
if (prefer === 'bundled' && bundled.length === 0) prefer = 'auto'
|
|
496
|
+
|
|
497
|
+
if (prefer === 'config') {
|
|
498
|
+
push(configured, [], 'config')
|
|
499
|
+
return out
|
|
500
|
+
}
|
|
501
|
+
if (prefer === 'bundled') {
|
|
502
|
+
for (const p of bundled) push(p, [], 'bundled')
|
|
503
|
+
return out
|
|
504
|
+
}
|
|
505
|
+
if (prefer === 'system') {
|
|
506
|
+
for (const [cmd, prefix] of system) push(cmd, prefix, 'system')
|
|
507
|
+
for (const p of bundled) push(p, [], 'bundled')
|
|
508
|
+
if (configured) push(configured, [], 'config')
|
|
509
|
+
return out
|
|
510
|
+
}
|
|
511
|
+
if (configured) push(configured, [], 'config')
|
|
512
|
+
for (const p of bundled) push(p, [], 'bundled')
|
|
513
|
+
for (const [cmd, prefix] of system) push(cmd, prefix, 'system')
|
|
514
|
+
return out
|
|
515
|
+
}
|
|
516
|
+
|
|
517
|
+
let probeCache = { at: 0, key: '', value: null }
|
|
518
|
+
|
|
519
|
+
/**
|
|
520
|
+
* Probe the converter chain once (cached for `cfg.probeTtlMs`).
|
|
521
|
+
* @returns {Promise<{chain: Array<object>, notes: string[]}>}
|
|
522
|
+
*/
|
|
523
|
+
export async function probeConverters(cfg, options = {}) {
|
|
524
|
+
const key = JSON.stringify([
|
|
525
|
+
cfg.pythonPath || '',
|
|
526
|
+
cfg.pythonPrefer || 'auto',
|
|
527
|
+
cfg.uvxExtras || '',
|
|
528
|
+
cfg.allowUvxDownload !== false,
|
|
529
|
+
cfg.fallbackEnabled !== false
|
|
530
|
+
])
|
|
531
|
+
const ttl = Number.isFinite(cfg.probeTtlMs) ? cfg.probeTtlMs : 600000
|
|
532
|
+
if (!options.force && probeCache.value && probeCache.key === key && Date.now() - probeCache.at < ttl) {
|
|
533
|
+
return probeCache.value
|
|
534
|
+
}
|
|
535
|
+
|
|
536
|
+
const chain = []
|
|
537
|
+
const notes = []
|
|
538
|
+
const probeTimeout = 45000
|
|
539
|
+
|
|
540
|
+
const uvx = await run('uvx', ['--version'], { timeoutMs: probeTimeout, signal: options.signal })
|
|
541
|
+
if (uvx.ok) {
|
|
542
|
+
chain.push({
|
|
543
|
+
id: 'uvx',
|
|
544
|
+
label: 'uvx markitdown(临时运行,无需永久安装)',
|
|
545
|
+
fidelity: 'high',
|
|
546
|
+
kind: 'uvx'
|
|
547
|
+
})
|
|
548
|
+
} else {
|
|
549
|
+
notes.push('未检测到 uvx(' + firstLine(uvx.stderr) + ')')
|
|
550
|
+
}
|
|
551
|
+
|
|
552
|
+
const cli = await run('markitdown', ['--help'], { timeoutMs: probeTimeout, signal: options.signal })
|
|
553
|
+
if (cli.ok) {
|
|
554
|
+
chain.push({
|
|
555
|
+
id: 'markitdown-cli',
|
|
556
|
+
label: 'markitdown 命令行',
|
|
557
|
+
fidelity: 'high',
|
|
558
|
+
kind: 'cli'
|
|
559
|
+
})
|
|
560
|
+
} else {
|
|
561
|
+
notes.push('未检测到 markitdown 命令(' + firstLine(cli.stderr) + ')')
|
|
562
|
+
}
|
|
563
|
+
|
|
564
|
+
let pythonHit = null
|
|
565
|
+
const pythonProbes = []
|
|
566
|
+
const candidates = pythonCandidates(cfg)
|
|
567
|
+
/* `force` (which is what `action:"status"` passes) audits every candidate so
|
|
568
|
+
* the user can see which interpreter actually holds markitdown; a normal
|
|
569
|
+
* conversion stops at the first hit to stay fast. */
|
|
570
|
+
const probeAll = options.probeAll === true || options.force === true
|
|
571
|
+
for (const cand of candidates) {
|
|
572
|
+
const r = await run(cand.cmd, [...cand.prefix, '-m', 'markitdown', '--help'], { timeoutMs: probeTimeout, signal: options.signal })
|
|
573
|
+
pythonProbes.push({ cmd: cand.cmd, source: cand.source, ok: !!r.ok, reason: r.ok ? '' : firstLine(r.stderr) })
|
|
574
|
+
if (r.ok) {
|
|
575
|
+
if (!pythonHit) {
|
|
576
|
+
pythonHit = { id: 'python-module', label: 'python -m markitdown(' + cand.cmd + ')', fidelity: 'high', kind: 'python', python: cand }
|
|
577
|
+
}
|
|
578
|
+
if (!probeAll) break
|
|
579
|
+
}
|
|
580
|
+
}
|
|
581
|
+
if (pythonHit) chain.push(pythonHit)
|
|
582
|
+
else if (candidates.length) notes.push('未检测到已安装 markitdown 的 Python 解释器(已检查 ' + candidates.length + ' 个候选,明细见下方)')
|
|
583
|
+
else notes.push('没有可用的 Python 解释器候选(pythonPrefer=' + String(cfg.pythonPrefer || 'auto') + ')')
|
|
584
|
+
|
|
585
|
+
if (cfg.fallbackEnabled !== false) {
|
|
586
|
+
const py = pythonHit && pythonHit.python ? pythonHit.python : pythonCandidates(cfg)[0]
|
|
587
|
+
if (py) {
|
|
588
|
+
chain.push({
|
|
589
|
+
id: 'builtin',
|
|
590
|
+
label: '插件内置 Python 兜底转换器(保真度有限)',
|
|
591
|
+
fidelity: 'limited',
|
|
592
|
+
kind: 'builtin',
|
|
593
|
+
python: py
|
|
594
|
+
})
|
|
595
|
+
} else {
|
|
596
|
+
notes.push('未检测到 Python 解释器,已跳过 Python 兜底转换器(不影响 Node 兜底)')
|
|
597
|
+
}
|
|
598
|
+
chain.push({
|
|
599
|
+
id: 'node-builtin',
|
|
600
|
+
label: '插件内置 Node 兜底转换器(纯 Node,无需 Python 与网络,保真度有限)',
|
|
601
|
+
fidelity: 'limited',
|
|
602
|
+
kind: 'inline'
|
|
603
|
+
})
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
const value = {
|
|
607
|
+
chain,
|
|
608
|
+
notes,
|
|
609
|
+
pythonProbes,
|
|
610
|
+
pythonPrefer: String(cfg.pythonPrefer || 'auto'),
|
|
611
|
+
probedAt: Date.now()
|
|
612
|
+
}
|
|
613
|
+
probeCache = { at: Date.now(), key, value }
|
|
614
|
+
if (options.force) probeCache.value = value
|
|
615
|
+
return value
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
function firstLine(text) {
|
|
619
|
+
const s = String(text || '').replace(/\r/g, '').trim()
|
|
620
|
+
const line = s.split('\n').filter(Boolean).pop() || s.split('\n')[0] || ''
|
|
621
|
+
return line.slice(0, 160)
|
|
622
|
+
}
|
|
623
|
+
|
|
624
|
+
/* ------------------------------------------------------------------ */
|
|
625
|
+
/* conversion */
|
|
626
|
+
/* ------------------------------------------------------------------ */
|
|
627
|
+
|
|
628
|
+
function buildSteps(converter, src, dst, cfg) {
|
|
629
|
+
const extras = cfg.uvxExtras || 'markitdown[all]'
|
|
630
|
+
switch (converter.kind) {
|
|
631
|
+
case 'uvx':
|
|
632
|
+
return [
|
|
633
|
+
['uvx', ['--from', extras, 'markitdown', src, '-o', dst]],
|
|
634
|
+
['uvx', [extras, src, '-o', dst]]
|
|
635
|
+
]
|
|
636
|
+
case 'cli':
|
|
637
|
+
return [['markitdown', [src, '-o', dst]]]
|
|
638
|
+
case 'python':
|
|
639
|
+
return [[converter.python.cmd, [...converter.python.prefix, '-m', 'markitdown', src, '-o', dst]]]
|
|
640
|
+
case 'builtin':
|
|
641
|
+
return [[
|
|
642
|
+
converter.python ? converter.python.cmd : 'python',
|
|
643
|
+
[
|
|
644
|
+
...(converter.python ? converter.python.prefix : []),
|
|
645
|
+
FALLBACK_PY,
|
|
646
|
+
'--input', src,
|
|
647
|
+
'--output', dst,
|
|
648
|
+
'--max-rows', String(cfg.maxRowsPerSheet || 400),
|
|
649
|
+
'--max-cols', String(cfg.maxTableCols || 24),
|
|
650
|
+
'--max-cells', String(cfg.maxCellsPerSheet || 20000)
|
|
651
|
+
]
|
|
652
|
+
]]
|
|
653
|
+
default:
|
|
654
|
+
return []
|
|
655
|
+
}
|
|
656
|
+
}
|
|
657
|
+
|
|
658
|
+
async function runSteps(steps, options) {
|
|
659
|
+
const failures = []
|
|
660
|
+
for (const [cmd, args] of steps) {
|
|
661
|
+
const r = await run(cmd, args, options)
|
|
662
|
+
if (r.ok) return { ok: true, failure: null, failures }
|
|
663
|
+
failures.push({ cmd: cmd + ' ' + args.join(' '), reason: r.timedOut ? '超时' : firstLine(r.stderr) || ('退出码 ' + r.code) })
|
|
664
|
+
if (r.aborted) break
|
|
665
|
+
}
|
|
666
|
+
return { ok: false, failures }
|
|
667
|
+
}
|
|
668
|
+
|
|
669
|
+
/** The pure-Node fallback runs in-process: no Python, no child process, no network. */
|
|
670
|
+
function runInline(converter, srcPath, dstPath, cfg) {
|
|
671
|
+
try { fs.rmSync(dstPath, { force: true }) } catch { /* ignore */ }
|
|
672
|
+
convertFileNode(srcPath, dstPath, {
|
|
673
|
+
maxRowsPerSheet: cfg.maxRowsPerSheet || NODE_DEFAULTS.maxRowsPerSheet,
|
|
674
|
+
maxTableCols: cfg.maxTableCols || NODE_DEFAULTS.maxTableCols,
|
|
675
|
+
maxCellsPerSheet: cfg.maxCellsPerSheet || NODE_DEFAULTS.maxCellsPerSheet
|
|
676
|
+
})
|
|
677
|
+
const stat = fs.statSync(dstPath)
|
|
678
|
+
if (!stat || stat.size === 0) throw new Error('转换结果为空文件')
|
|
679
|
+
return { ok: true, dstBytes: stat.size }
|
|
680
|
+
}
|
|
681
|
+
|
|
682
|
+
/**
|
|
683
|
+
* Write the provenance stamp and report the artifact's final size.
|
|
684
|
+
*
|
|
685
|
+
* A stamp that cannot be written is not a conversion failure — the Markdown
|
|
686
|
+
* itself is fine — so the size is still reported and only the reuse path loses
|
|
687
|
+
* its ability to say who produced the file.
|
|
688
|
+
*/
|
|
689
|
+
function stampArtifact(dstPath, converter, srcPath) {
|
|
690
|
+
let extra = null
|
|
691
|
+
try {
|
|
692
|
+
const st = fs.statSync(srcPath)
|
|
693
|
+
extra = { srcbytes: st.size, srchash: sourceContentHash(srcPath) || 'unknown' }
|
|
694
|
+
} catch {
|
|
695
|
+
extra = null
|
|
696
|
+
}
|
|
697
|
+
try {
|
|
698
|
+
return writeFidelityMarker(dstPath, converter.id, converter.fidelity, extra)
|
|
699
|
+
} catch {
|
|
700
|
+
try { return fs.statSync(dstPath).size } catch { return 0 }
|
|
701
|
+
}
|
|
702
|
+
}
|
|
703
|
+
|
|
704
|
+
/**
|
|
705
|
+
* Convert one file to Markdown.
|
|
706
|
+
*
|
|
707
|
+
* @returns {Promise<object>} a value object; `ok:false` carries a human error.
|
|
708
|
+
*/
|
|
709
|
+
export async function convertFile(ctx) {
|
|
710
|
+
const { srcPath, dstPath, cfg, signal, force } = ctx
|
|
711
|
+
const intended = cfg.converter && cfg.converter !== 'auto' ? cfg.converter : null
|
|
712
|
+
|
|
713
|
+
const probe = await probeConverters(cfg, { signal, force: !!force })
|
|
714
|
+
let chain = probe.chain
|
|
715
|
+
if (intended) {
|
|
716
|
+
const wanted = chain.filter((c) => c.kind === intended || c.id === intended)
|
|
717
|
+
if (wanted.length) chain = wanted
|
|
718
|
+
}
|
|
719
|
+
if (!chain.length) {
|
|
720
|
+
return { ok: false, error: '没有任何可用的转换器:uvx / markitdown 命令 / python -m markitdown / 内置兜底 全部不可用。', probe }
|
|
721
|
+
}
|
|
722
|
+
|
|
723
|
+
const attempts = []
|
|
724
|
+
for (const converter of chain) {
|
|
725
|
+
if (converter.kind === 'inline') {
|
|
726
|
+
try {
|
|
727
|
+
runInline(converter, srcPath, dstPath, cfg)
|
|
728
|
+
return {
|
|
729
|
+
ok: true,
|
|
730
|
+
converter: converter.id,
|
|
731
|
+
converterLabel: converter.label,
|
|
732
|
+
fidelity: converter.fidelity,
|
|
733
|
+
converterNote: '以上为插件内置 Node 兜底转换(纯 Node,未使用 MarkItDown),版式、图表、批注、图片、公式等可能缺失。',
|
|
734
|
+
attempts,
|
|
735
|
+
probeNotes: probe.notes,
|
|
736
|
+
dstBytes: stampArtifact(dstPath, converter, srcPath)
|
|
737
|
+
}
|
|
738
|
+
} catch (error) {
|
|
739
|
+
attempts.push({
|
|
740
|
+
converter: converter.id,
|
|
741
|
+
label: converter.label,
|
|
742
|
+
failures: [{ cmd: converter.label, reason: firstLine(String((error && error.message) || error)) }]
|
|
743
|
+
})
|
|
744
|
+
continue
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
|
|
748
|
+
const steps = buildSteps(converter, srcPath, dstPath, cfg)
|
|
749
|
+
if (!steps.length) continue
|
|
750
|
+
try { fs.rmSync(dstPath, { force: true }) } catch { /* ignore */ }
|
|
751
|
+
const result = await runSteps(steps, { timeoutMs: cfg.timeoutMs, signal })
|
|
752
|
+
if (!result.ok) {
|
|
753
|
+
attempts.push({ converter: converter.id, label: converter.label, failures: result.failures })
|
|
754
|
+
if (signal && signal.aborted) {
|
|
755
|
+
return { ok: false, error: '转换已取消。', attempts, probe, aborted: true }
|
|
756
|
+
}
|
|
757
|
+
continue
|
|
758
|
+
}
|
|
759
|
+
let stat = null
|
|
760
|
+
try { stat = fs.statSync(dstPath) } catch { /* ignore */ }
|
|
761
|
+
if (!stat || stat.size === 0) {
|
|
762
|
+
attempts.push({ converter: converter.id, label: converter.label, failures: [{ cmd: converter.label, reason: '转换结果为空文件' }] })
|
|
763
|
+
continue
|
|
764
|
+
}
|
|
765
|
+
return {
|
|
766
|
+
ok: true,
|
|
767
|
+
converter: converter.id,
|
|
768
|
+
converterLabel: converter.label,
|
|
769
|
+
fidelity: converter.fidelity,
|
|
770
|
+
converterNote: converter.kind === 'builtin'
|
|
771
|
+
? '以上为插件内置兜底转换(未使用 MarkItDown),版式、图表、批注、扫描件 OCR 等信息可能缺失。'
|
|
772
|
+
: null,
|
|
773
|
+
attempts,
|
|
774
|
+
probeNotes: probe.notes,
|
|
775
|
+
dstBytes: stampArtifact(dstPath, converter, srcPath)
|
|
776
|
+
}
|
|
777
|
+
}
|
|
778
|
+
|
|
779
|
+
const last = attempts[attempts.length - 1]
|
|
780
|
+
const detail = last && last.failures && last.failures.length
|
|
781
|
+
? last.failures.map((f) => ' - ' + f.cmd + ' → ' + f.reason).join('\n')
|
|
782
|
+
: ' - 未知原因'
|
|
783
|
+
return {
|
|
784
|
+
ok: false,
|
|
785
|
+
error: '所有可用转换器都失败了(最后尝试:' + ((last && last.label) || '无') + ')\n' + detail,
|
|
786
|
+
attempts,
|
|
787
|
+
probe,
|
|
788
|
+
probeNotes: probe.notes
|
|
789
|
+
}
|
|
790
|
+
}
|
|
791
|
+
|
|
792
|
+
export function ensureDirSync(dir) {
|
|
793
|
+
fs.mkdirSync(dir, { recursive: true })
|
|
794
|
+
}
|
|
795
|
+
|
|
796
|
+
export const paths = { HERE, FALLBACK_PY, FALLBACK_NODE }
|