dsh-plugin-office-markdown 1.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/convert.js ADDED
@@ -0,0 +1,796 @@
1
+ /*!
2
+ * dsh-plugin-office-markdown — converter core (host half)
3
+ *
4
+ * Responsibilities:
5
+ * - classify a workspace path as Office/PDF (needs conversion) or plain text;
6
+ * - probe the converter chain in the user's required priority order:
7
+ * 1. `uvx markitdown` (no permanent install)
8
+ * 2. local `markitdown` CLI
9
+ * 3. `python -m markitdown`
10
+ * 4. bundled fallback (fallback.py, needs a local Python, limited fidelity)
11
+ * 5. bundled pure-Node fallback (fallback-node.js, NO Python, NO network)
12
+ * - run the conversion into a workspace temp directory, never touching the
13
+ * source file, and report honestly which converter produced the output.
14
+ *
15
+ * Pure Node standard library: no @deepseek-ai imports, no npm dependencies.
16
+ */
17
+ import { spawn } from 'node:child_process'
18
+ import fs from 'node:fs'
19
+ import path from 'node:path'
20
+ import { createHash } from 'node:crypto'
21
+ import { StringDecoder } from 'node:string_decoder'
22
+ import { fileURLToPath } from 'node:url'
23
+
24
+ import { convertFileNode, NODE_DEFAULTS } from './fallback-node.js'
25
+ import { runtimesRoot } from './paths.js'
26
+
27
+ const HERE = path.dirname(fileURLToPath(import.meta.url))
28
+ const FALLBACK_PY = path.join(HERE, 'fallback.py')
29
+ const FALLBACK_NODE = path.join(HERE, 'fallback-node.js')
30
+
31
+ /** Binary Office / PDF containers: converting them is what saves tokens. */
32
+ export const OFFICE_EXTS = Object.freeze({
33
+ '.docx': 'Word 文档',
34
+ '.doc': 'Word 97-2003 文档',
35
+ '.docm': 'Word 宏文档',
36
+ '.xlsx': 'Excel 工作簿',
37
+ '.xlsm': 'Excel 宏工作簿',
38
+ '.xls': 'Excel 97-2003 工作簿',
39
+ '.pptx': 'PowerPoint 演示文稿',
40
+ '.pptm': 'PowerPoint 宏演示文稿',
41
+ '.ppt': 'PowerPoint 97-2003 演示文稿',
42
+ '.pdf': 'PDF 文档',
43
+ '.odt': 'OpenDocument 文本',
44
+ '.ods': 'OpenDocument 表格',
45
+ '.odp': 'OpenDocument 演示',
46
+ '.rtf': 'RTF 文档',
47
+ '.epub': 'EPUB 电子书',
48
+ '.msg': 'Outlook 邮件',
49
+ '.ipynb': 'Jupyter Notebook'
50
+ })
51
+
52
+ /** Text-ish formats the model may read directly (no conversion needed). */
53
+ export const PLAIN_EXTS = new Set([
54
+ '.txt', '.md', '.markdown', '.mdx', '.csv', '.tsv', '.json', '.jsonl', '.ndjson',
55
+ '.yaml', '.yml', '.xml', '.html', '.htm', '.log', '.ini', '.cfg', '.toml',
56
+ '.py', '.js', '.mjs', '.cjs', '.ts', '.tsx', '.jsx', '.sql', '.rst', '.tex',
57
+ '.srt', '.vtt', '.gitignore', '.env', '.properties'
58
+ ])
59
+
60
+ const CONVERTIBLE = new Set(Object.keys(OFFICE_EXTS))
61
+
62
+ export function extOf(filePath) {
63
+ const base = path.basename(String(filePath || ''))
64
+ const i = base.lastIndexOf('.')
65
+ return i <= 0 ? '' : base.slice(i).toLowerCase()
66
+ }
67
+
68
+ /**
69
+ * @returns {{kind:'office'|'plain'|'unknown', ext:string, label:string, convertible:boolean}}
70
+ */
71
+ export function classify(filePath) {
72
+ const ext = extOf(filePath)
73
+ if (CONVERTIBLE.has(ext)) {
74
+ return { kind: 'office', ext, label: OFFICE_EXTS[ext], convertible: true }
75
+ }
76
+ if (PLAIN_EXTS.has(ext)) return { kind: 'plain', ext, label: '纯文本', convertible: false }
77
+ return { kind: 'unknown', ext, label: ext ? ext + ' 文件' : '未知类型', convertible: false }
78
+ }
79
+
80
+ /** Rough token estimate (CJK counts ~0.75 token/char, latin ~1 token/3.8 chars). */
81
+ export function estimateTokens(text) {
82
+ if (!text) return 0
83
+ let cjk = 0
84
+ let other = 0
85
+ for (let i = 0; i < text.length; i++) {
86
+ const c = text.charCodeAt(i)
87
+ if (c >= 0x2e80 && c <= 0x9fff) cjk++
88
+ else if (c >= 0xd800 && c <= 0xdfff) {
89
+ cjk++
90
+ i++
91
+ } else if (c === 10 || c === 13 || c === 32 || c === 9) other++
92
+ else other++
93
+ }
94
+ return Math.ceil(other / 3.8 + cjk * 0.75)
95
+ }
96
+
97
+ export function sha8(input) {
98
+ return createHash('sha1').update(String(input)).digest('hex').slice(0, 8)
99
+ }
100
+
101
+ export function safeBaseName(filePath) {
102
+ const base = path.basename(String(filePath || 'file'))
103
+ const i = base.lastIndexOf('.')
104
+ const stem = i > 0 ? base.slice(0, i) : base
105
+ const cleaned = stem.replace(/[^0-9A-Za-z\u4e00-\u9fff._-]+/g, '_').replace(/^_+|_+$/g, '')
106
+ return (cleaned || 'file').slice(0, 80)
107
+ }
108
+
109
+ /* ------------------------------------------------------------------ */
110
+ /* artifact provenance */
111
+ /* ------------------------------------------------------------------ */
112
+
113
+ /** Opening of the comment every artifact this plugin writes starts with. */
114
+ export const MARKER_PREFIX = '<!-- dsh-office-markdown '
115
+
116
+ const MARKER_RE = /^<!--\s*dsh-office-markdown\s+([^>]*?)-->\s*\n?/
117
+
118
+ /** Human labels for every converter, usable without probing the machine. */
119
+ export const CONVERTER_LABELS = Object.freeze({
120
+ uvx: 'uvx markitdown(临时运行,无需永久安装)',
121
+ 'markitdown-cli': 'markitdown 命令行',
122
+ cli: 'markitdown 命令行',
123
+ 'python-module': 'python -m markitdown',
124
+ python: 'python -m markitdown',
125
+ builtin: '插件内置 Python 兜底转换器(保真度有限)',
126
+ 'node-builtin': '插件内置 Node 兜底转换器(纯 Node,无需 Python 与网络,保真度有限)'
127
+ })
128
+
129
+ /** Label for a converter id recorded in an artifact. */
130
+ export function converterLabelFor(id, fallback) {
131
+ const key = String(id || '')
132
+ return CONVERTER_LABELS[key] || fallback || key || '未知转换器'
133
+ }
134
+
135
+ /**
136
+ * Stamp the artifact with what actually produced it.
137
+ *
138
+ * Without this a reuse can only guess: `index.js` used to hard-code
139
+ * `fidelity:"high"` for every cache hit, so a file really produced by the
140
+ * limited-fidelity Node fallback reported the same confidence as a real
141
+ * MarkItDown run. The stamp is a Markdown comment, so it never renders.
142
+ *
143
+ * @returns {number} the artifact's new size in bytes
144
+ */
145
+ export function writeFidelityMarker(dstPath, converterId, fidelity, extra) {
146
+ const body = fs.readFileSync(dstPath, 'utf8')
147
+ const attrs = {
148
+ converter: String(converterId || 'unknown'),
149
+ fidelity: String(fidelity || 'unknown'),
150
+ at: new Date().toISOString()
151
+ }
152
+ /* `srcbytes` / `srchash` describe the SOURCE file this artifact came out of.
153
+ * They are what lets a moved mtime be checked against the real content
154
+ * instead of paying for a re-conversion. Values must stay whitespace-free:
155
+ * the marker is parsed by splitting on spaces. */
156
+ for (const key of Object.keys(extra || {})) {
157
+ const value = String(extra[key] == null ? '' : extra[key]).replace(/\s+/g, '')
158
+ if (value) attrs[key] = value
159
+ }
160
+ const marker = MARKER_PREFIX + Object.keys(attrs).map((k) => k + '=' + attrs[k]).join(' ') + ' -->\n'
161
+ fs.writeFileSync(dstPath, marker + body, 'utf8')
162
+ return Buffer.byteLength(marker, 'utf8') + Buffer.byteLength(body, 'utf8')
163
+ }
164
+
165
+ /** Read the stamp back. `null` for an artifact written by an earlier version. */
166
+ export function readFidelityMarker(absPath) {
167
+ let fd = null
168
+ try {
169
+ fd = fs.openSync(absPath, 'r')
170
+ const buf = Buffer.alloc(512)
171
+ const n = fs.readSync(fd, buf, 0, buf.length, 0)
172
+ const m = buf.subarray(0, n).toString('utf8').match(MARKER_RE)
173
+ if (!m) return null
174
+ const attrs = {}
175
+ for (const part of String(m[1]).trim().split(/\s+/)) {
176
+ const i = part.indexOf('=')
177
+ if (i > 0) attrs[part.slice(0, i)] = part.slice(i + 1)
178
+ }
179
+ return {
180
+ ...attrs,
181
+ converter: attrs.converter || '',
182
+ fidelity: attrs.fidelity || '',
183
+ at: attrs.at || ''
184
+ }
185
+ } catch {
186
+ return null
187
+ } finally {
188
+ if (fd !== null) { try { fs.closeSync(fd) } catch { /* ignore */ } }
189
+ }
190
+ }
191
+
192
+ /* ------------------------------------------------------------------ */
193
+ /* artifact bookkeeping */
194
+ /* ------------------------------------------------------------------ */
195
+
196
+ /** `^<safeBaseName(src)>-[0-9a-f]{8}\.md$` — the only names this plugin writes. */
197
+ export function artifactNamePattern(srcAbs) {
198
+ const stem = safeBaseName(srcAbs).replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
199
+ return new RegExp('^' + stem + '-[0-9a-f]{8}\\.md$')
200
+ }
201
+
202
+ /**
203
+ * Every artifact this plugin ever wrote for `srcAbs` inside `tmpDir`.
204
+ *
205
+ * Scoped on purpose: only that one directory, only files (never directories),
206
+ * only names matching the plugin's own pattern, never recursive. The same rules
207
+ * back `action:"clean"` and the optional auto-prune, so there is exactly one
208
+ * definition of "an artifact of this source file" to audit.
209
+ *
210
+ * @returns {Array<{path:string, name:string, size:number, mtimeMs:number}>} newest first
211
+ */
212
+ export function listArtifacts(srcAbs, tmpDir) {
213
+ const out = []
214
+ let names = []
215
+ try { names = fs.readdirSync(tmpDir) } catch { return out }
216
+ const re = artifactNamePattern(srcAbs)
217
+ for (const file of names) {
218
+ if (!re.test(file)) continue
219
+ const abs = path.join(tmpDir, file)
220
+ let st = null
221
+ try { st = fs.statSync(abs) } catch { continue }
222
+ if (!st || !st.isFile()) continue
223
+ out.push({ path: abs, name: file, size: st.size, mtimeMs: st.mtimeMs })
224
+ }
225
+ out.sort((a, b) => b.mtimeMs - a.mtimeMs)
226
+ return out
227
+ }
228
+
229
+ /** The most recently written artifact for `srcAbs`, or `null`. */
230
+ export function findLatestArtifact(srcAbs, tmpDir) {
231
+ const list = listArtifacts(srcAbs, tmpDir)
232
+ return list.length ? list[0] : null
233
+ }
234
+
235
+ /* ------------------------------------------------------------------ */
236
+ /* artifact analysis */
237
+ /* ------------------------------------------------------------------ */
238
+
239
+ const ANALYZE_CHUNK = 262144
240
+
241
+ /**
242
+ * Walk an artifact once, reporting its real byte size, line count and a token
243
+ * estimate — without ever materialising it as one string.
244
+ *
245
+ * `preview:0` (the default) used to `readFileSync` the whole Markdown anyway and
246
+ * then run a per-character loop over it: for a 50 MB workbook that is tens of
247
+ * megabytes of transient string on top of the file's own cost, on the default
248
+ * path, for a value the caller had just decided not to return. The estimate is
249
+ * the same one {@link estimateTokens} makes; only the retention changed.
250
+ *
251
+ * @returns {{bytes:number, lines:number, tokens:number, head:string, error:string}}
252
+ */
253
+ export function analyzeMarkdown(absPath, options = {}) {
254
+ const headChars = Math.max(0, Number(options.headChars) || 0)
255
+ const out = { bytes: 0, lines: 0, tokens: 0, head: '', error: '' }
256
+ let cjk = 0
257
+ let other = 0
258
+ let lines = 1
259
+ const count = (s) => {
260
+ for (let i = 0; i < s.length; i++) {
261
+ const c = s.charCodeAt(i)
262
+ if (c === 10) lines++
263
+ if (c >= 0x2e80 && c <= 0x9fff) cjk++
264
+ else if (c >= 0xd800 && c <= 0xdfff) { cjk++; i++ }
265
+ else other++
266
+ }
267
+ }
268
+ const take = (s) => {
269
+ if (headChars > 0 && out.head.length < headChars) out.head += s.slice(0, headChars - out.head.length)
270
+ }
271
+ let fd = null
272
+ try {
273
+ fd = fs.openSync(absPath, 'r')
274
+ const buf = Buffer.alloc(ANALYZE_CHUNK)
275
+ const dec = new StringDecoder('utf8')
276
+ for (;;) {
277
+ const n = fs.readSync(fd, buf, 0, buf.length, null)
278
+ if (n <= 0) break
279
+ out.bytes += n
280
+ const s = dec.write(buf.subarray(0, n))
281
+ take(s)
282
+ count(s)
283
+ }
284
+ const tail = dec.end()
285
+ if (tail) { take(tail); count(tail) }
286
+ out.lines = lines
287
+ out.tokens = Math.ceil(other / 3.8 + cjk * 0.75)
288
+ } catch (error) {
289
+ out.error = String((error && error.message) || error)
290
+ } finally {
291
+ if (fd !== null) { try { fs.closeSync(fd) } catch { /* ignore */ } }
292
+ }
293
+ return out
294
+ }
295
+
296
+ /**
297
+ * Streaming content hash of a source file.
298
+ *
299
+ * The cache key ends in the source's mtime, which `git checkout`, a copy or a
300
+ * restore changes without touching the content. Comparing this hash against the
301
+ * one recorded in an artifact is what tells "really edited" apart from "same
302
+ * bytes, new timestamp". Returns `''` when the file cannot be read.
303
+ */
304
+ export function sourceContentHash(absPath) {
305
+ let fd = null
306
+ try {
307
+ fd = fs.openSync(absPath, 'r')
308
+ const buf = Buffer.alloc(ANALYZE_CHUNK)
309
+ const hash = createHash('sha1')
310
+ for (;;) {
311
+ const n = fs.readSync(fd, buf, 0, buf.length, null)
312
+ if (n <= 0) break
313
+ hash.update(buf.subarray(0, n))
314
+ }
315
+ return hash.digest('hex').slice(0, 16)
316
+ } catch {
317
+ return ''
318
+ } finally {
319
+ if (fd !== null) { try { fs.closeSync(fd) } catch { /* ignore */ } }
320
+ }
321
+ }
322
+
323
+ /* ------------------------------------------------------------------ */
324
+ /* subprocess */
325
+ /* ------------------------------------------------------------------ */
326
+
327
+ const MAX_STDOUT = 262144
328
+ const MAX_STDERR = 65536
329
+
330
+ /**
331
+ * Run one child process. Never rejects: every failure is reported as a value
332
+ * so the caller can fall through the converter chain.
333
+ */
334
+ /**
335
+ * Kill a child process **and everything it spawned**.
336
+ *
337
+ * `child.kill()` on Windows is `TerminateProcess`, which stops only the process
338
+ * it is called on: `uvx` and the `python` it starts keep running, keep the
339
+ * output file open, and can even write into it after the caller has already
340
+ * declared the conversion dead. `taskkill /T /F` walks the whole tree. On POSIX
341
+ * the child is spawned into its own process group (`detached`) so the group can
342
+ * be signalled at once.
343
+ *
344
+ * Resolves once the tree is gone, or after `graceMs` at the latest, so the
345
+ * caller can safely delete and re-create the output file afterwards.
346
+ */
347
+ function killTree(child, graceMs = 2000) {
348
+ return new Promise((resolve) => {
349
+ let done = false
350
+ const settle = () => { if (!done) { done = true; clearTimeout(timer); resolve() } }
351
+ if (!child || typeof child.pid !== 'number') { resolve(); return }
352
+ const timer = setTimeout(settle, graceMs)
353
+ if (timer && typeof timer.unref === 'function') timer.unref()
354
+ if (process.platform === 'win32') {
355
+ try {
356
+ const killer = spawn('taskkill', ['/pid', String(child.pid), '/T', '/F'], { windowsHide: true, stdio: 'ignore' })
357
+ killer.on('error', () => { try { child.kill() } catch { /* ignore */ } ; settle() })
358
+ killer.on('close', settle)
359
+ } catch {
360
+ try { child.kill() } catch { /* ignore */ }
361
+ settle()
362
+ }
363
+ return
364
+ }
365
+ try { process.kill(-child.pid, 'SIGKILL') } catch { try { child.kill('SIGKILL') } catch { /* ignore */ } }
366
+ settle()
367
+ })
368
+ }
369
+
370
+ export function run(cmd, args, options = {}) {
371
+ const timeoutMs = Number.isFinite(options.timeoutMs) ? options.timeoutMs : 300000
372
+ const signal = options.signal
373
+ return new Promise((resolve) => {
374
+ if (signal && signal.aborted) {
375
+ resolve({ ok: false, code: -1, stdout: '', stderr: 'aborted before start', aborted: true })
376
+ return
377
+ }
378
+ let child
379
+ try {
380
+ child = spawn(cmd, args, {
381
+ cwd: options.cwd,
382
+ windowsHide: true,
383
+ shell: false,
384
+ /* Windows never gets `detached`: there the tree is taken down with
385
+ * `taskkill /T /F`, and a detached console child would only be harder
386
+ * to reach. */
387
+ detached: process.platform !== 'win32',
388
+ env: options.env ? { ...process.env, ...options.env } : process.env,
389
+ stdio: ['ignore', 'pipe', 'pipe']
390
+ })
391
+ } catch (error) {
392
+ resolve({ ok: false, code: -1, stdout: '', stderr: String((error && error.message) || error), spawnError: true })
393
+ return
394
+ }
395
+ let stdout = ''
396
+ let stderr = ''
397
+ let timedOut = false
398
+ let settled = false
399
+ let killPromise = null
400
+ const treeKill = () => {
401
+ if (!killPromise) killPromise = killTree(child)
402
+ return killPromise
403
+ }
404
+ const timer = setTimeout(() => {
405
+ timedOut = true
406
+ void treeKill()
407
+ }, timeoutMs)
408
+ const onAbort = () => { void treeKill() }
409
+ if (signal && typeof signal.addEventListener === 'function') signal.addEventListener('abort', onAbort, { once: true })
410
+
411
+ const finish = (result) => {
412
+ if (settled) return
413
+ settled = true
414
+ clearTimeout(timer)
415
+ if (signal && typeof signal.removeEventListener === 'function') signal.removeEventListener('abort', onAbort)
416
+ /* A timed-out or cancelled run must not report back before the tree is
417
+ * really gone: the caller deletes and re-creates the output file next. */
418
+ if (killPromise) killPromise.then(() => resolve(result))
419
+ else resolve(result)
420
+ }
421
+
422
+ /* `StringDecoder` holds the bytes of a multi-byte character together when
423
+ * one arrives split across two chunks. The old per-chunk `toString('utf8')`
424
+ * turned every such character into U+FFFD, and that text is what the failure
425
+ * message handed to the model is built from. Every chunk must be written
426
+ * through the decoder, even after the capture cap is reached, or the decoder
427
+ * loses track of where it was. */
428
+ const outDec = new StringDecoder('utf8')
429
+ const errDec = new StringDecoder('utf8')
430
+ if (child.stdout) child.stdout.on('data', (d) => {
431
+ const s = outDec.write(d)
432
+ if (stdout.length < MAX_STDOUT) stdout += s
433
+ })
434
+ if (child.stderr) child.stderr.on('data', (d) => {
435
+ const s = errDec.write(d)
436
+ if (stderr.length < MAX_STDERR) stderr += s
437
+ })
438
+ child.on('error', (error) => finish({ ok: false, code: -1, stdout, stderr: stderr + '\n' + String((error && error.message) || error), spawnError: true }))
439
+ child.on('close', (code) => finish({ ok: !timedOut && code === 0, code, stdout, stderr, timedOut, aborted: !!(signal && signal.aborted) }))
440
+ })
441
+ }
442
+
443
+ /* ------------------------------------------------------------------ */
444
+ /* converter probing */
445
+ /* ------------------------------------------------------------------ */
446
+
447
+ function discoverBundledPythons() {
448
+ const out = []
449
+ try {
450
+ const root = runtimesRoot()
451
+ for (const name of fs.readdirSync(root)) {
452
+ const p = path.join(root, name, 'dependencies', 'python', 'python.exe')
453
+ if (fs.existsSync(p)) out.push(p)
454
+ const p2 = path.join(root, name, 'dependencies', 'python', 'bin', 'python3')
455
+ if (fs.existsSync(p2)) out.push(p2)
456
+ }
457
+ } catch { /* no bundled runtime */ }
458
+ return out
459
+ }
460
+
461
+ /** Chinese label for where a Python candidate came from. */
462
+ export function pythonSourceLabel(source) {
463
+ if (source === 'config') return '配置 pythonPath'
464
+ if (source === 'bundled') return 'DSH 自带运行时'
465
+ return '系统 PATH'
466
+ }
467
+
468
+ export const PYTHON_PREFERS = ['auto', 'bundled', 'system', 'config']
469
+
470
+ /**
471
+ * Ordered list of Python interpreters to try, honouring `cfg.pythonPrefer`.
472
+ * Each entry carries its `source` so `action:"status"` can explain the order.
473
+ * @returns {Array<{cmd: string, prefix: string[], source: string}>}
474
+ */
475
+ export function pythonCandidates(cfg) {
476
+ const out = []
477
+ const seen = new Set()
478
+ const push = (cmd, prefix, source) => {
479
+ if (!cmd) return
480
+ const key = cmd + ' ' + prefix.join(' ')
481
+ if (seen.has(key)) return
482
+ seen.add(key)
483
+ out.push({ cmd, prefix, source: source || 'system' })
484
+ }
485
+
486
+ const configured = cfg && cfg.pythonPath ? String(cfg.pythonPath) : ''
487
+ const bundled = discoverBundledPythons()
488
+ const system = [['python', []], ['python3', []], ['py', ['-3']]]
489
+
490
+ let prefer = String((cfg && cfg.pythonPrefer) || 'auto').toLowerCase()
491
+ if (!PYTHON_PREFERS.includes(prefer)) prefer = 'auto'
492
+ /* A mode that cannot be satisfied degrades to the full order rather than
493
+ * leaving the plugin with no interpreter at all. */
494
+ if (prefer === 'config' && !configured) prefer = 'auto'
495
+ if (prefer === 'bundled' && bundled.length === 0) prefer = 'auto'
496
+
497
+ if (prefer === 'config') {
498
+ push(configured, [], 'config')
499
+ return out
500
+ }
501
+ if (prefer === 'bundled') {
502
+ for (const p of bundled) push(p, [], 'bundled')
503
+ return out
504
+ }
505
+ if (prefer === 'system') {
506
+ for (const [cmd, prefix] of system) push(cmd, prefix, 'system')
507
+ for (const p of bundled) push(p, [], 'bundled')
508
+ if (configured) push(configured, [], 'config')
509
+ return out
510
+ }
511
+ if (configured) push(configured, [], 'config')
512
+ for (const p of bundled) push(p, [], 'bundled')
513
+ for (const [cmd, prefix] of system) push(cmd, prefix, 'system')
514
+ return out
515
+ }
516
+
517
+ let probeCache = { at: 0, key: '', value: null }
518
+
519
+ /**
520
+ * Probe the converter chain once (cached for `cfg.probeTtlMs`).
521
+ * @returns {Promise<{chain: Array<object>, notes: string[]}>}
522
+ */
523
+ export async function probeConverters(cfg, options = {}) {
524
+ const key = JSON.stringify([
525
+ cfg.pythonPath || '',
526
+ cfg.pythonPrefer || 'auto',
527
+ cfg.uvxExtras || '',
528
+ cfg.allowUvxDownload !== false,
529
+ cfg.fallbackEnabled !== false
530
+ ])
531
+ const ttl = Number.isFinite(cfg.probeTtlMs) ? cfg.probeTtlMs : 600000
532
+ if (!options.force && probeCache.value && probeCache.key === key && Date.now() - probeCache.at < ttl) {
533
+ return probeCache.value
534
+ }
535
+
536
+ const chain = []
537
+ const notes = []
538
+ const probeTimeout = 45000
539
+
540
+ const uvx = await run('uvx', ['--version'], { timeoutMs: probeTimeout, signal: options.signal })
541
+ if (uvx.ok) {
542
+ chain.push({
543
+ id: 'uvx',
544
+ label: 'uvx markitdown(临时运行,无需永久安装)',
545
+ fidelity: 'high',
546
+ kind: 'uvx'
547
+ })
548
+ } else {
549
+ notes.push('未检测到 uvx(' + firstLine(uvx.stderr) + ')')
550
+ }
551
+
552
+ const cli = await run('markitdown', ['--help'], { timeoutMs: probeTimeout, signal: options.signal })
553
+ if (cli.ok) {
554
+ chain.push({
555
+ id: 'markitdown-cli',
556
+ label: 'markitdown 命令行',
557
+ fidelity: 'high',
558
+ kind: 'cli'
559
+ })
560
+ } else {
561
+ notes.push('未检测到 markitdown 命令(' + firstLine(cli.stderr) + ')')
562
+ }
563
+
564
+ let pythonHit = null
565
+ const pythonProbes = []
566
+ const candidates = pythonCandidates(cfg)
567
+ /* `force` (which is what `action:"status"` passes) audits every candidate so
568
+ * the user can see which interpreter actually holds markitdown; a normal
569
+ * conversion stops at the first hit to stay fast. */
570
+ const probeAll = options.probeAll === true || options.force === true
571
+ for (const cand of candidates) {
572
+ const r = await run(cand.cmd, [...cand.prefix, '-m', 'markitdown', '--help'], { timeoutMs: probeTimeout, signal: options.signal })
573
+ pythonProbes.push({ cmd: cand.cmd, source: cand.source, ok: !!r.ok, reason: r.ok ? '' : firstLine(r.stderr) })
574
+ if (r.ok) {
575
+ if (!pythonHit) {
576
+ pythonHit = { id: 'python-module', label: 'python -m markitdown(' + cand.cmd + ')', fidelity: 'high', kind: 'python', python: cand }
577
+ }
578
+ if (!probeAll) break
579
+ }
580
+ }
581
+ if (pythonHit) chain.push(pythonHit)
582
+ else if (candidates.length) notes.push('未检测到已安装 markitdown 的 Python 解释器(已检查 ' + candidates.length + ' 个候选,明细见下方)')
583
+ else notes.push('没有可用的 Python 解释器候选(pythonPrefer=' + String(cfg.pythonPrefer || 'auto') + ')')
584
+
585
+ if (cfg.fallbackEnabled !== false) {
586
+ const py = pythonHit && pythonHit.python ? pythonHit.python : pythonCandidates(cfg)[0]
587
+ if (py) {
588
+ chain.push({
589
+ id: 'builtin',
590
+ label: '插件内置 Python 兜底转换器(保真度有限)',
591
+ fidelity: 'limited',
592
+ kind: 'builtin',
593
+ python: py
594
+ })
595
+ } else {
596
+ notes.push('未检测到 Python 解释器,已跳过 Python 兜底转换器(不影响 Node 兜底)')
597
+ }
598
+ chain.push({
599
+ id: 'node-builtin',
600
+ label: '插件内置 Node 兜底转换器(纯 Node,无需 Python 与网络,保真度有限)',
601
+ fidelity: 'limited',
602
+ kind: 'inline'
603
+ })
604
+ }
605
+
606
+ const value = {
607
+ chain,
608
+ notes,
609
+ pythonProbes,
610
+ pythonPrefer: String(cfg.pythonPrefer || 'auto'),
611
+ probedAt: Date.now()
612
+ }
613
+ probeCache = { at: Date.now(), key, value }
614
+ if (options.force) probeCache.value = value
615
+ return value
616
+ }
617
+
618
+ function firstLine(text) {
619
+ const s = String(text || '').replace(/\r/g, '').trim()
620
+ const line = s.split('\n').filter(Boolean).pop() || s.split('\n')[0] || ''
621
+ return line.slice(0, 160)
622
+ }
623
+
624
+ /* ------------------------------------------------------------------ */
625
+ /* conversion */
626
+ /* ------------------------------------------------------------------ */
627
+
628
+ function buildSteps(converter, src, dst, cfg) {
629
+ const extras = cfg.uvxExtras || 'markitdown[all]'
630
+ switch (converter.kind) {
631
+ case 'uvx':
632
+ return [
633
+ ['uvx', ['--from', extras, 'markitdown', src, '-o', dst]],
634
+ ['uvx', [extras, src, '-o', dst]]
635
+ ]
636
+ case 'cli':
637
+ return [['markitdown', [src, '-o', dst]]]
638
+ case 'python':
639
+ return [[converter.python.cmd, [...converter.python.prefix, '-m', 'markitdown', src, '-o', dst]]]
640
+ case 'builtin':
641
+ return [[
642
+ converter.python ? converter.python.cmd : 'python',
643
+ [
644
+ ...(converter.python ? converter.python.prefix : []),
645
+ FALLBACK_PY,
646
+ '--input', src,
647
+ '--output', dst,
648
+ '--max-rows', String(cfg.maxRowsPerSheet || 400),
649
+ '--max-cols', String(cfg.maxTableCols || 24),
650
+ '--max-cells', String(cfg.maxCellsPerSheet || 20000)
651
+ ]
652
+ ]]
653
+ default:
654
+ return []
655
+ }
656
+ }
657
+
658
+ async function runSteps(steps, options) {
659
+ const failures = []
660
+ for (const [cmd, args] of steps) {
661
+ const r = await run(cmd, args, options)
662
+ if (r.ok) return { ok: true, failure: null, failures }
663
+ failures.push({ cmd: cmd + ' ' + args.join(' '), reason: r.timedOut ? '超时' : firstLine(r.stderr) || ('退出码 ' + r.code) })
664
+ if (r.aborted) break
665
+ }
666
+ return { ok: false, failures }
667
+ }
668
+
669
+ /** The pure-Node fallback runs in-process: no Python, no child process, no network. */
670
+ function runInline(converter, srcPath, dstPath, cfg) {
671
+ try { fs.rmSync(dstPath, { force: true }) } catch { /* ignore */ }
672
+ convertFileNode(srcPath, dstPath, {
673
+ maxRowsPerSheet: cfg.maxRowsPerSheet || NODE_DEFAULTS.maxRowsPerSheet,
674
+ maxTableCols: cfg.maxTableCols || NODE_DEFAULTS.maxTableCols,
675
+ maxCellsPerSheet: cfg.maxCellsPerSheet || NODE_DEFAULTS.maxCellsPerSheet
676
+ })
677
+ const stat = fs.statSync(dstPath)
678
+ if (!stat || stat.size === 0) throw new Error('转换结果为空文件')
679
+ return { ok: true, dstBytes: stat.size }
680
+ }
681
+
682
+ /**
683
+ * Write the provenance stamp and report the artifact's final size.
684
+ *
685
+ * A stamp that cannot be written is not a conversion failure — the Markdown
686
+ * itself is fine — so the size is still reported and only the reuse path loses
687
+ * its ability to say who produced the file.
688
+ */
689
+ function stampArtifact(dstPath, converter, srcPath) {
690
+ let extra = null
691
+ try {
692
+ const st = fs.statSync(srcPath)
693
+ extra = { srcbytes: st.size, srchash: sourceContentHash(srcPath) || 'unknown' }
694
+ } catch {
695
+ extra = null
696
+ }
697
+ try {
698
+ return writeFidelityMarker(dstPath, converter.id, converter.fidelity, extra)
699
+ } catch {
700
+ try { return fs.statSync(dstPath).size } catch { return 0 }
701
+ }
702
+ }
703
+
704
+ /**
705
+ * Convert one file to Markdown.
706
+ *
707
+ * @returns {Promise<object>} a value object; `ok:false` carries a human error.
708
+ */
709
+ export async function convertFile(ctx) {
710
+ const { srcPath, dstPath, cfg, signal, force } = ctx
711
+ const intended = cfg.converter && cfg.converter !== 'auto' ? cfg.converter : null
712
+
713
+ const probe = await probeConverters(cfg, { signal, force: !!force })
714
+ let chain = probe.chain
715
+ if (intended) {
716
+ const wanted = chain.filter((c) => c.kind === intended || c.id === intended)
717
+ if (wanted.length) chain = wanted
718
+ }
719
+ if (!chain.length) {
720
+ return { ok: false, error: '没有任何可用的转换器:uvx / markitdown 命令 / python -m markitdown / 内置兜底 全部不可用。', probe }
721
+ }
722
+
723
+ const attempts = []
724
+ for (const converter of chain) {
725
+ if (converter.kind === 'inline') {
726
+ try {
727
+ runInline(converter, srcPath, dstPath, cfg)
728
+ return {
729
+ ok: true,
730
+ converter: converter.id,
731
+ converterLabel: converter.label,
732
+ fidelity: converter.fidelity,
733
+ converterNote: '以上为插件内置 Node 兜底转换(纯 Node,未使用 MarkItDown),版式、图表、批注、图片、公式等可能缺失。',
734
+ attempts,
735
+ probeNotes: probe.notes,
736
+ dstBytes: stampArtifact(dstPath, converter, srcPath)
737
+ }
738
+ } catch (error) {
739
+ attempts.push({
740
+ converter: converter.id,
741
+ label: converter.label,
742
+ failures: [{ cmd: converter.label, reason: firstLine(String((error && error.message) || error)) }]
743
+ })
744
+ continue
745
+ }
746
+ }
747
+
748
+ const steps = buildSteps(converter, srcPath, dstPath, cfg)
749
+ if (!steps.length) continue
750
+ try { fs.rmSync(dstPath, { force: true }) } catch { /* ignore */ }
751
+ const result = await runSteps(steps, { timeoutMs: cfg.timeoutMs, signal })
752
+ if (!result.ok) {
753
+ attempts.push({ converter: converter.id, label: converter.label, failures: result.failures })
754
+ if (signal && signal.aborted) {
755
+ return { ok: false, error: '转换已取消。', attempts, probe, aborted: true }
756
+ }
757
+ continue
758
+ }
759
+ let stat = null
760
+ try { stat = fs.statSync(dstPath) } catch { /* ignore */ }
761
+ if (!stat || stat.size === 0) {
762
+ attempts.push({ converter: converter.id, label: converter.label, failures: [{ cmd: converter.label, reason: '转换结果为空文件' }] })
763
+ continue
764
+ }
765
+ return {
766
+ ok: true,
767
+ converter: converter.id,
768
+ converterLabel: converter.label,
769
+ fidelity: converter.fidelity,
770
+ converterNote: converter.kind === 'builtin'
771
+ ? '以上为插件内置兜底转换(未使用 MarkItDown),版式、图表、批注、扫描件 OCR 等信息可能缺失。'
772
+ : null,
773
+ attempts,
774
+ probeNotes: probe.notes,
775
+ dstBytes: stampArtifact(dstPath, converter, srcPath)
776
+ }
777
+ }
778
+
779
+ const last = attempts[attempts.length - 1]
780
+ const detail = last && last.failures && last.failures.length
781
+ ? last.failures.map((f) => ' - ' + f.cmd + ' → ' + f.reason).join('\n')
782
+ : ' - 未知原因'
783
+ return {
784
+ ok: false,
785
+ error: '所有可用转换器都失败了(最后尝试:' + ((last && last.label) || '无') + ')\n' + detail,
786
+ attempts,
787
+ probe,
788
+ probeNotes: probe.notes
789
+ }
790
+ }
791
+
792
+ export function ensureDirSync(dir) {
793
+ fs.mkdirSync(dir, { recursive: true })
794
+ }
795
+
796
+ export const paths = { HERE, FALLBACK_PY, FALLBACK_NODE }