dsh-plugin-office-markdown 1.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.en.md +156 -0
- package/README.md +144 -0
- package/cordis.patch.yml +35 -0
- package/lib/client.js +589 -0
- package/lib/convert.js +796 -0
- package/lib/env.js +602 -0
- package/lib/fallback-node.js +334 -0
- package/lib/fallback.py +579 -0
- package/lib/index.js +1443 -0
- package/lib/paths.js +76 -0
- package/lib/removal-watchdog.js +359 -0
- package/lib/settings-api.js +381 -0
- package/package.json +55 -0
|
@@ -0,0 +1,334 @@
|
|
|
1
|
+
/*!
|
|
2
|
+
* dsh-plugin-office-markdown — 纯 Node 兜底转换器
|
|
3
|
+
*
|
|
4
|
+
* 不依赖 Python、不联网、不调用任何 API:用 Node 内置的 zlib 解开 OOXML
|
|
5
|
+
* (.docx/.xlsx/.pptx 本质上是 zip + xml),再按 XML 结构抽取文字与表格,
|
|
6
|
+
* 生成 Markdown。保真度有限(无版式/图表/批注/图片/公式/OCR),但保证在
|
|
7
|
+
* “目标电脑既没有 Python 也没有网络”时仍能把文件读成 Markdown。
|
|
8
|
+
*
|
|
9
|
+
* 既可被 lib/convert.js 直接 import 调用,也可作为 CLI 使用:
|
|
10
|
+
* node fallback-node.js --input a.xlsx --output out.md
|
|
11
|
+
*/
|
|
12
|
+
import fs from 'node:fs'
|
|
13
|
+
import path from 'node:path'
|
|
14
|
+
import zlib from 'node:zlib'
|
|
15
|
+
import { pathToFileURL } from 'node:url'
|
|
16
|
+
|
|
17
|
+
export const NODE_SUPPORTED = new Set(['.docx', '.docm', '.xlsx', '.xlsm', '.pptx', '.pptm'])
|
|
18
|
+
|
|
19
|
+
const UNSUPPORTED = {
|
|
20
|
+
'.pdf': 'Node 兜底不支持 PDF(PDF 需要 MarkItDown 或本机 Python 兜底)。',
|
|
21
|
+
'.doc': '旧版二进制 .doc 需要 MarkItDown 转换,Node 兜底不支持。',
|
|
22
|
+
'.xls': '旧版二进制 .xls 需要 MarkItDown 转换,Node 兜底不支持。',
|
|
23
|
+
'.ppt': '旧版二进制 .ppt 需要 MarkItDown 转换,Node 兜底不支持。',
|
|
24
|
+
'.msg': 'Outlook .msg 需要 MarkItDown 转换,Node 兜底不支持。',
|
|
25
|
+
'.epub': 'EPUB 需要 MarkItDown 转换,Node 兜底不支持。',
|
|
26
|
+
'.odt': 'ODF/OpenDocument 格式需要 MarkItDown 转换,Node 兜底不支持。',
|
|
27
|
+
'.ods': 'ODF/OpenDocument 格式需要 MarkItDown 转换,Node 兜底不支持。',
|
|
28
|
+
'.odp': 'ODF/OpenDocument 格式需要 MarkItDown 转换,Node 兜底不支持。'
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/* ---------- zip ---------- */
|
|
32
|
+
|
|
33
|
+
const SIG_EOCD = 0x06054b50
|
|
34
|
+
const SIG_CENTRAL = 0x02014b50
|
|
35
|
+
|
|
36
|
+
function readZip(file) {
|
|
37
|
+
const buf = fs.readFileSync(file)
|
|
38
|
+
let eocd = -1
|
|
39
|
+
const floor = Math.max(0, buf.length - 66000)
|
|
40
|
+
for (let i = buf.length - 22; i >= floor; i--) {
|
|
41
|
+
if (buf.readUInt32LE(i) === SIG_EOCD) { eocd = i; break }
|
|
42
|
+
}
|
|
43
|
+
if (eocd < 0) throw new Error('无法解析为 zip 容器(文件可能已损坏或不是真正的 OOXML 文件)')
|
|
44
|
+
const count = buf.readUInt16LE(eocd + 10)
|
|
45
|
+
let off = buf.readUInt32LE(eocd + 16)
|
|
46
|
+
const entries = new Map()
|
|
47
|
+
for (let i = 0; i < count; i++) {
|
|
48
|
+
if (off + 46 > buf.length || buf.readUInt32LE(off) !== SIG_CENTRAL) break
|
|
49
|
+
const method = buf.readUInt16LE(off + 10)
|
|
50
|
+
const csize = buf.readUInt32LE(off + 20)
|
|
51
|
+
const nlen = buf.readUInt16LE(off + 28)
|
|
52
|
+
const elen = buf.readUInt16LE(off + 30)
|
|
53
|
+
const clen = buf.readUInt16LE(off + 32)
|
|
54
|
+
const lho = buf.readUInt32LE(off + 42)
|
|
55
|
+
const name = buf.toString('utf8', off + 46, off + 46 + nlen)
|
|
56
|
+
entries.set(name, { method, csize, lho })
|
|
57
|
+
off += 46 + nlen + elen + clen
|
|
58
|
+
}
|
|
59
|
+
return { buf, entries }
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function readEntry(buf, entry) {
|
|
63
|
+
const nlen = buf.readUInt16LE(entry.lho + 26)
|
|
64
|
+
const elen = buf.readUInt16LE(entry.lho + 28)
|
|
65
|
+
const start = entry.lho + 30 + nlen + elen
|
|
66
|
+
const raw = buf.subarray(start, start + entry.csize)
|
|
67
|
+
if (entry.method === 0) return raw
|
|
68
|
+
if (entry.method === 8) return zlib.inflateRawSync(raw)
|
|
69
|
+
throw new Error('zip 压缩方式 ' + entry.method + ' 不受支持')
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/* ---------- xml helpers ---------- */
|
|
73
|
+
|
|
74
|
+
function unescapeXml(s) {
|
|
75
|
+
return String(s)
|
|
76
|
+
.replace(/&#x([0-9a-fA-F]+);/g, (_, h) => String.fromCodePoint(parseInt(h, 16)))
|
|
77
|
+
.replace(/&#(\d+);/g, (_, d) => String.fromCodePoint(Number(d)))
|
|
78
|
+
.replace(/</g, '<')
|
|
79
|
+
.replace(/>/g, '>')
|
|
80
|
+
.replace(/"/g, '"')
|
|
81
|
+
.replace(/'/g, "'")
|
|
82
|
+
.replace(/&/g, '&')
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
const TEXT_RE = /<(?:[A-Za-z0-9_]+:)?t(?:\s[^>]*)?>([\s\S]*?)<\/(?:[A-Za-z0-9_]+:)?t>/g
|
|
86
|
+
|
|
87
|
+
function texts(xml) {
|
|
88
|
+
let out = ''
|
|
89
|
+
for (const m of String(xml).matchAll(TEXT_RE)) out += unescapeXml(m[1])
|
|
90
|
+
return out
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
function cell(text) {
|
|
94
|
+
return String(text).replace(/\|/g, '\\|').replace(/\s+/g, ' ').trim()
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
function mdTable(rows, opt) {
|
|
98
|
+
if (!rows.length) return ''
|
|
99
|
+
const width = Math.min(opt.maxTableCols, Math.max(...rows.map((r) => r.length)))
|
|
100
|
+
if (width <= 0) return ''
|
|
101
|
+
const lines = []
|
|
102
|
+
const limit = Math.min(rows.length, opt.maxRowsPerSheet)
|
|
103
|
+
for (let i = 0; i < limit; i++) {
|
|
104
|
+
const cells = []
|
|
105
|
+
for (let c = 0; c < width; c++) cells.push(rows[i][c] === undefined ? '' : cell(rows[i][c]))
|
|
106
|
+
lines.push('| ' + cells.join(' | ') + ' |')
|
|
107
|
+
if (i === 0) lines.push('| ' + cells.map(() => '---').join(' | ') + ' |')
|
|
108
|
+
}
|
|
109
|
+
if (rows.length > limit) lines.push('', `> 表格已截断:共 ${rows.length} 行,只显示前 ${limit} 行。`)
|
|
110
|
+
return lines.join('\n')
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/* ---------- docx ---------- */
|
|
114
|
+
|
|
115
|
+
function paraToMd(chunk) {
|
|
116
|
+
const text = texts(chunk).replace(/\s+$/, '')
|
|
117
|
+
if (!text.trim()) return ''
|
|
118
|
+
const style = (chunk.match(/<w:pStyle[^>]*w:val="([^"]*)"/) || [])[1] || ''
|
|
119
|
+
const heading = style.match(/^(?:Heading|heading|标题)\s*(\d+)$/)
|
|
120
|
+
if (heading) return '#'.repeat(Math.min(6, Number(heading[1]) || 1)) + ' ' + text
|
|
121
|
+
if (/^(?:Title|标题|Subtitle)$/i.test(style)) return '# ' + text
|
|
122
|
+
if (/<w:numPr\b/.test(chunk)) return '- ' + text
|
|
123
|
+
return text
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
function docxTableToMd(chunk, opt) {
|
|
127
|
+
const rows = []
|
|
128
|
+
for (const r of chunk.matchAll(/<w:tr\b[^>]*>([\s\S]*?)<\/w:tr>/g)) {
|
|
129
|
+
const cells = []
|
|
130
|
+
for (const c of r[1].matchAll(/<w:tc\b[^>]*>([\s\S]*?)<\/w:tc>/g)) cells.push(texts(c[1]))
|
|
131
|
+
if (cells.length) rows.push(cells)
|
|
132
|
+
}
|
|
133
|
+
return mdTable(rows, opt)
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
function docxToMarkdown(buf, entries, opt) {
|
|
137
|
+
const entry = entries.get('word/document.xml')
|
|
138
|
+
if (!entry) throw new Error('缺少 word/document.xml,可能不是有效的 .docx')
|
|
139
|
+
const xml = readEntry(buf, entry).toString('utf8')
|
|
140
|
+
const blockRe = /<w:tbl\b[\s\S]*?<\/w:tbl>|<w:p\b[^>]*\/>|<w:p\b[^>]*>[\s\S]*?<\/w:p>/g
|
|
141
|
+
const parts = []
|
|
142
|
+
for (const m of xml.matchAll(blockRe)) {
|
|
143
|
+
const chunk = m[0]
|
|
144
|
+
const piece = chunk.startsWith('<w:tbl') ? docxTableToMd(chunk, opt) : paraToMd(chunk)
|
|
145
|
+
if (piece && piece.trim()) parts.push(piece)
|
|
146
|
+
}
|
|
147
|
+
return parts.join('\n\n')
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/* ---------- xlsx ---------- */
|
|
151
|
+
|
|
152
|
+
function colIndex(ref) {
|
|
153
|
+
let n = 0
|
|
154
|
+
for (const ch of ref) {
|
|
155
|
+
const c = ch.charCodeAt(0)
|
|
156
|
+
if (c < 65 || c > 90) break
|
|
157
|
+
n = n * 26 + (c - 64)
|
|
158
|
+
}
|
|
159
|
+
return n - 1
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
function sheetOrder(buf, entries) {
|
|
163
|
+
const rels = {}
|
|
164
|
+
const relEntry = entries.get('xl/_rels/workbook.xml.rels')
|
|
165
|
+
if (relEntry) {
|
|
166
|
+
const xml = readEntry(buf, relEntry).toString('utf8')
|
|
167
|
+
for (const m of xml.matchAll(/<Relationship\b[^>]*>/g)) {
|
|
168
|
+
const id = (m[0].match(/Id="([^"]*)"/) || [])[1]
|
|
169
|
+
const target = (m[0].match(/Target="([^"]*)"/) || [])[1]
|
|
170
|
+
if (id && target) rels[id] = target.replace(/^\/?xl\//, '').replace(/^\.\//, '')
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
const out = []
|
|
174
|
+
const wbEntry = entries.get('xl/workbook.xml')
|
|
175
|
+
if (wbEntry) {
|
|
176
|
+
const xml = readEntry(buf, wbEntry).toString('utf8')
|
|
177
|
+
for (const m of xml.matchAll(/<sheet\b[^>]*>/g)) {
|
|
178
|
+
const name = (m[0].match(/name="([^"]*)"/) || [])[1] || ''
|
|
179
|
+
const rid = (m[0].match(/r:id="([^"]*)"/) || [])[1] || ''
|
|
180
|
+
const target = rels[rid]
|
|
181
|
+
if (target) out.push({ name: unescapeXml(name), target: 'xl/' + target })
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
if (!out.length) {
|
|
185
|
+
const names = [...entries.keys()]
|
|
186
|
+
.filter((n) => /^xl\/worksheets\/sheet\d+\.xml$/.test(n))
|
|
187
|
+
.sort((a, b) => Number(a.match(/(\d+)/)[1]) - Number(b.match(/(\d+)/)[1]))
|
|
188
|
+
names.forEach((n, i) => out.push({ name: 'Sheet' + (i + 1), target: n }))
|
|
189
|
+
}
|
|
190
|
+
return out
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
function xlsxToMarkdown(buf, entries, opt) {
|
|
194
|
+
const shared = []
|
|
195
|
+
const sharedEntry = entries.get('xl/sharedStrings.xml')
|
|
196
|
+
if (sharedEntry) {
|
|
197
|
+
const xml = readEntry(buf, sharedEntry).toString('utf8')
|
|
198
|
+
for (const m of xml.matchAll(/<si\b[^>]*>([\s\S]*?)<\/si>/g)) shared.push(texts(m[1]))
|
|
199
|
+
}
|
|
200
|
+
const parts = []
|
|
201
|
+
let budget = opt.maxCellsPerSheet
|
|
202
|
+
for (const sheet of sheetOrder(buf, entries)) {
|
|
203
|
+
const entry = entries.get(sheet.target)
|
|
204
|
+
if (!entry) continue
|
|
205
|
+
const xml = readEntry(buf, entry).toString('utf8')
|
|
206
|
+
const rows = []
|
|
207
|
+
for (const r of xml.matchAll(/<row\b[^>]*>([\s\S]*?)<\/row>/g)) {
|
|
208
|
+
if (rows.length >= opt.maxRowsPerSheet || budget <= 0) break
|
|
209
|
+
const arr = []
|
|
210
|
+
for (const c of r[1].matchAll(/<c\b([^>]*)>([\s\S]*?)<\/c>|<c\b([^>]*)\/>/g)) {
|
|
211
|
+
const attrs = c[1] || c[3] || ''
|
|
212
|
+
const inner = c[2] || ''
|
|
213
|
+
const type = (attrs.match(/t="([^"]*)"/) || [])[1] || ''
|
|
214
|
+
const ref = (attrs.match(/r="([A-Z]+)\d+"/) || [])[1] || ''
|
|
215
|
+
let value = ''
|
|
216
|
+
if (type === 's') {
|
|
217
|
+
const idx = (inner.match(/<v>([\s\S]*?)<\/v>/) || [])[1]
|
|
218
|
+
value = shared[Number(idx)] !== undefined ? shared[Number(idx)] : ''
|
|
219
|
+
} else if (type === 'inlineStr') {
|
|
220
|
+
value = texts(inner)
|
|
221
|
+
} else {
|
|
222
|
+
const v = (inner.match(/<v>([\s\S]*?)<\/v>/) || [])[1]
|
|
223
|
+
value = v === undefined ? '' : unescapeXml(v)
|
|
224
|
+
}
|
|
225
|
+
const at = ref ? colIndex(ref) : arr.length
|
|
226
|
+
if (at < 0) continue
|
|
227
|
+
while (arr.length < at) arr.push('')
|
|
228
|
+
arr[at] = value
|
|
229
|
+
budget--
|
|
230
|
+
if (budget <= 0) break
|
|
231
|
+
}
|
|
232
|
+
if (arr.length) rows.push(arr)
|
|
233
|
+
}
|
|
234
|
+
if (!rows.length) continue
|
|
235
|
+
const table = mdTable(rows, opt)
|
|
236
|
+
if (table) parts.push('## ' + (sheet.name || 'Sheet') + '\n\n' + table)
|
|
237
|
+
}
|
|
238
|
+
if (!parts.length) throw new Error('工作簿里没有可提取的单元格文本')
|
|
239
|
+
return parts.join('\n\n')
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
/* ---------- pptx ---------- */
|
|
243
|
+
|
|
244
|
+
function pptxToMarkdown(buf, entries) {
|
|
245
|
+
const names = [...entries.keys()]
|
|
246
|
+
.filter((n) => /^ppt\/slides\/slide\d+\.xml$/.test(n))
|
|
247
|
+
.sort((a, b) => Number(a.match(/(\d+)/)[1]) - Number(b.match(/(\d+)/)[1]))
|
|
248
|
+
const parts = []
|
|
249
|
+
names.forEach((n, i) => {
|
|
250
|
+
const xml = readEntry(buf, entries.get(n)).toString('utf8')
|
|
251
|
+
const lines = []
|
|
252
|
+
for (const m of xml.matchAll(/<a:p\b[^>]*>([\s\S]*?)<\/a:p>/g)) {
|
|
253
|
+
const t = texts(m[1]).trim()
|
|
254
|
+
if (t) lines.push(t)
|
|
255
|
+
}
|
|
256
|
+
if (!lines.length) return
|
|
257
|
+
const body = lines.slice(1).map((l) => '- ' + l).join('\n')
|
|
258
|
+
parts.push(`## 幻灯片 ${i + 1}:${lines[0]}` + (body ? '\n\n' + body : ''))
|
|
259
|
+
})
|
|
260
|
+
if (!parts.length) throw new Error('演示文稿里没有可提取的文本')
|
|
261
|
+
return parts.join('\n\n')
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
/* ---------- entry ---------- */
|
|
265
|
+
|
|
266
|
+
export const NODE_DEFAULTS = Object.freeze({
|
|
267
|
+
maxRowsPerSheet: 400,
|
|
268
|
+
maxTableCols: 24,
|
|
269
|
+
maxCellsPerSheet: 20000
|
|
270
|
+
})
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* 把 srcPath 指向的 OOXML 文件转成 Markdown 写入 dstPath(纯本地、无依赖)。
|
|
274
|
+
* @returns {{bytes:number, chars:number}}
|
|
275
|
+
*/
|
|
276
|
+
export function convertFileNode(srcPath, dstPath, options = {}) {
|
|
277
|
+
const opt = { ...NODE_DEFAULTS, ...options }
|
|
278
|
+
const ext = path.extname(srcPath).toLowerCase()
|
|
279
|
+
if (!NODE_SUPPORTED.has(ext)) {
|
|
280
|
+
throw new Error(UNSUPPORTED[ext] || `内置 Node 兜底转换器不支持 ${ext || '该'} 格式`)
|
|
281
|
+
}
|
|
282
|
+
const { buf, entries } = readZip(srcPath)
|
|
283
|
+
let body
|
|
284
|
+
if (ext === '.docx' || ext === '.docm') body = docxToMarkdown(buf, entries, opt)
|
|
285
|
+
else if (ext === '.xlsx' || ext === '.xlsm') body = xlsxToMarkdown(buf, entries, opt)
|
|
286
|
+
else body = pptxToMarkdown(buf, entries)
|
|
287
|
+
|
|
288
|
+
if (!body || !body.trim()) throw new Error('未能从文件中提取到任何文本内容')
|
|
289
|
+
|
|
290
|
+
const lines = [
|
|
291
|
+
'<!-- 由 dsh-plugin-office-markdown 内置 Node 兜底转换器生成(非 MarkItDown),版式/图表/批注/图片/公式等信息可能缺失,保真度有限。 -->',
|
|
292
|
+
`> 源文件:\`${path.basename(srcPath)}\``,
|
|
293
|
+
'> 转换器:内置兜底(fallback-node.js,纯 Node,无需 Python)',
|
|
294
|
+
'',
|
|
295
|
+
body,
|
|
296
|
+
'',
|
|
297
|
+
'---',
|
|
298
|
+
'',
|
|
299
|
+
'**转换提示**',
|
|
300
|
+
'',
|
|
301
|
+
'- 本转换完全在本地完成:未联网、未调用任何 API、未修改原文件。',
|
|
302
|
+
'- 纯 Node 解析 OOXML(zip + xml),不保留版式、图表、批注、图片、公式与扫描件 OCR。',
|
|
303
|
+
'- 如需更高保真度:有网络时用 `uvx markitdown`,或安装 `pip install "markitdown[all]"`。'
|
|
304
|
+
]
|
|
305
|
+
const text = lines.join('\n')
|
|
306
|
+
fs.mkdirSync(path.dirname(dstPath), { recursive: true })
|
|
307
|
+
fs.writeFileSync(dstPath, text, 'utf8')
|
|
308
|
+
return { bytes: Buffer.byteLength(text, 'utf8'), chars: text.length }
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
if (process.argv[1] && pathToFileURL(process.argv[1]).href === import.meta.url) {
|
|
312
|
+
const argv = process.argv.slice(2)
|
|
313
|
+
const arg = (flag, fallback) => {
|
|
314
|
+
const i = argv.indexOf(flag)
|
|
315
|
+
return i >= 0 && argv[i + 1] !== undefined ? argv[i + 1] : fallback
|
|
316
|
+
}
|
|
317
|
+
const input = arg('--input')
|
|
318
|
+
const output = arg('--output')
|
|
319
|
+
if (!input || !output) {
|
|
320
|
+
console.error('usage: node fallback-node.js --input <src> --output <dst> [--max-rows N] [--max-cols N] [--max-cells N]')
|
|
321
|
+
process.exit(2)
|
|
322
|
+
}
|
|
323
|
+
try {
|
|
324
|
+
const result = convertFileNode(input, output, {
|
|
325
|
+
maxRowsPerSheet: Number(arg('--max-rows', NODE_DEFAULTS.maxRowsPerSheet)),
|
|
326
|
+
maxTableCols: Number(arg('--max-cols', NODE_DEFAULTS.maxTableCols)),
|
|
327
|
+
maxCellsPerSheet: Number(arg('--max-cells', NODE_DEFAULTS.maxCellsPerSheet))
|
|
328
|
+
})
|
|
329
|
+
console.log(`OK ${result.bytes} bytes -> ${output}`)
|
|
330
|
+
} catch (error) {
|
|
331
|
+
console.error(String((error && error.message) || error))
|
|
332
|
+
process.exit(2)
|
|
333
|
+
}
|
|
334
|
+
}
|