dsh-plugin-office-markdown 1.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,334 @@
1
+ /*!
2
+ * dsh-plugin-office-markdown — 纯 Node 兜底转换器
3
+ *
4
+ * 不依赖 Python、不联网、不调用任何 API:用 Node 内置的 zlib 解开 OOXML
5
+ * (.docx/.xlsx/.pptx 本质上是 zip + xml),再按 XML 结构抽取文字与表格,
6
+ * 生成 Markdown。保真度有限(无版式/图表/批注/图片/公式/OCR),但保证在
7
+ * “目标电脑既没有 Python 也没有网络”时仍能把文件读成 Markdown。
8
+ *
9
+ * 既可被 lib/convert.js 直接 import 调用,也可作为 CLI 使用:
10
+ * node fallback-node.js --input a.xlsx --output out.md
11
+ */
12
+ import fs from 'node:fs'
13
+ import path from 'node:path'
14
+ import zlib from 'node:zlib'
15
+ import { pathToFileURL } from 'node:url'
16
+
17
+ export const NODE_SUPPORTED = new Set(['.docx', '.docm', '.xlsx', '.xlsm', '.pptx', '.pptm'])
18
+
19
+ const UNSUPPORTED = {
20
+ '.pdf': 'Node 兜底不支持 PDF(PDF 需要 MarkItDown 或本机 Python 兜底)。',
21
+ '.doc': '旧版二进制 .doc 需要 MarkItDown 转换,Node 兜底不支持。',
22
+ '.xls': '旧版二进制 .xls 需要 MarkItDown 转换,Node 兜底不支持。',
23
+ '.ppt': '旧版二进制 .ppt 需要 MarkItDown 转换,Node 兜底不支持。',
24
+ '.msg': 'Outlook .msg 需要 MarkItDown 转换,Node 兜底不支持。',
25
+ '.epub': 'EPUB 需要 MarkItDown 转换,Node 兜底不支持。',
26
+ '.odt': 'ODF/OpenDocument 格式需要 MarkItDown 转换,Node 兜底不支持。',
27
+ '.ods': 'ODF/OpenDocument 格式需要 MarkItDown 转换,Node 兜底不支持。',
28
+ '.odp': 'ODF/OpenDocument 格式需要 MarkItDown 转换,Node 兜底不支持。'
29
+ }
30
+
31
+ /* ---------- zip ---------- */
32
+
33
+ const SIG_EOCD = 0x06054b50
34
+ const SIG_CENTRAL = 0x02014b50
35
+
36
+ function readZip(file) {
37
+ const buf = fs.readFileSync(file)
38
+ let eocd = -1
39
+ const floor = Math.max(0, buf.length - 66000)
40
+ for (let i = buf.length - 22; i >= floor; i--) {
41
+ if (buf.readUInt32LE(i) === SIG_EOCD) { eocd = i; break }
42
+ }
43
+ if (eocd < 0) throw new Error('无法解析为 zip 容器(文件可能已损坏或不是真正的 OOXML 文件)')
44
+ const count = buf.readUInt16LE(eocd + 10)
45
+ let off = buf.readUInt32LE(eocd + 16)
46
+ const entries = new Map()
47
+ for (let i = 0; i < count; i++) {
48
+ if (off + 46 > buf.length || buf.readUInt32LE(off) !== SIG_CENTRAL) break
49
+ const method = buf.readUInt16LE(off + 10)
50
+ const csize = buf.readUInt32LE(off + 20)
51
+ const nlen = buf.readUInt16LE(off + 28)
52
+ const elen = buf.readUInt16LE(off + 30)
53
+ const clen = buf.readUInt16LE(off + 32)
54
+ const lho = buf.readUInt32LE(off + 42)
55
+ const name = buf.toString('utf8', off + 46, off + 46 + nlen)
56
+ entries.set(name, { method, csize, lho })
57
+ off += 46 + nlen + elen + clen
58
+ }
59
+ return { buf, entries }
60
+ }
61
+
62
+ function readEntry(buf, entry) {
63
+ const nlen = buf.readUInt16LE(entry.lho + 26)
64
+ const elen = buf.readUInt16LE(entry.lho + 28)
65
+ const start = entry.lho + 30 + nlen + elen
66
+ const raw = buf.subarray(start, start + entry.csize)
67
+ if (entry.method === 0) return raw
68
+ if (entry.method === 8) return zlib.inflateRawSync(raw)
69
+ throw new Error('zip 压缩方式 ' + entry.method + ' 不受支持')
70
+ }
71
+
72
+ /* ---------- xml helpers ---------- */
73
+
74
+ function unescapeXml(s) {
75
+ return String(s)
76
+ .replace(/&#x([0-9a-fA-F]+);/g, (_, h) => String.fromCodePoint(parseInt(h, 16)))
77
+ .replace(/&#(\d+);/g, (_, d) => String.fromCodePoint(Number(d)))
78
+ .replace(/&lt;/g, '<')
79
+ .replace(/&gt;/g, '>')
80
+ .replace(/&quot;/g, '"')
81
+ .replace(/&apos;/g, "'")
82
+ .replace(/&amp;/g, '&')
83
+ }
84
+
85
+ const TEXT_RE = /<(?:[A-Za-z0-9_]+:)?t(?:\s[^>]*)?>([\s\S]*?)<\/(?:[A-Za-z0-9_]+:)?t>/g
86
+
87
+ function texts(xml) {
88
+ let out = ''
89
+ for (const m of String(xml).matchAll(TEXT_RE)) out += unescapeXml(m[1])
90
+ return out
91
+ }
92
+
93
+ function cell(text) {
94
+ return String(text).replace(/\|/g, '\\|').replace(/\s+/g, ' ').trim()
95
+ }
96
+
97
+ function mdTable(rows, opt) {
98
+ if (!rows.length) return ''
99
+ const width = Math.min(opt.maxTableCols, Math.max(...rows.map((r) => r.length)))
100
+ if (width <= 0) return ''
101
+ const lines = []
102
+ const limit = Math.min(rows.length, opt.maxRowsPerSheet)
103
+ for (let i = 0; i < limit; i++) {
104
+ const cells = []
105
+ for (let c = 0; c < width; c++) cells.push(rows[i][c] === undefined ? '' : cell(rows[i][c]))
106
+ lines.push('| ' + cells.join(' | ') + ' |')
107
+ if (i === 0) lines.push('| ' + cells.map(() => '---').join(' | ') + ' |')
108
+ }
109
+ if (rows.length > limit) lines.push('', `> 表格已截断:共 ${rows.length} 行,只显示前 ${limit} 行。`)
110
+ return lines.join('\n')
111
+ }
112
+
113
+ /* ---------- docx ---------- */
114
+
115
+ function paraToMd(chunk) {
116
+ const text = texts(chunk).replace(/\s+$/, '')
117
+ if (!text.trim()) return ''
118
+ const style = (chunk.match(/<w:pStyle[^>]*w:val="([^"]*)"/) || [])[1] || ''
119
+ const heading = style.match(/^(?:Heading|heading|标题)\s*(\d+)$/)
120
+ if (heading) return '#'.repeat(Math.min(6, Number(heading[1]) || 1)) + ' ' + text
121
+ if (/^(?:Title|标题|Subtitle)$/i.test(style)) return '# ' + text
122
+ if (/<w:numPr\b/.test(chunk)) return '- ' + text
123
+ return text
124
+ }
125
+
126
+ function docxTableToMd(chunk, opt) {
127
+ const rows = []
128
+ for (const r of chunk.matchAll(/<w:tr\b[^>]*>([\s\S]*?)<\/w:tr>/g)) {
129
+ const cells = []
130
+ for (const c of r[1].matchAll(/<w:tc\b[^>]*>([\s\S]*?)<\/w:tc>/g)) cells.push(texts(c[1]))
131
+ if (cells.length) rows.push(cells)
132
+ }
133
+ return mdTable(rows, opt)
134
+ }
135
+
136
+ function docxToMarkdown(buf, entries, opt) {
137
+ const entry = entries.get('word/document.xml')
138
+ if (!entry) throw new Error('缺少 word/document.xml,可能不是有效的 .docx')
139
+ const xml = readEntry(buf, entry).toString('utf8')
140
+ const blockRe = /<w:tbl\b[\s\S]*?<\/w:tbl>|<w:p\b[^>]*\/>|<w:p\b[^>]*>[\s\S]*?<\/w:p>/g
141
+ const parts = []
142
+ for (const m of xml.matchAll(blockRe)) {
143
+ const chunk = m[0]
144
+ const piece = chunk.startsWith('<w:tbl') ? docxTableToMd(chunk, opt) : paraToMd(chunk)
145
+ if (piece && piece.trim()) parts.push(piece)
146
+ }
147
+ return parts.join('\n\n')
148
+ }
149
+
150
+ /* ---------- xlsx ---------- */
151
+
152
+ function colIndex(ref) {
153
+ let n = 0
154
+ for (const ch of ref) {
155
+ const c = ch.charCodeAt(0)
156
+ if (c < 65 || c > 90) break
157
+ n = n * 26 + (c - 64)
158
+ }
159
+ return n - 1
160
+ }
161
+
162
+ function sheetOrder(buf, entries) {
163
+ const rels = {}
164
+ const relEntry = entries.get('xl/_rels/workbook.xml.rels')
165
+ if (relEntry) {
166
+ const xml = readEntry(buf, relEntry).toString('utf8')
167
+ for (const m of xml.matchAll(/<Relationship\b[^>]*>/g)) {
168
+ const id = (m[0].match(/Id="([^"]*)"/) || [])[1]
169
+ const target = (m[0].match(/Target="([^"]*)"/) || [])[1]
170
+ if (id && target) rels[id] = target.replace(/^\/?xl\//, '').replace(/^\.\//, '')
171
+ }
172
+ }
173
+ const out = []
174
+ const wbEntry = entries.get('xl/workbook.xml')
175
+ if (wbEntry) {
176
+ const xml = readEntry(buf, wbEntry).toString('utf8')
177
+ for (const m of xml.matchAll(/<sheet\b[^>]*>/g)) {
178
+ const name = (m[0].match(/name="([^"]*)"/) || [])[1] || ''
179
+ const rid = (m[0].match(/r:id="([^"]*)"/) || [])[1] || ''
180
+ const target = rels[rid]
181
+ if (target) out.push({ name: unescapeXml(name), target: 'xl/' + target })
182
+ }
183
+ }
184
+ if (!out.length) {
185
+ const names = [...entries.keys()]
186
+ .filter((n) => /^xl\/worksheets\/sheet\d+\.xml$/.test(n))
187
+ .sort((a, b) => Number(a.match(/(\d+)/)[1]) - Number(b.match(/(\d+)/)[1]))
188
+ names.forEach((n, i) => out.push({ name: 'Sheet' + (i + 1), target: n }))
189
+ }
190
+ return out
191
+ }
192
+
193
+ function xlsxToMarkdown(buf, entries, opt) {
194
+ const shared = []
195
+ const sharedEntry = entries.get('xl/sharedStrings.xml')
196
+ if (sharedEntry) {
197
+ const xml = readEntry(buf, sharedEntry).toString('utf8')
198
+ for (const m of xml.matchAll(/<si\b[^>]*>([\s\S]*?)<\/si>/g)) shared.push(texts(m[1]))
199
+ }
200
+ const parts = []
201
+ let budget = opt.maxCellsPerSheet
202
+ for (const sheet of sheetOrder(buf, entries)) {
203
+ const entry = entries.get(sheet.target)
204
+ if (!entry) continue
205
+ const xml = readEntry(buf, entry).toString('utf8')
206
+ const rows = []
207
+ for (const r of xml.matchAll(/<row\b[^>]*>([\s\S]*?)<\/row>/g)) {
208
+ if (rows.length >= opt.maxRowsPerSheet || budget <= 0) break
209
+ const arr = []
210
+ for (const c of r[1].matchAll(/<c\b([^>]*)>([\s\S]*?)<\/c>|<c\b([^>]*)\/>/g)) {
211
+ const attrs = c[1] || c[3] || ''
212
+ const inner = c[2] || ''
213
+ const type = (attrs.match(/t="([^"]*)"/) || [])[1] || ''
214
+ const ref = (attrs.match(/r="([A-Z]+)\d+"/) || [])[1] || ''
215
+ let value = ''
216
+ if (type === 's') {
217
+ const idx = (inner.match(/<v>([\s\S]*?)<\/v>/) || [])[1]
218
+ value = shared[Number(idx)] !== undefined ? shared[Number(idx)] : ''
219
+ } else if (type === 'inlineStr') {
220
+ value = texts(inner)
221
+ } else {
222
+ const v = (inner.match(/<v>([\s\S]*?)<\/v>/) || [])[1]
223
+ value = v === undefined ? '' : unescapeXml(v)
224
+ }
225
+ const at = ref ? colIndex(ref) : arr.length
226
+ if (at < 0) continue
227
+ while (arr.length < at) arr.push('')
228
+ arr[at] = value
229
+ budget--
230
+ if (budget <= 0) break
231
+ }
232
+ if (arr.length) rows.push(arr)
233
+ }
234
+ if (!rows.length) continue
235
+ const table = mdTable(rows, opt)
236
+ if (table) parts.push('## ' + (sheet.name || 'Sheet') + '\n\n' + table)
237
+ }
238
+ if (!parts.length) throw new Error('工作簿里没有可提取的单元格文本')
239
+ return parts.join('\n\n')
240
+ }
241
+
242
+ /* ---------- pptx ---------- */
243
+
244
+ function pptxToMarkdown(buf, entries) {
245
+ const names = [...entries.keys()]
246
+ .filter((n) => /^ppt\/slides\/slide\d+\.xml$/.test(n))
247
+ .sort((a, b) => Number(a.match(/(\d+)/)[1]) - Number(b.match(/(\d+)/)[1]))
248
+ const parts = []
249
+ names.forEach((n, i) => {
250
+ const xml = readEntry(buf, entries.get(n)).toString('utf8')
251
+ const lines = []
252
+ for (const m of xml.matchAll(/<a:p\b[^>]*>([\s\S]*?)<\/a:p>/g)) {
253
+ const t = texts(m[1]).trim()
254
+ if (t) lines.push(t)
255
+ }
256
+ if (!lines.length) return
257
+ const body = lines.slice(1).map((l) => '- ' + l).join('\n')
258
+ parts.push(`## 幻灯片 ${i + 1}:${lines[0]}` + (body ? '\n\n' + body : ''))
259
+ })
260
+ if (!parts.length) throw new Error('演示文稿里没有可提取的文本')
261
+ return parts.join('\n\n')
262
+ }
263
+
264
+ /* ---------- entry ---------- */
265
+
266
+ export const NODE_DEFAULTS = Object.freeze({
267
+ maxRowsPerSheet: 400,
268
+ maxTableCols: 24,
269
+ maxCellsPerSheet: 20000
270
+ })
271
+
272
+ /**
273
+ * 把 srcPath 指向的 OOXML 文件转成 Markdown 写入 dstPath(纯本地、无依赖)。
274
+ * @returns {{bytes:number, chars:number}}
275
+ */
276
+ export function convertFileNode(srcPath, dstPath, options = {}) {
277
+ const opt = { ...NODE_DEFAULTS, ...options }
278
+ const ext = path.extname(srcPath).toLowerCase()
279
+ if (!NODE_SUPPORTED.has(ext)) {
280
+ throw new Error(UNSUPPORTED[ext] || `内置 Node 兜底转换器不支持 ${ext || '该'} 格式`)
281
+ }
282
+ const { buf, entries } = readZip(srcPath)
283
+ let body
284
+ if (ext === '.docx' || ext === '.docm') body = docxToMarkdown(buf, entries, opt)
285
+ else if (ext === '.xlsx' || ext === '.xlsm') body = xlsxToMarkdown(buf, entries, opt)
286
+ else body = pptxToMarkdown(buf, entries)
287
+
288
+ if (!body || !body.trim()) throw new Error('未能从文件中提取到任何文本内容')
289
+
290
+ const lines = [
291
+ '<!-- 由 dsh-plugin-office-markdown 内置 Node 兜底转换器生成(非 MarkItDown),版式/图表/批注/图片/公式等信息可能缺失,保真度有限。 -->',
292
+ `> 源文件:\`${path.basename(srcPath)}\``,
293
+ '> 转换器:内置兜底(fallback-node.js,纯 Node,无需 Python)',
294
+ '',
295
+ body,
296
+ '',
297
+ '---',
298
+ '',
299
+ '**转换提示**',
300
+ '',
301
+ '- 本转换完全在本地完成:未联网、未调用任何 API、未修改原文件。',
302
+ '- 纯 Node 解析 OOXML(zip + xml),不保留版式、图表、批注、图片、公式与扫描件 OCR。',
303
+ '- 如需更高保真度:有网络时用 `uvx markitdown`,或安装 `pip install "markitdown[all]"`。'
304
+ ]
305
+ const text = lines.join('\n')
306
+ fs.mkdirSync(path.dirname(dstPath), { recursive: true })
307
+ fs.writeFileSync(dstPath, text, 'utf8')
308
+ return { bytes: Buffer.byteLength(text, 'utf8'), chars: text.length }
309
+ }
310
+
311
+ if (process.argv[1] && pathToFileURL(process.argv[1]).href === import.meta.url) {
312
+ const argv = process.argv.slice(2)
313
+ const arg = (flag, fallback) => {
314
+ const i = argv.indexOf(flag)
315
+ return i >= 0 && argv[i + 1] !== undefined ? argv[i + 1] : fallback
316
+ }
317
+ const input = arg('--input')
318
+ const output = arg('--output')
319
+ if (!input || !output) {
320
+ console.error('usage: node fallback-node.js --input <src> --output <dst> [--max-rows N] [--max-cols N] [--max-cells N]')
321
+ process.exit(2)
322
+ }
323
+ try {
324
+ const result = convertFileNode(input, output, {
325
+ maxRowsPerSheet: Number(arg('--max-rows', NODE_DEFAULTS.maxRowsPerSheet)),
326
+ maxTableCols: Number(arg('--max-cols', NODE_DEFAULTS.maxTableCols)),
327
+ maxCellsPerSheet: Number(arg('--max-cells', NODE_DEFAULTS.maxCellsPerSheet))
328
+ })
329
+ console.log(`OK ${result.bytes} bytes -> ${output}`)
330
+ } catch (error) {
331
+ console.error(String((error && error.message) || error))
332
+ process.exit(2)
333
+ }
334
+ }