dsh-plugin-office-markdown 1.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.en.md +156 -0
- package/README.md +144 -0
- package/cordis.patch.yml +35 -0
- package/lib/client.js +589 -0
- package/lib/convert.js +796 -0
- package/lib/env.js +602 -0
- package/lib/fallback-node.js +334 -0
- package/lib/fallback.py +579 -0
- package/lib/index.js +1443 -0
- package/lib/paths.js +76 -0
- package/lib/removal-watchdog.js +359 -0
- package/lib/settings-api.js +381 -0
- package/package.json +55 -0
package/lib/fallback.py
ADDED
|
@@ -0,0 +1,579 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
"""dsh-plugin-office-markdown — bundled fallback converter (limited fidelity).
|
|
4
|
+
|
|
5
|
+
Used ONLY when `uvx markitdown`, the `markitdown` CLI and `python -m markitdown`
|
|
6
|
+
are all unavailable. It extracts text and tables from the Office containers with
|
|
7
|
+
the libraries that ship inside the DSH Python runtime (python-docx, openpyxl,
|
|
8
|
+
python-pptx) and emits Markdown.
|
|
9
|
+
|
|
10
|
+
Fidelity is intentionally declared "limited": styling, charts, images, comments,
|
|
11
|
+
tracked changes, footnotes and scanned-PDF OCR are not recovered.
|
|
12
|
+
|
|
13
|
+
Usage:
|
|
14
|
+
python fallback.py --input <source> --output <file.md>
|
|
15
|
+
[--max-rows N] [--max-cols N] [--max-cells N] [--max-slides N]
|
|
16
|
+
|
|
17
|
+
Exit codes: 0 ok, 2 unsupported format, 3 conversion error, 4 empty result.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import argparse
|
|
23
|
+
import csv
|
|
24
|
+
import io
|
|
25
|
+
import json
|
|
26
|
+
import os
|
|
27
|
+
import re
|
|
28
|
+
import sys
|
|
29
|
+
import zlib
|
|
30
|
+
|
|
31
|
+
BANNER = (
|
|
32
|
+
"<!-- 由 dsh-plugin-office-markdown 内置兜底转换器生成(非 MarkItDown),"
|
|
33
|
+
"版式/图表/批注/图片等信息可能缺失,保真度有限。 -->\n"
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
UNSUPPORTED = {
|
|
37
|
+
".doc": "旧版 Word 97-2003(.doc)",
|
|
38
|
+
".xls": "旧版 Excel 97-2003(.xls)",
|
|
39
|
+
".ppt": "旧版 PowerPoint 97-2003(.ppt)",
|
|
40
|
+
".msg": "Outlook 邮件(.msg)",
|
|
41
|
+
".epub": "EPUB 电子书(.epub)",
|
|
42
|
+
".odt": "OpenDocument 文本(.odt)",
|
|
43
|
+
".ods": "OpenDocument 表格(.ods)",
|
|
44
|
+
".odp": "OpenDocument 演示(.odp)",
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
# --------------------------------------------------------------------------
|
|
49
|
+
# helpers
|
|
50
|
+
# --------------------------------------------------------------------------
|
|
51
|
+
|
|
52
|
+
def cell(value) -> str:
|
|
53
|
+
if value is None:
|
|
54
|
+
return ""
|
|
55
|
+
text = str(value).replace("\r\n", "\n").replace("\r", "\n").strip()
|
|
56
|
+
text = text.replace("|", "\\|")
|
|
57
|
+
if "\n" in text:
|
|
58
|
+
text = text.replace("\n", "<br>")
|
|
59
|
+
return text
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def md_table(rows: list[list[str]], max_rows: int, max_cols: int, note: list[str], label: str) -> str:
|
|
63
|
+
rows = [r for r in rows if any(c for c in r)]
|
|
64
|
+
if not rows:
|
|
65
|
+
return ""
|
|
66
|
+
truncated_cols = False
|
|
67
|
+
if max_cols and any(len(r) > max_cols for r in rows):
|
|
68
|
+
rows = [r[:max_cols] for r in rows]
|
|
69
|
+
truncated_cols = True
|
|
70
|
+
width = max(len(r) for r in rows)
|
|
71
|
+
rows = [r + [""] * (width - len(r)) for r in rows]
|
|
72
|
+
truncated_rows = False
|
|
73
|
+
if max_rows and len(rows) > max_rows:
|
|
74
|
+
rows = rows[:max_rows]
|
|
75
|
+
truncated_rows = True
|
|
76
|
+
out = ["| " + " | ".join(rows[0]) + " |", "| " + " | ".join(["---"] * width) + " |"]
|
|
77
|
+
for r in rows[1:]:
|
|
78
|
+
out.append("| " + " | ".join(r) + " |")
|
|
79
|
+
if truncated_rows or truncated_cols:
|
|
80
|
+
bits = []
|
|
81
|
+
if truncated_rows:
|
|
82
|
+
bits.append("仅显示前 %d 行" % max_rows)
|
|
83
|
+
if truncated_cols:
|
|
84
|
+
bits.append("仅显示前 %d 列" % max_cols)
|
|
85
|
+
note.append("%s:%s。" % (label, "、".join(bits)))
|
|
86
|
+
return "\n".join(out)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
# --------------------------------------------------------------------------
|
|
90
|
+
# docx
|
|
91
|
+
# --------------------------------------------------------------------------
|
|
92
|
+
|
|
93
|
+
_HEADING_RE = re.compile(r"(?:heading|标题)\s*(\d+)", re.I)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def convert_docx(src: str, opt) -> tuple[str, list[str]]:
|
|
97
|
+
try:
|
|
98
|
+
import docx # python-docx
|
|
99
|
+
from docx.table import Table
|
|
100
|
+
from docx.text.paragraph import Paragraph
|
|
101
|
+
except Exception as exc: # pragma: no cover
|
|
102
|
+
raise RuntimeError("缺少 python-docx:%s" % exc)
|
|
103
|
+
|
|
104
|
+
document = docx.Document(src)
|
|
105
|
+
note: list[str] = []
|
|
106
|
+
out: list[str] = []
|
|
107
|
+
|
|
108
|
+
def emit_para(para) -> None:
|
|
109
|
+
text = (para.text or "").strip()
|
|
110
|
+
if not text:
|
|
111
|
+
if out and out[-1] != "":
|
|
112
|
+
out.append("")
|
|
113
|
+
return
|
|
114
|
+
style = ""
|
|
115
|
+
try:
|
|
116
|
+
style = (para.style.name or "")
|
|
117
|
+
except Exception:
|
|
118
|
+
style = ""
|
|
119
|
+
low = style.strip().lower()
|
|
120
|
+
m = _HEADING_RE.search(style)
|
|
121
|
+
if m:
|
|
122
|
+
level = max(1, min(int(m.group(1)), 6))
|
|
123
|
+
out.append("#" * level + " " + text)
|
|
124
|
+
elif low in ("title", "标题"):
|
|
125
|
+
# python-docx 的 add_heading(text, 0) 用的是内建样式 "Title",不含数字,
|
|
126
|
+
# 匹配不到 _HEADING_RE,旧版会退化成普通段落。
|
|
127
|
+
out.append("# " + text)
|
|
128
|
+
elif low in ("subtitle", "副标题"):
|
|
129
|
+
out.append("## " + text)
|
|
130
|
+
elif low.startswith("list") or "列表" in style:
|
|
131
|
+
out.append("- " + text)
|
|
132
|
+
else:
|
|
133
|
+
out.append(text)
|
|
134
|
+
|
|
135
|
+
def emit_table(table, index: int) -> None:
|
|
136
|
+
rows = []
|
|
137
|
+
for row in table.rows:
|
|
138
|
+
rows.append([cell(c.text) for c in row.cells])
|
|
139
|
+
md = md_table(rows, opt.max_rows, opt.max_cols, note, "表格 %d" % index)
|
|
140
|
+
if md:
|
|
141
|
+
out.append("")
|
|
142
|
+
out.append("**表格 %d**" % index)
|
|
143
|
+
out.append("")
|
|
144
|
+
out.append(md)
|
|
145
|
+
|
|
146
|
+
# 按正文真实顺序遍历段落与表格,让表格留在它原本出现的位置,
|
|
147
|
+
# 而不是像旧版那样把全部表格统一堆到文末。
|
|
148
|
+
table_index = 0
|
|
149
|
+
for child in document.element.body.iterchildren():
|
|
150
|
+
tag = child.tag.rsplit("}", 1)[-1]
|
|
151
|
+
if tag == "p":
|
|
152
|
+
emit_para(Paragraph(child, document))
|
|
153
|
+
elif tag == "tbl":
|
|
154
|
+
table_index += 1
|
|
155
|
+
emit_table(Table(child, document), table_index)
|
|
156
|
+
|
|
157
|
+
# 兜底:正文里什么都没取到(例如内容都在文本框内)时,
|
|
158
|
+
# 退回 python-docx 的段落 / 表格列表接口。
|
|
159
|
+
if not out:
|
|
160
|
+
for para in document.paragraphs:
|
|
161
|
+
emit_para(para)
|
|
162
|
+
for index, table in enumerate(document.tables, start=1):
|
|
163
|
+
emit_table(table, index)
|
|
164
|
+
|
|
165
|
+
return "\n".join(out), note
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
# --------------------------------------------------------------------------
|
|
169
|
+
# xlsx
|
|
170
|
+
# --------------------------------------------------------------------------
|
|
171
|
+
|
|
172
|
+
def convert_xlsx(src: str, opt) -> tuple[str, list[str]]:
|
|
173
|
+
try:
|
|
174
|
+
import openpyxl
|
|
175
|
+
except Exception as exc: # pragma: no cover
|
|
176
|
+
raise RuntimeError("缺少 openpyxl:%s" % exc)
|
|
177
|
+
|
|
178
|
+
workbook = openpyxl.load_workbook(src, read_only=True, data_only=True)
|
|
179
|
+
note: list[str] = []
|
|
180
|
+
out: list[str] = []
|
|
181
|
+
cells = 0
|
|
182
|
+
numeric_notes = ("openpyxl 以 data_only 读取:公式单元格只会显示上次由 Excel 保存的缓存值,"
|
|
183
|
+
"若线程中从未计算则显示为空。")
|
|
184
|
+
|
|
185
|
+
for sheet in workbook.worksheets:
|
|
186
|
+
out.append("## " + str(sheet.title))
|
|
187
|
+
out.append("")
|
|
188
|
+
rows: list[list[str]] = []
|
|
189
|
+
stopped = False
|
|
190
|
+
for raw in sheet.iter_rows(values_only=True):
|
|
191
|
+
if raw is None:
|
|
192
|
+
continue
|
|
193
|
+
cells += len(raw)
|
|
194
|
+
if opt.max_cells and cells > opt.max_cells:
|
|
195
|
+
stopped = True
|
|
196
|
+
break
|
|
197
|
+
rows.append([cell(v) for v in raw])
|
|
198
|
+
md = md_table(rows, opt.max_rows, opt.max_cols, note, "工作表「%s」" % sheet.title)
|
|
199
|
+
if md:
|
|
200
|
+
out.append(md)
|
|
201
|
+
elif not rows:
|
|
202
|
+
out.append("_(空工作表)_")
|
|
203
|
+
if stopped:
|
|
204
|
+
note.append("工作表「%s」:单元格总数超过上限 %d,已提前截断。" % (sheet.title, opt.max_cells))
|
|
205
|
+
out.append("")
|
|
206
|
+
|
|
207
|
+
try:
|
|
208
|
+
workbook.close()
|
|
209
|
+
except Exception:
|
|
210
|
+
pass
|
|
211
|
+
|
|
212
|
+
if any("#" in o for o in out):
|
|
213
|
+
note.append(numeric_notes)
|
|
214
|
+
return "\n".join(out), note
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
# --------------------------------------------------------------------------
|
|
218
|
+
# pptx
|
|
219
|
+
# --------------------------------------------------------------------------
|
|
220
|
+
|
|
221
|
+
def _shape_text(shape) -> list[str]:
|
|
222
|
+
parts: list[str] = []
|
|
223
|
+
if getattr(shape, "has_text_frame", False):
|
|
224
|
+
for para in shape.text_frame.paragraphs:
|
|
225
|
+
text = "".join(run.text or "" for run in para.runs).strip()
|
|
226
|
+
if not text:
|
|
227
|
+
text = (para.text or "").strip()
|
|
228
|
+
if text:
|
|
229
|
+
parts.append(text)
|
|
230
|
+
if getattr(shape, "has_table", False):
|
|
231
|
+
rows = []
|
|
232
|
+
try:
|
|
233
|
+
for row in shape.table.rows:
|
|
234
|
+
rows.append([cell(c.text) for c in row.cells])
|
|
235
|
+
except Exception:
|
|
236
|
+
rows = []
|
|
237
|
+
if rows:
|
|
238
|
+
parts.append("")
|
|
239
|
+
parts.append("| " + " | ".join(rows[0]) + " |")
|
|
240
|
+
parts.append("| " + " | ".join(["---"] * len(rows[0])) + " |")
|
|
241
|
+
for r in rows[1:]:
|
|
242
|
+
parts.append("| " + " | ".join(r) + " |")
|
|
243
|
+
return parts
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def convert_pptx(src: str, opt) -> tuple[str, list[str]]:
|
|
247
|
+
try:
|
|
248
|
+
from pptx import Presentation
|
|
249
|
+
except Exception as exc: # pragma: no cover
|
|
250
|
+
raise RuntimeError("缺少 python-pptx:%s" % exc)
|
|
251
|
+
|
|
252
|
+
presentation = Presentation(src)
|
|
253
|
+
note: list[str] = []
|
|
254
|
+
out: list[str] = []
|
|
255
|
+
total = len(presentation.slides)
|
|
256
|
+
|
|
257
|
+
for index, slide in enumerate(presentation.slides, start=1):
|
|
258
|
+
if opt.max_slides and index > opt.max_slides:
|
|
259
|
+
note.append("演示文稿共 %d 张幻灯片,仅转换前 %d 张。" % (total, opt.max_slides))
|
|
260
|
+
break
|
|
261
|
+
title_shape = None
|
|
262
|
+
try:
|
|
263
|
+
title_shape = slide.shapes.title
|
|
264
|
+
except Exception:
|
|
265
|
+
title_shape = None
|
|
266
|
+
title = ""
|
|
267
|
+
if title_shape is not None:
|
|
268
|
+
try:
|
|
269
|
+
title = (title_shape.text or "").strip()
|
|
270
|
+
except Exception:
|
|
271
|
+
title = ""
|
|
272
|
+
out.append("## 幻灯片 %d%s" % (index, (":" + title) if title else ""))
|
|
273
|
+
out.append("")
|
|
274
|
+
# python-pptx 每次访问 slide.shapes.title 都会重新包装同一段 XML,
|
|
275
|
+
# 用 `shape is slide.shapes.title` 判定会永远为假,导致标题文本重复输出一次;
|
|
276
|
+
# 改为按底层 lxml 元素判等。
|
|
277
|
+
title_element = getattr(title_shape, "_element", None)
|
|
278
|
+
for shape in slide.shapes:
|
|
279
|
+
if title_element is not None and getattr(shape, "_element", None) is title_element:
|
|
280
|
+
continue
|
|
281
|
+
parts = _shape_text(shape)
|
|
282
|
+
if parts:
|
|
283
|
+
out.extend(parts)
|
|
284
|
+
out.append("")
|
|
285
|
+
if slide.has_notes_slide:
|
|
286
|
+
try:
|
|
287
|
+
notes = (slide.notes_slide.notes_text_frame.text or "").strip()
|
|
288
|
+
except Exception:
|
|
289
|
+
notes = ""
|
|
290
|
+
if notes:
|
|
291
|
+
out.append("> 备注:" + notes.replace("\n", " "))
|
|
292
|
+
out.append("")
|
|
293
|
+
|
|
294
|
+
return "\n".join(out), note
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
# --------------------------------------------------------------------------
|
|
298
|
+
# delimited / plain text
|
|
299
|
+
# --------------------------------------------------------------------------
|
|
300
|
+
|
|
301
|
+
def sniff_delimiter(sample: str, default: str = ",") -> str:
|
|
302
|
+
try:
|
|
303
|
+
return csv.Sniffer().sniff(sample, delimiters=",;\t|").delimiter
|
|
304
|
+
except Exception:
|
|
305
|
+
return default
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def convert_delimited(src: str, opt, ext: str) -> tuple[str, list[str]]:
|
|
309
|
+
default = "\t" if ext == ".tsv" else ","
|
|
310
|
+
with open(src, "r", encoding="utf-8-sig", errors="replace", newline="") as handle:
|
|
311
|
+
sample = handle.read(8192)
|
|
312
|
+
handle.seek(0)
|
|
313
|
+
delimiter = sniff_delimiter(sample, default)
|
|
314
|
+
reader = csv.reader(handle, delimiter=delimiter)
|
|
315
|
+
rows = [[cell(v) for v in row] for row in reader]
|
|
316
|
+
note: list[str] = []
|
|
317
|
+
md = md_table(rows, opt.max_rows, opt.max_cols, note, "分隔符文本")
|
|
318
|
+
return md, note
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def convert_text(src: str, opt) -> tuple[str, list[str]]:
|
|
322
|
+
with open(src, "r", encoding="utf-8-sig", errors="replace") as handle:
|
|
323
|
+
return handle.read(), []
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def convert_json(src: str, opt) -> tuple[str, list[str]]:
|
|
327
|
+
with open(src, "r", encoding="utf-8-sig", errors="replace") as handle:
|
|
328
|
+
raw = handle.read()
|
|
329
|
+
try:
|
|
330
|
+
parsed = json.loads(raw)
|
|
331
|
+
pretty = json.dumps(parsed, ensure_ascii=False, indent=2)
|
|
332
|
+
except Exception:
|
|
333
|
+
pretty = raw
|
|
334
|
+
return "```json\n" + pretty + "\n```", []
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
# --------------------------------------------------------------------------
|
|
338
|
+
# html
|
|
339
|
+
# --------------------------------------------------------------------------
|
|
340
|
+
|
|
341
|
+
def convert_html(src: str, opt) -> tuple[str, list[str]]:
|
|
342
|
+
from html.parser import HTMLParser
|
|
343
|
+
|
|
344
|
+
class Extractor(HTMLParser):
|
|
345
|
+
def __init__(self) -> None:
|
|
346
|
+
super().__init__(convert_charrefs=True)
|
|
347
|
+
self.parts: list[str] = []
|
|
348
|
+
self.skip = 0
|
|
349
|
+
|
|
350
|
+
def handle_starttag(self, tag, attrs):
|
|
351
|
+
if tag in ("script", "style", "head"):
|
|
352
|
+
self.skip += 1
|
|
353
|
+
elif tag in ("p", "div", "br", "li", "tr", "h1", "h2", "h3", "h4", "h5", "h6"):
|
|
354
|
+
self.parts.append("\n")
|
|
355
|
+
if tag.startswith("h") and len(tag) == 2 and tag[1].isdigit():
|
|
356
|
+
self.parts.append("#" * int(tag[1]) + " ")
|
|
357
|
+
|
|
358
|
+
def handle_endtag(self, tag):
|
|
359
|
+
if tag in ("script", "style", "head") and self.skip:
|
|
360
|
+
self.skip -= 1
|
|
361
|
+
|
|
362
|
+
def handle_data(self, data):
|
|
363
|
+
if self.skip:
|
|
364
|
+
return
|
|
365
|
+
text = data.strip()
|
|
366
|
+
if text:
|
|
367
|
+
self.parts.append(text + " ")
|
|
368
|
+
|
|
369
|
+
extractor = Extractor()
|
|
370
|
+
with open(src, "r", encoding="utf-8-sig", errors="replace") as handle:
|
|
371
|
+
extractor.feed(handle.read())
|
|
372
|
+
text = re.sub(r"[ \t]+", " ", "".join(extractor.parts))
|
|
373
|
+
text = re.sub(r"\n{3,}", "\n\n", text)
|
|
374
|
+
return text.strip(), ["HTML 兜底提取仅保留可见文本,链接、图片与 CSS 布局丢失。"]
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
# --------------------------------------------------------------------------
|
|
378
|
+
# rtf
|
|
379
|
+
# --------------------------------------------------------------------------
|
|
380
|
+
|
|
381
|
+
def convert_rtf(src: str, opt) -> tuple[str, list[str]]:
|
|
382
|
+
with open(src, "rb") as handle:
|
|
383
|
+
data = handle.read().decode("latin-1", errors="replace")
|
|
384
|
+
data = re.sub(r"\\'([0-9a-fA-F]{2})", lambda m: chr(int(m.group(1), 16)), data)
|
|
385
|
+
data = re.sub(r"\\u(-?\d+)\?", lambda m: chr(int(m.group(1)) % 65536), data)
|
|
386
|
+
data = re.sub(r"\\par[d]?", "\n", data)
|
|
387
|
+
data = re.sub(r"\\[a-zA-Z]+-?\d* ?", "", data)
|
|
388
|
+
data = data.replace("{", "").replace("}", "")
|
|
389
|
+
data = re.sub(r"\n{3,}", "\n\n", data)
|
|
390
|
+
return data.strip(), ["RTF 兜底提取仅保留纯文本,格式、表格与图片丢失。"]
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
# --------------------------------------------------------------------------
|
|
394
|
+
# pdf
|
|
395
|
+
# --------------------------------------------------------------------------
|
|
396
|
+
|
|
397
|
+
_TOKEN_RE = re.compile(rb"\((?:[^()\\]|\\.)*\)|T\*|Td|TD|ET|Tj|TJ", re.S)
|
|
398
|
+
_OCTAL_RE = re.compile(rb"\\([0-7]{1,3})")
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def _decode_pdf_string(raw: bytes) -> str:
|
|
402
|
+
out = bytearray()
|
|
403
|
+
i = 0
|
|
404
|
+
n = len(raw)
|
|
405
|
+
simple = {ord("n"): 10, ord("r"): 13, ord("t"): 9, ord("b"): 8, ord("f"): 12,
|
|
406
|
+
ord("("): 40, ord(")"): 41, ord("\\"): 92}
|
|
407
|
+
while i < n:
|
|
408
|
+
ch = raw[i]
|
|
409
|
+
if ch == 0x5C and i + 1 < n:
|
|
410
|
+
nxt = raw[i + 1]
|
|
411
|
+
if nxt in simple:
|
|
412
|
+
out.append(simple[nxt])
|
|
413
|
+
i += 2
|
|
414
|
+
continue
|
|
415
|
+
m = _OCTAL_RE.match(raw, i)
|
|
416
|
+
if m:
|
|
417
|
+
out.append(int(m.group(1), 8) & 0xFF)
|
|
418
|
+
i = m.end()
|
|
419
|
+
continue
|
|
420
|
+
i += 2
|
|
421
|
+
continue
|
|
422
|
+
out.append(ch)
|
|
423
|
+
i += 1
|
|
424
|
+
return out.decode("latin-1", errors="replace")
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
def _iter_pdf_streams(data: bytes):
|
|
428
|
+
for match in re.finditer(rb"stream\r?\n", data):
|
|
429
|
+
start = match.end()
|
|
430
|
+
end = data.find(b"endstream", start)
|
|
431
|
+
if end < 0:
|
|
432
|
+
continue
|
|
433
|
+
yield data[start:end].rstrip(b"\r\n")
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def _inflate(raw: bytes):
|
|
437
|
+
try:
|
|
438
|
+
return zlib.decompress(raw)
|
|
439
|
+
except Exception:
|
|
440
|
+
pass
|
|
441
|
+
try:
|
|
442
|
+
return zlib.decompressobj().decompress(raw)
|
|
443
|
+
except Exception:
|
|
444
|
+
return None
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def _content_stream_text(stream: bytes) -> str:
|
|
448
|
+
lines: list[str] = []
|
|
449
|
+
current: list[str] = []
|
|
450
|
+
for match in _TOKEN_RE.finditer(stream):
|
|
451
|
+
token = match.group(0)
|
|
452
|
+
if token.startswith(b"("):
|
|
453
|
+
current.append(_decode_pdf_string(token[1:-1]))
|
|
454
|
+
elif token in (b"Td", b"TD", b"T*", b"ET"):
|
|
455
|
+
if current:
|
|
456
|
+
lines.append("".join(current))
|
|
457
|
+
current = []
|
|
458
|
+
if current:
|
|
459
|
+
lines.append("".join(current))
|
|
460
|
+
return "\n".join(lines)
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def convert_pdf(src: str, opt) -> tuple[str, list[str]]:
|
|
464
|
+
note: list[str] = []
|
|
465
|
+
|
|
466
|
+
for module, label in (("pypdf", "pypdf"), ("PyPDF2", "PyPDF2"), ("pdfminer", "pdfminer.six")):
|
|
467
|
+
try:
|
|
468
|
+
if module == "pdfminer":
|
|
469
|
+
from pdfminer.high_level import extract_text # type: ignore
|
|
470
|
+
text = extract_text(src)
|
|
471
|
+
else:
|
|
472
|
+
mod = __import__(module)
|
|
473
|
+
reader = mod.PdfReader(src)
|
|
474
|
+
pages = []
|
|
475
|
+
for page in reader.pages:
|
|
476
|
+
pages.append(page.extract_text() or "")
|
|
477
|
+
text = "\n\n".join(pages)
|
|
478
|
+
note.append("PDF 文本由 %s 提取;图片、表格结构与扫描件内容未提取。" % label)
|
|
479
|
+
return (text or "").strip(), note
|
|
480
|
+
except ImportError:
|
|
481
|
+
continue
|
|
482
|
+
except Exception as exc:
|
|
483
|
+
note.append("%s 提取 PDF 失败(%s),已改用内置极简提取。" % (label, exc))
|
|
484
|
+
break
|
|
485
|
+
|
|
486
|
+
with open(src, "rb") as handle:
|
|
487
|
+
data = handle.read()
|
|
488
|
+
chunks = []
|
|
489
|
+
for raw in _iter_pdf_streams(data):
|
|
490
|
+
stream = _inflate(raw)
|
|
491
|
+
if not stream:
|
|
492
|
+
continue
|
|
493
|
+
if b"Tj" not in stream and b"TJ" not in stream:
|
|
494
|
+
continue
|
|
495
|
+
text = _content_stream_text(stream)
|
|
496
|
+
if text.strip():
|
|
497
|
+
chunks.append(text)
|
|
498
|
+
note.append(
|
|
499
|
+
"内置极简 PDF 提取:只能识别简单 Latin 编码文本流,"
|
|
500
|
+
"中文/CID 字体、表格、图片与扫描件通常无法提取。安装 MarkItDown 可获得完整保真度。"
|
|
501
|
+
)
|
|
502
|
+
return "\n\n".join(chunks).strip(), note
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
# --------------------------------------------------------------------------
|
|
506
|
+
# dispatch
|
|
507
|
+
# --------------------------------------------------------------------------
|
|
508
|
+
|
|
509
|
+
CONVERTERS = {
|
|
510
|
+
".docx": convert_docx,
|
|
511
|
+
".docm": convert_docx,
|
|
512
|
+
".xlsx": convert_xlsx,
|
|
513
|
+
".xlsm": convert_xlsx,
|
|
514
|
+
".pptx": convert_pptx,
|
|
515
|
+
".pptm": convert_pptx,
|
|
516
|
+
".pdf": convert_pdf,
|
|
517
|
+
".csv": convert_delimited,
|
|
518
|
+
".tsv": convert_delimited,
|
|
519
|
+
".txt": convert_text,
|
|
520
|
+
".md": convert_text,
|
|
521
|
+
".html": convert_html,
|
|
522
|
+
".htm": convert_html,
|
|
523
|
+
".rtf": convert_rtf,
|
|
524
|
+
".json": convert_json,
|
|
525
|
+
}
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
def main() -> int:
|
|
529
|
+
parser = argparse.ArgumentParser(description="dsh-plugin-office-markdown fallback converter")
|
|
530
|
+
parser.add_argument("--input", required=True)
|
|
531
|
+
parser.add_argument("--output", required=True)
|
|
532
|
+
parser.add_argument("--max-rows", type=int, default=400)
|
|
533
|
+
parser.add_argument("--max-cols", type=int, default=24)
|
|
534
|
+
parser.add_argument("--max-cells", type=int, default=20000)
|
|
535
|
+
parser.add_argument("--max-slides", type=int, default=0)
|
|
536
|
+
opt = parser.parse_args()
|
|
537
|
+
|
|
538
|
+
src = os.path.abspath(opt.input)
|
|
539
|
+
if not os.path.isfile(src):
|
|
540
|
+
print("[fallback] 输入文件不存在:%s" % src, file=sys.stderr)
|
|
541
|
+
return 3
|
|
542
|
+
|
|
543
|
+
ext = os.path.splitext(src)[1].lower()
|
|
544
|
+
if ext in UNSUPPORTED:
|
|
545
|
+
print(
|
|
546
|
+
"[fallback] 不支持 %s;请安装 MarkItDown(uvx markitdown / pip install \"markitdown[all]\")后重试。"
|
|
547
|
+
% UNSUPPORTED[ext],
|
|
548
|
+
file=sys.stderr,
|
|
549
|
+
)
|
|
550
|
+
return 2
|
|
551
|
+
|
|
552
|
+
converter = CONVERTERS.get(ext)
|
|
553
|
+
if converter is None:
|
|
554
|
+
print("[fallback] 未知文件类型 %s,内置兜底转换器无法处理。" % (ext or "(无扩展名)"), file=sys.stderr)
|
|
555
|
+
return 2
|
|
556
|
+
|
|
557
|
+
try:
|
|
558
|
+
body, note = converter(src, opt)
|
|
559
|
+
except Exception as exc: # noqa: BLE001
|
|
560
|
+
print("[fallback] 转换失败:%s: %s" % (type(exc).__name__, exc), file=sys.stderr)
|
|
561
|
+
return 3
|
|
562
|
+
|
|
563
|
+
body = (body or "").strip()
|
|
564
|
+
if not body:
|
|
565
|
+
print("[fallback] 未能提取到任何文本(可能是扫描件、纯图片或特殊编码)。", file=sys.stderr)
|
|
566
|
+
return 4
|
|
567
|
+
|
|
568
|
+
header = BANNER + "> 源文件:`%s`\n> 转换器:内置兜底(fallback.py)\n\n" % os.path.basename(src)
|
|
569
|
+
footer = ""
|
|
570
|
+
if note:
|
|
571
|
+
footer = "\n\n---\n\n**转换提示**\n\n" + "\n".join("- " + n for n in note) + "\n"
|
|
572
|
+
|
|
573
|
+
with open(opt.output, "w", encoding="utf-8", newline="\n") as handle:
|
|
574
|
+
handle.write(header + body + footer)
|
|
575
|
+
return 0
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
if __name__ == "__main__":
|
|
579
|
+
sys.exit(main())
|