dsh-plugin-office-markdown 1.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,579 @@
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+ """dsh-plugin-office-markdown — bundled fallback converter (limited fidelity).
4
+
5
+ Used ONLY when `uvx markitdown`, the `markitdown` CLI and `python -m markitdown`
6
+ are all unavailable. It extracts text and tables from the Office containers with
7
+ the libraries that ship inside the DSH Python runtime (python-docx, openpyxl,
8
+ python-pptx) and emits Markdown.
9
+
10
+ Fidelity is intentionally declared "limited": styling, charts, images, comments,
11
+ tracked changes, footnotes and scanned-PDF OCR are not recovered.
12
+
13
+ Usage:
14
+ python fallback.py --input <source> --output <file.md>
15
+ [--max-rows N] [--max-cols N] [--max-cells N] [--max-slides N]
16
+
17
+ Exit codes: 0 ok, 2 unsupported format, 3 conversion error, 4 empty result.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import argparse
23
+ import csv
24
+ import io
25
+ import json
26
+ import os
27
+ import re
28
+ import sys
29
+ import zlib
30
+
31
+ BANNER = (
32
+ "<!-- 由 dsh-plugin-office-markdown 内置兜底转换器生成(非 MarkItDown),"
33
+ "版式/图表/批注/图片等信息可能缺失,保真度有限。 -->\n"
34
+ )
35
+
36
+ UNSUPPORTED = {
37
+ ".doc": "旧版 Word 97-2003(.doc)",
38
+ ".xls": "旧版 Excel 97-2003(.xls)",
39
+ ".ppt": "旧版 PowerPoint 97-2003(.ppt)",
40
+ ".msg": "Outlook 邮件(.msg)",
41
+ ".epub": "EPUB 电子书(.epub)",
42
+ ".odt": "OpenDocument 文本(.odt)",
43
+ ".ods": "OpenDocument 表格(.ods)",
44
+ ".odp": "OpenDocument 演示(.odp)",
45
+ }
46
+
47
+
48
+ # --------------------------------------------------------------------------
49
+ # helpers
50
+ # --------------------------------------------------------------------------
51
+
52
+ def cell(value) -> str:
53
+ if value is None:
54
+ return ""
55
+ text = str(value).replace("\r\n", "\n").replace("\r", "\n").strip()
56
+ text = text.replace("|", "\\|")
57
+ if "\n" in text:
58
+ text = text.replace("\n", "<br>")
59
+ return text
60
+
61
+
62
+ def md_table(rows: list[list[str]], max_rows: int, max_cols: int, note: list[str], label: str) -> str:
63
+ rows = [r for r in rows if any(c for c in r)]
64
+ if not rows:
65
+ return ""
66
+ truncated_cols = False
67
+ if max_cols and any(len(r) > max_cols for r in rows):
68
+ rows = [r[:max_cols] for r in rows]
69
+ truncated_cols = True
70
+ width = max(len(r) for r in rows)
71
+ rows = [r + [""] * (width - len(r)) for r in rows]
72
+ truncated_rows = False
73
+ if max_rows and len(rows) > max_rows:
74
+ rows = rows[:max_rows]
75
+ truncated_rows = True
76
+ out = ["| " + " | ".join(rows[0]) + " |", "| " + " | ".join(["---"] * width) + " |"]
77
+ for r in rows[1:]:
78
+ out.append("| " + " | ".join(r) + " |")
79
+ if truncated_rows or truncated_cols:
80
+ bits = []
81
+ if truncated_rows:
82
+ bits.append("仅显示前 %d 行" % max_rows)
83
+ if truncated_cols:
84
+ bits.append("仅显示前 %d 列" % max_cols)
85
+ note.append("%s:%s。" % (label, "、".join(bits)))
86
+ return "\n".join(out)
87
+
88
+
89
+ # --------------------------------------------------------------------------
90
+ # docx
91
+ # --------------------------------------------------------------------------
92
+
93
+ _HEADING_RE = re.compile(r"(?:heading|标题)\s*(\d+)", re.I)
94
+
95
+
96
+ def convert_docx(src: str, opt) -> tuple[str, list[str]]:
97
+ try:
98
+ import docx # python-docx
99
+ from docx.table import Table
100
+ from docx.text.paragraph import Paragraph
101
+ except Exception as exc: # pragma: no cover
102
+ raise RuntimeError("缺少 python-docx:%s" % exc)
103
+
104
+ document = docx.Document(src)
105
+ note: list[str] = []
106
+ out: list[str] = []
107
+
108
+ def emit_para(para) -> None:
109
+ text = (para.text or "").strip()
110
+ if not text:
111
+ if out and out[-1] != "":
112
+ out.append("")
113
+ return
114
+ style = ""
115
+ try:
116
+ style = (para.style.name or "")
117
+ except Exception:
118
+ style = ""
119
+ low = style.strip().lower()
120
+ m = _HEADING_RE.search(style)
121
+ if m:
122
+ level = max(1, min(int(m.group(1)), 6))
123
+ out.append("#" * level + " " + text)
124
+ elif low in ("title", "标题"):
125
+ # python-docx 的 add_heading(text, 0) 用的是内建样式 "Title",不含数字,
126
+ # 匹配不到 _HEADING_RE,旧版会退化成普通段落。
127
+ out.append("# " + text)
128
+ elif low in ("subtitle", "副标题"):
129
+ out.append("## " + text)
130
+ elif low.startswith("list") or "列表" in style:
131
+ out.append("- " + text)
132
+ else:
133
+ out.append(text)
134
+
135
+ def emit_table(table, index: int) -> None:
136
+ rows = []
137
+ for row in table.rows:
138
+ rows.append([cell(c.text) for c in row.cells])
139
+ md = md_table(rows, opt.max_rows, opt.max_cols, note, "表格 %d" % index)
140
+ if md:
141
+ out.append("")
142
+ out.append("**表格 %d**" % index)
143
+ out.append("")
144
+ out.append(md)
145
+
146
+ # 按正文真实顺序遍历段落与表格,让表格留在它原本出现的位置,
147
+ # 而不是像旧版那样把全部表格统一堆到文末。
148
+ table_index = 0
149
+ for child in document.element.body.iterchildren():
150
+ tag = child.tag.rsplit("}", 1)[-1]
151
+ if tag == "p":
152
+ emit_para(Paragraph(child, document))
153
+ elif tag == "tbl":
154
+ table_index += 1
155
+ emit_table(Table(child, document), table_index)
156
+
157
+ # 兜底:正文里什么都没取到(例如内容都在文本框内)时,
158
+ # 退回 python-docx 的段落 / 表格列表接口。
159
+ if not out:
160
+ for para in document.paragraphs:
161
+ emit_para(para)
162
+ for index, table in enumerate(document.tables, start=1):
163
+ emit_table(table, index)
164
+
165
+ return "\n".join(out), note
166
+
167
+
168
+ # --------------------------------------------------------------------------
169
+ # xlsx
170
+ # --------------------------------------------------------------------------
171
+
172
+ def convert_xlsx(src: str, opt) -> tuple[str, list[str]]:
173
+ try:
174
+ import openpyxl
175
+ except Exception as exc: # pragma: no cover
176
+ raise RuntimeError("缺少 openpyxl:%s" % exc)
177
+
178
+ workbook = openpyxl.load_workbook(src, read_only=True, data_only=True)
179
+ note: list[str] = []
180
+ out: list[str] = []
181
+ cells = 0
182
+ numeric_notes = ("openpyxl 以 data_only 读取:公式单元格只会显示上次由 Excel 保存的缓存值,"
183
+ "若线程中从未计算则显示为空。")
184
+
185
+ for sheet in workbook.worksheets:
186
+ out.append("## " + str(sheet.title))
187
+ out.append("")
188
+ rows: list[list[str]] = []
189
+ stopped = False
190
+ for raw in sheet.iter_rows(values_only=True):
191
+ if raw is None:
192
+ continue
193
+ cells += len(raw)
194
+ if opt.max_cells and cells > opt.max_cells:
195
+ stopped = True
196
+ break
197
+ rows.append([cell(v) for v in raw])
198
+ md = md_table(rows, opt.max_rows, opt.max_cols, note, "工作表「%s」" % sheet.title)
199
+ if md:
200
+ out.append(md)
201
+ elif not rows:
202
+ out.append("_(空工作表)_")
203
+ if stopped:
204
+ note.append("工作表「%s」:单元格总数超过上限 %d,已提前截断。" % (sheet.title, opt.max_cells))
205
+ out.append("")
206
+
207
+ try:
208
+ workbook.close()
209
+ except Exception:
210
+ pass
211
+
212
+ if any("#" in o for o in out):
213
+ note.append(numeric_notes)
214
+ return "\n".join(out), note
215
+
216
+
217
+ # --------------------------------------------------------------------------
218
+ # pptx
219
+ # --------------------------------------------------------------------------
220
+
221
+ def _shape_text(shape) -> list[str]:
222
+ parts: list[str] = []
223
+ if getattr(shape, "has_text_frame", False):
224
+ for para in shape.text_frame.paragraphs:
225
+ text = "".join(run.text or "" for run in para.runs).strip()
226
+ if not text:
227
+ text = (para.text or "").strip()
228
+ if text:
229
+ parts.append(text)
230
+ if getattr(shape, "has_table", False):
231
+ rows = []
232
+ try:
233
+ for row in shape.table.rows:
234
+ rows.append([cell(c.text) for c in row.cells])
235
+ except Exception:
236
+ rows = []
237
+ if rows:
238
+ parts.append("")
239
+ parts.append("| " + " | ".join(rows[0]) + " |")
240
+ parts.append("| " + " | ".join(["---"] * len(rows[0])) + " |")
241
+ for r in rows[1:]:
242
+ parts.append("| " + " | ".join(r) + " |")
243
+ return parts
244
+
245
+
246
+ def convert_pptx(src: str, opt) -> tuple[str, list[str]]:
247
+ try:
248
+ from pptx import Presentation
249
+ except Exception as exc: # pragma: no cover
250
+ raise RuntimeError("缺少 python-pptx:%s" % exc)
251
+
252
+ presentation = Presentation(src)
253
+ note: list[str] = []
254
+ out: list[str] = []
255
+ total = len(presentation.slides)
256
+
257
+ for index, slide in enumerate(presentation.slides, start=1):
258
+ if opt.max_slides and index > opt.max_slides:
259
+ note.append("演示文稿共 %d 张幻灯片,仅转换前 %d 张。" % (total, opt.max_slides))
260
+ break
261
+ title_shape = None
262
+ try:
263
+ title_shape = slide.shapes.title
264
+ except Exception:
265
+ title_shape = None
266
+ title = ""
267
+ if title_shape is not None:
268
+ try:
269
+ title = (title_shape.text or "").strip()
270
+ except Exception:
271
+ title = ""
272
+ out.append("## 幻灯片 %d%s" % (index, (":" + title) if title else ""))
273
+ out.append("")
274
+ # python-pptx 每次访问 slide.shapes.title 都会重新包装同一段 XML,
275
+ # 用 `shape is slide.shapes.title` 判定会永远为假,导致标题文本重复输出一次;
276
+ # 改为按底层 lxml 元素判等。
277
+ title_element = getattr(title_shape, "_element", None)
278
+ for shape in slide.shapes:
279
+ if title_element is not None and getattr(shape, "_element", None) is title_element:
280
+ continue
281
+ parts = _shape_text(shape)
282
+ if parts:
283
+ out.extend(parts)
284
+ out.append("")
285
+ if slide.has_notes_slide:
286
+ try:
287
+ notes = (slide.notes_slide.notes_text_frame.text or "").strip()
288
+ except Exception:
289
+ notes = ""
290
+ if notes:
291
+ out.append("> 备注:" + notes.replace("\n", " "))
292
+ out.append("")
293
+
294
+ return "\n".join(out), note
295
+
296
+
297
+ # --------------------------------------------------------------------------
298
+ # delimited / plain text
299
+ # --------------------------------------------------------------------------
300
+
301
+ def sniff_delimiter(sample: str, default: str = ",") -> str:
302
+ try:
303
+ return csv.Sniffer().sniff(sample, delimiters=",;\t|").delimiter
304
+ except Exception:
305
+ return default
306
+
307
+
308
+ def convert_delimited(src: str, opt, ext: str) -> tuple[str, list[str]]:
309
+ default = "\t" if ext == ".tsv" else ","
310
+ with open(src, "r", encoding="utf-8-sig", errors="replace", newline="") as handle:
311
+ sample = handle.read(8192)
312
+ handle.seek(0)
313
+ delimiter = sniff_delimiter(sample, default)
314
+ reader = csv.reader(handle, delimiter=delimiter)
315
+ rows = [[cell(v) for v in row] for row in reader]
316
+ note: list[str] = []
317
+ md = md_table(rows, opt.max_rows, opt.max_cols, note, "分隔符文本")
318
+ return md, note
319
+
320
+
321
+ def convert_text(src: str, opt) -> tuple[str, list[str]]:
322
+ with open(src, "r", encoding="utf-8-sig", errors="replace") as handle:
323
+ return handle.read(), []
324
+
325
+
326
+ def convert_json(src: str, opt) -> tuple[str, list[str]]:
327
+ with open(src, "r", encoding="utf-8-sig", errors="replace") as handle:
328
+ raw = handle.read()
329
+ try:
330
+ parsed = json.loads(raw)
331
+ pretty = json.dumps(parsed, ensure_ascii=False, indent=2)
332
+ except Exception:
333
+ pretty = raw
334
+ return "```json\n" + pretty + "\n```", []
335
+
336
+
337
+ # --------------------------------------------------------------------------
338
+ # html
339
+ # --------------------------------------------------------------------------
340
+
341
+ def convert_html(src: str, opt) -> tuple[str, list[str]]:
342
+ from html.parser import HTMLParser
343
+
344
+ class Extractor(HTMLParser):
345
+ def __init__(self) -> None:
346
+ super().__init__(convert_charrefs=True)
347
+ self.parts: list[str] = []
348
+ self.skip = 0
349
+
350
+ def handle_starttag(self, tag, attrs):
351
+ if tag in ("script", "style", "head"):
352
+ self.skip += 1
353
+ elif tag in ("p", "div", "br", "li", "tr", "h1", "h2", "h3", "h4", "h5", "h6"):
354
+ self.parts.append("\n")
355
+ if tag.startswith("h") and len(tag) == 2 and tag[1].isdigit():
356
+ self.parts.append("#" * int(tag[1]) + " ")
357
+
358
+ def handle_endtag(self, tag):
359
+ if tag in ("script", "style", "head") and self.skip:
360
+ self.skip -= 1
361
+
362
+ def handle_data(self, data):
363
+ if self.skip:
364
+ return
365
+ text = data.strip()
366
+ if text:
367
+ self.parts.append(text + " ")
368
+
369
+ extractor = Extractor()
370
+ with open(src, "r", encoding="utf-8-sig", errors="replace") as handle:
371
+ extractor.feed(handle.read())
372
+ text = re.sub(r"[ \t]+", " ", "".join(extractor.parts))
373
+ text = re.sub(r"\n{3,}", "\n\n", text)
374
+ return text.strip(), ["HTML 兜底提取仅保留可见文本,链接、图片与 CSS 布局丢失。"]
375
+
376
+
377
+ # --------------------------------------------------------------------------
378
+ # rtf
379
+ # --------------------------------------------------------------------------
380
+
381
+ def convert_rtf(src: str, opt) -> tuple[str, list[str]]:
382
+ with open(src, "rb") as handle:
383
+ data = handle.read().decode("latin-1", errors="replace")
384
+ data = re.sub(r"\\'([0-9a-fA-F]{2})", lambda m: chr(int(m.group(1), 16)), data)
385
+ data = re.sub(r"\\u(-?\d+)\?", lambda m: chr(int(m.group(1)) % 65536), data)
386
+ data = re.sub(r"\\par[d]?", "\n", data)
387
+ data = re.sub(r"\\[a-zA-Z]+-?\d* ?", "", data)
388
+ data = data.replace("{", "").replace("}", "")
389
+ data = re.sub(r"\n{3,}", "\n\n", data)
390
+ return data.strip(), ["RTF 兜底提取仅保留纯文本,格式、表格与图片丢失。"]
391
+
392
+
393
+ # --------------------------------------------------------------------------
394
+ # pdf
395
+ # --------------------------------------------------------------------------
396
+
397
+ _TOKEN_RE = re.compile(rb"\((?:[^()\\]|\\.)*\)|T\*|Td|TD|ET|Tj|TJ", re.S)
398
+ _OCTAL_RE = re.compile(rb"\\([0-7]{1,3})")
399
+
400
+
401
+ def _decode_pdf_string(raw: bytes) -> str:
402
+ out = bytearray()
403
+ i = 0
404
+ n = len(raw)
405
+ simple = {ord("n"): 10, ord("r"): 13, ord("t"): 9, ord("b"): 8, ord("f"): 12,
406
+ ord("("): 40, ord(")"): 41, ord("\\"): 92}
407
+ while i < n:
408
+ ch = raw[i]
409
+ if ch == 0x5C and i + 1 < n:
410
+ nxt = raw[i + 1]
411
+ if nxt in simple:
412
+ out.append(simple[nxt])
413
+ i += 2
414
+ continue
415
+ m = _OCTAL_RE.match(raw, i)
416
+ if m:
417
+ out.append(int(m.group(1), 8) & 0xFF)
418
+ i = m.end()
419
+ continue
420
+ i += 2
421
+ continue
422
+ out.append(ch)
423
+ i += 1
424
+ return out.decode("latin-1", errors="replace")
425
+
426
+
427
+ def _iter_pdf_streams(data: bytes):
428
+ for match in re.finditer(rb"stream\r?\n", data):
429
+ start = match.end()
430
+ end = data.find(b"endstream", start)
431
+ if end < 0:
432
+ continue
433
+ yield data[start:end].rstrip(b"\r\n")
434
+
435
+
436
+ def _inflate(raw: bytes):
437
+ try:
438
+ return zlib.decompress(raw)
439
+ except Exception:
440
+ pass
441
+ try:
442
+ return zlib.decompressobj().decompress(raw)
443
+ except Exception:
444
+ return None
445
+
446
+
447
+ def _content_stream_text(stream: bytes) -> str:
448
+ lines: list[str] = []
449
+ current: list[str] = []
450
+ for match in _TOKEN_RE.finditer(stream):
451
+ token = match.group(0)
452
+ if token.startswith(b"("):
453
+ current.append(_decode_pdf_string(token[1:-1]))
454
+ elif token in (b"Td", b"TD", b"T*", b"ET"):
455
+ if current:
456
+ lines.append("".join(current))
457
+ current = []
458
+ if current:
459
+ lines.append("".join(current))
460
+ return "\n".join(lines)
461
+
462
+
463
+ def convert_pdf(src: str, opt) -> tuple[str, list[str]]:
464
+ note: list[str] = []
465
+
466
+ for module, label in (("pypdf", "pypdf"), ("PyPDF2", "PyPDF2"), ("pdfminer", "pdfminer.six")):
467
+ try:
468
+ if module == "pdfminer":
469
+ from pdfminer.high_level import extract_text # type: ignore
470
+ text = extract_text(src)
471
+ else:
472
+ mod = __import__(module)
473
+ reader = mod.PdfReader(src)
474
+ pages = []
475
+ for page in reader.pages:
476
+ pages.append(page.extract_text() or "")
477
+ text = "\n\n".join(pages)
478
+ note.append("PDF 文本由 %s 提取;图片、表格结构与扫描件内容未提取。" % label)
479
+ return (text or "").strip(), note
480
+ except ImportError:
481
+ continue
482
+ except Exception as exc:
483
+ note.append("%s 提取 PDF 失败(%s),已改用内置极简提取。" % (label, exc))
484
+ break
485
+
486
+ with open(src, "rb") as handle:
487
+ data = handle.read()
488
+ chunks = []
489
+ for raw in _iter_pdf_streams(data):
490
+ stream = _inflate(raw)
491
+ if not stream:
492
+ continue
493
+ if b"Tj" not in stream and b"TJ" not in stream:
494
+ continue
495
+ text = _content_stream_text(stream)
496
+ if text.strip():
497
+ chunks.append(text)
498
+ note.append(
499
+ "内置极简 PDF 提取:只能识别简单 Latin 编码文本流,"
500
+ "中文/CID 字体、表格、图片与扫描件通常无法提取。安装 MarkItDown 可获得完整保真度。"
501
+ )
502
+ return "\n\n".join(chunks).strip(), note
503
+
504
+
505
+ # --------------------------------------------------------------------------
506
+ # dispatch
507
+ # --------------------------------------------------------------------------
508
+
509
+ CONVERTERS = {
510
+ ".docx": convert_docx,
511
+ ".docm": convert_docx,
512
+ ".xlsx": convert_xlsx,
513
+ ".xlsm": convert_xlsx,
514
+ ".pptx": convert_pptx,
515
+ ".pptm": convert_pptx,
516
+ ".pdf": convert_pdf,
517
+ ".csv": convert_delimited,
518
+ ".tsv": convert_delimited,
519
+ ".txt": convert_text,
520
+ ".md": convert_text,
521
+ ".html": convert_html,
522
+ ".htm": convert_html,
523
+ ".rtf": convert_rtf,
524
+ ".json": convert_json,
525
+ }
526
+
527
+
528
+ def main() -> int:
529
+ parser = argparse.ArgumentParser(description="dsh-plugin-office-markdown fallback converter")
530
+ parser.add_argument("--input", required=True)
531
+ parser.add_argument("--output", required=True)
532
+ parser.add_argument("--max-rows", type=int, default=400)
533
+ parser.add_argument("--max-cols", type=int, default=24)
534
+ parser.add_argument("--max-cells", type=int, default=20000)
535
+ parser.add_argument("--max-slides", type=int, default=0)
536
+ opt = parser.parse_args()
537
+
538
+ src = os.path.abspath(opt.input)
539
+ if not os.path.isfile(src):
540
+ print("[fallback] 输入文件不存在:%s" % src, file=sys.stderr)
541
+ return 3
542
+
543
+ ext = os.path.splitext(src)[1].lower()
544
+ if ext in UNSUPPORTED:
545
+ print(
546
+ "[fallback] 不支持 %s;请安装 MarkItDown(uvx markitdown / pip install \"markitdown[all]\")后重试。"
547
+ % UNSUPPORTED[ext],
548
+ file=sys.stderr,
549
+ )
550
+ return 2
551
+
552
+ converter = CONVERTERS.get(ext)
553
+ if converter is None:
554
+ print("[fallback] 未知文件类型 %s,内置兜底转换器无法处理。" % (ext or "(无扩展名)"), file=sys.stderr)
555
+ return 2
556
+
557
+ try:
558
+ body, note = converter(src, opt)
559
+ except Exception as exc: # noqa: BLE001
560
+ print("[fallback] 转换失败:%s: %s" % (type(exc).__name__, exc), file=sys.stderr)
561
+ return 3
562
+
563
+ body = (body or "").strip()
564
+ if not body:
565
+ print("[fallback] 未能提取到任何文本(可能是扫描件、纯图片或特殊编码)。", file=sys.stderr)
566
+ return 4
567
+
568
+ header = BANNER + "> 源文件:`%s`\n> 转换器:内置兜底(fallback.py)\n\n" % os.path.basename(src)
569
+ footer = ""
570
+ if note:
571
+ footer = "\n\n---\n\n**转换提示**\n\n" + "\n".join("- " + n for n in note) + "\n"
572
+
573
+ with open(opt.output, "w", encoding="utf-8", newline="\n") as handle:
574
+ handle.write(header + body + footer)
575
+ return 0
576
+
577
+
578
+ if __name__ == "__main__":
579
+ sys.exit(main())