pi-quiver 5.2.4 → 5.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/README.md +36 -10
- package/dist/bin/pi-quiver.js +702 -259
- package/dist/lib/unpdf-worker.js +58 -0
- package/extensions/doc_to_md.ts +52 -62
- package/lib/doc-to-md-bundle.ts +139 -0
- package/lib/doc-to-md-core.ts +261 -301
- package/lib/doc-to-md-handle.ts +108 -0
- package/lib/doc-to-md-options.ts +186 -0
- package/lib/unpdf-worker.ts +54 -0
- package/package.json +6 -4
- package/scripts/doc_to_md.py +369 -0
- package/scripts/pdf_to_md.py +0 -30
|
@@ -0,0 +1,369 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""doc_to_md child. argv[1] = mode (info | pdf-primary | pdf-fallback | xlsx); options JSON on stdin;
|
|
3
|
+
one JSON result on stdout. Exit 0 ok, 1 conversion failure (traceback on stderr), 3 user error
|
|
4
|
+
({"error", "pageCount"} on stdout). Library chatter is redirected to stderr so stdout is the result only.
|
|
5
|
+
Imports `pymupdf` / `pymupdf4llm` (never the deprecated `fitz` alias)."""
|
|
6
|
+
import contextlib
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import re
|
|
10
|
+
import shutil
|
|
11
|
+
import sys
|
|
12
|
+
import tempfile
|
|
13
|
+
import traceback
|
|
14
|
+
import warnings
|
|
15
|
+
|
|
16
|
+
SEP = "\n\n--- end of page.page_number={n} ---\n\n"
|
|
17
|
+
DEGRADED_NOTE = "degraded: PyMuPDF text extraction - layout/tables not preserved"
|
|
18
|
+
MARKDOWN_IMAGE_RE = re.compile(r"(!\[[^\]]*\]\(\s*)(?:<([^>]+)>|([^)]*?))(\s*\))")
|
|
19
|
+
HTML_IMAGE_RE = re.compile(
|
|
20
|
+
r"(<img\b[^>]*?\bsrc\s*=\s*)(?:\"([^\"]*)\"|'([^']*)'|([^\s\"'=<>`]+))", re.IGNORECASE
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def rewrite_image_destinations(md, sources):
|
|
25
|
+
def markdown(match):
|
|
26
|
+
dest = match.group(2) if match.group(2) is not None else match.group(3)
|
|
27
|
+
target = sources.get(dest)
|
|
28
|
+
if target is None:
|
|
29
|
+
return match.group(0)
|
|
30
|
+
if match.group(2) is not None:
|
|
31
|
+
return f"{match.group(1)}<{target}>{match.group(4)}"
|
|
32
|
+
return f"{match.group(1)}{target}{match.group(4)}"
|
|
33
|
+
|
|
34
|
+
def html(match):
|
|
35
|
+
dest = next(value for value in match.groups()[1:] if value is not None)
|
|
36
|
+
target = sources.get(dest)
|
|
37
|
+
if target is None:
|
|
38
|
+
return match.group(0)
|
|
39
|
+
quote = '"' if match.group(2) is not None else "'" if match.group(3) is not None else ""
|
|
40
|
+
return f"{match.group(1)}{quote}{target}{quote}"
|
|
41
|
+
|
|
42
|
+
return HTML_IMAGE_RE.sub(html, MARKDOWN_IMAGE_RE.sub(markdown, md))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def user_error(msg, page_count=None):
|
|
46
|
+
json.dump({"error": msg, **({"pageCount": page_count} if page_count is not None else {})}, sys.stdout)
|
|
47
|
+
return 3
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def check_pages(pages, page_count):
|
|
51
|
+
if pages is None:
|
|
52
|
+
return list(range(1, page_count + 1))
|
|
53
|
+
bad = [p for p in pages if p < 1 or p > page_count]
|
|
54
|
+
if bad:
|
|
55
|
+
raise UserError(f"pages out of range: {', '.join(map(str, bad))} (document has {page_count} pages)", page_count)
|
|
56
|
+
return pages
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class UserError(Exception):
|
|
60
|
+
def __init__(self, msg, page_count=None):
|
|
61
|
+
super().__init__(msg)
|
|
62
|
+
self.page_count = page_count
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def open_pdf(path):
|
|
66
|
+
import pymupdf
|
|
67
|
+
doc = pymupdf.open(path)
|
|
68
|
+
if doc.needs_pass:
|
|
69
|
+
raise UserError("Password-protected PDF", doc.page_count)
|
|
70
|
+
return doc
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def page_dir(staging, n):
|
|
74
|
+
d = os.path.join(staging, f"p{n}")
|
|
75
|
+
os.makedirs(d, exist_ok=True)
|
|
76
|
+
return d
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def mark_done(d):
|
|
80
|
+
open(os.path.join(d, ".done"), "w").close()
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def mode_info(o):
|
|
84
|
+
import pymupdf # noqa: F401
|
|
85
|
+
doc = open_pdf(o["path"])
|
|
86
|
+
meta = {k: v for k, v in (doc.metadata or {}).items() if v}
|
|
87
|
+
toc = [[lvl, title, page] for lvl, title, page in doc.get_toc()]
|
|
88
|
+
return {"pageCount": doc.page_count, "metadata": meta, "toc": toc}
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def mode_pdf_primary(o):
|
|
92
|
+
import pymupdf4llm
|
|
93
|
+
doc = open_pdf(o["path"])
|
|
94
|
+
pages = check_pages(o.get("pages"), doc.page_count)
|
|
95
|
+
staging, out, empty, failed, notes = o["stagingDir"], [], [], [], []
|
|
96
|
+
for n in pages:
|
|
97
|
+
d = page_dir(staging, n)
|
|
98
|
+
try:
|
|
99
|
+
# Space-free temp dir: pymupdf4llm's md_path() mangles paths containing spaces/parens.
|
|
100
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
101
|
+
md = pymupdf4llm.to_markdown(doc, pages=[n - 1], write_images=True, image_path=tmp,
|
|
102
|
+
image_format=o["imageFormat"], dpi=o["imageDpi"],
|
|
103
|
+
use_ocr=False, page_separators=False)
|
|
104
|
+
sources = {}
|
|
105
|
+
for i, f in enumerate(sorted(os.listdir(tmp)), 1):
|
|
106
|
+
dest = f"img{i}{os.path.splitext(f)[1].lower()}"
|
|
107
|
+
source = os.path.join(tmp, f)
|
|
108
|
+
target = f"p{n}/{dest}"
|
|
109
|
+
sources.update({source: target, os.path.realpath(source): target, f: target})
|
|
110
|
+
os.replace(source, os.path.join(d, dest))
|
|
111
|
+
md = rewrite_image_destinations(md, sources)
|
|
112
|
+
if not md.strip():
|
|
113
|
+
empty.append(n)
|
|
114
|
+
out.append(md.rstrip())
|
|
115
|
+
mark_done(d)
|
|
116
|
+
except Exception as exc: # noqa: BLE001
|
|
117
|
+
shutil.rmtree(d, ignore_errors=True)
|
|
118
|
+
failed.append({"page": n, "error": f"{type(exc).__name__}: {exc}"[:300]})
|
|
119
|
+
empty.append(n)
|
|
120
|
+
out.append("")
|
|
121
|
+
out.append(SEP.format(n=n).strip("\n"))
|
|
122
|
+
if failed and len(failed) == len(pages):
|
|
123
|
+
raise RuntimeError("every selected page failed: " + failed[0]["error"])
|
|
124
|
+
return {"markdown": "\n\n".join(out) + "\n", "pages": pages, "pageCount": doc.page_count,
|
|
125
|
+
"emptyPages": empty, "failedPages": failed, "notes": notes}
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def mode_pdf_fallback(o):
|
|
129
|
+
import pymupdf
|
|
130
|
+
doc = open_pdf(o["path"])
|
|
131
|
+
pages = check_pages(o.get("pages"), doc.page_count)
|
|
132
|
+
keep = {int(k): v for k, v in (o.get("keepPages") or {}).items()}
|
|
133
|
+
staging, out, empty, failed = o["stagingDir"], [], [], []
|
|
134
|
+
for n in pages:
|
|
135
|
+
links = [f"" for f in keep.get(n, [])]
|
|
136
|
+
text = ""
|
|
137
|
+
try:
|
|
138
|
+
page = doc[n - 1]
|
|
139
|
+
text = page.get_text("text").strip()
|
|
140
|
+
if n not in keep:
|
|
141
|
+
d = page_dir(staging, n)
|
|
142
|
+
i = 0
|
|
143
|
+
for info in page.get_image_info(xrefs=True):
|
|
144
|
+
i += 1
|
|
145
|
+
xref = info.get("xref", 0)
|
|
146
|
+
if xref > 0:
|
|
147
|
+
img = doc.extract_image(xref)
|
|
148
|
+
name = f"img{i}.{img['ext'].lower()}"
|
|
149
|
+
with open(os.path.join(d, name), "wb") as fh:
|
|
150
|
+
fh.write(img["image"])
|
|
151
|
+
else:
|
|
152
|
+
name = f"img{i}.{o['imageFormat']}"
|
|
153
|
+
page.get_pixmap(clip=pymupdf.Rect(info["bbox"]), dpi=o["imageDpi"]).save(os.path.join(d, name))
|
|
154
|
+
links.append(f"")
|
|
155
|
+
mark_done(d)
|
|
156
|
+
except Exception as exc: # noqa: BLE001
|
|
157
|
+
shutil.rmtree(os.path.join(staging, f"p{n}"), ignore_errors=True)
|
|
158
|
+
text = ""
|
|
159
|
+
links = [f"" for f in keep.get(n, [])]
|
|
160
|
+
failed.append({"page": n, "error": f"{type(exc).__name__}: {exc}"[:300]})
|
|
161
|
+
if not text:
|
|
162
|
+
empty.append(n)
|
|
163
|
+
out.append("\n\n".join(x for x in [text, "\n".join(links)] if x))
|
|
164
|
+
out.append(SEP.format(n=n).strip("\n"))
|
|
165
|
+
if failed and len(failed) == len(pages):
|
|
166
|
+
raise RuntimeError("every selected page failed: " + failed[0]["error"])
|
|
167
|
+
return {"markdown": "\n\n".join(out) + "\n", "pages": pages, "pageCount": doc.page_count,
|
|
168
|
+
"emptyPages": empty, "failedPages": failed, "notes": [DEGRADED_NOTE]}
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def esc(v):
|
|
172
|
+
if v is None:
|
|
173
|
+
return ""
|
|
174
|
+
if isinstance(v, bool):
|
|
175
|
+
return "TRUE" if v else "FALSE"
|
|
176
|
+
if isinstance(v, float):
|
|
177
|
+
return repr(v)
|
|
178
|
+
import datetime
|
|
179
|
+
if isinstance(v, (datetime.date, datetime.datetime)):
|
|
180
|
+
return v.isoformat()
|
|
181
|
+
return str(v).replace("\\", "\\\\").replace("|", "\\|").replace("\r\n", "<br>").replace("\n", "<br>")
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def col_letter(i):
|
|
185
|
+
from openpyxl.utils import get_column_letter
|
|
186
|
+
return get_column_letter(i)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def mode_xlsx(o):
|
|
190
|
+
path, staging, budget = o["path"], o["stagingDir"], o["maxCellsPerSheet"]
|
|
191
|
+
notes, images = [], []
|
|
192
|
+
if path.lower().endswith(".xls"):
|
|
193
|
+
return mode_xls(o)
|
|
194
|
+
import openpyxl
|
|
195
|
+
wb_f = openpyxl.load_workbook(path, data_only=False)
|
|
196
|
+
wb_v = openpyxl.load_workbook(path, data_only=True)
|
|
197
|
+
inventory, sections = [], []
|
|
198
|
+
for idx, ws in enumerate(wb_f.worksheets, 1):
|
|
199
|
+
wv = wb_v[ws.title]
|
|
200
|
+
rows, cols = ws.max_row, ws.max_column
|
|
201
|
+
r_lim, c_lim, truncated = rows, cols, False
|
|
202
|
+
if rows * cols > budget:
|
|
203
|
+
truncated = True
|
|
204
|
+
r_lim = max(1, budget // cols)
|
|
205
|
+
if r_lim == 1 and cols > budget:
|
|
206
|
+
c_lim = budget
|
|
207
|
+
hidden = ws.sheet_state != "visible"
|
|
208
|
+
inventory.append(f"- {idx}. {ws.title}{' hidden' if hidden else ''} - {rows} x {cols}{' truncated' if truncated else ''}")
|
|
209
|
+
lines = [f"## {ws.title}"]
|
|
210
|
+
if hidden:
|
|
211
|
+
lines.append("Hidden sheet")
|
|
212
|
+
merged = [str(r) for r in ws.merged_cells.ranges]
|
|
213
|
+
if merged:
|
|
214
|
+
lines.append("Merged: " + ", ".join(merged))
|
|
215
|
+
hr = [str(r) for r, dim in ws.row_dimensions.items() if dim.hidden]
|
|
216
|
+
if hr:
|
|
217
|
+
lines.append("Hidden rows: " + ",".join(hr))
|
|
218
|
+
hc = []
|
|
219
|
+
for dim in ws.column_dimensions.values():
|
|
220
|
+
if dim.hidden:
|
|
221
|
+
hc += [col_letter(i) for i in range(dim.min, dim.max + 1)]
|
|
222
|
+
if hc:
|
|
223
|
+
lines.append("Hidden cols: " + ",".join(hc))
|
|
224
|
+
if truncated:
|
|
225
|
+
lines.append(f"Truncated: showing rows 1-{r_lim} of {rows}, cols A-{col_letter(c_lim)} of {cols}")
|
|
226
|
+
continuation = set()
|
|
227
|
+
for rng in ws.merged_cells.ranges:
|
|
228
|
+
for r in range(rng.min_row, rng.max_row + 1):
|
|
229
|
+
for c in range(rng.min_col, rng.max_col + 1):
|
|
230
|
+
if (r, c) != (rng.min_row, rng.min_col):
|
|
231
|
+
continuation.add((r, c))
|
|
232
|
+
header = "| | " + " | ".join(col_letter(c) for c in range(1, c_lim + 1)) + " |"
|
|
233
|
+
lines += [header, "|---|" + "---|" * c_lim]
|
|
234
|
+
for r in range(1, r_lim + 1):
|
|
235
|
+
cells = []
|
|
236
|
+
for c in range(1, c_lim + 1):
|
|
237
|
+
if (r, c) in continuation:
|
|
238
|
+
cells.append("")
|
|
239
|
+
continue
|
|
240
|
+
f, v = ws.cell(r, c).value, wv.cell(r, c).value
|
|
241
|
+
if isinstance(f, str) and f.startswith("="):
|
|
242
|
+
cells.append(f"{esc(v) if v is not None else '(no cached result)'} ({esc(f)})")
|
|
243
|
+
else:
|
|
244
|
+
cells.append(esc(f))
|
|
245
|
+
lines.append(f"| {r} | " + " | ".join(cells) + " |")
|
|
246
|
+
for i, img in enumerate(getattr(ws, "_images", []), 1):
|
|
247
|
+
try:
|
|
248
|
+
fmt = (getattr(img, "format", None) or "png").lower()
|
|
249
|
+
name = f"s{idx}-{i}.{fmt}"
|
|
250
|
+
data = img._data() if callable(getattr(img, "_data", None)) else img.ref.getvalue()
|
|
251
|
+
with open(os.path.join(staging, name), "wb") as fh:
|
|
252
|
+
fh.write(data)
|
|
253
|
+
images.append({"sheetIndex": idx, "file": name})
|
|
254
|
+
lines.append(f"")
|
|
255
|
+
except Exception as exc: # noqa: BLE001
|
|
256
|
+
notes.append(f"sheet {ws.title}: image {i} not extracted ({type(exc).__name__})")
|
|
257
|
+
if not hasattr(ws, "_images"):
|
|
258
|
+
notes.append("Images: unavailable")
|
|
259
|
+
sections.append("\n".join(lines))
|
|
260
|
+
md = "## Sheets\n" + "\n".join(inventory) + "\n\n" + "\n\n".join(sections) + "\n"
|
|
261
|
+
return {"markdown": md, "images": images, "notes": notes}
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def mode_xls(o):
|
|
265
|
+
import xlrd
|
|
266
|
+
path, budget = o["path"], o["maxCellsPerSheet"]
|
|
267
|
+
book = xlrd.open_workbook(path, formatting_info=True, logfile=sys.stderr)
|
|
268
|
+
inventory, sections = [], []
|
|
269
|
+
for idx in range(book.nsheets):
|
|
270
|
+
sh = book.sheet_by_index(idx)
|
|
271
|
+
rows, cols = sh.nrows, sh.ncols
|
|
272
|
+
r_lim, c_lim, truncated = rows, cols, False
|
|
273
|
+
if rows * cols > budget:
|
|
274
|
+
truncated = True
|
|
275
|
+
r_lim = max(1, budget // max(cols, 1))
|
|
276
|
+
if r_lim == 1 and cols > budget:
|
|
277
|
+
c_lim = budget
|
|
278
|
+
hidden = sh.visibility != 0
|
|
279
|
+
inventory.append(f"- {idx + 1}. {sh.name}{' hidden' if hidden else ''} - {rows} x {cols}{' truncated' if truncated else ''}")
|
|
280
|
+
lines = [f"## {sh.name}", "Formulas: unavailable (.xls via xlrd); Images: unavailable"]
|
|
281
|
+
if hidden:
|
|
282
|
+
lines.append("Hidden sheet")
|
|
283
|
+
merged = [f"{col_letter(c0 + 1)}{r0 + 1}:{col_letter(c1)}{r1}" for r0, r1, c0, c1 in sh.merged_cells]
|
|
284
|
+
if merged:
|
|
285
|
+
lines.append("Merged: " + ", ".join(merged))
|
|
286
|
+
hr = [str(r + 1) for r, info in sh.rowinfo_map.items() if info.hidden]
|
|
287
|
+
if hr:
|
|
288
|
+
lines.append("Hidden rows: " + ",".join(hr))
|
|
289
|
+
hc = [col_letter(c + 1) for c, info in sh.colinfo_map.items() if info.hidden]
|
|
290
|
+
if hc:
|
|
291
|
+
lines.append("Hidden cols: " + ",".join(hc))
|
|
292
|
+
if truncated:
|
|
293
|
+
lines.append(f"Truncated: showing rows 1-{r_lim} of {rows}, cols A-{col_letter(c_lim)} of {cols}")
|
|
294
|
+
continuation = {(r, c) for r0, r1, c0, c1 in sh.merged_cells for r in range(r0, r1) for c in range(c0, c1) if (r, c) != (r0, c0)}
|
|
295
|
+
lines += ["| | " + " | ".join(col_letter(c + 1) for c in range(c_lim)) + " |", "|---|" + "---|" * c_lim]
|
|
296
|
+
for r in range(r_lim):
|
|
297
|
+
cells = []
|
|
298
|
+
for c in range(c_lim):
|
|
299
|
+
if (r, c) in continuation:
|
|
300
|
+
cells.append("")
|
|
301
|
+
continue
|
|
302
|
+
cell = sh.cell(r, c)
|
|
303
|
+
v = cell.value
|
|
304
|
+
if cell.ctype == xlrd.XL_CELL_DATE:
|
|
305
|
+
v = xlrd.xldate_as_datetime(v, book.datemode).isoformat()
|
|
306
|
+
elif cell.ctype == xlrd.XL_CELL_BOOLEAN:
|
|
307
|
+
v = bool(v)
|
|
308
|
+
elif cell.ctype == xlrd.XL_CELL_ERROR:
|
|
309
|
+
v = xlrd.error_text_from_code.get(int(v), f"#ERR{int(v)}")
|
|
310
|
+
elif cell.ctype == xlrd.XL_CELL_EMPTY:
|
|
311
|
+
v = None
|
|
312
|
+
cells.append(esc(v))
|
|
313
|
+
lines.append(f"| {r + 1} | " + " | ".join(cells) + " |")
|
|
314
|
+
sections.append("\n".join(lines))
|
|
315
|
+
md = "## Sheets\n" + "\n".join(inventory) + "\n\n" + "\n\n".join(sections) + "\n"
|
|
316
|
+
return {"markdown": md, "images": [], "notes": []}
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def mode_info_excel(o):
|
|
320
|
+
path = o["path"]
|
|
321
|
+
sheets = []
|
|
322
|
+
if path.lower().endswith(".xls"):
|
|
323
|
+
import xlrd
|
|
324
|
+
book = xlrd.open_workbook(path, formatting_info=True, logfile=sys.stderr)
|
|
325
|
+
for idx in range(book.nsheets):
|
|
326
|
+
sh = book.sheet_by_index(idx)
|
|
327
|
+
sheets.append({"name": sh.name, "index": idx + 1, "hidden": sh.visibility != 0, "rows": sh.nrows, "cols": sh.ncols,
|
|
328
|
+
"hiddenRows": sum(1 for i in sh.rowinfo_map.values() if i.hidden),
|
|
329
|
+
"hiddenCols": sum(1 for i in sh.colinfo_map.values() if i.hidden)})
|
|
330
|
+
else:
|
|
331
|
+
import openpyxl
|
|
332
|
+
wb = openpyxl.load_workbook(path, data_only=True)
|
|
333
|
+
for idx, ws in enumerate(wb.worksheets, 1):
|
|
334
|
+
hc = sum(d.max - d.min + 1 for d in ws.column_dimensions.values() if d.hidden)
|
|
335
|
+
sheets.append({"name": ws.title, "index": idx, "hidden": ws.sheet_state != "visible", "rows": ws.max_row, "cols": ws.max_column,
|
|
336
|
+
"hiddenRows": sum(1 for d in ws.row_dimensions.values() if d.hidden), "hiddenCols": hc})
|
|
337
|
+
return {"sheets": sheets}
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
MODES = {"info": None, "pdf-primary": mode_pdf_primary, "pdf-fallback": mode_pdf_fallback, "xlsx": mode_xlsx}
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def main():
|
|
344
|
+
if len(sys.argv) != 2 or sys.argv[1] not in MODES:
|
|
345
|
+
print("usage: doc_to_md.py <info|pdf-primary|pdf-fallback|xlsx> (options JSON on stdin)", file=sys.stderr)
|
|
346
|
+
return 1
|
|
347
|
+
mode = sys.argv[1]
|
|
348
|
+
o = json.loads(sys.stdin.read() or "{}")
|
|
349
|
+
try:
|
|
350
|
+
with warnings.catch_warnings(record=True) as caught:
|
|
351
|
+
warnings.simplefilter("always")
|
|
352
|
+
with contextlib.redirect_stdout(sys.stderr):
|
|
353
|
+
if mode == "info":
|
|
354
|
+
result = mode_info_excel(o) if o["path"].lower().endswith((".xlsx", ".xls")) else mode_info(o)
|
|
355
|
+
else:
|
|
356
|
+
result = MODES[mode](o)
|
|
357
|
+
if "notes" in result:
|
|
358
|
+
result["notes"] += sorted({str(w.message) for w in caught})
|
|
359
|
+
except UserError as ue:
|
|
360
|
+
return user_error(str(ue), ue.page_count)
|
|
361
|
+
except Exception: # noqa: BLE001
|
|
362
|
+
traceback.print_exc()
|
|
363
|
+
return 1
|
|
364
|
+
json.dump(result, sys.stdout)
|
|
365
|
+
return 0
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
if __name__ == "__main__":
|
|
369
|
+
sys.exit(main())
|
package/scripts/pdf_to_md.py
DELETED
|
@@ -1,30 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env python3
|
|
2
|
-
"""Convert a PDF to Markdown via pymupdf4llm. One argv: the PDF path.
|
|
3
|
-
Writes Markdown to stdout (exit 0) or an error to stderr (non-zero).
|
|
4
|
-
|
|
5
|
-
pymupdf4llm/PyMuPDF emit diagnostic chatter to stdout; the conversion result
|
|
6
|
-
is the RETURN value of to_markdown(). We redirect library stdout into a sink so
|
|
7
|
-
stdout carries only the Markdown, honoring the caller's verbatim contract."""
|
|
8
|
-
import contextlib
|
|
9
|
-
import io
|
|
10
|
-
import sys
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
def main() -> int:
|
|
14
|
-
if len(sys.argv) != 2:
|
|
15
|
-
print("usage: pdf_to_md.py <pdf-path>", file=sys.stderr)
|
|
16
|
-
return 2
|
|
17
|
-
try:
|
|
18
|
-
sink = io.StringIO()
|
|
19
|
-
with contextlib.redirect_stdout(sink):
|
|
20
|
-
import pymupdf4llm
|
|
21
|
-
md = pymupdf4llm.to_markdown(sys.argv[1])
|
|
22
|
-
except Exception as exc: # noqa: BLE001 — surface any failure to the caller
|
|
23
|
-
print(f"pymupdf4llm conversion failed: {exc}", file=sys.stderr)
|
|
24
|
-
return 1
|
|
25
|
-
sys.stdout.write(md)
|
|
26
|
-
return 0
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
if __name__ == "__main__":
|
|
30
|
-
sys.exit(main())
|