pi-quiver 5.2.4 → 5.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,369 @@
1
+ #!/usr/bin/env python3
2
+ """doc_to_md child. argv[1] = mode (info | pdf-primary | pdf-fallback | xlsx); options JSON on stdin;
3
+ one JSON result on stdout. Exit 0 ok, 1 conversion failure (traceback on stderr), 3 user error
4
+ ({"error", "pageCount"} on stdout). Library chatter is redirected to stderr so stdout is the result only.
5
+ Imports `pymupdf` / `pymupdf4llm` (never the deprecated `fitz` alias)."""
6
+ import contextlib
7
+ import json
8
+ import os
9
+ import re
10
+ import shutil
11
+ import sys
12
+ import tempfile
13
+ import traceback
14
+ import warnings
15
+
16
+ SEP = "\n\n--- end of page.page_number={n} ---\n\n"
17
+ DEGRADED_NOTE = "degraded: PyMuPDF text extraction - layout/tables not preserved"
18
+ MARKDOWN_IMAGE_RE = re.compile(r"(!\[[^\]]*\]\(\s*)(?:<([^>]+)>|([^)]*?))(\s*\))")
19
+ HTML_IMAGE_RE = re.compile(
20
+ r"(<img\b[^>]*?\bsrc\s*=\s*)(?:\"([^\"]*)\"|'([^']*)'|([^\s\"'=<>`]+))", re.IGNORECASE
21
+ )
22
+
23
+
24
+ def rewrite_image_destinations(md, sources):
25
+ def markdown(match):
26
+ dest = match.group(2) if match.group(2) is not None else match.group(3)
27
+ target = sources.get(dest)
28
+ if target is None:
29
+ return match.group(0)
30
+ if match.group(2) is not None:
31
+ return f"{match.group(1)}<{target}>{match.group(4)}"
32
+ return f"{match.group(1)}{target}{match.group(4)}"
33
+
34
+ def html(match):
35
+ dest = next(value for value in match.groups()[1:] if value is not None)
36
+ target = sources.get(dest)
37
+ if target is None:
38
+ return match.group(0)
39
+ quote = '"' if match.group(2) is not None else "'" if match.group(3) is not None else ""
40
+ return f"{match.group(1)}{quote}{target}{quote}"
41
+
42
+ return HTML_IMAGE_RE.sub(html, MARKDOWN_IMAGE_RE.sub(markdown, md))
43
+
44
+
45
+ def user_error(msg, page_count=None):
46
+ json.dump({"error": msg, **({"pageCount": page_count} if page_count is not None else {})}, sys.stdout)
47
+ return 3
48
+
49
+
50
+ def check_pages(pages, page_count):
51
+ if pages is None:
52
+ return list(range(1, page_count + 1))
53
+ bad = [p for p in pages if p < 1 or p > page_count]
54
+ if bad:
55
+ raise UserError(f"pages out of range: {', '.join(map(str, bad))} (document has {page_count} pages)", page_count)
56
+ return pages
57
+
58
+
59
+ class UserError(Exception):
60
+ def __init__(self, msg, page_count=None):
61
+ super().__init__(msg)
62
+ self.page_count = page_count
63
+
64
+
65
+ def open_pdf(path):
66
+ import pymupdf
67
+ doc = pymupdf.open(path)
68
+ if doc.needs_pass:
69
+ raise UserError("Password-protected PDF", doc.page_count)
70
+ return doc
71
+
72
+
73
+ def page_dir(staging, n):
74
+ d = os.path.join(staging, f"p{n}")
75
+ os.makedirs(d, exist_ok=True)
76
+ return d
77
+
78
+
79
+ def mark_done(d):
80
+ open(os.path.join(d, ".done"), "w").close()
81
+
82
+
83
+ def mode_info(o):
84
+ import pymupdf # noqa: F401
85
+ doc = open_pdf(o["path"])
86
+ meta = {k: v for k, v in (doc.metadata or {}).items() if v}
87
+ toc = [[lvl, title, page] for lvl, title, page in doc.get_toc()]
88
+ return {"pageCount": doc.page_count, "metadata": meta, "toc": toc}
89
+
90
+
91
+ def mode_pdf_primary(o):
92
+ import pymupdf4llm
93
+ doc = open_pdf(o["path"])
94
+ pages = check_pages(o.get("pages"), doc.page_count)
95
+ staging, out, empty, failed, notes = o["stagingDir"], [], [], [], []
96
+ for n in pages:
97
+ d = page_dir(staging, n)
98
+ try:
99
+ # Space-free temp dir: pymupdf4llm's md_path() mangles paths containing spaces/parens.
100
+ with tempfile.TemporaryDirectory() as tmp:
101
+ md = pymupdf4llm.to_markdown(doc, pages=[n - 1], write_images=True, image_path=tmp,
102
+ image_format=o["imageFormat"], dpi=o["imageDpi"],
103
+ use_ocr=False, page_separators=False)
104
+ sources = {}
105
+ for i, f in enumerate(sorted(os.listdir(tmp)), 1):
106
+ dest = f"img{i}{os.path.splitext(f)[1].lower()}"
107
+ source = os.path.join(tmp, f)
108
+ target = f"p{n}/{dest}"
109
+ sources.update({source: target, os.path.realpath(source): target, f: target})
110
+ os.replace(source, os.path.join(d, dest))
111
+ md = rewrite_image_destinations(md, sources)
112
+ if not md.strip():
113
+ empty.append(n)
114
+ out.append(md.rstrip())
115
+ mark_done(d)
116
+ except Exception as exc: # noqa: BLE001
117
+ shutil.rmtree(d, ignore_errors=True)
118
+ failed.append({"page": n, "error": f"{type(exc).__name__}: {exc}"[:300]})
119
+ empty.append(n)
120
+ out.append("")
121
+ out.append(SEP.format(n=n).strip("\n"))
122
+ if failed and len(failed) == len(pages):
123
+ raise RuntimeError("every selected page failed: " + failed[0]["error"])
124
+ return {"markdown": "\n\n".join(out) + "\n", "pages": pages, "pageCount": doc.page_count,
125
+ "emptyPages": empty, "failedPages": failed, "notes": notes}
126
+
127
+
128
+ def mode_pdf_fallback(o):
129
+ import pymupdf
130
+ doc = open_pdf(o["path"])
131
+ pages = check_pages(o.get("pages"), doc.page_count)
132
+ keep = {int(k): v for k, v in (o.get("keepPages") or {}).items()}
133
+ staging, out, empty, failed = o["stagingDir"], [], [], []
134
+ for n in pages:
135
+ links = [f"![](images/{f})" for f in keep.get(n, [])]
136
+ text = ""
137
+ try:
138
+ page = doc[n - 1]
139
+ text = page.get_text("text").strip()
140
+ if n not in keep:
141
+ d = page_dir(staging, n)
142
+ i = 0
143
+ for info in page.get_image_info(xrefs=True):
144
+ i += 1
145
+ xref = info.get("xref", 0)
146
+ if xref > 0:
147
+ img = doc.extract_image(xref)
148
+ name = f"img{i}.{img['ext'].lower()}"
149
+ with open(os.path.join(d, name), "wb") as fh:
150
+ fh.write(img["image"])
151
+ else:
152
+ name = f"img{i}.{o['imageFormat']}"
153
+ page.get_pixmap(clip=pymupdf.Rect(info["bbox"]), dpi=o["imageDpi"]).save(os.path.join(d, name))
154
+ links.append(f"![](p{n}/{name})")
155
+ mark_done(d)
156
+ except Exception as exc: # noqa: BLE001
157
+ shutil.rmtree(os.path.join(staging, f"p{n}"), ignore_errors=True)
158
+ text = ""
159
+ links = [f"![](images/{f})" for f in keep.get(n, [])]
160
+ failed.append({"page": n, "error": f"{type(exc).__name__}: {exc}"[:300]})
161
+ if not text:
162
+ empty.append(n)
163
+ out.append("\n\n".join(x for x in [text, "\n".join(links)] if x))
164
+ out.append(SEP.format(n=n).strip("\n"))
165
+ if failed and len(failed) == len(pages):
166
+ raise RuntimeError("every selected page failed: " + failed[0]["error"])
167
+ return {"markdown": "\n\n".join(out) + "\n", "pages": pages, "pageCount": doc.page_count,
168
+ "emptyPages": empty, "failedPages": failed, "notes": [DEGRADED_NOTE]}
169
+
170
+
171
+ def esc(v):
172
+ if v is None:
173
+ return ""
174
+ if isinstance(v, bool):
175
+ return "TRUE" if v else "FALSE"
176
+ if isinstance(v, float):
177
+ return repr(v)
178
+ import datetime
179
+ if isinstance(v, (datetime.date, datetime.datetime)):
180
+ return v.isoformat()
181
+ return str(v).replace("\\", "\\\\").replace("|", "\\|").replace("\r\n", "<br>").replace("\n", "<br>")
182
+
183
+
184
+ def col_letter(i):
185
+ from openpyxl.utils import get_column_letter
186
+ return get_column_letter(i)
187
+
188
+
189
+ def mode_xlsx(o):
190
+ path, staging, budget = o["path"], o["stagingDir"], o["maxCellsPerSheet"]
191
+ notes, images = [], []
192
+ if path.lower().endswith(".xls"):
193
+ return mode_xls(o)
194
+ import openpyxl
195
+ wb_f = openpyxl.load_workbook(path, data_only=False)
196
+ wb_v = openpyxl.load_workbook(path, data_only=True)
197
+ inventory, sections = [], []
198
+ for idx, ws in enumerate(wb_f.worksheets, 1):
199
+ wv = wb_v[ws.title]
200
+ rows, cols = ws.max_row, ws.max_column
201
+ r_lim, c_lim, truncated = rows, cols, False
202
+ if rows * cols > budget:
203
+ truncated = True
204
+ r_lim = max(1, budget // cols)
205
+ if r_lim == 1 and cols > budget:
206
+ c_lim = budget
207
+ hidden = ws.sheet_state != "visible"
208
+ inventory.append(f"- {idx}. {ws.title}{' hidden' if hidden else ''} - {rows} x {cols}{' truncated' if truncated else ''}")
209
+ lines = [f"## {ws.title}"]
210
+ if hidden:
211
+ lines.append("Hidden sheet")
212
+ merged = [str(r) for r in ws.merged_cells.ranges]
213
+ if merged:
214
+ lines.append("Merged: " + ", ".join(merged))
215
+ hr = [str(r) for r, dim in ws.row_dimensions.items() if dim.hidden]
216
+ if hr:
217
+ lines.append("Hidden rows: " + ",".join(hr))
218
+ hc = []
219
+ for dim in ws.column_dimensions.values():
220
+ if dim.hidden:
221
+ hc += [col_letter(i) for i in range(dim.min, dim.max + 1)]
222
+ if hc:
223
+ lines.append("Hidden cols: " + ",".join(hc))
224
+ if truncated:
225
+ lines.append(f"Truncated: showing rows 1-{r_lim} of {rows}, cols A-{col_letter(c_lim)} of {cols}")
226
+ continuation = set()
227
+ for rng in ws.merged_cells.ranges:
228
+ for r in range(rng.min_row, rng.max_row + 1):
229
+ for c in range(rng.min_col, rng.max_col + 1):
230
+ if (r, c) != (rng.min_row, rng.min_col):
231
+ continuation.add((r, c))
232
+ header = "| | " + " | ".join(col_letter(c) for c in range(1, c_lim + 1)) + " |"
233
+ lines += [header, "|---|" + "---|" * c_lim]
234
+ for r in range(1, r_lim + 1):
235
+ cells = []
236
+ for c in range(1, c_lim + 1):
237
+ if (r, c) in continuation:
238
+ cells.append("")
239
+ continue
240
+ f, v = ws.cell(r, c).value, wv.cell(r, c).value
241
+ if isinstance(f, str) and f.startswith("="):
242
+ cells.append(f"{esc(v) if v is not None else '(no cached result)'} ({esc(f)})")
243
+ else:
244
+ cells.append(esc(f))
245
+ lines.append(f"| {r} | " + " | ".join(cells) + " |")
246
+ for i, img in enumerate(getattr(ws, "_images", []), 1):
247
+ try:
248
+ fmt = (getattr(img, "format", None) or "png").lower()
249
+ name = f"s{idx}-{i}.{fmt}"
250
+ data = img._data() if callable(getattr(img, "_data", None)) else img.ref.getvalue()
251
+ with open(os.path.join(staging, name), "wb") as fh:
252
+ fh.write(data)
253
+ images.append({"sheetIndex": idx, "file": name})
254
+ lines.append(f"![]({name})")
255
+ except Exception as exc: # noqa: BLE001
256
+ notes.append(f"sheet {ws.title}: image {i} not extracted ({type(exc).__name__})")
257
+ if not hasattr(ws, "_images"):
258
+ notes.append("Images: unavailable")
259
+ sections.append("\n".join(lines))
260
+ md = "## Sheets\n" + "\n".join(inventory) + "\n\n" + "\n\n".join(sections) + "\n"
261
+ return {"markdown": md, "images": images, "notes": notes}
262
+
263
+
264
+ def mode_xls(o):
265
+ import xlrd
266
+ path, budget = o["path"], o["maxCellsPerSheet"]
267
+ book = xlrd.open_workbook(path, formatting_info=True, logfile=sys.stderr)
268
+ inventory, sections = [], []
269
+ for idx in range(book.nsheets):
270
+ sh = book.sheet_by_index(idx)
271
+ rows, cols = sh.nrows, sh.ncols
272
+ r_lim, c_lim, truncated = rows, cols, False
273
+ if rows * cols > budget:
274
+ truncated = True
275
+ r_lim = max(1, budget // max(cols, 1))
276
+ if r_lim == 1 and cols > budget:
277
+ c_lim = budget
278
+ hidden = sh.visibility != 0
279
+ inventory.append(f"- {idx + 1}. {sh.name}{' hidden' if hidden else ''} - {rows} x {cols}{' truncated' if truncated else ''}")
280
+ lines = [f"## {sh.name}", "Formulas: unavailable (.xls via xlrd); Images: unavailable"]
281
+ if hidden:
282
+ lines.append("Hidden sheet")
283
+ merged = [f"{col_letter(c0 + 1)}{r0 + 1}:{col_letter(c1)}{r1}" for r0, r1, c0, c1 in sh.merged_cells]
284
+ if merged:
285
+ lines.append("Merged: " + ", ".join(merged))
286
+ hr = [str(r + 1) for r, info in sh.rowinfo_map.items() if info.hidden]
287
+ if hr:
288
+ lines.append("Hidden rows: " + ",".join(hr))
289
+ hc = [col_letter(c + 1) for c, info in sh.colinfo_map.items() if info.hidden]
290
+ if hc:
291
+ lines.append("Hidden cols: " + ",".join(hc))
292
+ if truncated:
293
+ lines.append(f"Truncated: showing rows 1-{r_lim} of {rows}, cols A-{col_letter(c_lim)} of {cols}")
294
+ continuation = {(r, c) for r0, r1, c0, c1 in sh.merged_cells for r in range(r0, r1) for c in range(c0, c1) if (r, c) != (r0, c0)}
295
+ lines += ["| | " + " | ".join(col_letter(c + 1) for c in range(c_lim)) + " |", "|---|" + "---|" * c_lim]
296
+ for r in range(r_lim):
297
+ cells = []
298
+ for c in range(c_lim):
299
+ if (r, c) in continuation:
300
+ cells.append("")
301
+ continue
302
+ cell = sh.cell(r, c)
303
+ v = cell.value
304
+ if cell.ctype == xlrd.XL_CELL_DATE:
305
+ v = xlrd.xldate_as_datetime(v, book.datemode).isoformat()
306
+ elif cell.ctype == xlrd.XL_CELL_BOOLEAN:
307
+ v = bool(v)
308
+ elif cell.ctype == xlrd.XL_CELL_ERROR:
309
+ v = xlrd.error_text_from_code.get(int(v), f"#ERR{int(v)}")
310
+ elif cell.ctype == xlrd.XL_CELL_EMPTY:
311
+ v = None
312
+ cells.append(esc(v))
313
+ lines.append(f"| {r + 1} | " + " | ".join(cells) + " |")
314
+ sections.append("\n".join(lines))
315
+ md = "## Sheets\n" + "\n".join(inventory) + "\n\n" + "\n\n".join(sections) + "\n"
316
+ return {"markdown": md, "images": [], "notes": []}
317
+
318
+
319
+ def mode_info_excel(o):
320
+ path = o["path"]
321
+ sheets = []
322
+ if path.lower().endswith(".xls"):
323
+ import xlrd
324
+ book = xlrd.open_workbook(path, formatting_info=True, logfile=sys.stderr)
325
+ for idx in range(book.nsheets):
326
+ sh = book.sheet_by_index(idx)
327
+ sheets.append({"name": sh.name, "index": idx + 1, "hidden": sh.visibility != 0, "rows": sh.nrows, "cols": sh.ncols,
328
+ "hiddenRows": sum(1 for i in sh.rowinfo_map.values() if i.hidden),
329
+ "hiddenCols": sum(1 for i in sh.colinfo_map.values() if i.hidden)})
330
+ else:
331
+ import openpyxl
332
+ wb = openpyxl.load_workbook(path, data_only=True)
333
+ for idx, ws in enumerate(wb.worksheets, 1):
334
+ hc = sum(d.max - d.min + 1 for d in ws.column_dimensions.values() if d.hidden)
335
+ sheets.append({"name": ws.title, "index": idx, "hidden": ws.sheet_state != "visible", "rows": ws.max_row, "cols": ws.max_column,
336
+ "hiddenRows": sum(1 for d in ws.row_dimensions.values() if d.hidden), "hiddenCols": hc})
337
+ return {"sheets": sheets}
338
+
339
+
340
+ MODES = {"info": None, "pdf-primary": mode_pdf_primary, "pdf-fallback": mode_pdf_fallback, "xlsx": mode_xlsx}
341
+
342
+
343
+ def main():
344
+ if len(sys.argv) != 2 or sys.argv[1] not in MODES:
345
+ print("usage: doc_to_md.py <info|pdf-primary|pdf-fallback|xlsx> (options JSON on stdin)", file=sys.stderr)
346
+ return 1
347
+ mode = sys.argv[1]
348
+ o = json.loads(sys.stdin.read() or "{}")
349
+ try:
350
+ with warnings.catch_warnings(record=True) as caught:
351
+ warnings.simplefilter("always")
352
+ with contextlib.redirect_stdout(sys.stderr):
353
+ if mode == "info":
354
+ result = mode_info_excel(o) if o["path"].lower().endswith((".xlsx", ".xls")) else mode_info(o)
355
+ else:
356
+ result = MODES[mode](o)
357
+ if "notes" in result:
358
+ result["notes"] += sorted({str(w.message) for w in caught})
359
+ except UserError as ue:
360
+ return user_error(str(ue), ue.page_count)
361
+ except Exception: # noqa: BLE001
362
+ traceback.print_exc()
363
+ return 1
364
+ json.dump(result, sys.stdout)
365
+ return 0
366
+
367
+
368
+ if __name__ == "__main__":
369
+ sys.exit(main())
@@ -1,30 +0,0 @@
1
- #!/usr/bin/env python3
2
- """Convert a PDF to Markdown via pymupdf4llm. One argv: the PDF path.
3
- Writes Markdown to stdout (exit 0) or an error to stderr (non-zero).
4
-
5
- pymupdf4llm/PyMuPDF emit diagnostic chatter to stdout; the conversion result
6
- is the RETURN value of to_markdown(). We redirect library stdout into a sink so
7
- stdout carries only the Markdown, honoring the caller's verbatim contract."""
8
- import contextlib
9
- import io
10
- import sys
11
-
12
-
13
- def main() -> int:
14
- if len(sys.argv) != 2:
15
- print("usage: pdf_to_md.py <pdf-path>", file=sys.stderr)
16
- return 2
17
- try:
18
- sink = io.StringIO()
19
- with contextlib.redirect_stdout(sink):
20
- import pymupdf4llm
21
- md = pymupdf4llm.to_markdown(sys.argv[1])
22
- except Exception as exc: # noqa: BLE001 — surface any failure to the caller
23
- print(f"pymupdf4llm conversion failed: {exc}", file=sys.stderr)
24
- return 1
25
- sys.stdout.write(md)
26
- return 0
27
-
28
-
29
- if __name__ == "__main__":
30
- sys.exit(main())