dot-parser 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dot_parser/__init__.py +49 -0
- dot_parser/backends/__init__.py +26 -0
- dot_parser/backends/_base.py +79 -0
- dot_parser/backends/docling.py +59 -0
- dot_parser/backends/llama.py +76 -0
- dot_parser/backends/mistral.py +622 -0
- dot_parser/backends/pymu.py +56 -0
- dot_parser/chunking.py +257 -0
- dot_parser/docx_images.py +633 -0
- dot_parser/image_utils.py +152 -0
- dot_parser/images.py +125 -0
- dot_parser/markdown_utils.py +48 -0
- dot_parser/models.py +22 -0
- dot_parser/parsers.py +325 -0
- dot_parser/pricing.py +58 -0
- dot_parser/tokens.py +6 -0
- dot_parser/vlms.py +203 -0
- dot_parser-2.0.0.dist-info/METADATA +274 -0
- dot_parser-2.0.0.dist-info/RECORD +21 -0
- dot_parser-2.0.0.dist-info/WHEEL +4 -0
- dot_parser-2.0.0.dist-info/licenses/LICENSE.md +660 -0
|
@@ -0,0 +1,633 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
"""Office documents (DOCX/PPTX) → Markdown with per-image interpretation.
|
|
5
|
+
|
|
6
|
+
Runs markitdown to produce the markdown body — which already inlines
|
|
7
|
+
each embedded image as ```` at the
|
|
8
|
+
correct document position — then walks those data URLs in order,
|
|
9
|
+
rewriting them into stable ```` anchors and emitting one
|
|
10
|
+
:class:`ExtractedImage` per match. Each image is then handed to the
|
|
11
|
+
caller-supplied VLM for a textual interpretation.
|
|
12
|
+
|
|
13
|
+
Going through markitdown (instead of the previous hand-written
|
|
14
|
+
python-docx walker) means the text rendering matches the text-only
|
|
15
|
+
``parse(format='docx')`` path exactly — same markdown surface with or
|
|
16
|
+
without ``parse_images=True`` — and we inherit markitdown's handling of
|
|
17
|
+
formatting, lists, tables, headers, and footers for free.
|
|
18
|
+
|
|
19
|
+
The pipeline is format-agnostic: markitdown dispatches on the file
|
|
20
|
+
extension and both its DOCX and PPTX converters honour
|
|
21
|
+
``keep_data_uris=True``. For PPTX, markitdown additionally emits a
|
|
22
|
+
``<!-- Slide number: N -->`` comment before each slide and the slide
|
|
23
|
+
title as an ``#`` heading. Note that the PPTX path only sees *embedded
|
|
24
|
+
pictures* — diagrams drawn natively in PowerPoint (shapes, arrows,
|
|
25
|
+
SmartArt) have no image payload and surface as text fragments only;
|
|
26
|
+
use the Mistral OCR path (``backend=Mistral()``) when those matter.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
import base64
|
|
30
|
+
import io
|
|
31
|
+
import logging
|
|
32
|
+
import re
|
|
33
|
+
import tempfile
|
|
34
|
+
import time
|
|
35
|
+
from pathlib import Path
|
|
36
|
+
|
|
37
|
+
from dot_parser.image_utils import dedupe_titles
|
|
38
|
+
from dot_parser.images import (
|
|
39
|
+
VLM,
|
|
40
|
+
ExtractedImage,
|
|
41
|
+
ImageDescription,
|
|
42
|
+
ParseResult,
|
|
43
|
+
)
|
|
44
|
+
from dot_parser.markdown_utils import strip_markdown_tables
|
|
45
|
+
from dot_parser.models import ParseError
|
|
46
|
+
|
|
47
|
+
_logger = logging.getLogger(__name__)
|
|
48
|
+
|
|
49
|
+
# Markdown blocks around each image anchor that we feed back to the VLM
|
|
50
|
+
# as context. We look both before and after the anchor: French/EU
|
|
51
|
+
# technical documents place figure captions ("Figure 1: …") directly
|
|
52
|
+
# *below* the image, so a preceding-only window would miss them.
|
|
53
|
+
# TODO: ENH since I did not really optimize these numbers, maybe we should adapt them
|
|
54
|
+
# based on some parameters and the multimodel model of choice.)
|
|
55
|
+
_CONTEXT_BLOCKS_BEFORE = 3
|
|
56
|
+
_CONTEXT_BLOCKS_AFTER = 2
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
# Reported in `unsupported_reason` for images the deadline cut short.
|
|
60
|
+
# Deliberately reuses the "why is there no interpretation" channel that
|
|
61
|
+
# undecodable formats already populate, so existing consumers render it
|
|
62
|
+
# with no change.
|
|
63
|
+
SKIPPED_DEADLINE_REASON = (
|
|
64
|
+
"Not interpreted: the document's image-analysis time budget ran out "
|
|
65
|
+
"before this image was reached."
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
# MIME types Mistral's chat-vision models currently accept (Mistral
|
|
69
|
+
# Large 3 / Medium 3.1 / Small 3.2 / Ministral 3 — Pixtral is
|
|
70
|
+
# deprecated). Anything outside this set is first handed to Pillow for a
|
|
71
|
+
# rescue conversion (see the interpretation loop); only if Pillow can't
|
|
72
|
+
# decode it do we surface it as unsupported.
|
|
73
|
+
_VLM_SUPPORTED_MIMES = frozenset(
|
|
74
|
+
{
|
|
75
|
+
"image/png",
|
|
76
|
+
"image/jpeg",
|
|
77
|
+
"image/jpg",
|
|
78
|
+
"image/webp",
|
|
79
|
+
"image/gif",
|
|
80
|
+
}
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
# Aliases for the canonical MIME types above. Data URLs don't use a
|
|
84
|
+
# single canonical spelling — the same bytes show up as "image/x-png",
|
|
85
|
+
# "image/x-ms-bmp", etc. depending on the producer. We normalise these
|
|
86
|
+
# before the membership / conversion decisions so a non-canonical
|
|
87
|
+
# spelling never gets misrouted to the unsupported branch.
|
|
88
|
+
_MIME_ALIASES: dict[str, str] = {
|
|
89
|
+
"image/x-png": "image/png",
|
|
90
|
+
"image/pjpeg": "image/jpeg",
|
|
91
|
+
"image/x-ms-bmp": "image/bmp",
|
|
92
|
+
"image/x-bmp": "image/bmp",
|
|
93
|
+
"image/x-tiff": "image/tiff",
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _normalize_mime(mime_type: str) -> str:
|
|
98
|
+
"""Collapse known MIME aliases to their canonical spelling."""
|
|
99
|
+
return _MIME_ALIASES.get(mime_type.lower(), mime_type.lower())
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
# Per-format messages surfaced as ExtractedImage.unsupported_reason
|
|
103
|
+
# so downstream consumers can explain to end-users why an image was
|
|
104
|
+
# not interpreted, and what external tool could recover it.
|
|
105
|
+
#
|
|
106
|
+
# Each comment below records (a) when this MIME actually appears in
|
|
107
|
+
# real DOCX files and (b) the recommended fix if we ever decide to
|
|
108
|
+
# add full support. Every non-VLM-native image is first attempted via
|
|
109
|
+
# Pillow (_convert_raster_to_png); this dict is only consulted when that
|
|
110
|
+
# decode returns None — so raster formats (BMP, TIFF, and any
|
|
111
|
+
# plugin-registered format) only land here on an actual decode failure.
|
|
112
|
+
_UNSUPPORTED_REASONS: dict[str, str] = {
|
|
113
|
+
# Seen in: Word docs containing Visio drawings, SmartArt, grouped
|
|
114
|
+
# shapes, or charts — Word inlines an EMF as a fallback raster of
|
|
115
|
+
# the vector content. The "real" source often lives in
|
|
116
|
+
# word/embeddings/*.bin (OLE-embedded Visio), which is a separate
|
|
117
|
+
# extraction problem.
|
|
118
|
+
# Fix: shell out to `soffice --headless --convert-to png`
|
|
119
|
+
"image/x-emf": (
|
|
120
|
+
"Windows Enhanced Metafile (.emf): vector instruction stream "
|
|
121
|
+
"embedded by Word as a fallback preview for Visio drawings, "
|
|
122
|
+
"SmartArt or grouped shapes. Not accepted by the vision API. "
|
|
123
|
+
"Rendering requires an external tool such as LibreOffice "
|
|
124
|
+
"(`soffice --headless --convert-to png`) or Inkscape. The raw "
|
|
125
|
+
"base64 bytes are still returned for download or external "
|
|
126
|
+
"processing."
|
|
127
|
+
),
|
|
128
|
+
# Seen in: legacy/old Office documents (pre-2007 era) and content
|
|
129
|
+
# copy-pasted from older applications. Much rarer than EMF today.
|
|
130
|
+
# Fix: same toolchain as EMF — both LibreOffice and Inkscape read
|
|
131
|
+
# WMF. No separate code path needed if EMF support lands.
|
|
132
|
+
"image/x-wmf": (
|
|
133
|
+
"Windows Metafile (.wmf): legacy Windows vector format, often "
|
|
134
|
+
"carried inside older Office documents. Not accepted by the "
|
|
135
|
+
"vision API. Convert via LibreOffice or Inkscape to render."
|
|
136
|
+
),
|
|
137
|
+
# Seen in: modern Word (Office 365 inserts SVG natively when you
|
|
138
|
+
# paste icons or vector graphics) and exports from Figma /
|
|
139
|
+
# Illustrator / draw.io. Browsers already render SVG data-URLs
|
|
140
|
+
# directly, so the frontend visual is fine without us — the only
|
|
141
|
+
# gain from local conversion is the VLM caption.
|
|
142
|
+
# Fix: `resvg-py` (Rust binding, ships with wheels, no system
|
|
143
|
+
# libs) is the low-friction option; `cairosvg` has broader SVG
|
|
144
|
+
# feature coverage but pulls in cairo/pango as OS libs.
|
|
145
|
+
"image/svg+xml": (
|
|
146
|
+
"Scalable Vector Graphics: browser-renderable but not accepted "
|
|
147
|
+
"by the vision API. Convert to PNG (e.g. cairosvg) for "
|
|
148
|
+
"interpretation, or render client-side from the base64 bytes."
|
|
149
|
+
),
|
|
150
|
+
# Reached only when _convert_raster_to_png() fails on a TIFF
|
|
151
|
+
# payload — typically unsupported compression (e.g. CCITT G4 from
|
|
152
|
+
# scanners without libtiff support in the wheel) or multi-page
|
|
153
|
+
# documents where we currently keep only the first frame.
|
|
154
|
+
# Fix: extend _convert_raster_to_png to iterate `n_frames` and
|
|
155
|
+
# emit one ExtractedImage per page; install Pillow with full
|
|
156
|
+
# libtiff support for exotic compressions.
|
|
157
|
+
"image/tiff": (
|
|
158
|
+
"TIFF: automatic conversion to PNG via Pillow failed for this "
|
|
159
|
+
"image (e.g. unsupported compression or multi-page payload). The "
|
|
160
|
+
"raw base64 bytes are still returned for external recovery."
|
|
161
|
+
),
|
|
162
|
+
# Reached only when _convert_raster_to_png() fails on a BMP — very
|
|
163
|
+
# rare in practice; mostly indicates a corrupt or truncated
|
|
164
|
+
# payload extracted from a damaged DOCX. No code change needed
|
|
165
|
+
# beyond surfacing this reason.
|
|
166
|
+
"image/bmp": (
|
|
167
|
+
"BMP: automatic conversion to PNG via Pillow failed for this "
|
|
168
|
+
"image. The raw base64 bytes are still returned for external "
|
|
169
|
+
"recovery."
|
|
170
|
+
),
|
|
171
|
+
# Seen in: almost never in real DOCX files — AVIF is a web format
|
|
172
|
+
# (Chrome/Safari era), not something Office tooling produces.
|
|
173
|
+
# Listed defensively in case a user pastes web content.
|
|
174
|
+
# Fix: add `pillow-avif-plugin` to deps and import it once at
|
|
175
|
+
# startup; the plugin auto-registers a Pillow decoder, after which
|
|
176
|
+
# the rescue-conversion path picks AVIF up with no other changes.
|
|
177
|
+
"image/avif": (
|
|
178
|
+
"AVIF: not accepted by the vision API. Convert to PNG or "
|
|
179
|
+
"WebP using Pillow (with pillow-avif-plugin) for interpretation."
|
|
180
|
+
),
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
# Formats not in _UNSUPPORTED_REASONS yet (they fall through to the
|
|
184
|
+
# generic message in _unsupported_reason). Because the rescue path now
|
|
185
|
+
# tries Pillow on anything non-native, "adding support" for a raster
|
|
186
|
+
# format is purely a matter of making Pillow able to decode it — no
|
|
187
|
+
# MIME-set edit is needed:
|
|
188
|
+
#
|
|
189
|
+
# image/heic, image/heif
|
|
190
|
+
# Seen in: iPhone screenshots and photos pasted into Word — rising
|
|
191
|
+
# fast as a real-world source.
|
|
192
|
+
# Fix: add `pillow-heif` dep + import it once at startup. The plugin
|
|
193
|
+
# registers a Pillow decoder and the rescue path handles the rest.
|
|
194
|
+
#
|
|
195
|
+
# image/jp2 (JPEG 2000)
|
|
196
|
+
# Seen in: scientific/archival pipelines; vanishingly rare in
|
|
197
|
+
# Office documents.
|
|
198
|
+
# Fix: works out of the box if the Pillow wheel was built with
|
|
199
|
+
# OpenJPEG support (most modern wheels are) — nothing to add.
|
|
200
|
+
#
|
|
201
|
+
# image/x-icon (favicons / Windows ICO)
|
|
202
|
+
# Seen in: occasional paste from a browser into Word.
|
|
203
|
+
# Fix: Pillow has native ICO support, so the rescue path already
|
|
204
|
+
# handles it. Multi-resolution ICOs collapse to the largest frame,
|
|
205
|
+
# which is what we want for VLM interpretation.
|
|
206
|
+
#
|
|
207
|
+
# image/eps (encapsulated PostScript)
|
|
208
|
+
# Seen in: academic/print-publishing pipelines, very rare in
|
|
209
|
+
# modern Office.
|
|
210
|
+
# Fix: Pillow can decode EPS but only if Ghostscript is installed
|
|
211
|
+
# on the host. Better treated like EMF (optional extra + binary
|
|
212
|
+
# probe).
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _convert_raster_to_png(b64: str) -> tuple[str, str] | None:
|
|
216
|
+
"""Re-encode a raster image (BMP/TIFF) as PNG via Pillow.
|
|
217
|
+
|
|
218
|
+
Returns ``(new_b64, "image/png")`` on success, or ``None`` if the
|
|
219
|
+
payload can't be decoded (corrupt bytes, unsupported TIFF
|
|
220
|
+
compression, missing Pillow). Multi-page TIFFs collapse to the
|
|
221
|
+
first frame — good enough for document images, where TIFFs are
|
|
222
|
+
almost always single-page scans.
|
|
223
|
+
"""
|
|
224
|
+
try:
|
|
225
|
+
from PIL import Image
|
|
226
|
+
except ImportError:
|
|
227
|
+
_logger.warning("Pillow not installed; skipping raster→PNG conversion.")
|
|
228
|
+
return None
|
|
229
|
+
try:
|
|
230
|
+
raw = base64.b64decode(b64)
|
|
231
|
+
with Image.open(io.BytesIO(raw)) as im:
|
|
232
|
+
im.load()
|
|
233
|
+
if im.mode not in ("RGB", "RGBA", "L", "LA", "P"):
|
|
234
|
+
im = im.convert("RGBA")
|
|
235
|
+
out = io.BytesIO()
|
|
236
|
+
im.save(out, format="PNG")
|
|
237
|
+
return base64.b64encode(out.getvalue()).decode("ascii"), "image/png"
|
|
238
|
+
except Exception as exc:
|
|
239
|
+
_logger.warning("Raster→PNG conversion failed: %s", exc)
|
|
240
|
+
return None
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _unsupported_reason(mime_type: str) -> str:
|
|
244
|
+
"""Return a human-readable explanation for an unsupported MIME.
|
|
245
|
+
|
|
246
|
+
Looks up a per-format message; falls back to a generic one
|
|
247
|
+
pointing at the offending MIME so consumers always get *some*
|
|
248
|
+
actionable detail rather than a bare None.
|
|
249
|
+
"""
|
|
250
|
+
if mime_type in _UNSUPPORTED_REASONS:
|
|
251
|
+
return _UNSUPPORTED_REASONS[mime_type]
|
|
252
|
+
return (
|
|
253
|
+
f"Image format '{mime_type}' is not currently supported by "
|
|
254
|
+
"the image interpretation pipeline. The raw base64 bytes are "
|
|
255
|
+
"still returned for download or external processing."
|
|
256
|
+
)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
# Matches markitdown's inline data-URL images. markitdown emits them
|
|
260
|
+
# on a single line so we don't need DOTALL. Alt text is captured so
|
|
261
|
+
# we can carry it through as `original_caption` for the title fallback.
|
|
262
|
+
_DATA_URL_IMG_RE = re.compile(
|
|
263
|
+
r"!\[(?P<alt>[^\]]*)\]"
|
|
264
|
+
r"\(data:image/(?P<subtype>[A-Za-z0-9.+\-]+);base64,(?P<b64>[A-Za-z0-9+/=]+)\)"
|
|
265
|
+
)
|
|
266
|
+
|
|
267
|
+
_MIME_SUBTYPE_TO_EXT: dict[str, str] = {
|
|
268
|
+
"png": "png",
|
|
269
|
+
"x-png": "png",
|
|
270
|
+
"jpeg": "jpg",
|
|
271
|
+
"jpg": "jpg",
|
|
272
|
+
"webp": "webp",
|
|
273
|
+
"gif": "gif",
|
|
274
|
+
"bmp": "bmp",
|
|
275
|
+
"tiff": "tiff",
|
|
276
|
+
"svg+xml": "svg",
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _ext_for_subtype(subtype: str) -> str:
|
|
281
|
+
return _MIME_SUBTYPE_TO_EXT.get(subtype.lower(), "png")
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def _extract_data_url_images(
|
|
285
|
+
raw_md: str,
|
|
286
|
+
) -> tuple[str, list[ExtractedImage]]:
|
|
287
|
+
"""Rewrite markitdown's inline data-URL images as stable anchors.
|
|
288
|
+
|
|
289
|
+
Walks data-URL matches in document order, mints ``img-NNN.<ext>``
|
|
290
|
+
names, and returns the rewritten markdown alongside one
|
|
291
|
+
:class:`ExtractedImage` per match (carrying the raw base64). The
|
|
292
|
+
image's alt text is preserved as ``original_caption`` so it remains
|
|
293
|
+
available as a title fallback in :func:`_interpret_images`.
|
|
294
|
+
"""
|
|
295
|
+
images: list[ExtractedImage] = []
|
|
296
|
+
|
|
297
|
+
def _sub(match: re.Match) -> str:
|
|
298
|
+
alt = match.group("alt") or None
|
|
299
|
+
subtype = match.group("subtype")
|
|
300
|
+
b64 = match.group("b64")
|
|
301
|
+
mime = f"image/{subtype}"
|
|
302
|
+
ext = _ext_for_subtype(subtype)
|
|
303
|
+
name = f"img-{len(images) + 1:03d}.{ext}"
|
|
304
|
+
images.append(
|
|
305
|
+
ExtractedImage(
|
|
306
|
+
name=name,
|
|
307
|
+
base64=b64,
|
|
308
|
+
mime_type=mime,
|
|
309
|
+
original_caption=alt,
|
|
310
|
+
)
|
|
311
|
+
)
|
|
312
|
+
return f""
|
|
313
|
+
|
|
314
|
+
rewritten = _DATA_URL_IMG_RE.sub(_sub, raw_md)
|
|
315
|
+
return rewritten, images
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _slugify_caption(caption: str | None) -> str | None:
|
|
319
|
+
"""Derive a kebab-case slug from a free-form caption string.
|
|
320
|
+
|
|
321
|
+
Returns None for empty / unusable captions. Caps length at 8 words so
|
|
322
|
+
captions like "Figure 3: The full bill-of-materials of the assembly,
|
|
323
|
+
including..." produce something compact rather than a sentence-long
|
|
324
|
+
filename.
|
|
325
|
+
"""
|
|
326
|
+
if not caption:
|
|
327
|
+
return None
|
|
328
|
+
# Strip common "Figure N:" / "Fig. N -" prefixes Word users type.
|
|
329
|
+
stripped = re.sub(
|
|
330
|
+
r"^\s*(figure|fig\.?|table|image|img\.?|exhibit)\s*\d*[:.\-]?\s*",
|
|
331
|
+
"",
|
|
332
|
+
caption,
|
|
333
|
+
flags=re.IGNORECASE,
|
|
334
|
+
)
|
|
335
|
+
slug = re.sub(r"[^a-zA-Z0-9]+", "-", stripped).strip("-").lower()
|
|
336
|
+
if not slug:
|
|
337
|
+
return None
|
|
338
|
+
parts = slug.split("-")
|
|
339
|
+
return "-".join(parts[:8])
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def _interpret_images(
|
|
343
|
+
markdown: str,
|
|
344
|
+
images: list[ExtractedImage],
|
|
345
|
+
vlm: VLM,
|
|
346
|
+
*,
|
|
347
|
+
caption_in_context: bool = False,
|
|
348
|
+
deadline: float | None = None,
|
|
349
|
+
) -> list[ExtractedImage]:
|
|
350
|
+
"""Call ``vlm.describe_image`` for each image and assign titles.
|
|
351
|
+
|
|
352
|
+
When ``caption_in_context`` is True, an image's ``original_caption``
|
|
353
|
+
is prepended to the VLM context, explicitly labelled. The caption is
|
|
354
|
+
the strongest grounding available — it is the author's own statement
|
|
355
|
+
of what the figure shows — so it is passed explicitly rather than
|
|
356
|
+
relying on it happening to appear in the sliced markdown
|
|
357
|
+
neighbourhood. Both production call sites (the DOCX markitdown path
|
|
358
|
+
and :func:`dot_parser.parsers.interpret_images`) enable it; the flag
|
|
359
|
+
exists so images without captions cost nothing and so the grounding
|
|
360
|
+
can be switched off in isolation if it ever misbehaves.
|
|
361
|
+
|
|
362
|
+
``deadline`` is an absolute :func:`time.monotonic` timestamp bounding
|
|
363
|
+
the interpretation phase. Calls are issued one at a time, each taking
|
|
364
|
+
a few seconds, so a document with many figures can otherwise run for
|
|
365
|
+
longer than any caller is willing to wait — an image-heavy report
|
|
366
|
+
carries 100-200 of them. Once the deadline passes, the images not yet
|
|
367
|
+
reached are returned with their raw bytes intact and
|
|
368
|
+
:data:`SKIPPED_DEADLINE_REASON` in ``unsupported_reason``, the same
|
|
369
|
+
channel already used for undecodable formats, so consumers need no
|
|
370
|
+
new branch. That makes an oversized document a partial result rather
|
|
371
|
+
than a failed request. The deadline is checked before each call, so
|
|
372
|
+
at most one in-progress call overruns it; bound that with a per-call
|
|
373
|
+
timeout on the VLM itself.
|
|
374
|
+
|
|
375
|
+
Title resolution order per image:
|
|
376
|
+
1. VLM-generated kebab-case slug (preferred)
|
|
377
|
+
2. Slug derived from the document's original caption, if any
|
|
378
|
+
3. None (leave it to consumers to display the technical ``name``)
|
|
379
|
+
|
|
380
|
+
Duplicate titles across the document are disambiguated with
|
|
381
|
+
``-2``, ``-3``, ... suffixes so that the title is safe to use as a
|
|
382
|
+
display label.
|
|
383
|
+
|
|
384
|
+
Backwards-compatible with VLM implementations that return a plain
|
|
385
|
+
``str``: those simply produce ``interpretation`` with no title.
|
|
386
|
+
Per-image VLM failures are absorbed (interpretation stays None).
|
|
387
|
+
"""
|
|
388
|
+
blocks = markdown.split("\n\n")
|
|
389
|
+
# Locate each anchor block once to slice surrounding context cheaply.
|
|
390
|
+
anchor_positions: dict[str, int] = {}
|
|
391
|
+
for i, block in enumerate(blocks):
|
|
392
|
+
for img in images:
|
|
393
|
+
anchor = f""
|
|
394
|
+
if anchor in block and img.name not in anchor_positions:
|
|
395
|
+
anchor_positions[img.name] = i
|
|
396
|
+
|
|
397
|
+
interpretations: list[str | None] = []
|
|
398
|
+
raw_titles: list[str | None] = []
|
|
399
|
+
unsupported_reasons: list[str | None] = []
|
|
400
|
+
effective_b64s: list[str | None] = []
|
|
401
|
+
effective_mimes: list[str | None] = []
|
|
402
|
+
for index, img in enumerate(images):
|
|
403
|
+
if deadline is not None and time.monotonic() >= deadline:
|
|
404
|
+
# Out of time: hand back everything still untouched rather
|
|
405
|
+
# than letting the caller's request run over.
|
|
406
|
+
remaining = len(images) - index
|
|
407
|
+
_logger.warning(
|
|
408
|
+
"image interpretation deadline reached: %d of %d image(s) left uninterpreted",
|
|
409
|
+
remaining,
|
|
410
|
+
len(images),
|
|
411
|
+
)
|
|
412
|
+
for _ in range(remaining):
|
|
413
|
+
interpretations.append(None)
|
|
414
|
+
raw_titles.append(None)
|
|
415
|
+
unsupported_reasons.append(SKIPPED_DEADLINE_REASON)
|
|
416
|
+
effective_b64s.append(None)
|
|
417
|
+
effective_mimes.append(None)
|
|
418
|
+
break
|
|
419
|
+
|
|
420
|
+
vlm_b64 = img.base64
|
|
421
|
+
vlm_mime = img.mime_type
|
|
422
|
+
converted: tuple[str, str] | None = None
|
|
423
|
+
|
|
424
|
+
# Anything the VLM doesn't natively accept is handed to Pillow
|
|
425
|
+
# for a rescue conversion to PNG — BMP/TIFF and any raster format
|
|
426
|
+
# Pillow can decode (incl. plugin-registered HEIC/AVIF) come back
|
|
427
|
+
# as PNG and enter the regular VLM path. We let Pillow be the
|
|
428
|
+
# judge rather than gating on a hardcoded MIME set: a successful
|
|
429
|
+
# decode *is* the support signal, so non-canonical spellings
|
|
430
|
+
# (image/x-ms-bmp, …) and future formats work without code
|
|
431
|
+
# changes. Formats Pillow can't decode (EMF/WMF/SVG/…) return
|
|
432
|
+
# None and fall through to the unsupported branch below, with the
|
|
433
|
+
# original bytes preserved for download or external processing.
|
|
434
|
+
if _normalize_mime(vlm_mime) not in _VLM_SUPPORTED_MIMES:
|
|
435
|
+
converted = _convert_raster_to_png(img.base64)
|
|
436
|
+
if converted is not None:
|
|
437
|
+
vlm_b64, vlm_mime = converted
|
|
438
|
+
else:
|
|
439
|
+
interpretations.append(None)
|
|
440
|
+
raw_titles.append(None)
|
|
441
|
+
unsupported_reasons.append(_unsupported_reason(img.mime_type))
|
|
442
|
+
effective_b64s.append(None)
|
|
443
|
+
effective_mimes.append(None)
|
|
444
|
+
continue
|
|
445
|
+
|
|
446
|
+
pos = anchor_positions.get(img.name)
|
|
447
|
+
context_parts: list[str] = []
|
|
448
|
+
if caption_in_context and img.original_caption:
|
|
449
|
+
context_parts.append(f"Figure caption from the document: {img.original_caption}")
|
|
450
|
+
if pos is not None:
|
|
451
|
+
start = max(0, pos - _CONTEXT_BLOCKS_BEFORE)
|
|
452
|
+
end = pos + 1 + _CONTEXT_BLOCKS_AFTER
|
|
453
|
+
before = [b for b in blocks[start:pos] if b.strip()]
|
|
454
|
+
after = [b for b in blocks[pos + 1 : end] if b.strip()]
|
|
455
|
+
context_parts.extend(before + after)
|
|
456
|
+
context: str | None = "\n\n".join(context_parts) or None
|
|
457
|
+
try:
|
|
458
|
+
result = vlm.describe_image(
|
|
459
|
+
vlm_b64,
|
|
460
|
+
mime_type=vlm_mime,
|
|
461
|
+
context=context,
|
|
462
|
+
)
|
|
463
|
+
except Exception:
|
|
464
|
+
result = None
|
|
465
|
+
|
|
466
|
+
if isinstance(result, ImageDescription):
|
|
467
|
+
interpretations.append(result.interpretation or None)
|
|
468
|
+
raw_titles.append(result.title)
|
|
469
|
+
elif isinstance(result, str):
|
|
470
|
+
interpretations.append(result or None)
|
|
471
|
+
raw_titles.append(None)
|
|
472
|
+
else:
|
|
473
|
+
interpretations.append(None)
|
|
474
|
+
raw_titles.append(None)
|
|
475
|
+
unsupported_reasons.append(None)
|
|
476
|
+
# Surface the converted PNG bytes on the emitted ExtractedImage
|
|
477
|
+
# so the frontend can render the image without browser-side
|
|
478
|
+
# decoder support for the original format.
|
|
479
|
+
effective_b64s.append(converted[0] if converted else None)
|
|
480
|
+
effective_mimes.append(converted[1] if converted else None)
|
|
481
|
+
|
|
482
|
+
# Apply caption fallback for any image whose VLM didn't supply a title.
|
|
483
|
+
resolved_titles: list[str | None] = [
|
|
484
|
+
rt if rt else _slugify_caption(img.original_caption)
|
|
485
|
+
for rt, img in zip(raw_titles, images, strict=True)
|
|
486
|
+
]
|
|
487
|
+
final_titles = dedupe_titles(resolved_titles)
|
|
488
|
+
|
|
489
|
+
return [
|
|
490
|
+
ExtractedImage(
|
|
491
|
+
name=img.name,
|
|
492
|
+
base64=eff_b64 or img.base64,
|
|
493
|
+
mime_type=eff_mime or img.mime_type,
|
|
494
|
+
page=img.page,
|
|
495
|
+
original_caption=img.original_caption,
|
|
496
|
+
interpretation=interp,
|
|
497
|
+
title=title,
|
|
498
|
+
unsupported_reason=reason,
|
|
499
|
+
)
|
|
500
|
+
for img, interp, title, reason, eff_b64, eff_mime in zip(
|
|
501
|
+
images,
|
|
502
|
+
interpretations,
|
|
503
|
+
final_titles,
|
|
504
|
+
unsupported_reasons,
|
|
505
|
+
effective_b64s,
|
|
506
|
+
effective_mimes,
|
|
507
|
+
strict=True,
|
|
508
|
+
)
|
|
509
|
+
]
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
def _office_to_markdown_via_markitdown(file_path: str) -> str:
|
|
513
|
+
"""Convert an office document (DOCX/PPTX) to markdown via markitdown.
|
|
514
|
+
|
|
515
|
+
markitdown dispatches on the file extension and emits embedded
|
|
516
|
+
images inline as ```` at the
|
|
517
|
+
correct document position; we hand that off to
|
|
518
|
+
:func:`_extract_data_url_images` to rewrite the anchors and pull
|
|
519
|
+
the bytes out.
|
|
520
|
+
|
|
521
|
+
``keep_data_uris=True`` is required: by default markitdown's
|
|
522
|
+
``_CustomMarkdownify.convert_img`` truncates ``data:`` URLs at the
|
|
523
|
+
first comma (replacing the base64 payload with ``...``) to keep
|
|
524
|
+
markdown small for the LLM-summarisation use case it was designed
|
|
525
|
+
for. We need the full payload to extract the image bytes.
|
|
526
|
+
"""
|
|
527
|
+
try:
|
|
528
|
+
import markitdown
|
|
529
|
+
except ImportError as e:
|
|
530
|
+
raise ParseError(
|
|
531
|
+
"Office image extraction requires markitdown. Install with: pip install dot-parser"
|
|
532
|
+
) from e
|
|
533
|
+
return markitdown.MarkItDown().convert(file_path, keep_data_uris=True).text_content
|
|
534
|
+
|
|
535
|
+
|
|
536
|
+
def _parse_office_with_images(
|
|
537
|
+
source: str | Path | bytes,
|
|
538
|
+
vlm: VLM,
|
|
539
|
+
*,
|
|
540
|
+
suffix: str,
|
|
541
|
+
include_tables: bool,
|
|
542
|
+
deadline: float | None = None,
|
|
543
|
+
) -> ParseResult:
|
|
544
|
+
"""Shared markitdown+VLM pipeline for DOCX and PPTX sources."""
|
|
545
|
+
if isinstance(source, bytes):
|
|
546
|
+
with tempfile.NamedTemporaryFile(suffix=suffix, delete=False) as tmp:
|
|
547
|
+
tmp.write(source)
|
|
548
|
+
tmp.flush()
|
|
549
|
+
path = Path(tmp.name)
|
|
550
|
+
try:
|
|
551
|
+
raw_md = _office_to_markdown_via_markitdown(str(path))
|
|
552
|
+
finally:
|
|
553
|
+
path.unlink(missing_ok=True)
|
|
554
|
+
else:
|
|
555
|
+
raw_md = _office_to_markdown_via_markitdown(str(source))
|
|
556
|
+
|
|
557
|
+
markdown, images = _extract_data_url_images(raw_md)
|
|
558
|
+
if not include_tables:
|
|
559
|
+
markdown = strip_markdown_tables(markdown)
|
|
560
|
+
if images:
|
|
561
|
+
images = _interpret_images(
|
|
562
|
+
markdown,
|
|
563
|
+
images,
|
|
564
|
+
vlm,
|
|
565
|
+
caption_in_context=True,
|
|
566
|
+
deadline=deadline,
|
|
567
|
+
)
|
|
568
|
+
return ParseResult(markdown=markdown, images=images)
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
def parse_docx_with_images(
|
|
572
|
+
source: str | Path | bytes,
|
|
573
|
+
vlm: VLM,
|
|
574
|
+
*,
|
|
575
|
+
include_tables: bool = True,
|
|
576
|
+
deadline: float | None = None,
|
|
577
|
+
) -> ParseResult:
|
|
578
|
+
"""Parse a DOCX into markdown + interpreted images.
|
|
579
|
+
|
|
580
|
+
Text rendering matches the text-only ``parse(format='docx')`` path
|
|
581
|
+
because both routes go through markitdown. Image bytes are surfaced
|
|
582
|
+
out of markitdown's inline data URLs and rewritten as stable
|
|
583
|
+
```` anchors at the same document positions.
|
|
584
|
+
|
|
585
|
+
``vlm`` is invoked once per image with surrounding markdown context
|
|
586
|
+
(±3 blocks before, ±2 after), one image at a time.
|
|
587
|
+
Per-image VLM failures are absorbed (interpretation stays None) so
|
|
588
|
+
that a single bad image cannot fail the whole document; ``deadline``
|
|
589
|
+
bounds the interpretation phase and leaves any images it cuts short
|
|
590
|
+
uninterpreted rather than failing. See :func:`_interpret_images`.
|
|
591
|
+
|
|
592
|
+
When ``include_tables`` is False, GFM-style tables are stripped from
|
|
593
|
+
the resulting markdown — markitdown has no native flag to suppress
|
|
594
|
+
them, so this is a post-processing pass shared with the PDF path.
|
|
595
|
+
"""
|
|
596
|
+
return _parse_office_with_images(
|
|
597
|
+
source,
|
|
598
|
+
vlm,
|
|
599
|
+
suffix=".docx",
|
|
600
|
+
include_tables=include_tables,
|
|
601
|
+
deadline=deadline,
|
|
602
|
+
)
|
|
603
|
+
|
|
604
|
+
|
|
605
|
+
def parse_pptx_with_images(
|
|
606
|
+
source: str | Path | bytes,
|
|
607
|
+
vlm: VLM,
|
|
608
|
+
*,
|
|
609
|
+
include_tables: bool = True,
|
|
610
|
+
deadline: float | None = None,
|
|
611
|
+
) -> ParseResult:
|
|
612
|
+
"""Parse a PPTX into markdown + interpreted embedded pictures.
|
|
613
|
+
|
|
614
|
+
Same pipeline and guarantees as :func:`parse_docx_with_images`. The
|
|
615
|
+
markdown carries markitdown's ``<!-- Slide number: N -->`` markers
|
|
616
|
+
and slide titles as ``#`` headings, so slide-aware chunkers can key
|
|
617
|
+
on them.
|
|
618
|
+
|
|
619
|
+
Only *embedded picture files* are extracted and interpreted —
|
|
620
|
+
shapes, SmartArt and other content drawn natively in PowerPoint
|
|
621
|
+
have no image payload on this path (their text fragments still
|
|
622
|
+
appear in the markdown; charts come out as data tables). For decks
|
|
623
|
+
where drawn diagrams matter, prefer the OCR path
|
|
624
|
+
(``parse_with_images(..., backend=Mistral())``), which rasterises
|
|
625
|
+
each slide.
|
|
626
|
+
"""
|
|
627
|
+
return _parse_office_with_images(
|
|
628
|
+
source,
|
|
629
|
+
vlm,
|
|
630
|
+
suffix=".pptx",
|
|
631
|
+
include_tables=include_tables,
|
|
632
|
+
deadline=deadline,
|
|
633
|
+
)
|