dot-parser 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,633 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ """Office documents (DOCX/PPTX) → Markdown with per-image interpretation.
5
+
6
+ Runs markitdown to produce the markdown body — which already inlines
7
+ each embedded image as ``![alt](data:image/<mime>;base64,...)`` at the
8
+ correct document position — then walks those data URLs in order,
9
+ rewriting them into stable ``![name](name)`` anchors and emitting one
10
+ :class:`ExtractedImage` per match. Each image is then handed to the
11
+ caller-supplied VLM for a textual interpretation.
12
+
13
+ Going through markitdown (instead of the previous hand-written
14
+ python-docx walker) means the text rendering matches the text-only
15
+ ``parse(format='docx')`` path exactly — same markdown surface with or
16
+ without ``parse_images=True`` — and we inherit markitdown's handling of
17
+ formatting, lists, tables, headers, and footers for free.
18
+
19
+ The pipeline is format-agnostic: markitdown dispatches on the file
20
+ extension and both its DOCX and PPTX converters honour
21
+ ``keep_data_uris=True``. For PPTX, markitdown additionally emits a
22
+ ``<!-- Slide number: N -->`` comment before each slide and the slide
23
+ title as an ``#`` heading. Note that the PPTX path only sees *embedded
24
+ pictures* — diagrams drawn natively in PowerPoint (shapes, arrows,
25
+ SmartArt) have no image payload and surface as text fragments only;
26
+ use the Mistral OCR path (``backend=Mistral()``) when those matter.
27
+ """
28
+
29
+ import base64
30
+ import io
31
+ import logging
32
+ import re
33
+ import tempfile
34
+ import time
35
+ from pathlib import Path
36
+
37
+ from dot_parser.image_utils import dedupe_titles
38
+ from dot_parser.images import (
39
+ VLM,
40
+ ExtractedImage,
41
+ ImageDescription,
42
+ ParseResult,
43
+ )
44
+ from dot_parser.markdown_utils import strip_markdown_tables
45
+ from dot_parser.models import ParseError
46
+
47
+ _logger = logging.getLogger(__name__)
48
+
49
+ # Markdown blocks around each image anchor that we feed back to the VLM
50
+ # as context. We look both before and after the anchor: French/EU
51
+ # technical documents place figure captions ("Figure 1: …") directly
52
+ # *below* the image, so a preceding-only window would miss them.
53
+ # TODO: ENH since I did not really optimize these numbers, maybe we should adapt them
54
+ # based on some parameters and the multimodel model of choice.)
55
+ _CONTEXT_BLOCKS_BEFORE = 3
56
+ _CONTEXT_BLOCKS_AFTER = 2
57
+
58
+
59
+ # Reported in `unsupported_reason` for images the deadline cut short.
60
+ # Deliberately reuses the "why is there no interpretation" channel that
61
+ # undecodable formats already populate, so existing consumers render it
62
+ # with no change.
63
+ SKIPPED_DEADLINE_REASON = (
64
+ "Not interpreted: the document's image-analysis time budget ran out "
65
+ "before this image was reached."
66
+ )
67
+
68
+ # MIME types Mistral's chat-vision models currently accept (Mistral
69
+ # Large 3 / Medium 3.1 / Small 3.2 / Ministral 3 — Pixtral is
70
+ # deprecated). Anything outside this set is first handed to Pillow for a
71
+ # rescue conversion (see the interpretation loop); only if Pillow can't
72
+ # decode it do we surface it as unsupported.
73
+ _VLM_SUPPORTED_MIMES = frozenset(
74
+ {
75
+ "image/png",
76
+ "image/jpeg",
77
+ "image/jpg",
78
+ "image/webp",
79
+ "image/gif",
80
+ }
81
+ )
82
+
83
+ # Aliases for the canonical MIME types above. Data URLs don't use a
84
+ # single canonical spelling — the same bytes show up as "image/x-png",
85
+ # "image/x-ms-bmp", etc. depending on the producer. We normalise these
86
+ # before the membership / conversion decisions so a non-canonical
87
+ # spelling never gets misrouted to the unsupported branch.
88
+ _MIME_ALIASES: dict[str, str] = {
89
+ "image/x-png": "image/png",
90
+ "image/pjpeg": "image/jpeg",
91
+ "image/x-ms-bmp": "image/bmp",
92
+ "image/x-bmp": "image/bmp",
93
+ "image/x-tiff": "image/tiff",
94
+ }
95
+
96
+
97
+ def _normalize_mime(mime_type: str) -> str:
98
+ """Collapse known MIME aliases to their canonical spelling."""
99
+ return _MIME_ALIASES.get(mime_type.lower(), mime_type.lower())
100
+
101
+
102
+ # Per-format messages surfaced as ExtractedImage.unsupported_reason
103
+ # so downstream consumers can explain to end-users why an image was
104
+ # not interpreted, and what external tool could recover it.
105
+ #
106
+ # Each comment below records (a) when this MIME actually appears in
107
+ # real DOCX files and (b) the recommended fix if we ever decide to
108
+ # add full support. Every non-VLM-native image is first attempted via
109
+ # Pillow (_convert_raster_to_png); this dict is only consulted when that
110
+ # decode returns None — so raster formats (BMP, TIFF, and any
111
+ # plugin-registered format) only land here on an actual decode failure.
112
+ _UNSUPPORTED_REASONS: dict[str, str] = {
113
+ # Seen in: Word docs containing Visio drawings, SmartArt, grouped
114
+ # shapes, or charts — Word inlines an EMF as a fallback raster of
115
+ # the vector content. The "real" source often lives in
116
+ # word/embeddings/*.bin (OLE-embedded Visio), which is a separate
117
+ # extraction problem.
118
+ # Fix: shell out to `soffice --headless --convert-to png`
119
+ "image/x-emf": (
120
+ "Windows Enhanced Metafile (.emf): vector instruction stream "
121
+ "embedded by Word as a fallback preview for Visio drawings, "
122
+ "SmartArt or grouped shapes. Not accepted by the vision API. "
123
+ "Rendering requires an external tool such as LibreOffice "
124
+ "(`soffice --headless --convert-to png`) or Inkscape. The raw "
125
+ "base64 bytes are still returned for download or external "
126
+ "processing."
127
+ ),
128
+ # Seen in: legacy/old Office documents (pre-2007 era) and content
129
+ # copy-pasted from older applications. Much rarer than EMF today.
130
+ # Fix: same toolchain as EMF — both LibreOffice and Inkscape read
131
+ # WMF. No separate code path needed if EMF support lands.
132
+ "image/x-wmf": (
133
+ "Windows Metafile (.wmf): legacy Windows vector format, often "
134
+ "carried inside older Office documents. Not accepted by the "
135
+ "vision API. Convert via LibreOffice or Inkscape to render."
136
+ ),
137
+ # Seen in: modern Word (Office 365 inserts SVG natively when you
138
+ # paste icons or vector graphics) and exports from Figma /
139
+ # Illustrator / draw.io. Browsers already render SVG data-URLs
140
+ # directly, so the frontend visual is fine without us — the only
141
+ # gain from local conversion is the VLM caption.
142
+ # Fix: `resvg-py` (Rust binding, ships with wheels, no system
143
+ # libs) is the low-friction option; `cairosvg` has broader SVG
144
+ # feature coverage but pulls in cairo/pango as OS libs.
145
+ "image/svg+xml": (
146
+ "Scalable Vector Graphics: browser-renderable but not accepted "
147
+ "by the vision API. Convert to PNG (e.g. cairosvg) for "
148
+ "interpretation, or render client-side from the base64 bytes."
149
+ ),
150
+ # Reached only when _convert_raster_to_png() fails on a TIFF
151
+ # payload — typically unsupported compression (e.g. CCITT G4 from
152
+ # scanners without libtiff support in the wheel) or multi-page
153
+ # documents where we currently keep only the first frame.
154
+ # Fix: extend _convert_raster_to_png to iterate `n_frames` and
155
+ # emit one ExtractedImage per page; install Pillow with full
156
+ # libtiff support for exotic compressions.
157
+ "image/tiff": (
158
+ "TIFF: automatic conversion to PNG via Pillow failed for this "
159
+ "image (e.g. unsupported compression or multi-page payload). The "
160
+ "raw base64 bytes are still returned for external recovery."
161
+ ),
162
+ # Reached only when _convert_raster_to_png() fails on a BMP — very
163
+ # rare in practice; mostly indicates a corrupt or truncated
164
+ # payload extracted from a damaged DOCX. No code change needed
165
+ # beyond surfacing this reason.
166
+ "image/bmp": (
167
+ "BMP: automatic conversion to PNG via Pillow failed for this "
168
+ "image. The raw base64 bytes are still returned for external "
169
+ "recovery."
170
+ ),
171
+ # Seen in: almost never in real DOCX files — AVIF is a web format
172
+ # (Chrome/Safari era), not something Office tooling produces.
173
+ # Listed defensively in case a user pastes web content.
174
+ # Fix: add `pillow-avif-plugin` to deps and import it once at
175
+ # startup; the plugin auto-registers a Pillow decoder, after which
176
+ # the rescue-conversion path picks AVIF up with no other changes.
177
+ "image/avif": (
178
+ "AVIF: not accepted by the vision API. Convert to PNG or "
179
+ "WebP using Pillow (with pillow-avif-plugin) for interpretation."
180
+ ),
181
+ }
182
+
183
+ # Formats not in _UNSUPPORTED_REASONS yet (they fall through to the
184
+ # generic message in _unsupported_reason). Because the rescue path now
185
+ # tries Pillow on anything non-native, "adding support" for a raster
186
+ # format is purely a matter of making Pillow able to decode it — no
187
+ # MIME-set edit is needed:
188
+ #
189
+ # image/heic, image/heif
190
+ # Seen in: iPhone screenshots and photos pasted into Word — rising
191
+ # fast as a real-world source.
192
+ # Fix: add `pillow-heif` dep + import it once at startup. The plugin
193
+ # registers a Pillow decoder and the rescue path handles the rest.
194
+ #
195
+ # image/jp2 (JPEG 2000)
196
+ # Seen in: scientific/archival pipelines; vanishingly rare in
197
+ # Office documents.
198
+ # Fix: works out of the box if the Pillow wheel was built with
199
+ # OpenJPEG support (most modern wheels are) — nothing to add.
200
+ #
201
+ # image/x-icon (favicons / Windows ICO)
202
+ # Seen in: occasional paste from a browser into Word.
203
+ # Fix: Pillow has native ICO support, so the rescue path already
204
+ # handles it. Multi-resolution ICOs collapse to the largest frame,
205
+ # which is what we want for VLM interpretation.
206
+ #
207
+ # image/eps (encapsulated PostScript)
208
+ # Seen in: academic/print-publishing pipelines, very rare in
209
+ # modern Office.
210
+ # Fix: Pillow can decode EPS but only if Ghostscript is installed
211
+ # on the host. Better treated like EMF (optional extra + binary
212
+ # probe).
213
+
214
+
215
+ def _convert_raster_to_png(b64: str) -> tuple[str, str] | None:
216
+ """Re-encode a raster image (BMP/TIFF) as PNG via Pillow.
217
+
218
+ Returns ``(new_b64, "image/png")`` on success, or ``None`` if the
219
+ payload can't be decoded (corrupt bytes, unsupported TIFF
220
+ compression, missing Pillow). Multi-page TIFFs collapse to the
221
+ first frame — good enough for document images, where TIFFs are
222
+ almost always single-page scans.
223
+ """
224
+ try:
225
+ from PIL import Image
226
+ except ImportError:
227
+ _logger.warning("Pillow not installed; skipping raster→PNG conversion.")
228
+ return None
229
+ try:
230
+ raw = base64.b64decode(b64)
231
+ with Image.open(io.BytesIO(raw)) as im:
232
+ im.load()
233
+ if im.mode not in ("RGB", "RGBA", "L", "LA", "P"):
234
+ im = im.convert("RGBA")
235
+ out = io.BytesIO()
236
+ im.save(out, format="PNG")
237
+ return base64.b64encode(out.getvalue()).decode("ascii"), "image/png"
238
+ except Exception as exc:
239
+ _logger.warning("Raster→PNG conversion failed: %s", exc)
240
+ return None
241
+
242
+
243
+ def _unsupported_reason(mime_type: str) -> str:
244
+ """Return a human-readable explanation for an unsupported MIME.
245
+
246
+ Looks up a per-format message; falls back to a generic one
247
+ pointing at the offending MIME so consumers always get *some*
248
+ actionable detail rather than a bare None.
249
+ """
250
+ if mime_type in _UNSUPPORTED_REASONS:
251
+ return _UNSUPPORTED_REASONS[mime_type]
252
+ return (
253
+ f"Image format '{mime_type}' is not currently supported by "
254
+ "the image interpretation pipeline. The raw base64 bytes are "
255
+ "still returned for download or external processing."
256
+ )
257
+
258
+
259
+ # Matches markitdown's inline data-URL images. markitdown emits them
260
+ # on a single line so we don't need DOTALL. Alt text is captured so
261
+ # we can carry it through as `original_caption` for the title fallback.
262
+ _DATA_URL_IMG_RE = re.compile(
263
+ r"!\[(?P<alt>[^\]]*)\]"
264
+ r"\(data:image/(?P<subtype>[A-Za-z0-9.+\-]+);base64,(?P<b64>[A-Za-z0-9+/=]+)\)"
265
+ )
266
+
267
+ _MIME_SUBTYPE_TO_EXT: dict[str, str] = {
268
+ "png": "png",
269
+ "x-png": "png",
270
+ "jpeg": "jpg",
271
+ "jpg": "jpg",
272
+ "webp": "webp",
273
+ "gif": "gif",
274
+ "bmp": "bmp",
275
+ "tiff": "tiff",
276
+ "svg+xml": "svg",
277
+ }
278
+
279
+
280
+ def _ext_for_subtype(subtype: str) -> str:
281
+ return _MIME_SUBTYPE_TO_EXT.get(subtype.lower(), "png")
282
+
283
+
284
+ def _extract_data_url_images(
285
+ raw_md: str,
286
+ ) -> tuple[str, list[ExtractedImage]]:
287
+ """Rewrite markitdown's inline data-URL images as stable anchors.
288
+
289
+ Walks data-URL matches in document order, mints ``img-NNN.<ext>``
290
+ names, and returns the rewritten markdown alongside one
291
+ :class:`ExtractedImage` per match (carrying the raw base64). The
292
+ image's alt text is preserved as ``original_caption`` so it remains
293
+ available as a title fallback in :func:`_interpret_images`.
294
+ """
295
+ images: list[ExtractedImage] = []
296
+
297
+ def _sub(match: re.Match) -> str:
298
+ alt = match.group("alt") or None
299
+ subtype = match.group("subtype")
300
+ b64 = match.group("b64")
301
+ mime = f"image/{subtype}"
302
+ ext = _ext_for_subtype(subtype)
303
+ name = f"img-{len(images) + 1:03d}.{ext}"
304
+ images.append(
305
+ ExtractedImage(
306
+ name=name,
307
+ base64=b64,
308
+ mime_type=mime,
309
+ original_caption=alt,
310
+ )
311
+ )
312
+ return f"![{name}]({name})"
313
+
314
+ rewritten = _DATA_URL_IMG_RE.sub(_sub, raw_md)
315
+ return rewritten, images
316
+
317
+
318
+ def _slugify_caption(caption: str | None) -> str | None:
319
+ """Derive a kebab-case slug from a free-form caption string.
320
+
321
+ Returns None for empty / unusable captions. Caps length at 8 words so
322
+ captions like "Figure 3: The full bill-of-materials of the assembly,
323
+ including..." produce something compact rather than a sentence-long
324
+ filename.
325
+ """
326
+ if not caption:
327
+ return None
328
+ # Strip common "Figure N:" / "Fig. N -" prefixes Word users type.
329
+ stripped = re.sub(
330
+ r"^\s*(figure|fig\.?|table|image|img\.?|exhibit)\s*\d*[:.\-]?\s*",
331
+ "",
332
+ caption,
333
+ flags=re.IGNORECASE,
334
+ )
335
+ slug = re.sub(r"[^a-zA-Z0-9]+", "-", stripped).strip("-").lower()
336
+ if not slug:
337
+ return None
338
+ parts = slug.split("-")
339
+ return "-".join(parts[:8])
340
+
341
+
342
+ def _interpret_images(
343
+ markdown: str,
344
+ images: list[ExtractedImage],
345
+ vlm: VLM,
346
+ *,
347
+ caption_in_context: bool = False,
348
+ deadline: float | None = None,
349
+ ) -> list[ExtractedImage]:
350
+ """Call ``vlm.describe_image`` for each image and assign titles.
351
+
352
+ When ``caption_in_context`` is True, an image's ``original_caption``
353
+ is prepended to the VLM context, explicitly labelled. The caption is
354
+ the strongest grounding available — it is the author's own statement
355
+ of what the figure shows — so it is passed explicitly rather than
356
+ relying on it happening to appear in the sliced markdown
357
+ neighbourhood. Both production call sites (the DOCX markitdown path
358
+ and :func:`dot_parser.parsers.interpret_images`) enable it; the flag
359
+ exists so images without captions cost nothing and so the grounding
360
+ can be switched off in isolation if it ever misbehaves.
361
+
362
+ ``deadline`` is an absolute :func:`time.monotonic` timestamp bounding
363
+ the interpretation phase. Calls are issued one at a time, each taking
364
+ a few seconds, so a document with many figures can otherwise run for
365
+ longer than any caller is willing to wait — an image-heavy report
366
+ carries 100-200 of them. Once the deadline passes, the images not yet
367
+ reached are returned with their raw bytes intact and
368
+ :data:`SKIPPED_DEADLINE_REASON` in ``unsupported_reason``, the same
369
+ channel already used for undecodable formats, so consumers need no
370
+ new branch. That makes an oversized document a partial result rather
371
+ than a failed request. The deadline is checked before each call, so
372
+ at most one in-progress call overruns it; bound that with a per-call
373
+ timeout on the VLM itself.
374
+
375
+ Title resolution order per image:
376
+ 1. VLM-generated kebab-case slug (preferred)
377
+ 2. Slug derived from the document's original caption, if any
378
+ 3. None (leave it to consumers to display the technical ``name``)
379
+
380
+ Duplicate titles across the document are disambiguated with
381
+ ``-2``, ``-3``, ... suffixes so that the title is safe to use as a
382
+ display label.
383
+
384
+ Backwards-compatible with VLM implementations that return a plain
385
+ ``str``: those simply produce ``interpretation`` with no title.
386
+ Per-image VLM failures are absorbed (interpretation stays None).
387
+ """
388
+ blocks = markdown.split("\n\n")
389
+ # Locate each anchor block once to slice surrounding context cheaply.
390
+ anchor_positions: dict[str, int] = {}
391
+ for i, block in enumerate(blocks):
392
+ for img in images:
393
+ anchor = f"![{img.name}]({img.name})"
394
+ if anchor in block and img.name not in anchor_positions:
395
+ anchor_positions[img.name] = i
396
+
397
+ interpretations: list[str | None] = []
398
+ raw_titles: list[str | None] = []
399
+ unsupported_reasons: list[str | None] = []
400
+ effective_b64s: list[str | None] = []
401
+ effective_mimes: list[str | None] = []
402
+ for index, img in enumerate(images):
403
+ if deadline is not None and time.monotonic() >= deadline:
404
+ # Out of time: hand back everything still untouched rather
405
+ # than letting the caller's request run over.
406
+ remaining = len(images) - index
407
+ _logger.warning(
408
+ "image interpretation deadline reached: %d of %d image(s) left uninterpreted",
409
+ remaining,
410
+ len(images),
411
+ )
412
+ for _ in range(remaining):
413
+ interpretations.append(None)
414
+ raw_titles.append(None)
415
+ unsupported_reasons.append(SKIPPED_DEADLINE_REASON)
416
+ effective_b64s.append(None)
417
+ effective_mimes.append(None)
418
+ break
419
+
420
+ vlm_b64 = img.base64
421
+ vlm_mime = img.mime_type
422
+ converted: tuple[str, str] | None = None
423
+
424
+ # Anything the VLM doesn't natively accept is handed to Pillow
425
+ # for a rescue conversion to PNG — BMP/TIFF and any raster format
426
+ # Pillow can decode (incl. plugin-registered HEIC/AVIF) come back
427
+ # as PNG and enter the regular VLM path. We let Pillow be the
428
+ # judge rather than gating on a hardcoded MIME set: a successful
429
+ # decode *is* the support signal, so non-canonical spellings
430
+ # (image/x-ms-bmp, …) and future formats work without code
431
+ # changes. Formats Pillow can't decode (EMF/WMF/SVG/…) return
432
+ # None and fall through to the unsupported branch below, with the
433
+ # original bytes preserved for download or external processing.
434
+ if _normalize_mime(vlm_mime) not in _VLM_SUPPORTED_MIMES:
435
+ converted = _convert_raster_to_png(img.base64)
436
+ if converted is not None:
437
+ vlm_b64, vlm_mime = converted
438
+ else:
439
+ interpretations.append(None)
440
+ raw_titles.append(None)
441
+ unsupported_reasons.append(_unsupported_reason(img.mime_type))
442
+ effective_b64s.append(None)
443
+ effective_mimes.append(None)
444
+ continue
445
+
446
+ pos = anchor_positions.get(img.name)
447
+ context_parts: list[str] = []
448
+ if caption_in_context and img.original_caption:
449
+ context_parts.append(f"Figure caption from the document: {img.original_caption}")
450
+ if pos is not None:
451
+ start = max(0, pos - _CONTEXT_BLOCKS_BEFORE)
452
+ end = pos + 1 + _CONTEXT_BLOCKS_AFTER
453
+ before = [b for b in blocks[start:pos] if b.strip()]
454
+ after = [b for b in blocks[pos + 1 : end] if b.strip()]
455
+ context_parts.extend(before + after)
456
+ context: str | None = "\n\n".join(context_parts) or None
457
+ try:
458
+ result = vlm.describe_image(
459
+ vlm_b64,
460
+ mime_type=vlm_mime,
461
+ context=context,
462
+ )
463
+ except Exception:
464
+ result = None
465
+
466
+ if isinstance(result, ImageDescription):
467
+ interpretations.append(result.interpretation or None)
468
+ raw_titles.append(result.title)
469
+ elif isinstance(result, str):
470
+ interpretations.append(result or None)
471
+ raw_titles.append(None)
472
+ else:
473
+ interpretations.append(None)
474
+ raw_titles.append(None)
475
+ unsupported_reasons.append(None)
476
+ # Surface the converted PNG bytes on the emitted ExtractedImage
477
+ # so the frontend can render the image without browser-side
478
+ # decoder support for the original format.
479
+ effective_b64s.append(converted[0] if converted else None)
480
+ effective_mimes.append(converted[1] if converted else None)
481
+
482
+ # Apply caption fallback for any image whose VLM didn't supply a title.
483
+ resolved_titles: list[str | None] = [
484
+ rt if rt else _slugify_caption(img.original_caption)
485
+ for rt, img in zip(raw_titles, images, strict=True)
486
+ ]
487
+ final_titles = dedupe_titles(resolved_titles)
488
+
489
+ return [
490
+ ExtractedImage(
491
+ name=img.name,
492
+ base64=eff_b64 or img.base64,
493
+ mime_type=eff_mime or img.mime_type,
494
+ page=img.page,
495
+ original_caption=img.original_caption,
496
+ interpretation=interp,
497
+ title=title,
498
+ unsupported_reason=reason,
499
+ )
500
+ for img, interp, title, reason, eff_b64, eff_mime in zip(
501
+ images,
502
+ interpretations,
503
+ final_titles,
504
+ unsupported_reasons,
505
+ effective_b64s,
506
+ effective_mimes,
507
+ strict=True,
508
+ )
509
+ ]
510
+
511
+
512
+ def _office_to_markdown_via_markitdown(file_path: str) -> str:
513
+ """Convert an office document (DOCX/PPTX) to markdown via markitdown.
514
+
515
+ markitdown dispatches on the file extension and emits embedded
516
+ images inline as ``![alt](data:image/<mime>;base64,<b64>)`` at the
517
+ correct document position; we hand that off to
518
+ :func:`_extract_data_url_images` to rewrite the anchors and pull
519
+ the bytes out.
520
+
521
+ ``keep_data_uris=True`` is required: by default markitdown's
522
+ ``_CustomMarkdownify.convert_img`` truncates ``data:`` URLs at the
523
+ first comma (replacing the base64 payload with ``...``) to keep
524
+ markdown small for the LLM-summarisation use case it was designed
525
+ for. We need the full payload to extract the image bytes.
526
+ """
527
+ try:
528
+ import markitdown
529
+ except ImportError as e:
530
+ raise ParseError(
531
+ "Office image extraction requires markitdown. Install with: pip install dot-parser"
532
+ ) from e
533
+ return markitdown.MarkItDown().convert(file_path, keep_data_uris=True).text_content
534
+
535
+
536
+ def _parse_office_with_images(
537
+ source: str | Path | bytes,
538
+ vlm: VLM,
539
+ *,
540
+ suffix: str,
541
+ include_tables: bool,
542
+ deadline: float | None = None,
543
+ ) -> ParseResult:
544
+ """Shared markitdown+VLM pipeline for DOCX and PPTX sources."""
545
+ if isinstance(source, bytes):
546
+ with tempfile.NamedTemporaryFile(suffix=suffix, delete=False) as tmp:
547
+ tmp.write(source)
548
+ tmp.flush()
549
+ path = Path(tmp.name)
550
+ try:
551
+ raw_md = _office_to_markdown_via_markitdown(str(path))
552
+ finally:
553
+ path.unlink(missing_ok=True)
554
+ else:
555
+ raw_md = _office_to_markdown_via_markitdown(str(source))
556
+
557
+ markdown, images = _extract_data_url_images(raw_md)
558
+ if not include_tables:
559
+ markdown = strip_markdown_tables(markdown)
560
+ if images:
561
+ images = _interpret_images(
562
+ markdown,
563
+ images,
564
+ vlm,
565
+ caption_in_context=True,
566
+ deadline=deadline,
567
+ )
568
+ return ParseResult(markdown=markdown, images=images)
569
+
570
+
571
+ def parse_docx_with_images(
572
+ source: str | Path | bytes,
573
+ vlm: VLM,
574
+ *,
575
+ include_tables: bool = True,
576
+ deadline: float | None = None,
577
+ ) -> ParseResult:
578
+ """Parse a DOCX into markdown + interpreted images.
579
+
580
+ Text rendering matches the text-only ``parse(format='docx')`` path
581
+ because both routes go through markitdown. Image bytes are surfaced
582
+ out of markitdown's inline data URLs and rewritten as stable
583
+ ``![name](name)`` anchors at the same document positions.
584
+
585
+ ``vlm`` is invoked once per image with surrounding markdown context
586
+ (±3 blocks before, ±2 after), one image at a time.
587
+ Per-image VLM failures are absorbed (interpretation stays None) so
588
+ that a single bad image cannot fail the whole document; ``deadline``
589
+ bounds the interpretation phase and leaves any images it cuts short
590
+ uninterpreted rather than failing. See :func:`_interpret_images`.
591
+
592
+ When ``include_tables`` is False, GFM-style tables are stripped from
593
+ the resulting markdown — markitdown has no native flag to suppress
594
+ them, so this is a post-processing pass shared with the PDF path.
595
+ """
596
+ return _parse_office_with_images(
597
+ source,
598
+ vlm,
599
+ suffix=".docx",
600
+ include_tables=include_tables,
601
+ deadline=deadline,
602
+ )
603
+
604
+
605
+ def parse_pptx_with_images(
606
+ source: str | Path | bytes,
607
+ vlm: VLM,
608
+ *,
609
+ include_tables: bool = True,
610
+ deadline: float | None = None,
611
+ ) -> ParseResult:
612
+ """Parse a PPTX into markdown + interpreted embedded pictures.
613
+
614
+ Same pipeline and guarantees as :func:`parse_docx_with_images`. The
615
+ markdown carries markitdown's ``<!-- Slide number: N -->`` markers
616
+ and slide titles as ``#`` headings, so slide-aware chunkers can key
617
+ on them.
618
+
619
+ Only *embedded picture files* are extracted and interpreted —
620
+ shapes, SmartArt and other content drawn natively in PowerPoint
621
+ have no image payload on this path (their text fragments still
622
+ appear in the markdown; charts come out as data tables). For decks
623
+ where drawn diagrams matter, prefer the OCR path
624
+ (``parse_with_images(..., backend=Mistral())``), which rasterises
625
+ each slide.
626
+ """
627
+ return _parse_office_with_images(
628
+ source,
629
+ vlm,
630
+ suffix=".pptx",
631
+ include_tables=include_tables,
632
+ deadline=deadline,
633
+ )