dot-parser 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,152 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ """Shared helpers for image-aware parsing.
5
+
6
+ Lives outside `backends/` because both the PDF (Mistral OCR) and the
7
+ DOCX (markitdown + VLM) paths need the same primitives:
8
+
9
+ - Title normalisation into a safe kebab-case slug (``clean_slug``).
10
+ - Cross-document title disambiguation (``dedupe_titles``) — two images
11
+ with the same model-supplied title get ``-2``, ``-3`` suffixes.
12
+ - MIME type / data-URL helpers for parsing Mistral OCR's image payloads
13
+ (``guess_mime_type``, ``strip_data_url_prefix``).
14
+ - A single robust parser for ``{title, description}`` JSON annotations
15
+ (``parse_image_annotation``) — used both for Mistral OCR's
16
+ ``image_annotation`` field and for the chat-vision response in
17
+ :class:`dot_parser.vlms.MistralVLM`.
18
+
19
+ Keeping these here avoids three near-identical copies drifting in
20
+ ``backends/mistral.py``, ``vlms.py``, and ``docx_images.py``.
21
+ """
22
+
23
+ import json
24
+ import re
25
+
26
+ from dot_parser.images import ImageDescription
27
+
28
+ _SLUG_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+){0,15}$")
29
+
30
+
31
+ def clean_slug(value: object) -> str | None:
32
+ """Normalise a model-supplied title into a safe kebab-case slug.
33
+
34
+ Accepts the title even if it includes spaces, underscores, title-case
35
+ or trailing punctuation: lowercases and collapses any run of
36
+ non-alphanumerics into a single hyphen. Returns None when the result
37
+ is empty or would exceed the kebab-case length cap — better no title
38
+ than an unwieldy filename-like string in the UI.
39
+ """
40
+ if not isinstance(value, str):
41
+ return None
42
+ candidate = value.strip().lower()
43
+ if not candidate:
44
+ return None
45
+ candidate = re.sub(r"[^a-z0-9]+", "-", candidate).strip("-")
46
+ if not candidate or not _SLUG_RE.match(candidate):
47
+ return None
48
+ return candidate
49
+
50
+
51
+ def dedupe_titles(titles: list[str | None]) -> list[str | None]:
52
+ """Append ``-2``, ``-3``, ... to repeated titles, preserving Nones.
53
+
54
+ Two images that the model labelled ``workflow-diagram`` come back as
55
+ ``workflow-diagram`` and ``workflow-diagram-2`` so the title remains
56
+ safe to use as a display label or filename.
57
+ """
58
+ seen: dict[str, int] = {}
59
+ out: list[str | None] = []
60
+ for t in titles:
61
+ if t is None:
62
+ out.append(None)
63
+ continue
64
+ seen[t] = seen.get(t, 0) + 1
65
+ out.append(t if seen[t] == 1 else f"{t}-{seen[t]}")
66
+ return out
67
+
68
+
69
+ def guess_mime_type(image_id: str) -> str:
70
+ """Infer MIME type from a Mistral OCR image id like ``img-0.jpeg``.
71
+
72
+ Mistral encodes the format in the id's extension. Defaults to
73
+ ``image/png`` when the extension is missing or unrecognised.
74
+ """
75
+ ext = image_id.rsplit(".", 1)[-1].lower() if "." in image_id else ""
76
+ if ext in ("jpg", "jpeg"):
77
+ return "image/jpeg"
78
+ if ext == "webp":
79
+ return "image/webp"
80
+ if ext == "gif":
81
+ return "image/gif"
82
+ return "image/png"
83
+
84
+
85
+ def strip_data_url_prefix(b64: str) -> str:
86
+ """Strip an optional ``data:image/...;base64,`` prefix.
87
+
88
+ Mistral OCR sometimes returns the image as a full data URL rather
89
+ than bare base64. Callers always want the raw payload.
90
+ """
91
+ if b64.startswith("data:") and "," in b64:
92
+ return b64.split(",", 1)[1]
93
+ return b64
94
+
95
+
96
+ def parse_image_annotation(raw: str | None) -> ImageDescription:
97
+ """Parse a ``{title, description}`` JSON annotation into a record.
98
+
99
+ Used for two distinct input flavours:
100
+
101
+ - Mistral OCR's ``OCRImageObject.image_annotation`` field, which is
102
+ strict JSON when ``bbox_annotation_format`` is set.
103
+ - The content of a chat-vision response with a JSON schema
104
+ ``response_format`` — generally strict JSON, but the model
105
+ occasionally emits ``"description":`` followed by unescaped
106
+ multi-line text that breaks ``json.loads``.
107
+
108
+ Tries strict ``json.loads`` first; on failure, falls back to regex
109
+ extraction so a usable title is still surfaced. Non-JSON input is
110
+ returned as ``interpretation=raw, title=None`` (legacy passthrough).
111
+ Empty / None input yields an empty :class:`ImageDescription`.
112
+ """
113
+ if not raw:
114
+ return ImageDescription(interpretation="", title=None)
115
+
116
+ fenced = re.match(r"^```(?:json)?\s*(.*?)\s*```$", raw, flags=re.DOTALL)
117
+ candidate = fenced.group(1) if fenced else raw
118
+
119
+ start = candidate.find("{")
120
+ end = candidate.rfind("}")
121
+ if start != -1 and end > start:
122
+ snippet = candidate[start : end + 1]
123
+ try:
124
+ parsed = json.loads(snippet)
125
+ except (json.JSONDecodeError, ValueError):
126
+ parsed = None
127
+ if isinstance(parsed, dict):
128
+ desc_raw = parsed.get("description") or parsed.get("interpretation")
129
+ description = desc_raw.strip() if isinstance(desc_raw, str) else ""
130
+ title = clean_slug(parsed.get("title"))
131
+ # Surface any structured signal we got. When only a title
132
+ # comes back, interpretation is "" so callers can `or None`
133
+ # coerce uniformly without leaking the raw JSON blob.
134
+ if description or title:
135
+ return ImageDescription(interpretation=description, title=title)
136
+
137
+ # JSON looked promising but didn't yield usable fields — recover
138
+ # via regex. Title is always a short single-line string;
139
+ # description is whatever sits between ``"description":`` and
140
+ # the closing brace.
141
+ title_match = re.search(r'"title"\s*:\s*"([^"\n]*)"', snippet)
142
+ desc_match = re.search(
143
+ r'"description"\s*:\s*"?(.*?)"?\s*\}\s*$',
144
+ snippet,
145
+ flags=re.DOTALL,
146
+ )
147
+ recovered_title = clean_slug(title_match.group(1)) if title_match else None
148
+ recovered_desc = desc_match.group(1).strip() if desc_match else ""
149
+ if recovered_title or recovered_desc:
150
+ return ImageDescription(interpretation=recovered_desc, title=recovered_title)
151
+
152
+ return ImageDescription(interpretation=raw, title=None)
dot_parser/images.py ADDED
@@ -0,0 +1,125 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ """Public types for image-aware parsing.
5
+
6
+ `parse_with_images()` returns a `ParseResult` that pairs the markdown text
7
+ (with `![name](name)` anchors at each image position) with a list of
8
+ `ExtractedImage` records carrying the image bytes and, when available,
9
+ a textual interpretation.
10
+
11
+ The `VLM` protocol lets callers inject a vision-language model for the
12
+ DOCX path; the PDF path uses Mistral OCR's native per-image annotations
13
+ instead, so no VLM is required there.
14
+ """
15
+
16
+ from dataclasses import dataclass, field
17
+ from typing import Protocol, runtime_checkable
18
+
19
+
20
+ @dataclass(frozen=True)
21
+ class ExtractedImage:
22
+ """A single image extracted from a document.
23
+
24
+ `name` matches the `![name](name)` placeholder embedded at the
25
+ image's position inside the markdown returned alongside it; it
26
+ is a stable technical identifier (e.g. `img-003.png`).
27
+
28
+ `title` is a short human-readable label describing what the image
29
+ *depicts* (e.g. `gear-assembly-exploded-view`) — produced by the
30
+ VLM in the same call as `interpretation`. Use it for display and
31
+ `name` for cross-referencing with anchors / chunks.
32
+
33
+ `unsupported_reason` is populated when the extraction pipeline
34
+ deliberately did not call a VLM on this image because its
35
+ `mime_type` is outside the supported allowlist (e.g. EMF, WMF,
36
+ SVG, TIFF). The string carries a human-readable explanation of
37
+ the format and a hint at what external tool could recover it.
38
+ When set, `interpretation` and `title` will both be None; the
39
+ raw `base64` bytes are still populated so consumers can offer a
40
+ download or run their own conversion.
41
+ """
42
+
43
+ name: str
44
+ base64: str
45
+ mime_type: str
46
+ page: int | None = None
47
+ original_caption: str | None = None
48
+ interpretation: str | None = None
49
+ title: str | None = None
50
+ unsupported_reason: str | None = None
51
+
52
+
53
+ @dataclass(frozen=True)
54
+ class ImageDescription:
55
+ """Structured VLM response for a single image.
56
+
57
+ `title` is a short kebab-case label (4–8 words max) describing
58
+ what the image depicts. `interpretation` is the long-form
59
+ RAG-oriented description. `title` may be None when the VLM
60
+ cannot produce one (e.g. failure to parse a structured response).
61
+ """
62
+
63
+ interpretation: str
64
+ title: str | None = None
65
+
66
+
67
+ @dataclass(frozen=True)
68
+ class PageInfo:
69
+ """Per-page metadata a backend can report alongside the markdown.
70
+
71
+ Populated only when the caller opts in — see the ``Mistral``
72
+ backend's ``extract_headers_footers`` and ``confidence_scores``
73
+ options. Fields left None mean "not requested" rather than "not
74
+ present in the document", so an empty ``ParseResult.pages`` is the
75
+ default and carries no information either way.
76
+
77
+ ``header`` / ``footer`` hold the running page furniture (e.g.
78
+ "Confidential -- page 4 of 27") that OCR pulled out of the main
79
+ content, so it can be excluded from RAG chunks.
80
+
81
+ ``average_confidence`` / ``minimum_confidence`` are OCR self-reported
82
+ scores for the page. Calibrate thresholds against your own corpus:
83
+ measured across the benchmark PDFs, ``average_confidence`` stayed in
84
+ 0.985-0.989 and ``minimum_confidence`` in 0.23-0.46 regardless of
85
+ document quality. ``minimum_confidence`` reflects the single worst
86
+ region on the page, so it reads low even on clean pages and is not
87
+ on its own a signal that the page went badly.
88
+ """
89
+
90
+ page: int
91
+ header: str | None = None
92
+ footer: str | None = None
93
+ average_confidence: float | None = None
94
+ minimum_confidence: float | None = None
95
+
96
+
97
+ @dataclass(frozen=True)
98
+ class ParseResult:
99
+ """Markdown plus the images extracted from the source document."""
100
+
101
+ markdown: str
102
+ images: list[ExtractedImage] = field(default_factory=list)
103
+ pages: list[PageInfo] = field(default_factory=list)
104
+
105
+
106
+ @runtime_checkable
107
+ class VLM(Protocol):
108
+ """Vision-language model used to interpret images extracted from DOCX.
109
+
110
+ Implementations should return either a plain description string
111
+ (legacy) or a structured :class:`ImageDescription` carrying both
112
+ a short title and the longer interpretation. `context` is the
113
+ surrounding markdown text (e.g. the paragraph immediately before
114
+ the image) that the implementation may use to ground its
115
+ description; treat it as optional context, not as part of the
116
+ prompt's instructions.
117
+ """
118
+
119
+ def describe_image(
120
+ self,
121
+ image_base64: str,
122
+ *,
123
+ mime_type: str = "image/png",
124
+ context: str | None = None,
125
+ ) -> "str | ImageDescription": ...
@@ -0,0 +1,48 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ """Shared markdown post-processing utilities used across backends.
5
+
6
+ Lives outside `backends/` because both the PDF (Mistral OCR) and the
7
+ DOCX (markitdown) paths need the same table-stripping pass: neither
8
+ underlying tool exposes a native "drop tables" flag, so the caller is
9
+ responsible for removing them after the fact when ``include_tables``
10
+ is False.
11
+ """
12
+
13
+ import re
14
+
15
+ _TABLE_ROW_RE = re.compile(r"^\s*\|.*\|\s*$")
16
+ _TABLE_SEP_RE = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(\|\s*:?-{3,}:?\s*)+\|?\s*$")
17
+
18
+
19
+ def strip_markdown_tables(markdown: str) -> str:
20
+ """Drop GFM-style markdown tables from a body of text.
21
+
22
+ A table block here means: two or more consecutive lines where the
23
+ first matches a pipe-row and the second is a ``| --- | --- |``
24
+ separator. Operates line-by-line so it survives tables that aren't
25
+ surrounded by blank lines (Mistral OCR sometimes packs them tight
26
+ against surrounding prose; markitdown leaves a blank line, but the
27
+ same code path handles both).
28
+
29
+ Heuristic by design — we don't try to parse the full GFM table
30
+ grammar — but it cleanly removes the structures both Mistral OCR
31
+ and markitdown produce today and leaves non-table pipe lines alone.
32
+ """
33
+ lines = markdown.split("\n")
34
+ out: list[str] = []
35
+ i = 0
36
+ while i < len(lines):
37
+ if (
38
+ i + 1 < len(lines)
39
+ and _TABLE_ROW_RE.match(lines[i])
40
+ and _TABLE_SEP_RE.match(lines[i + 1])
41
+ ):
42
+ i += 2
43
+ while i < len(lines) and _TABLE_ROW_RE.match(lines[i]):
44
+ i += 1
45
+ continue
46
+ out.append(lines[i])
47
+ i += 1
48
+ return re.sub(r"\n{3,}", "\n\n", "\n".join(out)).strip("\n")
dot_parser/models.py ADDED
@@ -0,0 +1,22 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ import re
5
+ from dataclasses import dataclass, field
6
+
7
+ _HEADING_RE = re.compile(r"^(#{1,6})\s+(.+)$")
8
+
9
+
10
+ class ParseError(Exception):
11
+ """Raised when a document cannot be parsed."""
12
+
13
+
14
+ @dataclass(frozen=True)
15
+ class Chunk:
16
+ content: str
17
+ section_path: list[str] = field(default_factory=list)
18
+
19
+ @property
20
+ def heading(self) -> str | None:
21
+ first_line = self.content.split("\n", 1)[0]
22
+ return first_line if _HEADING_RE.match(first_line) else None
dot_parser/parsers.py ADDED
@@ -0,0 +1,325 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ import logging
5
+ import tempfile
6
+ from pathlib import Path
7
+
8
+ import markitdown
9
+
10
+ from dot_parser.backends import (
11
+ Backend,
12
+ BatchBackend,
13
+ DocxImageBackend,
14
+ ImageBackend,
15
+ Mistral,
16
+ PptxImageBackend,
17
+ Pymu,
18
+ )
19
+ from dot_parser.images import VLM, ParseResult
20
+ from dot_parser.models import ParseError
21
+
22
+ _log = logging.getLogger(__name__)
23
+
24
+ PDF_EXTENSIONS = {".pdf"}
25
+ PASSTHROUGH_EXTENSIONS = {".md", ".txt"}
26
+ MARKITDOWN_EXTENSIONS = {".docx", ".pptx", ".html", ".xhtml", ".htm", ".xlsx", ".csv"}
27
+ ALL_EXTENSIONS = PDF_EXTENSIONS | PASSTHROUGH_EXTENSIONS | MARKITDOWN_EXTENSIONS
28
+
29
+ _EXTENSION_MAP: dict[str, str] = {
30
+ "pdf": ".pdf",
31
+ "md": ".md",
32
+ "markdown": ".md",
33
+ "txt": ".txt",
34
+ "text": ".txt",
35
+ "docx": ".docx",
36
+ "pptx": ".pptx",
37
+ "html": ".html",
38
+ "xhtml": ".xhtml",
39
+ "htm": ".htm",
40
+ "xlsx": ".xlsx",
41
+ "csv": ".csv",
42
+ }
43
+
44
+
45
+ def _resolve_extension(source: str | Path | bytes, format: str | None) -> str:
46
+ if format is not None:
47
+ ext = _EXTENSION_MAP.get(format.lower())
48
+ if ext is None:
49
+ raise ParseError(f"Unknown format: {format!r}")
50
+ return ext
51
+
52
+ if isinstance(source, bytes):
53
+ raise ParseError("format is required when source is bytes")
54
+
55
+ ext = Path(source).suffix.lower()
56
+ if not ext:
57
+ raise ParseError(f"Cannot determine format for {source!r}: no file extension")
58
+ if ext not in ALL_EXTENSIONS:
59
+ raise ParseError(f"Unsupported file extension: {ext!r}")
60
+ return ext
61
+
62
+
63
+ def _read_text(source: str | Path | bytes) -> str:
64
+ if isinstance(source, bytes):
65
+ return source.decode("utf-8")
66
+ return Path(source).read_text(encoding="utf-8")
67
+
68
+
69
+ def _parse_with_markitdown(source: str | Path | bytes, ext: str) -> str:
70
+ md = markitdown.MarkItDown()
71
+ if isinstance(source, bytes):
72
+ with tempfile.NamedTemporaryFile(suffix=ext, delete=True) as tmp:
73
+ tmp.write(source)
74
+ tmp.flush()
75
+ result = md.convert(tmp.name)
76
+ else:
77
+ result = md.convert(str(source))
78
+ return result.text_content
79
+
80
+
81
+ def parse(
82
+ source: str | Path | bytes,
83
+ format: str | None = None,
84
+ backend: Backend | None = None,
85
+ ) -> str:
86
+ """Convert a document to Markdown.
87
+
88
+ Args:
89
+ source: file path (str or Path) or raw bytes.
90
+ format: explicit format hint (e.g. "pdf"). Required when source is bytes.
91
+ backend: optional PDF backend (Pymu, Docling, Mistral, Llama). When None,
92
+ uses Pymu (pymupdf4llm) for PDFs. Backends only apply to PDF input;
93
+ passing one for a non-PDF format raises ParseError.
94
+
95
+ Returns:
96
+ Markdown string.
97
+
98
+ Raises:
99
+ ParseError: on unsupported format, missing file, or backend mismatch.
100
+ """
101
+ ext = _resolve_extension(source, format)
102
+
103
+ if backend is not None and ext not in PDF_EXTENSIONS:
104
+ raise ParseError(f"backend= is only supported for PDF input (got {ext!r})")
105
+
106
+ try:
107
+ if ext in PDF_EXTENSIONS:
108
+ return (backend or Pymu()).parse_pdf(source)
109
+ elif ext in PASSTHROUGH_EXTENSIONS:
110
+ return _read_text(source)
111
+ else:
112
+ return _parse_with_markitdown(source, ext)
113
+ except ParseError:
114
+ raise
115
+ except Exception as e:
116
+ label = "<bytes>" if isinstance(source, bytes) else str(source)
117
+ raise ParseError(f"Could not parse {label}: {e}") from e
118
+
119
+
120
+ def parse_with_images(
121
+ source: str | Path | bytes,
122
+ format: str | None = None,
123
+ *,
124
+ backend: ImageBackend | DocxImageBackend | PptxImageBackend | None = None,
125
+ vlm: VLM | None = None,
126
+ annotate_images: bool = True,
127
+ include_tables: bool = True,
128
+ deadline: float | None = None,
129
+ ) -> ParseResult:
130
+ """Convert a document to Markdown and extract its images.
131
+
132
+ Args:
133
+ source: file path or raw bytes. Only ``.pdf``, ``.docx`` and
134
+ ``.pptx`` are supported by this function; other formats
135
+ raise ParseError.
136
+ format: explicit format hint. Required when source is bytes.
137
+ backend: optional image-capable backend. For PDF, defaults to a
138
+ fresh ``Mistral()`` instance and must expose
139
+ ``parse_pdf_with_images``. For DOCX/PPTX, when a backend
140
+ exposing ``parse_docx_with_images`` /
141
+ ``parse_pptx_with_images`` is passed (e.g. ``Mistral()``),
142
+ the document goes through OCR end-to-end — no VLM needed.
143
+ vlm: vision-language model used to interpret each DOCX/PPTX
144
+ image in the markitdown path. Required for DOCX/PPTX when
145
+ no format-capable backend is provided; ignored for PDF and
146
+ for the OCR path (Mistral OCR annotates images inline via
147
+ ``annotate_images``).
148
+ annotate_images: Applies to any OCR-backed path (PDF, or
149
+ DOCX/PPTX with ``backend=Mistral()``). When True (default),
150
+ Mistral OCR is asked to return a short description per
151
+ image in the same call. Set False to receive raw images
152
+ without interpretation. Ignored on the markitdown+VLM path
153
+ (the VLM is always invoked there).
154
+ include_tables: When False, drop tabular content from the
155
+ returned markdown.
156
+ deadline: absolute ``time.monotonic()`` timestamp bounding
157
+ per-image interpretation on the markitdown+VLM path. Images
158
+ not reached in time come back uninterpreted with
159
+ ``unsupported_reason`` set, instead of the call overrunning.
160
+ Ignored on OCR-backed paths.
161
+
162
+ Returns:
163
+ A :class:`ParseResult` with the markdown (containing
164
+ ``![name](name)`` anchors at image positions) and an
165
+ :class:`ExtractedImage` per image, in document order.
166
+
167
+ Raises:
168
+ ParseError: on unsupported format, missing backend/vlm, or when
169
+ a backend doesn't expose the relevant image-extraction
170
+ method for the input format.
171
+ """
172
+ ext = _resolve_extension(source, format)
173
+
174
+ if ext == ".pdf":
175
+ be = backend or Mistral()
176
+ if not isinstance(be, ImageBackend):
177
+ raise ParseError(
178
+ "Selected backend does not support image extraction. "
179
+ "Use Mistral() or another ImageBackend implementation."
180
+ )
181
+ try:
182
+ return be.parse_pdf_with_images(
183
+ source,
184
+ annotate_images=annotate_images,
185
+ include_tables=include_tables,
186
+ )
187
+ except ParseError:
188
+ raise
189
+ except Exception as e:
190
+ label = "<bytes>" if isinstance(source, bytes) else str(source)
191
+ raise ParseError(f"Could not parse {label}: {e}") from e
192
+
193
+ if ext in (".docx", ".pptx"):
194
+ # DOCX and PPTX each have two interchangeable paths. The OCR
195
+ # path wins when a backend that supports the format is passed
196
+ # in; otherwise we fall back to the markitdown + per-image VLM
197
+ # path.
198
+ fmt = ext[1:]
199
+ backend_method = f"parse_{fmt}_with_images"
200
+ if backend is not None and hasattr(backend, backend_method):
201
+ try:
202
+ return getattr(backend, backend_method)(
203
+ source,
204
+ annotate_images=annotate_images,
205
+ include_tables=include_tables,
206
+ )
207
+ except ParseError:
208
+ raise
209
+ except Exception as e:
210
+ label = "<bytes>" if isinstance(source, bytes) else str(source)
211
+ raise ParseError(f"Could not parse {label}: {e}") from e
212
+
213
+ if vlm is None:
214
+ raise ParseError(
215
+ f"{fmt.upper()} image extraction requires either a "
216
+ f"{fmt.upper()}-capable backend (e.g. backend=Mistral()) "
217
+ "or a VLM (e.g. vlm=MistralVLM()) for the markitdown path."
218
+ )
219
+ # Imported lazily so importing dot_parser.parsers doesn't pull
220
+ # in python-docx for callers that only use the text-only API.
221
+ from dot_parser import docx_images as _office_images
222
+
223
+ _parse_office = getattr(_office_images, f"parse_{fmt}_with_images")
224
+ try:
225
+ return _parse_office(
226
+ source,
227
+ vlm=vlm,
228
+ include_tables=include_tables,
229
+ deadline=deadline,
230
+ )
231
+ except ParseError:
232
+ raise
233
+ except Exception as e:
234
+ label = "<bytes>" if isinstance(source, bytes) else str(source)
235
+ raise ParseError(f"Could not parse {label}: {e}") from e
236
+
237
+ raise ParseError(f"parse_with_images only supports .pdf, .docx and .pptx (got {ext!r})")
238
+
239
+
240
+ def interpret_images(
241
+ result: ParseResult,
242
+ vlm: VLM,
243
+ *,
244
+ deadline: float | None = None,
245
+ ) -> ParseResult:
246
+ """Second-pass VLM interpretation of a ``ParseResult``'s images.
247
+
248
+ Fills ``interpretation`` (and ``title``) on each image by calling
249
+ ``vlm.describe_image`` once per image, grounding the call with the
250
+ image's ``original_caption`` (explicitly labelled) plus the markdown
251
+ surrounding its ``![name](name)`` anchor.
252
+
253
+ This exists because Mistral OCR's inline annotations
254
+ (``annotate_images=True``) are produced from the cropped image alone —
255
+ the annotating model sees no caption and no page text, which on
256
+ domain-specific figures produces confidently wrong descriptions. The
257
+ grounded recipe is::
258
+
259
+ result = parse_with_images(
260
+ "doc.pdf",
261
+ backend=Mistral(extract_captions=True),
262
+ annotate_images=False, # skip the ungrounded inline pass
263
+ )
264
+ result = interpret_images(result, MistralVLM())
265
+
266
+ Same number of VLM calls as ``annotate_images=True`` (one per image,
267
+ just client-side), but each call carries the figure's own caption.
268
+ Note the latency shape differs even though the count matches:
269
+ ``annotate_images=True`` annotates every image inside the single
270
+ ``ocr.process`` request, whereas this pass issues its own calls one
271
+ after another — so on image-heavy documents pass a ``deadline``.
272
+
273
+ Returns a new ``ParseResult``; markdown and pages are unchanged.
274
+ Per-image VLM failures are absorbed (that image's interpretation
275
+ stays None), as are images the ``deadline`` cuts short — those carry
276
+ ``unsupported_reason`` instead. Existing interpretations are
277
+ overwritten.
278
+ """
279
+ from dot_parser.docx_images import _interpret_images
280
+
281
+ images = _interpret_images(
282
+ result.markdown,
283
+ result.images,
284
+ vlm,
285
+ caption_in_context=True,
286
+ deadline=deadline,
287
+ )
288
+ return ParseResult(markdown=result.markdown, images=images, pages=result.pages)
289
+
290
+
291
+ def parse_pdfs(
292
+ sources: list[str | Path | bytes],
293
+ backend: Backend | None = None,
294
+ ) -> list[str | None]:
295
+ """Convert multiple PDFs to Markdown, in input order.
296
+
297
+ Backends that implement a ``parse_pdfs`` method (e.g. ``Mistral`` via the
298
+ Batch API) handle the whole list in one optimized call. For backends
299
+ without that method, falls back to looping ``parse()`` on each source.
300
+
301
+ Args:
302
+ sources: list of PDF file paths or raw bytes.
303
+ backend: optional PDF backend. Defaults to Pymu().
304
+
305
+ Returns:
306
+ A list of Markdown strings, same length and order as ``sources``.
307
+ Failed items are returned as ``None`` (no exception raised).
308
+ """
309
+ be = backend or Pymu()
310
+ if isinstance(be, BatchBackend):
311
+ return be.parse_pdfs(sources)
312
+ results: list[str | None] = []
313
+ for src in sources:
314
+ try:
315
+ # parse_pdfs is PDF-only, so pin format="pdf": callers pass raw bytes
316
+ # (in-memory PDFs) which parse() can't infer an extension from, and
317
+ # would otherwise reject with "format is required when source is bytes".
318
+ results.append(parse(src, format="pdf", backend=be))
319
+ except ParseError as exc:
320
+ # Loop-fallback backends (e.g. Pymu) fail per file; log why before
321
+ # dropping to None so the caller isn't left with an unexplained gap.
322
+ label = "<bytes>" if isinstance(src, bytes) else str(src)
323
+ _log.error("parse_pdfs: could not parse %s: %s", label, exc)
324
+ results.append(None)
325
+ return results