dot-parser 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dot_parser/__init__.py +49 -0
- dot_parser/backends/__init__.py +26 -0
- dot_parser/backends/_base.py +79 -0
- dot_parser/backends/docling.py +59 -0
- dot_parser/backends/llama.py +76 -0
- dot_parser/backends/mistral.py +622 -0
- dot_parser/backends/pymu.py +56 -0
- dot_parser/chunking.py +257 -0
- dot_parser/docx_images.py +633 -0
- dot_parser/image_utils.py +152 -0
- dot_parser/images.py +125 -0
- dot_parser/markdown_utils.py +48 -0
- dot_parser/models.py +22 -0
- dot_parser/parsers.py +325 -0
- dot_parser/pricing.py +58 -0
- dot_parser/tokens.py +6 -0
- dot_parser/vlms.py +203 -0
- dot_parser-2.0.0.dist-info/METADATA +274 -0
- dot_parser-2.0.0.dist-info/RECORD +21 -0
- dot_parser-2.0.0.dist-info/WHEEL +4 -0
- dot_parser-2.0.0.dist-info/licenses/LICENSE.md +660 -0
|
@@ -0,0 +1,622 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
import inspect
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
import os
|
|
8
|
+
import time
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import TYPE_CHECKING, Literal
|
|
11
|
+
|
|
12
|
+
from dot_parser.image_utils import (
|
|
13
|
+
dedupe_titles,
|
|
14
|
+
guess_mime_type,
|
|
15
|
+
parse_image_annotation,
|
|
16
|
+
strip_data_url_prefix,
|
|
17
|
+
)
|
|
18
|
+
from dot_parser.images import ExtractedImage, PageInfo, ParseResult
|
|
19
|
+
from dot_parser.markdown_utils import strip_markdown_tables
|
|
20
|
+
|
|
21
|
+
if TYPE_CHECKING:
|
|
22
|
+
from mistralai.client.models import BatchJob, OCRPageObject, ResponseFormat
|
|
23
|
+
|
|
24
|
+
_BATCH_TERMINAL = {"SUCCESS", "FAILED", "TIMEOUT_EXCEEDED", "CANCELLED"}
|
|
25
|
+
_BATCH_POLL_INTERVAL_S = 5
|
|
26
|
+
|
|
27
|
+
_log = logging.getLogger(__name__)
|
|
28
|
+
|
|
29
|
+
# JSON schema passed as `bbox_annotation_format` so each detected image
|
|
30
|
+
# is annotated by Mistral OCR in the same call. Kept intentionally small:
|
|
31
|
+
# a `title` slug for UI display and a free-form `description` for RAG —
|
|
32
|
+
# a tight schema avoids burning tokens on speculative fields.
|
|
33
|
+
_IMAGE_ANNOTATION_SCHEMA: dict = {
|
|
34
|
+
"type": "object",
|
|
35
|
+
"properties": {
|
|
36
|
+
"title": {
|
|
37
|
+
"type": "string",
|
|
38
|
+
"description": (
|
|
39
|
+
"Short kebab-case slug (3-8 words, lowercase, "
|
|
40
|
+
"hyphen-separated, no file extension) naming what the "
|
|
41
|
+
"image depicts, e.g. 'system-architecture-diagram' or "
|
|
42
|
+
"'monthly-revenue-bar-chart'."
|
|
43
|
+
),
|
|
44
|
+
},
|
|
45
|
+
"description": {
|
|
46
|
+
"type": "string",
|
|
47
|
+
"description": (
|
|
48
|
+
"Concise description of the image content, including any "
|
|
49
|
+
"text visible in the image and the relationship to "
|
|
50
|
+
"surrounding document context."
|
|
51
|
+
),
|
|
52
|
+
},
|
|
53
|
+
},
|
|
54
|
+
"required": ["title", "description"],
|
|
55
|
+
"additionalProperties": False,
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _annotation_schema(language: str | None) -> dict:
|
|
60
|
+
"""Image-annotation JSON schema, optionally fixing the output language.
|
|
61
|
+
|
|
62
|
+
When ``language`` is set, the ``description`` field instruction is
|
|
63
|
+
extended to demand prose in that language while leaving verbatim
|
|
64
|
+
transcribed text in its original language. Returns a fresh dict so the
|
|
65
|
+
module-level ``_IMAGE_ANNOTATION_SCHEMA`` is never mutated.
|
|
66
|
+
"""
|
|
67
|
+
if not language:
|
|
68
|
+
return _IMAGE_ANNOTATION_SCHEMA
|
|
69
|
+
schema = json.loads(json.dumps(_IMAGE_ANNOTATION_SCHEMA))
|
|
70
|
+
schema["properties"]["description"]["description"] += (
|
|
71
|
+
f" Write this description in {language}; keep any text "
|
|
72
|
+
"transcribed from the image in its original language."
|
|
73
|
+
)
|
|
74
|
+
return schema
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _sdk_supports(param: str) -> bool:
|
|
78
|
+
"""Whether the installed `mistralai` exposes `param` on `ocr.process`.
|
|
79
|
+
|
|
80
|
+
Block extraction (`include_blocks`) landed in mistralai 2.8; the
|
|
81
|
+
package floor here is `>=2.0`, so a client can legitimately be on a
|
|
82
|
+
release that predates it. Feature-detecting keeps that client working
|
|
83
|
+
— the corresponding enrichment is skipped rather than the call
|
|
84
|
+
failing — and avoids forcing a dependency bump on everyone.
|
|
85
|
+
|
|
86
|
+
Imported lazily and tolerant of any failure: an inspection problem
|
|
87
|
+
must degrade to "unsupported", never break parsing.
|
|
88
|
+
"""
|
|
89
|
+
try:
|
|
90
|
+
from mistralai.client.ocr import Ocr
|
|
91
|
+
|
|
92
|
+
return param in inspect.signature(Ocr.process).parameters
|
|
93
|
+
except Exception:
|
|
94
|
+
return False
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _caption_by_image_id(page: "OCRPageObject") -> dict[str, str]:
|
|
98
|
+
"""Map ``image_id`` -> caption text, from a page's block list.
|
|
99
|
+
|
|
100
|
+
Requires ``include_blocks``; returns empty when blocks are absent
|
|
101
|
+
(older SDK, older OCR model, or the option switched off).
|
|
102
|
+
|
|
103
|
+
Blocks arrive in reading order, so a figure's caption is the nearest
|
|
104
|
+
`caption` block to its `image` block. We prefer the one immediately
|
|
105
|
+
after (captions typically sit below the figure) and fall back to the
|
|
106
|
+
one immediately before (some layouts put them above). Neighbours are
|
|
107
|
+
only considered when directly adjacent, so an unrelated caption
|
|
108
|
+
further down the page is never attached to the wrong image.
|
|
109
|
+
"""
|
|
110
|
+
blocks = getattr(page, "blocks", None) or []
|
|
111
|
+
captions: dict[str, str] = {}
|
|
112
|
+
for i, block in enumerate(blocks):
|
|
113
|
+
if getattr(block, "type", None) != "image":
|
|
114
|
+
continue
|
|
115
|
+
image_id = getattr(block, "image_id", None)
|
|
116
|
+
if not image_id:
|
|
117
|
+
continue
|
|
118
|
+
for neighbour in (i + 1, i - 1):
|
|
119
|
+
if not 0 <= neighbour < len(blocks):
|
|
120
|
+
continue
|
|
121
|
+
candidate = blocks[neighbour]
|
|
122
|
+
if getattr(candidate, "type", None) == "caption":
|
|
123
|
+
text = (getattr(candidate, "content", "") or "").strip()
|
|
124
|
+
if text:
|
|
125
|
+
captions[image_id] = text
|
|
126
|
+
break
|
|
127
|
+
return captions
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _page_info(page: "OCRPageObject") -> PageInfo:
|
|
131
|
+
"""Collect the optional per-page extras OCR returned for one page.
|
|
132
|
+
|
|
133
|
+
Every field stays None when the matching option wasn't requested, so
|
|
134
|
+
this is safe to call unconditionally.
|
|
135
|
+
"""
|
|
136
|
+
scores = getattr(page, "confidence_scores", None)
|
|
137
|
+
return PageInfo(
|
|
138
|
+
page=page.index + 1,
|
|
139
|
+
header=(getattr(page, "header", None) or None),
|
|
140
|
+
footer=(getattr(page, "footer", None) or None),
|
|
141
|
+
average_confidence=getattr(scores, "average_page_confidence_score", None),
|
|
142
|
+
minimum_confidence=getattr(scores, "minimum_page_confidence_score", None),
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _build_bbox_annotation_format(language: str | None = None) -> "ResponseFormat":
|
|
147
|
+
"""Build the Mistral `bbox_annotation_format` for per-image descriptions.
|
|
148
|
+
|
|
149
|
+
Imported lazily so the module remains importable without `mistralai`
|
|
150
|
+
installed; callers reach this only via `Mistral.parse_pdf_with_images`.
|
|
151
|
+
"""
|
|
152
|
+
from mistralai.client.models import JSONSchema, ResponseFormat
|
|
153
|
+
|
|
154
|
+
return ResponseFormat(
|
|
155
|
+
type="json_schema",
|
|
156
|
+
json_schema=JSONSchema(
|
|
157
|
+
name="ImageAnnotation",
|
|
158
|
+
schema_definition=_annotation_schema(language),
|
|
159
|
+
strict=True,
|
|
160
|
+
),
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
class Mistral:
|
|
165
|
+
"""Mistral OCR backend (cloud).
|
|
166
|
+
|
|
167
|
+
Calls the Mistral OCR API. Robust on image-only and broken-encoding PDFs,
|
|
168
|
+
preserves LaTeX equations and structured tables in Markdown.
|
|
169
|
+
|
|
170
|
+
Requires a Mistral API key. Reads ``MISTRAL_API_KEY`` from environment
|
|
171
|
+
by default, or accepts it as the ``api_key`` argument.
|
|
172
|
+
|
|
173
|
+
Single-PDF: ``parse_pdf(source)`` calls the sync OCR endpoint
|
|
174
|
+
(~$4/1k pages).
|
|
175
|
+
|
|
176
|
+
Multi-PDF: ``parse_pdfs(sources)`` uses the Batch API in a single
|
|
177
|
+
optimized job (50%% off, ~$2/1k pages). Blocks until the batch completes
|
|
178
|
+
(typically tens of seconds for small batches; can be longer under load).
|
|
179
|
+
Failed items are returned as ``None`` in input order.
|
|
180
|
+
|
|
181
|
+
Three optional enrichments are available on the ``*_with_images``
|
|
182
|
+
methods, all off by default so the returned markdown is unchanged
|
|
183
|
+
unless asked for:
|
|
184
|
+
|
|
185
|
+
``extract_headers_footers``
|
|
186
|
+
Pull running page furniture out of the page markdown and into
|
|
187
|
+
``ParseResult.pages[i].header`` / ``.footer``. Useful before
|
|
188
|
+
chunking for retrieval, where a repeated "page 4 of 27" is noise.
|
|
189
|
+
Note this *removes* that text from the markdown.
|
|
190
|
+
|
|
191
|
+
``confidence_scores``
|
|
192
|
+
``"page"`` or ``"word"``. Populates the confidence fields on
|
|
193
|
+
``ParseResult.pages``. Thresholds need calibrating per corpus —
|
|
194
|
+
on the benchmark PDFs these scores did not separate clean
|
|
195
|
+
documents from degraded ones (see ``PageInfo``).
|
|
196
|
+
|
|
197
|
+
``extract_captions``
|
|
198
|
+
Fill ``ExtractedImage.original_caption`` with the figure caption
|
|
199
|
+
printed next to each image. Requires OCR 4+ and ``mistralai>=2.8``;
|
|
200
|
+
on older SDKs it logs a warning and leaves captions None.
|
|
201
|
+
|
|
202
|
+
Install with::
|
|
203
|
+
|
|
204
|
+
pip install dot-parser[mistral]
|
|
205
|
+
"""
|
|
206
|
+
|
|
207
|
+
# Class-level defaults so an instance built without ``__init__`` still
|
|
208
|
+
# behaves: ``Mistral.__new__(Mistral)`` (used to skip the API-key check)
|
|
209
|
+
# and instances unpickled from a version predating these options would
|
|
210
|
+
# otherwise raise AttributeError mid-parse.
|
|
211
|
+
_language: str | None = None
|
|
212
|
+
_extract_headers_footers: bool = False
|
|
213
|
+
_confidence_scores: "Literal['page', 'word'] | None" = None
|
|
214
|
+
_extract_captions: bool = False
|
|
215
|
+
|
|
216
|
+
def __init__(
|
|
217
|
+
self,
|
|
218
|
+
*,
|
|
219
|
+
api_key: str | None = None,
|
|
220
|
+
model: str = "mistral-ocr-4-0",
|
|
221
|
+
language: str | None = None,
|
|
222
|
+
extract_headers_footers: bool = False,
|
|
223
|
+
confidence_scores: Literal["page", "word"] | None = None,
|
|
224
|
+
extract_captions: bool = False,
|
|
225
|
+
) -> None:
|
|
226
|
+
try:
|
|
227
|
+
from mistralai.client import Mistral as _Mistral
|
|
228
|
+
except ImportError as e:
|
|
229
|
+
raise ImportError(
|
|
230
|
+
"Mistral backend requires `mistralai`. "
|
|
231
|
+
"Install with: pip install dot-parser[mistral]"
|
|
232
|
+
) from e
|
|
233
|
+
|
|
234
|
+
key = api_key or os.environ.get("MISTRAL_API_KEY")
|
|
235
|
+
if not key:
|
|
236
|
+
raise ValueError(
|
|
237
|
+
"Mistral backend requires an API key. Set MISTRAL_API_KEY or pass api_key=..."
|
|
238
|
+
)
|
|
239
|
+
self._client = _Mistral(api_key=key)
|
|
240
|
+
self._model = model
|
|
241
|
+
# When set (e.g. "French"), per-image annotations are requested in
|
|
242
|
+
# this language via the bbox_annotation schema; None keeps Mistral's
|
|
243
|
+
# default (English-leaning) behaviour.
|
|
244
|
+
self._language = language
|
|
245
|
+
# The three options below all default to today's behaviour: each
|
|
246
|
+
# one adds a field to the OCR request, so leaving them off keeps
|
|
247
|
+
# the request — and therefore the response — byte-identical to
|
|
248
|
+
# what callers got before they existed.
|
|
249
|
+
self._extract_headers_footers = extract_headers_footers
|
|
250
|
+
self._confidence_scores = confidence_scores
|
|
251
|
+
# Captions need block extraction, which requires both OCR 4+ and
|
|
252
|
+
# mistralai>=2.8. Resolved once here so an old SDK degrades to
|
|
253
|
+
# "no captions" instead of erroring on every call.
|
|
254
|
+
self._extract_captions = extract_captions and _sdk_supports("include_blocks")
|
|
255
|
+
if extract_captions and not self._extract_captions:
|
|
256
|
+
_log.warning(
|
|
257
|
+
"extract_captions=True ignored: the installed `mistralai` has no "
|
|
258
|
+
"include_blocks support on ocr.process (needs >=2.8). Image "
|
|
259
|
+
"captions will stay None."
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
def parse_pdf(self, source: str | Path | bytes) -> str:
|
|
263
|
+
if isinstance(source, bytes):
|
|
264
|
+
content = source
|
|
265
|
+
file_name = "document.pdf"
|
|
266
|
+
else:
|
|
267
|
+
path = Path(source)
|
|
268
|
+
content = path.read_bytes()
|
|
269
|
+
file_name = path.name
|
|
270
|
+
|
|
271
|
+
uploaded = self._client.files.upload(
|
|
272
|
+
file={"file_name": file_name, "content": content},
|
|
273
|
+
purpose="ocr",
|
|
274
|
+
)
|
|
275
|
+
signed = self._client.files.get_signed_url(file_id=uploaded.id)
|
|
276
|
+
resp = self._client.ocr.process(
|
|
277
|
+
model=self._model,
|
|
278
|
+
document={"type": "document_url", "document_url": signed.url},
|
|
279
|
+
)
|
|
280
|
+
return "\n\n".join(p.markdown for p in resp.pages)
|
|
281
|
+
|
|
282
|
+
def parse_pdf_with_images(
|
|
283
|
+
self,
|
|
284
|
+
source: str | Path | bytes,
|
|
285
|
+
*,
|
|
286
|
+
annotate_images: bool = True,
|
|
287
|
+
include_tables: bool = True,
|
|
288
|
+
) -> ParseResult:
|
|
289
|
+
"""Parse a PDF and return markdown together with extracted images.
|
|
290
|
+
|
|
291
|
+
Each image is requested with `include_image_base64=True` and, when
|
|
292
|
+
`annotate_images` is True, with a structured `bbox_annotation_format`
|
|
293
|
+
so Mistral writes a short description per image in the same OCR call
|
|
294
|
+
(no separate VLM round-trip). If that annotated call fails the
|
|
295
|
+
method automatically retries once without the annotation schema,
|
|
296
|
+
salvaging the markdown + raw images (interpretations stay None).
|
|
297
|
+
|
|
298
|
+
The returned markdown preserves Mistral's native ``
|
|
299
|
+
placeholders at image positions; `ExtractedImage.name` matches
|
|
300
|
+
the `id` in those placeholders so callers can correlate text and
|
|
301
|
+
image without re-parsing the markdown.
|
|
302
|
+
|
|
303
|
+
When ``include_tables`` is False, GFM-style tables are stripped
|
|
304
|
+
from the resulting markdown — Mistral OCR has no native flag to
|
|
305
|
+
suppress them, so this is a post-processing pass.
|
|
306
|
+
"""
|
|
307
|
+
content, file_name = self._read_source(source, default_name="document.pdf")
|
|
308
|
+
return self._ocr_document_with_images(
|
|
309
|
+
content,
|
|
310
|
+
file_name,
|
|
311
|
+
annotate_images=annotate_images,
|
|
312
|
+
include_tables=include_tables,
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
def parse_docx_with_images(
|
|
316
|
+
self,
|
|
317
|
+
source: str | Path | bytes,
|
|
318
|
+
*,
|
|
319
|
+
annotate_images: bool = True,
|
|
320
|
+
include_tables: bool = True,
|
|
321
|
+
) -> ParseResult:
|
|
322
|
+
"""Parse a DOCX directly through Mistral OCR (no markitdown / VLM).
|
|
323
|
+
|
|
324
|
+
Mirrors :meth:`parse_pdf_with_images` — the OCR API accepts ``.docx``
|
|
325
|
+
natively via ``document_url`` (Mistral rasters the document to pages
|
|
326
|
+
before OCR). The returned shape is identical to the PDF path: markdown
|
|
327
|
+
with ```` placeholders at image positions, plus one
|
|
328
|
+
:class:`ExtractedImage` per detected image with optional per-image
|
|
329
|
+
annotation when ``annotate_images=True``.
|
|
330
|
+
|
|
331
|
+
Exists as an A/B alternative to
|
|
332
|
+
:func:`dot_parser.docx_images.parse_docx_with_images` (markitdown +
|
|
333
|
+
VLM). Both paths are selectable from :func:`parse_with_images` —
|
|
334
|
+
pass ``backend=Mistral()`` for this OCR path, or pass ``vlm=...`` for
|
|
335
|
+
the markitdown path.
|
|
336
|
+
"""
|
|
337
|
+
content, file_name = self._read_source(source, default_name="document.docx")
|
|
338
|
+
return self._ocr_document_with_images(
|
|
339
|
+
content,
|
|
340
|
+
file_name,
|
|
341
|
+
annotate_images=annotate_images,
|
|
342
|
+
include_tables=include_tables,
|
|
343
|
+
)
|
|
344
|
+
|
|
345
|
+
def parse_pptx_with_images(
|
|
346
|
+
self,
|
|
347
|
+
source: str | Path | bytes,
|
|
348
|
+
*,
|
|
349
|
+
annotate_images: bool = True,
|
|
350
|
+
include_tables: bool = True,
|
|
351
|
+
) -> ParseResult:
|
|
352
|
+
"""Parse a PPTX directly through Mistral OCR (no markitdown / VLM).
|
|
353
|
+
|
|
354
|
+
Same pipeline as :meth:`parse_docx_with_images`. Because the OCR
|
|
355
|
+
API rasters the deck page by page, one OCR page = one slide:
|
|
356
|
+
``ExtractedImage.page`` is the slide number, and the per-page
|
|
357
|
+
markdowns are joined with ``<!-- Slide number: N -->`` markers —
|
|
358
|
+
the same markers markitdown emits for PPTX — so slide-aware
|
|
359
|
+
chunkers see an identical surface on both paths.
|
|
360
|
+
|
|
361
|
+
Unlike the markitdown+VLM path, rasterising the slide means
|
|
362
|
+
diagrams *drawn* in PowerPoint (shapes, arrows, SmartArt) are
|
|
363
|
+
seen and annotated, not just embedded picture files.
|
|
364
|
+
"""
|
|
365
|
+
content, file_name = self._read_source(source, default_name="document.pptx")
|
|
366
|
+
return self._ocr_document_with_images(
|
|
367
|
+
content,
|
|
368
|
+
file_name,
|
|
369
|
+
annotate_images=annotate_images,
|
|
370
|
+
include_tables=include_tables,
|
|
371
|
+
page_marker_template="<!-- Slide number: {n} -->",
|
|
372
|
+
)
|
|
373
|
+
|
|
374
|
+
@staticmethod
|
|
375
|
+
def _read_source(source: str | Path | bytes, *, default_name: str) -> tuple[bytes, str]:
|
|
376
|
+
if isinstance(source, bytes):
|
|
377
|
+
return source, default_name
|
|
378
|
+
path = Path(source)
|
|
379
|
+
return path.read_bytes(), path.name
|
|
380
|
+
|
|
381
|
+
def _ocr_document_with_images(
|
|
382
|
+
self,
|
|
383
|
+
content: bytes,
|
|
384
|
+
file_name: str,
|
|
385
|
+
*,
|
|
386
|
+
annotate_images: bool,
|
|
387
|
+
include_tables: bool,
|
|
388
|
+
page_marker_template: str | None = None,
|
|
389
|
+
) -> ParseResult:
|
|
390
|
+
"""Shared OCR-with-images pipeline for any document_url-accepting input.
|
|
391
|
+
|
|
392
|
+
Mistral OCR's behavior is the same whether the source is PDF, DOCX,
|
|
393
|
+
PPTX, etc.: it rasters the document to pages, then runs text + image
|
|
394
|
+
+ (optional) annotation extraction. This helper centralises the
|
|
395
|
+
upload → process → assemble flow so the per-format entry points
|
|
396
|
+
differ only in default filename and, optionally, a
|
|
397
|
+
``page_marker_template`` (containing ``{n}``) emitted before each
|
|
398
|
+
page's markdown — used by the PPTX path to mark slide boundaries.
|
|
399
|
+
"""
|
|
400
|
+
uploaded = self._client.files.upload(
|
|
401
|
+
file={"file_name": file_name, "content": content},
|
|
402
|
+
purpose="ocr",
|
|
403
|
+
)
|
|
404
|
+
signed = self._client.files.get_signed_url(file_id=uploaded.id)
|
|
405
|
+
|
|
406
|
+
base_kwargs: dict = {
|
|
407
|
+
"model": self._model,
|
|
408
|
+
"document": {"type": "document_url", "document_url": signed.url},
|
|
409
|
+
"include_image_base64": True,
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
# Everything beyond markdown + raw images is opt-in and grouped
|
|
413
|
+
# here, because they share one failure policy: any of them can be
|
|
414
|
+
# rejected by an older OCR model, and none is worth losing the
|
|
415
|
+
# document over.
|
|
416
|
+
optional_kwargs: dict = {}
|
|
417
|
+
if annotate_images:
|
|
418
|
+
optional_kwargs["bbox_annotation_format"] = _build_bbox_annotation_format(
|
|
419
|
+
self._language
|
|
420
|
+
)
|
|
421
|
+
if self._extract_headers_footers:
|
|
422
|
+
optional_kwargs["extract_header"] = True
|
|
423
|
+
optional_kwargs["extract_footer"] = True
|
|
424
|
+
if self._confidence_scores:
|
|
425
|
+
optional_kwargs["confidence_scores_granularity"] = self._confidence_scores
|
|
426
|
+
if self._extract_captions:
|
|
427
|
+
optional_kwargs["include_blocks"] = True
|
|
428
|
+
|
|
429
|
+
if optional_kwargs:
|
|
430
|
+
try:
|
|
431
|
+
resp = self._client.ocr.process(**base_kwargs, **optional_kwargs)
|
|
432
|
+
except Exception as exc:
|
|
433
|
+
# Retry once with the extras dropped so one unsupported or
|
|
434
|
+
# flaky enrichment doesn't lose the whole document. The
|
|
435
|
+
# caller still gets markdown + raw images, minus the
|
|
436
|
+
# annotations / headers / confidence / captions.
|
|
437
|
+
_log.warning(
|
|
438
|
+
"Mistral OCR failed with optional features %s (%s); retrying without them",
|
|
439
|
+
sorted(optional_kwargs),
|
|
440
|
+
exc,
|
|
441
|
+
)
|
|
442
|
+
resp = self._client.ocr.process(**base_kwargs)
|
|
443
|
+
else:
|
|
444
|
+
resp = self._client.ocr.process(**base_kwargs)
|
|
445
|
+
|
|
446
|
+
if page_marker_template is None:
|
|
447
|
+
markdown = "\n\n".join(p.markdown for p in resp.pages)
|
|
448
|
+
else:
|
|
449
|
+
markdown = "\n\n".join(
|
|
450
|
+
f"{page_marker_template.format(n=p.index + 1)}\n{p.markdown}" for p in resp.pages
|
|
451
|
+
)
|
|
452
|
+
if not include_tables:
|
|
453
|
+
markdown = strip_markdown_tables(markdown)
|
|
454
|
+
|
|
455
|
+
# Two-pass: gather raw fields, then dedupe titles document-wide so
|
|
456
|
+
# repeated diagrams ("workflow-diagram" twice) get -2/-3 suffixes.
|
|
457
|
+
raw_records: list[tuple[ExtractedImage, str | None]] = []
|
|
458
|
+
pages: list[PageInfo] = []
|
|
459
|
+
for page in resp.pages:
|
|
460
|
+
if self._extract_headers_footers or self._confidence_scores:
|
|
461
|
+
pages.append(_page_info(page))
|
|
462
|
+
captions = _caption_by_image_id(page) if self._extract_captions else {}
|
|
463
|
+
for img in page.images or []:
|
|
464
|
+
if not img.image_base64:
|
|
465
|
+
continue
|
|
466
|
+
annotation = parse_image_annotation(img.image_annotation)
|
|
467
|
+
raw_records.append(
|
|
468
|
+
(
|
|
469
|
+
ExtractedImage(
|
|
470
|
+
name=img.id,
|
|
471
|
+
base64=strip_data_url_prefix(img.image_base64),
|
|
472
|
+
mime_type=guess_mime_type(img.id),
|
|
473
|
+
page=page.index + 1,
|
|
474
|
+
original_caption=captions.get(img.id),
|
|
475
|
+
interpretation=annotation.interpretation or None,
|
|
476
|
+
title=None,
|
|
477
|
+
),
|
|
478
|
+
annotation.title,
|
|
479
|
+
)
|
|
480
|
+
)
|
|
481
|
+
|
|
482
|
+
final_titles = dedupe_titles([t for _, t in raw_records])
|
|
483
|
+
images: list[ExtractedImage] = []
|
|
484
|
+
for (record, _), title in zip(raw_records, final_titles, strict=True):
|
|
485
|
+
images.append(
|
|
486
|
+
ExtractedImage(
|
|
487
|
+
name=record.name,
|
|
488
|
+
base64=record.base64,
|
|
489
|
+
mime_type=record.mime_type,
|
|
490
|
+
page=record.page,
|
|
491
|
+
original_caption=record.original_caption,
|
|
492
|
+
interpretation=record.interpretation,
|
|
493
|
+
title=title,
|
|
494
|
+
)
|
|
495
|
+
)
|
|
496
|
+
return ParseResult(markdown=markdown, images=images, pages=pages)
|
|
497
|
+
|
|
498
|
+
def _log_batch_error_file(self, job: "BatchJob") -> None:
|
|
499
|
+
"""Log the batch job's ``error_file`` (Mistral's per-request failure
|
|
500
|
+
details), if it produced one. Best-effort: never raises into the caller,
|
|
501
|
+
and caps the dump so a large batch can't flood the logs."""
|
|
502
|
+
err_file = getattr(job, "error_file", None)
|
|
503
|
+
if not err_file:
|
|
504
|
+
return
|
|
505
|
+
try:
|
|
506
|
+
content = self._client.files.download(file_id=err_file).read().decode()
|
|
507
|
+
except Exception as exc:
|
|
508
|
+
_log.warning("could not download Mistral batch error_file %s: %s", err_file, exc)
|
|
509
|
+
return
|
|
510
|
+
for line in content.strip().splitlines()[:20]:
|
|
511
|
+
_log.error("Mistral batch error_file: %s", line)
|
|
512
|
+
|
|
513
|
+
def parse_pdfs(self, sources: list[str | Path | bytes]) -> list[str | None]:
|
|
514
|
+
"""Parse multiple PDFs in a single Batch API job (50% off vs sync).
|
|
515
|
+
|
|
516
|
+
Returns markdowns in input order. Failed items are ``None``.
|
|
517
|
+
"""
|
|
518
|
+
if not sources:
|
|
519
|
+
return []
|
|
520
|
+
|
|
521
|
+
# 1. Upload each PDF + get signed URL.
|
|
522
|
+
signed_urls: list[str] = []
|
|
523
|
+
for i, src in enumerate(sources):
|
|
524
|
+
if isinstance(src, bytes):
|
|
525
|
+
content = src
|
|
526
|
+
file_name = f"document_{i}.pdf"
|
|
527
|
+
else:
|
|
528
|
+
path = Path(src)
|
|
529
|
+
content = path.read_bytes()
|
|
530
|
+
file_name = path.name
|
|
531
|
+
uploaded = self._client.files.upload(
|
|
532
|
+
file={"file_name": file_name, "content": content},
|
|
533
|
+
purpose="ocr",
|
|
534
|
+
)
|
|
535
|
+
signed = self._client.files.get_signed_url(file_id=uploaded.id)
|
|
536
|
+
signed_urls.append(signed.url)
|
|
537
|
+
|
|
538
|
+
# 2. Build JSONL with one request per PDF, indexed by position.
|
|
539
|
+
lines = []
|
|
540
|
+
for i, url in enumerate(signed_urls):
|
|
541
|
+
lines.append(
|
|
542
|
+
json.dumps(
|
|
543
|
+
{
|
|
544
|
+
"custom_id": str(i),
|
|
545
|
+
"body": {
|
|
546
|
+
"model": self._model,
|
|
547
|
+
"document": {"type": "document_url", "document_url": url},
|
|
548
|
+
},
|
|
549
|
+
}
|
|
550
|
+
)
|
|
551
|
+
)
|
|
552
|
+
jsonl_bytes = ("\n".join(lines) + "\n").encode()
|
|
553
|
+
|
|
554
|
+
# 3. Upload JSONL + create batch job.
|
|
555
|
+
batch_file = self._client.files.upload(
|
|
556
|
+
file={"file_name": "batch_ocr.jsonl", "content": jsonl_bytes},
|
|
557
|
+
purpose="batch",
|
|
558
|
+
)
|
|
559
|
+
job = self._client.batch.jobs.create(
|
|
560
|
+
endpoint="/v1/ocr",
|
|
561
|
+
input_files=[batch_file.id],
|
|
562
|
+
model=self._model,
|
|
563
|
+
)
|
|
564
|
+
|
|
565
|
+
# 4. Poll until terminal.
|
|
566
|
+
while job.status not in _BATCH_TERMINAL:
|
|
567
|
+
time.sleep(_BATCH_POLL_INTERVAL_S)
|
|
568
|
+
job = self._client.batch.jobs.get(job_id=job.id)
|
|
569
|
+
|
|
570
|
+
results: list[str | None] = [None] * len(sources)
|
|
571
|
+
# A job yields no usable output if it didn't succeed, or if it "succeeded"
|
|
572
|
+
# but has no output_file (every request errored — the details are then in
|
|
573
|
+
# the error_file). Guard the download: calling it with output_file=None
|
|
574
|
+
# raises a cryptic SDK ValidationError that hides the real reason.
|
|
575
|
+
output_file = getattr(job, "output_file", None)
|
|
576
|
+
if job.status != "SUCCESS" or not output_file:
|
|
577
|
+
_log.error(
|
|
578
|
+
"Mistral batch job %s ended in %s with no usable output "
|
|
579
|
+
"(output_file=%s); all %d PDF(s) failed. job.errors=%s",
|
|
580
|
+
job.id,
|
|
581
|
+
job.status,
|
|
582
|
+
output_file,
|
|
583
|
+
len(sources),
|
|
584
|
+
getattr(job, "errors", None),
|
|
585
|
+
)
|
|
586
|
+
self._log_batch_error_file(job)
|
|
587
|
+
return results
|
|
588
|
+
|
|
589
|
+
# 5. Download output JSONL + map back to input order via custom_id.
|
|
590
|
+
output_bytes = self._client.files.download(file_id=output_file).read()
|
|
591
|
+
for line in output_bytes.decode().strip().splitlines():
|
|
592
|
+
rec = json.loads(line)
|
|
593
|
+
try:
|
|
594
|
+
idx = int(rec["custom_id"])
|
|
595
|
+
except (KeyError, ValueError):
|
|
596
|
+
_log.warning("Mistral batch output line has no valid custom_id: %.200s", line)
|
|
597
|
+
continue
|
|
598
|
+
if not (0 <= idx < len(sources)):
|
|
599
|
+
continue
|
|
600
|
+
# A per-item OCR failure comes back as an error on the record rather
|
|
601
|
+
# than pages; surface it instead of silently leaving a None.
|
|
602
|
+
error = rec.get("error") or (rec.get("response") or {}).get("body", {}).get("error")
|
|
603
|
+
if error is not None:
|
|
604
|
+
_log.error("Mistral OCR failed for PDF #%d: %s", idx, error)
|
|
605
|
+
continue
|
|
606
|
+
try:
|
|
607
|
+
pages = rec["response"]["body"]["pages"]
|
|
608
|
+
results[idx] = "\n\n".join(p["markdown"] for p in pages)
|
|
609
|
+
except (KeyError, TypeError):
|
|
610
|
+
_log.error("Mistral OCR returned no pages for PDF #%d: %.300s", idx, line)
|
|
611
|
+
|
|
612
|
+
missing = [i for i, md in enumerate(results) if md is None]
|
|
613
|
+
if missing:
|
|
614
|
+
self._log_batch_error_file(job)
|
|
615
|
+
_log.warning(
|
|
616
|
+
"Mistral batch %s: %d/%d PDF(s) produced no markdown (indices %s)",
|
|
617
|
+
job.id,
|
|
618
|
+
len(missing),
|
|
619
|
+
len(sources),
|
|
620
|
+
missing,
|
|
621
|
+
)
|
|
622
|
+
return results
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
import tempfile
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from types import ModuleType
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class Pymu:
|
|
10
|
+
"""Default backend: pymupdf4llm.
|
|
11
|
+
|
|
12
|
+
Fast, CPU-only, lightweight. Works on natively-text PDFs. Fails on
|
|
13
|
+
image-only PDFs and PDFs with broken ToUnicode CMaps unless an OCR
|
|
14
|
+
engine (tesseract, rapidocr or paddleocr) is available — pymupdf4llm
|
|
15
|
+
auto-detects and falls back to OCR for problematic pages.
|
|
16
|
+
|
|
17
|
+
Args:
|
|
18
|
+
ocr_fallback: when True (default), pymupdf4llm uses its built-in
|
|
19
|
+
OCR fallback if an engine is available. When False, force the
|
|
20
|
+
no-OCR code path (matches a runtime image without tesseract,
|
|
21
|
+
useful for reproducing production where OCR is not installed).
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
def __init__(self, *, ocr_fallback: bool = True) -> None:
|
|
25
|
+
self._ocr_fallback = ocr_fallback
|
|
26
|
+
|
|
27
|
+
def parse_pdf(self, source: str | Path | bytes) -> str:
|
|
28
|
+
import pymupdf4llm
|
|
29
|
+
|
|
30
|
+
if self._ocr_fallback:
|
|
31
|
+
return self._call(pymupdf4llm, source)
|
|
32
|
+
|
|
33
|
+
# Reproduce a runtime without an OCR engine available (le-lab
|
|
34
|
+
# production). pymupdf4llm gates OCR on the result of
|
|
35
|
+
# ``select_ocr_function()`` — when it returns None, ``document.use_ocr``
|
|
36
|
+
# is forced to ``OCRMode.NEVER`` and neither full-page nor text-only
|
|
37
|
+
# OCR is invoked.
|
|
38
|
+
from pymupdf4llm.helpers import document_layout as _dl
|
|
39
|
+
|
|
40
|
+
original = _dl.select_ocr_function
|
|
41
|
+
# Deliberate monkeypatch of a third-party module function; returning
|
|
42
|
+
# None is precisely the signal pymupdf4llm reads to disable OCR.
|
|
43
|
+
_dl.select_ocr_function = lambda: None # ty: ignore[invalid-assignment]
|
|
44
|
+
try:
|
|
45
|
+
return self._call(pymupdf4llm, source)
|
|
46
|
+
finally:
|
|
47
|
+
_dl.select_ocr_function = original
|
|
48
|
+
|
|
49
|
+
@staticmethod
|
|
50
|
+
def _call(pymupdf4llm: ModuleType, source: str | Path | bytes) -> str:
|
|
51
|
+
if isinstance(source, bytes):
|
|
52
|
+
with tempfile.NamedTemporaryFile(suffix=".pdf", delete=True) as tmp:
|
|
53
|
+
tmp.write(source)
|
|
54
|
+
tmp.flush()
|
|
55
|
+
return pymupdf4llm.to_markdown(tmp.name)
|
|
56
|
+
return pymupdf4llm.to_markdown(str(source))
|