dot-parser 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,622 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ import inspect
5
+ import json
6
+ import logging
7
+ import os
8
+ import time
9
+ from pathlib import Path
10
+ from typing import TYPE_CHECKING, Literal
11
+
12
+ from dot_parser.image_utils import (
13
+ dedupe_titles,
14
+ guess_mime_type,
15
+ parse_image_annotation,
16
+ strip_data_url_prefix,
17
+ )
18
+ from dot_parser.images import ExtractedImage, PageInfo, ParseResult
19
+ from dot_parser.markdown_utils import strip_markdown_tables
20
+
21
+ if TYPE_CHECKING:
22
+ from mistralai.client.models import BatchJob, OCRPageObject, ResponseFormat
23
+
24
+ _BATCH_TERMINAL = {"SUCCESS", "FAILED", "TIMEOUT_EXCEEDED", "CANCELLED"}
25
+ _BATCH_POLL_INTERVAL_S = 5
26
+
27
+ _log = logging.getLogger(__name__)
28
+
29
+ # JSON schema passed as `bbox_annotation_format` so each detected image
30
+ # is annotated by Mistral OCR in the same call. Kept intentionally small:
31
+ # a `title` slug for UI display and a free-form `description` for RAG —
32
+ # a tight schema avoids burning tokens on speculative fields.
33
+ _IMAGE_ANNOTATION_SCHEMA: dict = {
34
+ "type": "object",
35
+ "properties": {
36
+ "title": {
37
+ "type": "string",
38
+ "description": (
39
+ "Short kebab-case slug (3-8 words, lowercase, "
40
+ "hyphen-separated, no file extension) naming what the "
41
+ "image depicts, e.g. 'system-architecture-diagram' or "
42
+ "'monthly-revenue-bar-chart'."
43
+ ),
44
+ },
45
+ "description": {
46
+ "type": "string",
47
+ "description": (
48
+ "Concise description of the image content, including any "
49
+ "text visible in the image and the relationship to "
50
+ "surrounding document context."
51
+ ),
52
+ },
53
+ },
54
+ "required": ["title", "description"],
55
+ "additionalProperties": False,
56
+ }
57
+
58
+
59
+ def _annotation_schema(language: str | None) -> dict:
60
+ """Image-annotation JSON schema, optionally fixing the output language.
61
+
62
+ When ``language`` is set, the ``description`` field instruction is
63
+ extended to demand prose in that language while leaving verbatim
64
+ transcribed text in its original language. Returns a fresh dict so the
65
+ module-level ``_IMAGE_ANNOTATION_SCHEMA`` is never mutated.
66
+ """
67
+ if not language:
68
+ return _IMAGE_ANNOTATION_SCHEMA
69
+ schema = json.loads(json.dumps(_IMAGE_ANNOTATION_SCHEMA))
70
+ schema["properties"]["description"]["description"] += (
71
+ f" Write this description in {language}; keep any text "
72
+ "transcribed from the image in its original language."
73
+ )
74
+ return schema
75
+
76
+
77
+ def _sdk_supports(param: str) -> bool:
78
+ """Whether the installed `mistralai` exposes `param` on `ocr.process`.
79
+
80
+ Block extraction (`include_blocks`) landed in mistralai 2.8; the
81
+ package floor here is `>=2.0`, so a client can legitimately be on a
82
+ release that predates it. Feature-detecting keeps that client working
83
+ — the corresponding enrichment is skipped rather than the call
84
+ failing — and avoids forcing a dependency bump on everyone.
85
+
86
+ Imported lazily and tolerant of any failure: an inspection problem
87
+ must degrade to "unsupported", never break parsing.
88
+ """
89
+ try:
90
+ from mistralai.client.ocr import Ocr
91
+
92
+ return param in inspect.signature(Ocr.process).parameters
93
+ except Exception:
94
+ return False
95
+
96
+
97
+ def _caption_by_image_id(page: "OCRPageObject") -> dict[str, str]:
98
+ """Map ``image_id`` -> caption text, from a page's block list.
99
+
100
+ Requires ``include_blocks``; returns empty when blocks are absent
101
+ (older SDK, older OCR model, or the option switched off).
102
+
103
+ Blocks arrive in reading order, so a figure's caption is the nearest
104
+ `caption` block to its `image` block. We prefer the one immediately
105
+ after (captions typically sit below the figure) and fall back to the
106
+ one immediately before (some layouts put them above). Neighbours are
107
+ only considered when directly adjacent, so an unrelated caption
108
+ further down the page is never attached to the wrong image.
109
+ """
110
+ blocks = getattr(page, "blocks", None) or []
111
+ captions: dict[str, str] = {}
112
+ for i, block in enumerate(blocks):
113
+ if getattr(block, "type", None) != "image":
114
+ continue
115
+ image_id = getattr(block, "image_id", None)
116
+ if not image_id:
117
+ continue
118
+ for neighbour in (i + 1, i - 1):
119
+ if not 0 <= neighbour < len(blocks):
120
+ continue
121
+ candidate = blocks[neighbour]
122
+ if getattr(candidate, "type", None) == "caption":
123
+ text = (getattr(candidate, "content", "") or "").strip()
124
+ if text:
125
+ captions[image_id] = text
126
+ break
127
+ return captions
128
+
129
+
130
+ def _page_info(page: "OCRPageObject") -> PageInfo:
131
+ """Collect the optional per-page extras OCR returned for one page.
132
+
133
+ Every field stays None when the matching option wasn't requested, so
134
+ this is safe to call unconditionally.
135
+ """
136
+ scores = getattr(page, "confidence_scores", None)
137
+ return PageInfo(
138
+ page=page.index + 1,
139
+ header=(getattr(page, "header", None) or None),
140
+ footer=(getattr(page, "footer", None) or None),
141
+ average_confidence=getattr(scores, "average_page_confidence_score", None),
142
+ minimum_confidence=getattr(scores, "minimum_page_confidence_score", None),
143
+ )
144
+
145
+
146
+ def _build_bbox_annotation_format(language: str | None = None) -> "ResponseFormat":
147
+ """Build the Mistral `bbox_annotation_format` for per-image descriptions.
148
+
149
+ Imported lazily so the module remains importable without `mistralai`
150
+ installed; callers reach this only via `Mistral.parse_pdf_with_images`.
151
+ """
152
+ from mistralai.client.models import JSONSchema, ResponseFormat
153
+
154
+ return ResponseFormat(
155
+ type="json_schema",
156
+ json_schema=JSONSchema(
157
+ name="ImageAnnotation",
158
+ schema_definition=_annotation_schema(language),
159
+ strict=True,
160
+ ),
161
+ )
162
+
163
+
164
+ class Mistral:
165
+ """Mistral OCR backend (cloud).
166
+
167
+ Calls the Mistral OCR API. Robust on image-only and broken-encoding PDFs,
168
+ preserves LaTeX equations and structured tables in Markdown.
169
+
170
+ Requires a Mistral API key. Reads ``MISTRAL_API_KEY`` from environment
171
+ by default, or accepts it as the ``api_key`` argument.
172
+
173
+ Single-PDF: ``parse_pdf(source)`` calls the sync OCR endpoint
174
+ (~$4/1k pages).
175
+
176
+ Multi-PDF: ``parse_pdfs(sources)`` uses the Batch API in a single
177
+ optimized job (50%% off, ~$2/1k pages). Blocks until the batch completes
178
+ (typically tens of seconds for small batches; can be longer under load).
179
+ Failed items are returned as ``None`` in input order.
180
+
181
+ Three optional enrichments are available on the ``*_with_images``
182
+ methods, all off by default so the returned markdown is unchanged
183
+ unless asked for:
184
+
185
+ ``extract_headers_footers``
186
+ Pull running page furniture out of the page markdown and into
187
+ ``ParseResult.pages[i].header`` / ``.footer``. Useful before
188
+ chunking for retrieval, where a repeated "page 4 of 27" is noise.
189
+ Note this *removes* that text from the markdown.
190
+
191
+ ``confidence_scores``
192
+ ``"page"`` or ``"word"``. Populates the confidence fields on
193
+ ``ParseResult.pages``. Thresholds need calibrating per corpus —
194
+ on the benchmark PDFs these scores did not separate clean
195
+ documents from degraded ones (see ``PageInfo``).
196
+
197
+ ``extract_captions``
198
+ Fill ``ExtractedImage.original_caption`` with the figure caption
199
+ printed next to each image. Requires OCR 4+ and ``mistralai>=2.8``;
200
+ on older SDKs it logs a warning and leaves captions None.
201
+
202
+ Install with::
203
+
204
+ pip install dot-parser[mistral]
205
+ """
206
+
207
+ # Class-level defaults so an instance built without ``__init__`` still
208
+ # behaves: ``Mistral.__new__(Mistral)`` (used to skip the API-key check)
209
+ # and instances unpickled from a version predating these options would
210
+ # otherwise raise AttributeError mid-parse.
211
+ _language: str | None = None
212
+ _extract_headers_footers: bool = False
213
+ _confidence_scores: "Literal['page', 'word'] | None" = None
214
+ _extract_captions: bool = False
215
+
216
+ def __init__(
217
+ self,
218
+ *,
219
+ api_key: str | None = None,
220
+ model: str = "mistral-ocr-4-0",
221
+ language: str | None = None,
222
+ extract_headers_footers: bool = False,
223
+ confidence_scores: Literal["page", "word"] | None = None,
224
+ extract_captions: bool = False,
225
+ ) -> None:
226
+ try:
227
+ from mistralai.client import Mistral as _Mistral
228
+ except ImportError as e:
229
+ raise ImportError(
230
+ "Mistral backend requires `mistralai`. "
231
+ "Install with: pip install dot-parser[mistral]"
232
+ ) from e
233
+
234
+ key = api_key or os.environ.get("MISTRAL_API_KEY")
235
+ if not key:
236
+ raise ValueError(
237
+ "Mistral backend requires an API key. Set MISTRAL_API_KEY or pass api_key=..."
238
+ )
239
+ self._client = _Mistral(api_key=key)
240
+ self._model = model
241
+ # When set (e.g. "French"), per-image annotations are requested in
242
+ # this language via the bbox_annotation schema; None keeps Mistral's
243
+ # default (English-leaning) behaviour.
244
+ self._language = language
245
+ # The three options below all default to today's behaviour: each
246
+ # one adds a field to the OCR request, so leaving them off keeps
247
+ # the request — and therefore the response — byte-identical to
248
+ # what callers got before they existed.
249
+ self._extract_headers_footers = extract_headers_footers
250
+ self._confidence_scores = confidence_scores
251
+ # Captions need block extraction, which requires both OCR 4+ and
252
+ # mistralai>=2.8. Resolved once here so an old SDK degrades to
253
+ # "no captions" instead of erroring on every call.
254
+ self._extract_captions = extract_captions and _sdk_supports("include_blocks")
255
+ if extract_captions and not self._extract_captions:
256
+ _log.warning(
257
+ "extract_captions=True ignored: the installed `mistralai` has no "
258
+ "include_blocks support on ocr.process (needs >=2.8). Image "
259
+ "captions will stay None."
260
+ )
261
+
262
+ def parse_pdf(self, source: str | Path | bytes) -> str:
263
+ if isinstance(source, bytes):
264
+ content = source
265
+ file_name = "document.pdf"
266
+ else:
267
+ path = Path(source)
268
+ content = path.read_bytes()
269
+ file_name = path.name
270
+
271
+ uploaded = self._client.files.upload(
272
+ file={"file_name": file_name, "content": content},
273
+ purpose="ocr",
274
+ )
275
+ signed = self._client.files.get_signed_url(file_id=uploaded.id)
276
+ resp = self._client.ocr.process(
277
+ model=self._model,
278
+ document={"type": "document_url", "document_url": signed.url},
279
+ )
280
+ return "\n\n".join(p.markdown for p in resp.pages)
281
+
282
+ def parse_pdf_with_images(
283
+ self,
284
+ source: str | Path | bytes,
285
+ *,
286
+ annotate_images: bool = True,
287
+ include_tables: bool = True,
288
+ ) -> ParseResult:
289
+ """Parse a PDF and return markdown together with extracted images.
290
+
291
+ Each image is requested with `include_image_base64=True` and, when
292
+ `annotate_images` is True, with a structured `bbox_annotation_format`
293
+ so Mistral writes a short description per image in the same OCR call
294
+ (no separate VLM round-trip). If that annotated call fails the
295
+ method automatically retries once without the annotation schema,
296
+ salvaging the markdown + raw images (interpretations stay None).
297
+
298
+ The returned markdown preserves Mistral's native `![id](id)`
299
+ placeholders at image positions; `ExtractedImage.name` matches
300
+ the `id` in those placeholders so callers can correlate text and
301
+ image without re-parsing the markdown.
302
+
303
+ When ``include_tables`` is False, GFM-style tables are stripped
304
+ from the resulting markdown — Mistral OCR has no native flag to
305
+ suppress them, so this is a post-processing pass.
306
+ """
307
+ content, file_name = self._read_source(source, default_name="document.pdf")
308
+ return self._ocr_document_with_images(
309
+ content,
310
+ file_name,
311
+ annotate_images=annotate_images,
312
+ include_tables=include_tables,
313
+ )
314
+
315
+ def parse_docx_with_images(
316
+ self,
317
+ source: str | Path | bytes,
318
+ *,
319
+ annotate_images: bool = True,
320
+ include_tables: bool = True,
321
+ ) -> ParseResult:
322
+ """Parse a DOCX directly through Mistral OCR (no markitdown / VLM).
323
+
324
+ Mirrors :meth:`parse_pdf_with_images` — the OCR API accepts ``.docx``
325
+ natively via ``document_url`` (Mistral rasters the document to pages
326
+ before OCR). The returned shape is identical to the PDF path: markdown
327
+ with ``![id](id)`` placeholders at image positions, plus one
328
+ :class:`ExtractedImage` per detected image with optional per-image
329
+ annotation when ``annotate_images=True``.
330
+
331
+ Exists as an A/B alternative to
332
+ :func:`dot_parser.docx_images.parse_docx_with_images` (markitdown +
333
+ VLM). Both paths are selectable from :func:`parse_with_images` —
334
+ pass ``backend=Mistral()`` for this OCR path, or pass ``vlm=...`` for
335
+ the markitdown path.
336
+ """
337
+ content, file_name = self._read_source(source, default_name="document.docx")
338
+ return self._ocr_document_with_images(
339
+ content,
340
+ file_name,
341
+ annotate_images=annotate_images,
342
+ include_tables=include_tables,
343
+ )
344
+
345
+ def parse_pptx_with_images(
346
+ self,
347
+ source: str | Path | bytes,
348
+ *,
349
+ annotate_images: bool = True,
350
+ include_tables: bool = True,
351
+ ) -> ParseResult:
352
+ """Parse a PPTX directly through Mistral OCR (no markitdown / VLM).
353
+
354
+ Same pipeline as :meth:`parse_docx_with_images`. Because the OCR
355
+ API rasters the deck page by page, one OCR page = one slide:
356
+ ``ExtractedImage.page`` is the slide number, and the per-page
357
+ markdowns are joined with ``<!-- Slide number: N -->`` markers —
358
+ the same markers markitdown emits for PPTX — so slide-aware
359
+ chunkers see an identical surface on both paths.
360
+
361
+ Unlike the markitdown+VLM path, rasterising the slide means
362
+ diagrams *drawn* in PowerPoint (shapes, arrows, SmartArt) are
363
+ seen and annotated, not just embedded picture files.
364
+ """
365
+ content, file_name = self._read_source(source, default_name="document.pptx")
366
+ return self._ocr_document_with_images(
367
+ content,
368
+ file_name,
369
+ annotate_images=annotate_images,
370
+ include_tables=include_tables,
371
+ page_marker_template="<!-- Slide number: {n} -->",
372
+ )
373
+
374
+ @staticmethod
375
+ def _read_source(source: str | Path | bytes, *, default_name: str) -> tuple[bytes, str]:
376
+ if isinstance(source, bytes):
377
+ return source, default_name
378
+ path = Path(source)
379
+ return path.read_bytes(), path.name
380
+
381
+ def _ocr_document_with_images(
382
+ self,
383
+ content: bytes,
384
+ file_name: str,
385
+ *,
386
+ annotate_images: bool,
387
+ include_tables: bool,
388
+ page_marker_template: str | None = None,
389
+ ) -> ParseResult:
390
+ """Shared OCR-with-images pipeline for any document_url-accepting input.
391
+
392
+ Mistral OCR's behavior is the same whether the source is PDF, DOCX,
393
+ PPTX, etc.: it rasters the document to pages, then runs text + image
394
+ + (optional) annotation extraction. This helper centralises the
395
+ upload → process → assemble flow so the per-format entry points
396
+ differ only in default filename and, optionally, a
397
+ ``page_marker_template`` (containing ``{n}``) emitted before each
398
+ page's markdown — used by the PPTX path to mark slide boundaries.
399
+ """
400
+ uploaded = self._client.files.upload(
401
+ file={"file_name": file_name, "content": content},
402
+ purpose="ocr",
403
+ )
404
+ signed = self._client.files.get_signed_url(file_id=uploaded.id)
405
+
406
+ base_kwargs: dict = {
407
+ "model": self._model,
408
+ "document": {"type": "document_url", "document_url": signed.url},
409
+ "include_image_base64": True,
410
+ }
411
+
412
+ # Everything beyond markdown + raw images is opt-in and grouped
413
+ # here, because they share one failure policy: any of them can be
414
+ # rejected by an older OCR model, and none is worth losing the
415
+ # document over.
416
+ optional_kwargs: dict = {}
417
+ if annotate_images:
418
+ optional_kwargs["bbox_annotation_format"] = _build_bbox_annotation_format(
419
+ self._language
420
+ )
421
+ if self._extract_headers_footers:
422
+ optional_kwargs["extract_header"] = True
423
+ optional_kwargs["extract_footer"] = True
424
+ if self._confidence_scores:
425
+ optional_kwargs["confidence_scores_granularity"] = self._confidence_scores
426
+ if self._extract_captions:
427
+ optional_kwargs["include_blocks"] = True
428
+
429
+ if optional_kwargs:
430
+ try:
431
+ resp = self._client.ocr.process(**base_kwargs, **optional_kwargs)
432
+ except Exception as exc:
433
+ # Retry once with the extras dropped so one unsupported or
434
+ # flaky enrichment doesn't lose the whole document. The
435
+ # caller still gets markdown + raw images, minus the
436
+ # annotations / headers / confidence / captions.
437
+ _log.warning(
438
+ "Mistral OCR failed with optional features %s (%s); retrying without them",
439
+ sorted(optional_kwargs),
440
+ exc,
441
+ )
442
+ resp = self._client.ocr.process(**base_kwargs)
443
+ else:
444
+ resp = self._client.ocr.process(**base_kwargs)
445
+
446
+ if page_marker_template is None:
447
+ markdown = "\n\n".join(p.markdown for p in resp.pages)
448
+ else:
449
+ markdown = "\n\n".join(
450
+ f"{page_marker_template.format(n=p.index + 1)}\n{p.markdown}" for p in resp.pages
451
+ )
452
+ if not include_tables:
453
+ markdown = strip_markdown_tables(markdown)
454
+
455
+ # Two-pass: gather raw fields, then dedupe titles document-wide so
456
+ # repeated diagrams ("workflow-diagram" twice) get -2/-3 suffixes.
457
+ raw_records: list[tuple[ExtractedImage, str | None]] = []
458
+ pages: list[PageInfo] = []
459
+ for page in resp.pages:
460
+ if self._extract_headers_footers or self._confidence_scores:
461
+ pages.append(_page_info(page))
462
+ captions = _caption_by_image_id(page) if self._extract_captions else {}
463
+ for img in page.images or []:
464
+ if not img.image_base64:
465
+ continue
466
+ annotation = parse_image_annotation(img.image_annotation)
467
+ raw_records.append(
468
+ (
469
+ ExtractedImage(
470
+ name=img.id,
471
+ base64=strip_data_url_prefix(img.image_base64),
472
+ mime_type=guess_mime_type(img.id),
473
+ page=page.index + 1,
474
+ original_caption=captions.get(img.id),
475
+ interpretation=annotation.interpretation or None,
476
+ title=None,
477
+ ),
478
+ annotation.title,
479
+ )
480
+ )
481
+
482
+ final_titles = dedupe_titles([t for _, t in raw_records])
483
+ images: list[ExtractedImage] = []
484
+ for (record, _), title in zip(raw_records, final_titles, strict=True):
485
+ images.append(
486
+ ExtractedImage(
487
+ name=record.name,
488
+ base64=record.base64,
489
+ mime_type=record.mime_type,
490
+ page=record.page,
491
+ original_caption=record.original_caption,
492
+ interpretation=record.interpretation,
493
+ title=title,
494
+ )
495
+ )
496
+ return ParseResult(markdown=markdown, images=images, pages=pages)
497
+
498
+ def _log_batch_error_file(self, job: "BatchJob") -> None:
499
+ """Log the batch job's ``error_file`` (Mistral's per-request failure
500
+ details), if it produced one. Best-effort: never raises into the caller,
501
+ and caps the dump so a large batch can't flood the logs."""
502
+ err_file = getattr(job, "error_file", None)
503
+ if not err_file:
504
+ return
505
+ try:
506
+ content = self._client.files.download(file_id=err_file).read().decode()
507
+ except Exception as exc:
508
+ _log.warning("could not download Mistral batch error_file %s: %s", err_file, exc)
509
+ return
510
+ for line in content.strip().splitlines()[:20]:
511
+ _log.error("Mistral batch error_file: %s", line)
512
+
513
+ def parse_pdfs(self, sources: list[str | Path | bytes]) -> list[str | None]:
514
+ """Parse multiple PDFs in a single Batch API job (50% off vs sync).
515
+
516
+ Returns markdowns in input order. Failed items are ``None``.
517
+ """
518
+ if not sources:
519
+ return []
520
+
521
+ # 1. Upload each PDF + get signed URL.
522
+ signed_urls: list[str] = []
523
+ for i, src in enumerate(sources):
524
+ if isinstance(src, bytes):
525
+ content = src
526
+ file_name = f"document_{i}.pdf"
527
+ else:
528
+ path = Path(src)
529
+ content = path.read_bytes()
530
+ file_name = path.name
531
+ uploaded = self._client.files.upload(
532
+ file={"file_name": file_name, "content": content},
533
+ purpose="ocr",
534
+ )
535
+ signed = self._client.files.get_signed_url(file_id=uploaded.id)
536
+ signed_urls.append(signed.url)
537
+
538
+ # 2. Build JSONL with one request per PDF, indexed by position.
539
+ lines = []
540
+ for i, url in enumerate(signed_urls):
541
+ lines.append(
542
+ json.dumps(
543
+ {
544
+ "custom_id": str(i),
545
+ "body": {
546
+ "model": self._model,
547
+ "document": {"type": "document_url", "document_url": url},
548
+ },
549
+ }
550
+ )
551
+ )
552
+ jsonl_bytes = ("\n".join(lines) + "\n").encode()
553
+
554
+ # 3. Upload JSONL + create batch job.
555
+ batch_file = self._client.files.upload(
556
+ file={"file_name": "batch_ocr.jsonl", "content": jsonl_bytes},
557
+ purpose="batch",
558
+ )
559
+ job = self._client.batch.jobs.create(
560
+ endpoint="/v1/ocr",
561
+ input_files=[batch_file.id],
562
+ model=self._model,
563
+ )
564
+
565
+ # 4. Poll until terminal.
566
+ while job.status not in _BATCH_TERMINAL:
567
+ time.sleep(_BATCH_POLL_INTERVAL_S)
568
+ job = self._client.batch.jobs.get(job_id=job.id)
569
+
570
+ results: list[str | None] = [None] * len(sources)
571
+ # A job yields no usable output if it didn't succeed, or if it "succeeded"
572
+ # but has no output_file (every request errored — the details are then in
573
+ # the error_file). Guard the download: calling it with output_file=None
574
+ # raises a cryptic SDK ValidationError that hides the real reason.
575
+ output_file = getattr(job, "output_file", None)
576
+ if job.status != "SUCCESS" or not output_file:
577
+ _log.error(
578
+ "Mistral batch job %s ended in %s with no usable output "
579
+ "(output_file=%s); all %d PDF(s) failed. job.errors=%s",
580
+ job.id,
581
+ job.status,
582
+ output_file,
583
+ len(sources),
584
+ getattr(job, "errors", None),
585
+ )
586
+ self._log_batch_error_file(job)
587
+ return results
588
+
589
+ # 5. Download output JSONL + map back to input order via custom_id.
590
+ output_bytes = self._client.files.download(file_id=output_file).read()
591
+ for line in output_bytes.decode().strip().splitlines():
592
+ rec = json.loads(line)
593
+ try:
594
+ idx = int(rec["custom_id"])
595
+ except (KeyError, ValueError):
596
+ _log.warning("Mistral batch output line has no valid custom_id: %.200s", line)
597
+ continue
598
+ if not (0 <= idx < len(sources)):
599
+ continue
600
+ # A per-item OCR failure comes back as an error on the record rather
601
+ # than pages; surface it instead of silently leaving a None.
602
+ error = rec.get("error") or (rec.get("response") or {}).get("body", {}).get("error")
603
+ if error is not None:
604
+ _log.error("Mistral OCR failed for PDF #%d: %s", idx, error)
605
+ continue
606
+ try:
607
+ pages = rec["response"]["body"]["pages"]
608
+ results[idx] = "\n\n".join(p["markdown"] for p in pages)
609
+ except (KeyError, TypeError):
610
+ _log.error("Mistral OCR returned no pages for PDF #%d: %.300s", idx, line)
611
+
612
+ missing = [i for i, md in enumerate(results) if md is None]
613
+ if missing:
614
+ self._log_batch_error_file(job)
615
+ _log.warning(
616
+ "Mistral batch %s: %d/%d PDF(s) produced no markdown (indices %s)",
617
+ job.id,
618
+ len(missing),
619
+ len(sources),
620
+ missing,
621
+ )
622
+ return results
@@ -0,0 +1,56 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ import tempfile
5
+ from pathlib import Path
6
+ from types import ModuleType
7
+
8
+
9
+ class Pymu:
10
+ """Default backend: pymupdf4llm.
11
+
12
+ Fast, CPU-only, lightweight. Works on natively-text PDFs. Fails on
13
+ image-only PDFs and PDFs with broken ToUnicode CMaps unless an OCR
14
+ engine (tesseract, rapidocr or paddleocr) is available — pymupdf4llm
15
+ auto-detects and falls back to OCR for problematic pages.
16
+
17
+ Args:
18
+ ocr_fallback: when True (default), pymupdf4llm uses its built-in
19
+ OCR fallback if an engine is available. When False, force the
20
+ no-OCR code path (matches a runtime image without tesseract,
21
+ useful for reproducing production where OCR is not installed).
22
+ """
23
+
24
+ def __init__(self, *, ocr_fallback: bool = True) -> None:
25
+ self._ocr_fallback = ocr_fallback
26
+
27
+ def parse_pdf(self, source: str | Path | bytes) -> str:
28
+ import pymupdf4llm
29
+
30
+ if self._ocr_fallback:
31
+ return self._call(pymupdf4llm, source)
32
+
33
+ # Reproduce a runtime without an OCR engine available (le-lab
34
+ # production). pymupdf4llm gates OCR on the result of
35
+ # ``select_ocr_function()`` — when it returns None, ``document.use_ocr``
36
+ # is forced to ``OCRMode.NEVER`` and neither full-page nor text-only
37
+ # OCR is invoked.
38
+ from pymupdf4llm.helpers import document_layout as _dl
39
+
40
+ original = _dl.select_ocr_function
41
+ # Deliberate monkeypatch of a third-party module function; returning
42
+ # None is precisely the signal pymupdf4llm reads to disable OCR.
43
+ _dl.select_ocr_function = lambda: None # ty: ignore[invalid-assignment]
44
+ try:
45
+ return self._call(pymupdf4llm, source)
46
+ finally:
47
+ _dl.select_ocr_function = original
48
+
49
+ @staticmethod
50
+ def _call(pymupdf4llm: ModuleType, source: str | Path | bytes) -> str:
51
+ if isinstance(source, bytes):
52
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=True) as tmp:
53
+ tmp.write(source)
54
+ tmp.flush()
55
+ return pymupdf4llm.to_markdown(tmp.name)
56
+ return pymupdf4llm.to_markdown(str(source))