dot-parser 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dot_parser/__init__.py +49 -0
- dot_parser/backends/__init__.py +26 -0
- dot_parser/backends/_base.py +79 -0
- dot_parser/backends/docling.py +59 -0
- dot_parser/backends/llama.py +76 -0
- dot_parser/backends/mistral.py +622 -0
- dot_parser/backends/pymu.py +56 -0
- dot_parser/chunking.py +257 -0
- dot_parser/docx_images.py +633 -0
- dot_parser/image_utils.py +152 -0
- dot_parser/images.py +125 -0
- dot_parser/markdown_utils.py +48 -0
- dot_parser/models.py +22 -0
- dot_parser/parsers.py +325 -0
- dot_parser/pricing.py +58 -0
- dot_parser/tokens.py +6 -0
- dot_parser/vlms.py +203 -0
- dot_parser-2.0.0.dist-info/METADATA +274 -0
- dot_parser-2.0.0.dist-info/RECORD +21 -0
- dot_parser-2.0.0.dist-info/WHEEL +4 -0
- dot_parser-2.0.0.dist-info/licenses/LICENSE.md +660 -0
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
"""Shared helpers for image-aware parsing.
|
|
5
|
+
|
|
6
|
+
Lives outside `backends/` because both the PDF (Mistral OCR) and the
|
|
7
|
+
DOCX (markitdown + VLM) paths need the same primitives:
|
|
8
|
+
|
|
9
|
+
- Title normalisation into a safe kebab-case slug (``clean_slug``).
|
|
10
|
+
- Cross-document title disambiguation (``dedupe_titles``) — two images
|
|
11
|
+
with the same model-supplied title get ``-2``, ``-3`` suffixes.
|
|
12
|
+
- MIME type / data-URL helpers for parsing Mistral OCR's image payloads
|
|
13
|
+
(``guess_mime_type``, ``strip_data_url_prefix``).
|
|
14
|
+
- A single robust parser for ``{title, description}`` JSON annotations
|
|
15
|
+
(``parse_image_annotation``) — used both for Mistral OCR's
|
|
16
|
+
``image_annotation`` field and for the chat-vision response in
|
|
17
|
+
:class:`dot_parser.vlms.MistralVLM`.
|
|
18
|
+
|
|
19
|
+
Keeping these here avoids three near-identical copies drifting in
|
|
20
|
+
``backends/mistral.py``, ``vlms.py``, and ``docx_images.py``.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import json
|
|
24
|
+
import re
|
|
25
|
+
|
|
26
|
+
from dot_parser.images import ImageDescription
|
|
27
|
+
|
|
28
|
+
_SLUG_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+){0,15}$")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def clean_slug(value: object) -> str | None:
|
|
32
|
+
"""Normalise a model-supplied title into a safe kebab-case slug.
|
|
33
|
+
|
|
34
|
+
Accepts the title even if it includes spaces, underscores, title-case
|
|
35
|
+
or trailing punctuation: lowercases and collapses any run of
|
|
36
|
+
non-alphanumerics into a single hyphen. Returns None when the result
|
|
37
|
+
is empty or would exceed the kebab-case length cap — better no title
|
|
38
|
+
than an unwieldy filename-like string in the UI.
|
|
39
|
+
"""
|
|
40
|
+
if not isinstance(value, str):
|
|
41
|
+
return None
|
|
42
|
+
candidate = value.strip().lower()
|
|
43
|
+
if not candidate:
|
|
44
|
+
return None
|
|
45
|
+
candidate = re.sub(r"[^a-z0-9]+", "-", candidate).strip("-")
|
|
46
|
+
if not candidate or not _SLUG_RE.match(candidate):
|
|
47
|
+
return None
|
|
48
|
+
return candidate
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def dedupe_titles(titles: list[str | None]) -> list[str | None]:
|
|
52
|
+
"""Append ``-2``, ``-3``, ... to repeated titles, preserving Nones.
|
|
53
|
+
|
|
54
|
+
Two images that the model labelled ``workflow-diagram`` come back as
|
|
55
|
+
``workflow-diagram`` and ``workflow-diagram-2`` so the title remains
|
|
56
|
+
safe to use as a display label or filename.
|
|
57
|
+
"""
|
|
58
|
+
seen: dict[str, int] = {}
|
|
59
|
+
out: list[str | None] = []
|
|
60
|
+
for t in titles:
|
|
61
|
+
if t is None:
|
|
62
|
+
out.append(None)
|
|
63
|
+
continue
|
|
64
|
+
seen[t] = seen.get(t, 0) + 1
|
|
65
|
+
out.append(t if seen[t] == 1 else f"{t}-{seen[t]}")
|
|
66
|
+
return out
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def guess_mime_type(image_id: str) -> str:
|
|
70
|
+
"""Infer MIME type from a Mistral OCR image id like ``img-0.jpeg``.
|
|
71
|
+
|
|
72
|
+
Mistral encodes the format in the id's extension. Defaults to
|
|
73
|
+
``image/png`` when the extension is missing or unrecognised.
|
|
74
|
+
"""
|
|
75
|
+
ext = image_id.rsplit(".", 1)[-1].lower() if "." in image_id else ""
|
|
76
|
+
if ext in ("jpg", "jpeg"):
|
|
77
|
+
return "image/jpeg"
|
|
78
|
+
if ext == "webp":
|
|
79
|
+
return "image/webp"
|
|
80
|
+
if ext == "gif":
|
|
81
|
+
return "image/gif"
|
|
82
|
+
return "image/png"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def strip_data_url_prefix(b64: str) -> str:
|
|
86
|
+
"""Strip an optional ``data:image/...;base64,`` prefix.
|
|
87
|
+
|
|
88
|
+
Mistral OCR sometimes returns the image as a full data URL rather
|
|
89
|
+
than bare base64. Callers always want the raw payload.
|
|
90
|
+
"""
|
|
91
|
+
if b64.startswith("data:") and "," in b64:
|
|
92
|
+
return b64.split(",", 1)[1]
|
|
93
|
+
return b64
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def parse_image_annotation(raw: str | None) -> ImageDescription:
|
|
97
|
+
"""Parse a ``{title, description}`` JSON annotation into a record.
|
|
98
|
+
|
|
99
|
+
Used for two distinct input flavours:
|
|
100
|
+
|
|
101
|
+
- Mistral OCR's ``OCRImageObject.image_annotation`` field, which is
|
|
102
|
+
strict JSON when ``bbox_annotation_format`` is set.
|
|
103
|
+
- The content of a chat-vision response with a JSON schema
|
|
104
|
+
``response_format`` — generally strict JSON, but the model
|
|
105
|
+
occasionally emits ``"description":`` followed by unescaped
|
|
106
|
+
multi-line text that breaks ``json.loads``.
|
|
107
|
+
|
|
108
|
+
Tries strict ``json.loads`` first; on failure, falls back to regex
|
|
109
|
+
extraction so a usable title is still surfaced. Non-JSON input is
|
|
110
|
+
returned as ``interpretation=raw, title=None`` (legacy passthrough).
|
|
111
|
+
Empty / None input yields an empty :class:`ImageDescription`.
|
|
112
|
+
"""
|
|
113
|
+
if not raw:
|
|
114
|
+
return ImageDescription(interpretation="", title=None)
|
|
115
|
+
|
|
116
|
+
fenced = re.match(r"^```(?:json)?\s*(.*?)\s*```$", raw, flags=re.DOTALL)
|
|
117
|
+
candidate = fenced.group(1) if fenced else raw
|
|
118
|
+
|
|
119
|
+
start = candidate.find("{")
|
|
120
|
+
end = candidate.rfind("}")
|
|
121
|
+
if start != -1 and end > start:
|
|
122
|
+
snippet = candidate[start : end + 1]
|
|
123
|
+
try:
|
|
124
|
+
parsed = json.loads(snippet)
|
|
125
|
+
except (json.JSONDecodeError, ValueError):
|
|
126
|
+
parsed = None
|
|
127
|
+
if isinstance(parsed, dict):
|
|
128
|
+
desc_raw = parsed.get("description") or parsed.get("interpretation")
|
|
129
|
+
description = desc_raw.strip() if isinstance(desc_raw, str) else ""
|
|
130
|
+
title = clean_slug(parsed.get("title"))
|
|
131
|
+
# Surface any structured signal we got. When only a title
|
|
132
|
+
# comes back, interpretation is "" so callers can `or None`
|
|
133
|
+
# coerce uniformly without leaking the raw JSON blob.
|
|
134
|
+
if description or title:
|
|
135
|
+
return ImageDescription(interpretation=description, title=title)
|
|
136
|
+
|
|
137
|
+
# JSON looked promising but didn't yield usable fields — recover
|
|
138
|
+
# via regex. Title is always a short single-line string;
|
|
139
|
+
# description is whatever sits between ``"description":`` and
|
|
140
|
+
# the closing brace.
|
|
141
|
+
title_match = re.search(r'"title"\s*:\s*"([^"\n]*)"', snippet)
|
|
142
|
+
desc_match = re.search(
|
|
143
|
+
r'"description"\s*:\s*"?(.*?)"?\s*\}\s*$',
|
|
144
|
+
snippet,
|
|
145
|
+
flags=re.DOTALL,
|
|
146
|
+
)
|
|
147
|
+
recovered_title = clean_slug(title_match.group(1)) if title_match else None
|
|
148
|
+
recovered_desc = desc_match.group(1).strip() if desc_match else ""
|
|
149
|
+
if recovered_title or recovered_desc:
|
|
150
|
+
return ImageDescription(interpretation=recovered_desc, title=recovered_title)
|
|
151
|
+
|
|
152
|
+
return ImageDescription(interpretation=raw, title=None)
|
dot_parser/images.py
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
"""Public types for image-aware parsing.
|
|
5
|
+
|
|
6
|
+
`parse_with_images()` returns a `ParseResult` that pairs the markdown text
|
|
7
|
+
(with `` anchors at each image position) with a list of
|
|
8
|
+
`ExtractedImage` records carrying the image bytes and, when available,
|
|
9
|
+
a textual interpretation.
|
|
10
|
+
|
|
11
|
+
The `VLM` protocol lets callers inject a vision-language model for the
|
|
12
|
+
DOCX path; the PDF path uses Mistral OCR's native per-image annotations
|
|
13
|
+
instead, so no VLM is required there.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from typing import Protocol, runtime_checkable
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class ExtractedImage:
|
|
22
|
+
"""A single image extracted from a document.
|
|
23
|
+
|
|
24
|
+
`name` matches the `` placeholder embedded at the
|
|
25
|
+
image's position inside the markdown returned alongside it; it
|
|
26
|
+
is a stable technical identifier (e.g. `img-003.png`).
|
|
27
|
+
|
|
28
|
+
`title` is a short human-readable label describing what the image
|
|
29
|
+
*depicts* (e.g. `gear-assembly-exploded-view`) — produced by the
|
|
30
|
+
VLM in the same call as `interpretation`. Use it for display and
|
|
31
|
+
`name` for cross-referencing with anchors / chunks.
|
|
32
|
+
|
|
33
|
+
`unsupported_reason` is populated when the extraction pipeline
|
|
34
|
+
deliberately did not call a VLM on this image because its
|
|
35
|
+
`mime_type` is outside the supported allowlist (e.g. EMF, WMF,
|
|
36
|
+
SVG, TIFF). The string carries a human-readable explanation of
|
|
37
|
+
the format and a hint at what external tool could recover it.
|
|
38
|
+
When set, `interpretation` and `title` will both be None; the
|
|
39
|
+
raw `base64` bytes are still populated so consumers can offer a
|
|
40
|
+
download or run their own conversion.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
name: str
|
|
44
|
+
base64: str
|
|
45
|
+
mime_type: str
|
|
46
|
+
page: int | None = None
|
|
47
|
+
original_caption: str | None = None
|
|
48
|
+
interpretation: str | None = None
|
|
49
|
+
title: str | None = None
|
|
50
|
+
unsupported_reason: str | None = None
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True)
|
|
54
|
+
class ImageDescription:
|
|
55
|
+
"""Structured VLM response for a single image.
|
|
56
|
+
|
|
57
|
+
`title` is a short kebab-case label (4–8 words max) describing
|
|
58
|
+
what the image depicts. `interpretation` is the long-form
|
|
59
|
+
RAG-oriented description. `title` may be None when the VLM
|
|
60
|
+
cannot produce one (e.g. failure to parse a structured response).
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
interpretation: str
|
|
64
|
+
title: str | None = None
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass(frozen=True)
|
|
68
|
+
class PageInfo:
|
|
69
|
+
"""Per-page metadata a backend can report alongside the markdown.
|
|
70
|
+
|
|
71
|
+
Populated only when the caller opts in — see the ``Mistral``
|
|
72
|
+
backend's ``extract_headers_footers`` and ``confidence_scores``
|
|
73
|
+
options. Fields left None mean "not requested" rather than "not
|
|
74
|
+
present in the document", so an empty ``ParseResult.pages`` is the
|
|
75
|
+
default and carries no information either way.
|
|
76
|
+
|
|
77
|
+
``header`` / ``footer`` hold the running page furniture (e.g.
|
|
78
|
+
"Confidential -- page 4 of 27") that OCR pulled out of the main
|
|
79
|
+
content, so it can be excluded from RAG chunks.
|
|
80
|
+
|
|
81
|
+
``average_confidence`` / ``minimum_confidence`` are OCR self-reported
|
|
82
|
+
scores for the page. Calibrate thresholds against your own corpus:
|
|
83
|
+
measured across the benchmark PDFs, ``average_confidence`` stayed in
|
|
84
|
+
0.985-0.989 and ``minimum_confidence`` in 0.23-0.46 regardless of
|
|
85
|
+
document quality. ``minimum_confidence`` reflects the single worst
|
|
86
|
+
region on the page, so it reads low even on clean pages and is not
|
|
87
|
+
on its own a signal that the page went badly.
|
|
88
|
+
"""
|
|
89
|
+
|
|
90
|
+
page: int
|
|
91
|
+
header: str | None = None
|
|
92
|
+
footer: str | None = None
|
|
93
|
+
average_confidence: float | None = None
|
|
94
|
+
minimum_confidence: float | None = None
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
@dataclass(frozen=True)
|
|
98
|
+
class ParseResult:
|
|
99
|
+
"""Markdown plus the images extracted from the source document."""
|
|
100
|
+
|
|
101
|
+
markdown: str
|
|
102
|
+
images: list[ExtractedImage] = field(default_factory=list)
|
|
103
|
+
pages: list[PageInfo] = field(default_factory=list)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@runtime_checkable
|
|
107
|
+
class VLM(Protocol):
|
|
108
|
+
"""Vision-language model used to interpret images extracted from DOCX.
|
|
109
|
+
|
|
110
|
+
Implementations should return either a plain description string
|
|
111
|
+
(legacy) or a structured :class:`ImageDescription` carrying both
|
|
112
|
+
a short title and the longer interpretation. `context` is the
|
|
113
|
+
surrounding markdown text (e.g. the paragraph immediately before
|
|
114
|
+
the image) that the implementation may use to ground its
|
|
115
|
+
description; treat it as optional context, not as part of the
|
|
116
|
+
prompt's instructions.
|
|
117
|
+
"""
|
|
118
|
+
|
|
119
|
+
def describe_image(
|
|
120
|
+
self,
|
|
121
|
+
image_base64: str,
|
|
122
|
+
*,
|
|
123
|
+
mime_type: str = "image/png",
|
|
124
|
+
context: str | None = None,
|
|
125
|
+
) -> "str | ImageDescription": ...
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
"""Shared markdown post-processing utilities used across backends.
|
|
5
|
+
|
|
6
|
+
Lives outside `backends/` because both the PDF (Mistral OCR) and the
|
|
7
|
+
DOCX (markitdown) paths need the same table-stripping pass: neither
|
|
8
|
+
underlying tool exposes a native "drop tables" flag, so the caller is
|
|
9
|
+
responsible for removing them after the fact when ``include_tables``
|
|
10
|
+
is False.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
|
|
15
|
+
_TABLE_ROW_RE = re.compile(r"^\s*\|.*\|\s*$")
|
|
16
|
+
_TABLE_SEP_RE = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(\|\s*:?-{3,}:?\s*)+\|?\s*$")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def strip_markdown_tables(markdown: str) -> str:
|
|
20
|
+
"""Drop GFM-style markdown tables from a body of text.
|
|
21
|
+
|
|
22
|
+
A table block here means: two or more consecutive lines where the
|
|
23
|
+
first matches a pipe-row and the second is a ``| --- | --- |``
|
|
24
|
+
separator. Operates line-by-line so it survives tables that aren't
|
|
25
|
+
surrounded by blank lines (Mistral OCR sometimes packs them tight
|
|
26
|
+
against surrounding prose; markitdown leaves a blank line, but the
|
|
27
|
+
same code path handles both).
|
|
28
|
+
|
|
29
|
+
Heuristic by design — we don't try to parse the full GFM table
|
|
30
|
+
grammar — but it cleanly removes the structures both Mistral OCR
|
|
31
|
+
and markitdown produce today and leaves non-table pipe lines alone.
|
|
32
|
+
"""
|
|
33
|
+
lines = markdown.split("\n")
|
|
34
|
+
out: list[str] = []
|
|
35
|
+
i = 0
|
|
36
|
+
while i < len(lines):
|
|
37
|
+
if (
|
|
38
|
+
i + 1 < len(lines)
|
|
39
|
+
and _TABLE_ROW_RE.match(lines[i])
|
|
40
|
+
and _TABLE_SEP_RE.match(lines[i + 1])
|
|
41
|
+
):
|
|
42
|
+
i += 2
|
|
43
|
+
while i < len(lines) and _TABLE_ROW_RE.match(lines[i]):
|
|
44
|
+
i += 1
|
|
45
|
+
continue
|
|
46
|
+
out.append(lines[i])
|
|
47
|
+
i += 1
|
|
48
|
+
return re.sub(r"\n{3,}", "\n\n", "\n".join(out)).strip("\n")
|
dot_parser/models.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
import re
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
_HEADING_RE = re.compile(r"^(#{1,6})\s+(.+)$")
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ParseError(Exception):
|
|
11
|
+
"""Raised when a document cannot be parsed."""
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class Chunk:
|
|
16
|
+
content: str
|
|
17
|
+
section_path: list[str] = field(default_factory=list)
|
|
18
|
+
|
|
19
|
+
@property
|
|
20
|
+
def heading(self) -> str | None:
|
|
21
|
+
first_line = self.content.split("\n", 1)[0]
|
|
22
|
+
return first_line if _HEADING_RE.match(first_line) else None
|
dot_parser/parsers.py
ADDED
|
@@ -0,0 +1,325 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
import logging
|
|
5
|
+
import tempfile
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
import markitdown
|
|
9
|
+
|
|
10
|
+
from dot_parser.backends import (
|
|
11
|
+
Backend,
|
|
12
|
+
BatchBackend,
|
|
13
|
+
DocxImageBackend,
|
|
14
|
+
ImageBackend,
|
|
15
|
+
Mistral,
|
|
16
|
+
PptxImageBackend,
|
|
17
|
+
Pymu,
|
|
18
|
+
)
|
|
19
|
+
from dot_parser.images import VLM, ParseResult
|
|
20
|
+
from dot_parser.models import ParseError
|
|
21
|
+
|
|
22
|
+
_log = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
PDF_EXTENSIONS = {".pdf"}
|
|
25
|
+
PASSTHROUGH_EXTENSIONS = {".md", ".txt"}
|
|
26
|
+
MARKITDOWN_EXTENSIONS = {".docx", ".pptx", ".html", ".xhtml", ".htm", ".xlsx", ".csv"}
|
|
27
|
+
ALL_EXTENSIONS = PDF_EXTENSIONS | PASSTHROUGH_EXTENSIONS | MARKITDOWN_EXTENSIONS
|
|
28
|
+
|
|
29
|
+
_EXTENSION_MAP: dict[str, str] = {
|
|
30
|
+
"pdf": ".pdf",
|
|
31
|
+
"md": ".md",
|
|
32
|
+
"markdown": ".md",
|
|
33
|
+
"txt": ".txt",
|
|
34
|
+
"text": ".txt",
|
|
35
|
+
"docx": ".docx",
|
|
36
|
+
"pptx": ".pptx",
|
|
37
|
+
"html": ".html",
|
|
38
|
+
"xhtml": ".xhtml",
|
|
39
|
+
"htm": ".htm",
|
|
40
|
+
"xlsx": ".xlsx",
|
|
41
|
+
"csv": ".csv",
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _resolve_extension(source: str | Path | bytes, format: str | None) -> str:
|
|
46
|
+
if format is not None:
|
|
47
|
+
ext = _EXTENSION_MAP.get(format.lower())
|
|
48
|
+
if ext is None:
|
|
49
|
+
raise ParseError(f"Unknown format: {format!r}")
|
|
50
|
+
return ext
|
|
51
|
+
|
|
52
|
+
if isinstance(source, bytes):
|
|
53
|
+
raise ParseError("format is required when source is bytes")
|
|
54
|
+
|
|
55
|
+
ext = Path(source).suffix.lower()
|
|
56
|
+
if not ext:
|
|
57
|
+
raise ParseError(f"Cannot determine format for {source!r}: no file extension")
|
|
58
|
+
if ext not in ALL_EXTENSIONS:
|
|
59
|
+
raise ParseError(f"Unsupported file extension: {ext!r}")
|
|
60
|
+
return ext
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _read_text(source: str | Path | bytes) -> str:
|
|
64
|
+
if isinstance(source, bytes):
|
|
65
|
+
return source.decode("utf-8")
|
|
66
|
+
return Path(source).read_text(encoding="utf-8")
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _parse_with_markitdown(source: str | Path | bytes, ext: str) -> str:
|
|
70
|
+
md = markitdown.MarkItDown()
|
|
71
|
+
if isinstance(source, bytes):
|
|
72
|
+
with tempfile.NamedTemporaryFile(suffix=ext, delete=True) as tmp:
|
|
73
|
+
tmp.write(source)
|
|
74
|
+
tmp.flush()
|
|
75
|
+
result = md.convert(tmp.name)
|
|
76
|
+
else:
|
|
77
|
+
result = md.convert(str(source))
|
|
78
|
+
return result.text_content
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def parse(
|
|
82
|
+
source: str | Path | bytes,
|
|
83
|
+
format: str | None = None,
|
|
84
|
+
backend: Backend | None = None,
|
|
85
|
+
) -> str:
|
|
86
|
+
"""Convert a document to Markdown.
|
|
87
|
+
|
|
88
|
+
Args:
|
|
89
|
+
source: file path (str or Path) or raw bytes.
|
|
90
|
+
format: explicit format hint (e.g. "pdf"). Required when source is bytes.
|
|
91
|
+
backend: optional PDF backend (Pymu, Docling, Mistral, Llama). When None,
|
|
92
|
+
uses Pymu (pymupdf4llm) for PDFs. Backends only apply to PDF input;
|
|
93
|
+
passing one for a non-PDF format raises ParseError.
|
|
94
|
+
|
|
95
|
+
Returns:
|
|
96
|
+
Markdown string.
|
|
97
|
+
|
|
98
|
+
Raises:
|
|
99
|
+
ParseError: on unsupported format, missing file, or backend mismatch.
|
|
100
|
+
"""
|
|
101
|
+
ext = _resolve_extension(source, format)
|
|
102
|
+
|
|
103
|
+
if backend is not None and ext not in PDF_EXTENSIONS:
|
|
104
|
+
raise ParseError(f"backend= is only supported for PDF input (got {ext!r})")
|
|
105
|
+
|
|
106
|
+
try:
|
|
107
|
+
if ext in PDF_EXTENSIONS:
|
|
108
|
+
return (backend or Pymu()).parse_pdf(source)
|
|
109
|
+
elif ext in PASSTHROUGH_EXTENSIONS:
|
|
110
|
+
return _read_text(source)
|
|
111
|
+
else:
|
|
112
|
+
return _parse_with_markitdown(source, ext)
|
|
113
|
+
except ParseError:
|
|
114
|
+
raise
|
|
115
|
+
except Exception as e:
|
|
116
|
+
label = "<bytes>" if isinstance(source, bytes) else str(source)
|
|
117
|
+
raise ParseError(f"Could not parse {label}: {e}") from e
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def parse_with_images(
|
|
121
|
+
source: str | Path | bytes,
|
|
122
|
+
format: str | None = None,
|
|
123
|
+
*,
|
|
124
|
+
backend: ImageBackend | DocxImageBackend | PptxImageBackend | None = None,
|
|
125
|
+
vlm: VLM | None = None,
|
|
126
|
+
annotate_images: bool = True,
|
|
127
|
+
include_tables: bool = True,
|
|
128
|
+
deadline: float | None = None,
|
|
129
|
+
) -> ParseResult:
|
|
130
|
+
"""Convert a document to Markdown and extract its images.
|
|
131
|
+
|
|
132
|
+
Args:
|
|
133
|
+
source: file path or raw bytes. Only ``.pdf``, ``.docx`` and
|
|
134
|
+
``.pptx`` are supported by this function; other formats
|
|
135
|
+
raise ParseError.
|
|
136
|
+
format: explicit format hint. Required when source is bytes.
|
|
137
|
+
backend: optional image-capable backend. For PDF, defaults to a
|
|
138
|
+
fresh ``Mistral()`` instance and must expose
|
|
139
|
+
``parse_pdf_with_images``. For DOCX/PPTX, when a backend
|
|
140
|
+
exposing ``parse_docx_with_images`` /
|
|
141
|
+
``parse_pptx_with_images`` is passed (e.g. ``Mistral()``),
|
|
142
|
+
the document goes through OCR end-to-end — no VLM needed.
|
|
143
|
+
vlm: vision-language model used to interpret each DOCX/PPTX
|
|
144
|
+
image in the markitdown path. Required for DOCX/PPTX when
|
|
145
|
+
no format-capable backend is provided; ignored for PDF and
|
|
146
|
+
for the OCR path (Mistral OCR annotates images inline via
|
|
147
|
+
``annotate_images``).
|
|
148
|
+
annotate_images: Applies to any OCR-backed path (PDF, or
|
|
149
|
+
DOCX/PPTX with ``backend=Mistral()``). When True (default),
|
|
150
|
+
Mistral OCR is asked to return a short description per
|
|
151
|
+
image in the same call. Set False to receive raw images
|
|
152
|
+
without interpretation. Ignored on the markitdown+VLM path
|
|
153
|
+
(the VLM is always invoked there).
|
|
154
|
+
include_tables: When False, drop tabular content from the
|
|
155
|
+
returned markdown.
|
|
156
|
+
deadline: absolute ``time.monotonic()`` timestamp bounding
|
|
157
|
+
per-image interpretation on the markitdown+VLM path. Images
|
|
158
|
+
not reached in time come back uninterpreted with
|
|
159
|
+
``unsupported_reason`` set, instead of the call overrunning.
|
|
160
|
+
Ignored on OCR-backed paths.
|
|
161
|
+
|
|
162
|
+
Returns:
|
|
163
|
+
A :class:`ParseResult` with the markdown (containing
|
|
164
|
+
```` anchors at image positions) and an
|
|
165
|
+
:class:`ExtractedImage` per image, in document order.
|
|
166
|
+
|
|
167
|
+
Raises:
|
|
168
|
+
ParseError: on unsupported format, missing backend/vlm, or when
|
|
169
|
+
a backend doesn't expose the relevant image-extraction
|
|
170
|
+
method for the input format.
|
|
171
|
+
"""
|
|
172
|
+
ext = _resolve_extension(source, format)
|
|
173
|
+
|
|
174
|
+
if ext == ".pdf":
|
|
175
|
+
be = backend or Mistral()
|
|
176
|
+
if not isinstance(be, ImageBackend):
|
|
177
|
+
raise ParseError(
|
|
178
|
+
"Selected backend does not support image extraction. "
|
|
179
|
+
"Use Mistral() or another ImageBackend implementation."
|
|
180
|
+
)
|
|
181
|
+
try:
|
|
182
|
+
return be.parse_pdf_with_images(
|
|
183
|
+
source,
|
|
184
|
+
annotate_images=annotate_images,
|
|
185
|
+
include_tables=include_tables,
|
|
186
|
+
)
|
|
187
|
+
except ParseError:
|
|
188
|
+
raise
|
|
189
|
+
except Exception as e:
|
|
190
|
+
label = "<bytes>" if isinstance(source, bytes) else str(source)
|
|
191
|
+
raise ParseError(f"Could not parse {label}: {e}") from e
|
|
192
|
+
|
|
193
|
+
if ext in (".docx", ".pptx"):
|
|
194
|
+
# DOCX and PPTX each have two interchangeable paths. The OCR
|
|
195
|
+
# path wins when a backend that supports the format is passed
|
|
196
|
+
# in; otherwise we fall back to the markitdown + per-image VLM
|
|
197
|
+
# path.
|
|
198
|
+
fmt = ext[1:]
|
|
199
|
+
backend_method = f"parse_{fmt}_with_images"
|
|
200
|
+
if backend is not None and hasattr(backend, backend_method):
|
|
201
|
+
try:
|
|
202
|
+
return getattr(backend, backend_method)(
|
|
203
|
+
source,
|
|
204
|
+
annotate_images=annotate_images,
|
|
205
|
+
include_tables=include_tables,
|
|
206
|
+
)
|
|
207
|
+
except ParseError:
|
|
208
|
+
raise
|
|
209
|
+
except Exception as e:
|
|
210
|
+
label = "<bytes>" if isinstance(source, bytes) else str(source)
|
|
211
|
+
raise ParseError(f"Could not parse {label}: {e}") from e
|
|
212
|
+
|
|
213
|
+
if vlm is None:
|
|
214
|
+
raise ParseError(
|
|
215
|
+
f"{fmt.upper()} image extraction requires either a "
|
|
216
|
+
f"{fmt.upper()}-capable backend (e.g. backend=Mistral()) "
|
|
217
|
+
"or a VLM (e.g. vlm=MistralVLM()) for the markitdown path."
|
|
218
|
+
)
|
|
219
|
+
# Imported lazily so importing dot_parser.parsers doesn't pull
|
|
220
|
+
# in python-docx for callers that only use the text-only API.
|
|
221
|
+
from dot_parser import docx_images as _office_images
|
|
222
|
+
|
|
223
|
+
_parse_office = getattr(_office_images, f"parse_{fmt}_with_images")
|
|
224
|
+
try:
|
|
225
|
+
return _parse_office(
|
|
226
|
+
source,
|
|
227
|
+
vlm=vlm,
|
|
228
|
+
include_tables=include_tables,
|
|
229
|
+
deadline=deadline,
|
|
230
|
+
)
|
|
231
|
+
except ParseError:
|
|
232
|
+
raise
|
|
233
|
+
except Exception as e:
|
|
234
|
+
label = "<bytes>" if isinstance(source, bytes) else str(source)
|
|
235
|
+
raise ParseError(f"Could not parse {label}: {e}") from e
|
|
236
|
+
|
|
237
|
+
raise ParseError(f"parse_with_images only supports .pdf, .docx and .pptx (got {ext!r})")
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def interpret_images(
|
|
241
|
+
result: ParseResult,
|
|
242
|
+
vlm: VLM,
|
|
243
|
+
*,
|
|
244
|
+
deadline: float | None = None,
|
|
245
|
+
) -> ParseResult:
|
|
246
|
+
"""Second-pass VLM interpretation of a ``ParseResult``'s images.
|
|
247
|
+
|
|
248
|
+
Fills ``interpretation`` (and ``title``) on each image by calling
|
|
249
|
+
``vlm.describe_image`` once per image, grounding the call with the
|
|
250
|
+
image's ``original_caption`` (explicitly labelled) plus the markdown
|
|
251
|
+
surrounding its ```` anchor.
|
|
252
|
+
|
|
253
|
+
This exists because Mistral OCR's inline annotations
|
|
254
|
+
(``annotate_images=True``) are produced from the cropped image alone —
|
|
255
|
+
the annotating model sees no caption and no page text, which on
|
|
256
|
+
domain-specific figures produces confidently wrong descriptions. The
|
|
257
|
+
grounded recipe is::
|
|
258
|
+
|
|
259
|
+
result = parse_with_images(
|
|
260
|
+
"doc.pdf",
|
|
261
|
+
backend=Mistral(extract_captions=True),
|
|
262
|
+
annotate_images=False, # skip the ungrounded inline pass
|
|
263
|
+
)
|
|
264
|
+
result = interpret_images(result, MistralVLM())
|
|
265
|
+
|
|
266
|
+
Same number of VLM calls as ``annotate_images=True`` (one per image,
|
|
267
|
+
just client-side), but each call carries the figure's own caption.
|
|
268
|
+
Note the latency shape differs even though the count matches:
|
|
269
|
+
``annotate_images=True`` annotates every image inside the single
|
|
270
|
+
``ocr.process`` request, whereas this pass issues its own calls one
|
|
271
|
+
after another — so on image-heavy documents pass a ``deadline``.
|
|
272
|
+
|
|
273
|
+
Returns a new ``ParseResult``; markdown and pages are unchanged.
|
|
274
|
+
Per-image VLM failures are absorbed (that image's interpretation
|
|
275
|
+
stays None), as are images the ``deadline`` cuts short — those carry
|
|
276
|
+
``unsupported_reason`` instead. Existing interpretations are
|
|
277
|
+
overwritten.
|
|
278
|
+
"""
|
|
279
|
+
from dot_parser.docx_images import _interpret_images
|
|
280
|
+
|
|
281
|
+
images = _interpret_images(
|
|
282
|
+
result.markdown,
|
|
283
|
+
result.images,
|
|
284
|
+
vlm,
|
|
285
|
+
caption_in_context=True,
|
|
286
|
+
deadline=deadline,
|
|
287
|
+
)
|
|
288
|
+
return ParseResult(markdown=result.markdown, images=images, pages=result.pages)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def parse_pdfs(
|
|
292
|
+
sources: list[str | Path | bytes],
|
|
293
|
+
backend: Backend | None = None,
|
|
294
|
+
) -> list[str | None]:
|
|
295
|
+
"""Convert multiple PDFs to Markdown, in input order.
|
|
296
|
+
|
|
297
|
+
Backends that implement a ``parse_pdfs`` method (e.g. ``Mistral`` via the
|
|
298
|
+
Batch API) handle the whole list in one optimized call. For backends
|
|
299
|
+
without that method, falls back to looping ``parse()`` on each source.
|
|
300
|
+
|
|
301
|
+
Args:
|
|
302
|
+
sources: list of PDF file paths or raw bytes.
|
|
303
|
+
backend: optional PDF backend. Defaults to Pymu().
|
|
304
|
+
|
|
305
|
+
Returns:
|
|
306
|
+
A list of Markdown strings, same length and order as ``sources``.
|
|
307
|
+
Failed items are returned as ``None`` (no exception raised).
|
|
308
|
+
"""
|
|
309
|
+
be = backend or Pymu()
|
|
310
|
+
if isinstance(be, BatchBackend):
|
|
311
|
+
return be.parse_pdfs(sources)
|
|
312
|
+
results: list[str | None] = []
|
|
313
|
+
for src in sources:
|
|
314
|
+
try:
|
|
315
|
+
# parse_pdfs is PDF-only, so pin format="pdf": callers pass raw bytes
|
|
316
|
+
# (in-memory PDFs) which parse() can't infer an extension from, and
|
|
317
|
+
# would otherwise reject with "format is required when source is bytes".
|
|
318
|
+
results.append(parse(src, format="pdf", backend=be))
|
|
319
|
+
except ParseError as exc:
|
|
320
|
+
# Loop-fallback backends (e.g. Pymu) fail per file; log why before
|
|
321
|
+
# dropping to None so the caller isn't left with an unexplained gap.
|
|
322
|
+
label = "<bytes>" if isinstance(src, bytes) else str(src)
|
|
323
|
+
_log.error("parse_pdfs: could not parse %s: %s", label, exc)
|
|
324
|
+
results.append(None)
|
|
325
|
+
return results
|