dot-parser 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dot_parser/__init__.py +49 -0
- dot_parser/backends/__init__.py +26 -0
- dot_parser/backends/_base.py +79 -0
- dot_parser/backends/docling.py +59 -0
- dot_parser/backends/llama.py +76 -0
- dot_parser/backends/mistral.py +622 -0
- dot_parser/backends/pymu.py +56 -0
- dot_parser/chunking.py +257 -0
- dot_parser/docx_images.py +633 -0
- dot_parser/image_utils.py +152 -0
- dot_parser/images.py +125 -0
- dot_parser/markdown_utils.py +48 -0
- dot_parser/models.py +22 -0
- dot_parser/parsers.py +325 -0
- dot_parser/pricing.py +58 -0
- dot_parser/tokens.py +6 -0
- dot_parser/vlms.py +203 -0
- dot_parser-2.0.0.dist-info/METADATA +274 -0
- dot_parser-2.0.0.dist-info/RECORD +21 -0
- dot_parser-2.0.0.dist-info/WHEEL +4 -0
- dot_parser-2.0.0.dist-info/licenses/LICENSE.md +660 -0
dot_parser/pricing.py
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
"""Per-backend pricing helpers.
|
|
5
|
+
|
|
6
|
+
Exposes ``cost_per_1k_pages`` and ``estimate_cost``. Name-based so callers
|
|
7
|
+
can query pricing without instantiating a backend (no API key needed).
|
|
8
|
+
|
|
9
|
+
Sources: https://mistral.ai/news/ocr-4/ and the LlamaCloud pricing page.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from typing import Literal
|
|
13
|
+
|
|
14
|
+
BackendName = Literal["pymu", "docling", "mistral", "llama"]
|
|
15
|
+
LlamaTier = Literal["fast", "cost_effective", "agentic", "agentic_plus"]
|
|
16
|
+
|
|
17
|
+
_MISTRAL_SYNC = 4.0
|
|
18
|
+
_MISTRAL_BATCH = 2.0
|
|
19
|
+
|
|
20
|
+
# LlamaCloud bills in credits; rates below are $/1k pages assuming
|
|
21
|
+
# $0.001/credit (the public list price at the time of writing).
|
|
22
|
+
_LLAMA_PER_TIER: dict[str, float] = {
|
|
23
|
+
"fast": 1.0,
|
|
24
|
+
"cost_effective": 3.0,
|
|
25
|
+
"agentic": 10.0,
|
|
26
|
+
"agentic_plus": 45.0,
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def cost_per_1k_pages(
|
|
31
|
+
name: BackendName,
|
|
32
|
+
*,
|
|
33
|
+
tier: LlamaTier = "cost_effective",
|
|
34
|
+
batch: bool = False,
|
|
35
|
+
) -> float:
|
|
36
|
+
"""Return USD cost per 1k pages for a backend.
|
|
37
|
+
|
|
38
|
+
``tier`` applies only to ``llama``; ``batch=True`` applies only to
|
|
39
|
+
``mistral`` (Batch API, 50% off sync). Both are ignored otherwise.
|
|
40
|
+
"""
|
|
41
|
+
if name in ("pymu", "docling"):
|
|
42
|
+
return 0.0
|
|
43
|
+
if name == "mistral":
|
|
44
|
+
return _MISTRAL_BATCH if batch else _MISTRAL_SYNC
|
|
45
|
+
if name == "llama":
|
|
46
|
+
return _LLAMA_PER_TIER[tier]
|
|
47
|
+
raise ValueError(f"unknown backend: {name}")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def estimate_cost(
|
|
51
|
+
name: BackendName,
|
|
52
|
+
pages: int,
|
|
53
|
+
*,
|
|
54
|
+
tier: LlamaTier = "cost_effective",
|
|
55
|
+
batch: bool = False,
|
|
56
|
+
) -> float:
|
|
57
|
+
"""Return total USD cost for ``pages`` pages on ``name``."""
|
|
58
|
+
return pages * cost_per_1k_pages(name, tier=tier, batch=batch) / 1000
|
dot_parser/tokens.py
ADDED
dot_parser/vlms.py
ADDED
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
"""Ready-made `VLM` implementations.
|
|
5
|
+
|
|
6
|
+
`MistralVLM` is the implementation used for the DOCX image-extraction
|
|
7
|
+
path: it sends each image to Mistral's chat-completion API with a
|
|
8
|
+
short, deterministic prompt and returns a description string.
|
|
9
|
+
|
|
10
|
+
Kept separate from the Mistral OCR backend (`backends/mistral.py`)
|
|
11
|
+
because the two use different Mistral endpoints (`ocr.process` vs.
|
|
12
|
+
`chat.complete`) and have independent install/model requirements.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import os
|
|
16
|
+
|
|
17
|
+
from pydantic import BaseModel, Field
|
|
18
|
+
|
|
19
|
+
from dot_parser.image_utils import parse_image_annotation
|
|
20
|
+
from dot_parser.images import VLM, ImageDescription
|
|
21
|
+
|
|
22
|
+
# Per-call ceiling for a single image description. The Mistral SDK's own
|
|
23
|
+
# default is 300_000 ms, which is longer than most callers' entire
|
|
24
|
+
# request budget — one stuck image could consume all of it and starve
|
|
25
|
+
# every image behind it. Measured descriptions land at 0.7-8.5 s, so 60 s
|
|
26
|
+
# is roughly 7x the slowest observed call: generous enough never to cut
|
|
27
|
+
# off a legitimately slow one, short enough to fail fast. Pass
|
|
28
|
+
# ``timeout_ms=None`` to restore the SDK default.
|
|
29
|
+
DEFAULT_VLM_TIMEOUT_MS = 60_000
|
|
30
|
+
|
|
31
|
+
_DEFAULT_PROMPT = (
|
|
32
|
+
"Describe this image for a downstream RAG system. Populate the two "
|
|
33
|
+
"fields of the structured response as follows.\n\n"
|
|
34
|
+
"title: a short kebab-case slug (3-8 words, lowercase, "
|
|
35
|
+
"hyphen-separated, no file extension) naming what the image "
|
|
36
|
+
'depicts (e.g. "system-architecture-diagram", '
|
|
37
|
+
'"exploded-gear-assembly-view", "monthly-revenue-bar-chart").\n\n'
|
|
38
|
+
"description: let image complexity set the length — do not pad "
|
|
39
|
+
"simple images and do not compress complex ones.\n"
|
|
40
|
+
" - Simple imagery (logo, photo, icon, single screenshot): 2-3 "
|
|
41
|
+
"concrete sentences.\n"
|
|
42
|
+
" - Diagrams, schematics, architecture views, flowcharts, "
|
|
43
|
+
"process maps: a structured paragraph that (a) names every "
|
|
44
|
+
"labeled node/block/component, (b) describes each connection/"
|
|
45
|
+
"arrow/flow (source -> target, with relationship label if any), "
|
|
46
|
+
"and (c) transcribes all visible text verbatim.\n"
|
|
47
|
+
" - Charts and plots: state the chart type, axes (with units), "
|
|
48
|
+
"series, and the notable values, trends, or extrema. Transcribe "
|
|
49
|
+
"the title, legend, and axis labels verbatim.\n"
|
|
50
|
+
" - Tables: transcribe the header row verbatim and summarize the "
|
|
51
|
+
"rows; if the table is small (<= 15 rows), transcribe it in full "
|
|
52
|
+
"as markdown inside the description.\n"
|
|
53
|
+
" - Screenshots of UI or code: transcribe all readable text "
|
|
54
|
+
"verbatim and name the visible UI elements or code constructs.\n\n"
|
|
55
|
+
"Always: be concrete, transcribe text exactly as it appears, do "
|
|
56
|
+
"not speculate about anything not visible."
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _language_directive(language: str) -> str:
|
|
61
|
+
"""Instruction appended to a description prompt to fix the output language.
|
|
62
|
+
|
|
63
|
+
Only the model's own prose (``description``, and the ``title`` slug)
|
|
64
|
+
is forced into ``language``; text the model transcribes from the image
|
|
65
|
+
must stay in whatever language it appears in, so quoted labels and
|
|
66
|
+
captions are not silently translated.
|
|
67
|
+
"""
|
|
68
|
+
return (
|
|
69
|
+
f"\n\nWrite the description and the title slug in {language}. "
|
|
70
|
+
"Text you transcribe verbatim from the image must stay in its "
|
|
71
|
+
"original language — only your own prose description is in "
|
|
72
|
+
f"{language}."
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class _ImageDescriptionSchema(BaseModel):
|
|
77
|
+
"""Structured-output schema for `chat.parse`.
|
|
78
|
+
|
|
79
|
+
Sending this class as `response_format=` puts the model in
|
|
80
|
+
constrained-decoding mode, so the emitted JSON is guaranteed to be
|
|
81
|
+
syntactically valid (newlines escaped) and to contain both keys
|
|
82
|
+
with string values. This is what prevents the long-description
|
|
83
|
+
failure mode where unescaped newlines broke `json.loads`.
|
|
84
|
+
"""
|
|
85
|
+
|
|
86
|
+
title: str = Field(
|
|
87
|
+
description=(
|
|
88
|
+
"Short kebab-case slug, 3-8 words, lowercase, "
|
|
89
|
+
"hyphen-separated, no file extension. "
|
|
90
|
+
'Example: "system-architecture-diagram".'
|
|
91
|
+
)
|
|
92
|
+
)
|
|
93
|
+
description: str = Field(
|
|
94
|
+
description=(
|
|
95
|
+
"Interpretation of the image content; length scales with "
|
|
96
|
+
"image complexity per the user prompt."
|
|
97
|
+
)
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class MistralVLM(VLM):
|
|
102
|
+
"""Vision-language model backed by Mistral's chat-completion API.
|
|
103
|
+
|
|
104
|
+
Used by `parse_with_images(format="docx", vlm=MistralVLM())` to
|
|
105
|
+
interpret each embedded image. PDF parsing through the Mistral OCR
|
|
106
|
+
backend does not need this — annotations there come back inline
|
|
107
|
+
from the OCR call itself.
|
|
108
|
+
"""
|
|
109
|
+
|
|
110
|
+
def __init__(
|
|
111
|
+
self,
|
|
112
|
+
*,
|
|
113
|
+
api_key: str | None = None,
|
|
114
|
+
model: str = "mistral-medium-2604",
|
|
115
|
+
prompt: str = _DEFAULT_PROMPT,
|
|
116
|
+
language: str | None = None,
|
|
117
|
+
timeout_ms: int | None = DEFAULT_VLM_TIMEOUT_MS,
|
|
118
|
+
reasoning_effort: str | None = None,
|
|
119
|
+
) -> None:
|
|
120
|
+
try:
|
|
121
|
+
from mistralai.client import Mistral as _Mistral
|
|
122
|
+
from mistralai.extra.utils.response_format import (
|
|
123
|
+
response_format_from_pydantic_model,
|
|
124
|
+
)
|
|
125
|
+
except ImportError as e:
|
|
126
|
+
raise ImportError(
|
|
127
|
+
"MistralVLM requires `mistralai`. Install with: pip install dot-parser[mistral]"
|
|
128
|
+
) from e
|
|
129
|
+
|
|
130
|
+
key = api_key or os.environ.get("MISTRAL_API_KEY")
|
|
131
|
+
if not key:
|
|
132
|
+
raise ValueError(
|
|
133
|
+
"MistralVLM requires an API key. Set MISTRAL_API_KEY or pass api_key=..."
|
|
134
|
+
)
|
|
135
|
+
self._client = _Mistral(api_key=key)
|
|
136
|
+
self._model = model
|
|
137
|
+
self._timeout_ms = timeout_ms
|
|
138
|
+
# Only sent when set: the accepted values are per-model (reasoning
|
|
139
|
+
# models such as magistral/mistral-small-latest take "none" or
|
|
140
|
+
# "high" and reject "minimal"/"low"/"medium"; non-reasoning models
|
|
141
|
+
# reject the field outright), so an unconditional default would
|
|
142
|
+
# 400 on some models. None keeps the model's own default.
|
|
143
|
+
self._reasoning_effort = reasoning_effort
|
|
144
|
+
# `language`, when set, fixes the output language of the
|
|
145
|
+
# description (e.g. "French"); appended once here so it survives
|
|
146
|
+
# whether or not the caller also overrode `prompt`.
|
|
147
|
+
self._prompt = prompt + (_language_directive(language) if language else "")
|
|
148
|
+
# Precomputed JSON-schema response_format. Sending this on every
|
|
149
|
+
# chat.complete call asks the API for constrained decoding so the
|
|
150
|
+
# model emits a JSON object matching _ImageDescriptionSchema.
|
|
151
|
+
self._response_format = response_format_from_pydantic_model(_ImageDescriptionSchema)
|
|
152
|
+
|
|
153
|
+
def describe_image(
|
|
154
|
+
self,
|
|
155
|
+
image_base64: str,
|
|
156
|
+
*,
|
|
157
|
+
mime_type: str = "image/png",
|
|
158
|
+
context: str | None = None,
|
|
159
|
+
) -> ImageDescription:
|
|
160
|
+
prompt = self._prompt
|
|
161
|
+
if context:
|
|
162
|
+
prompt = (
|
|
163
|
+
f"{prompt}\n\nThe image appears in a document near the "
|
|
164
|
+
f"following text (use it only to ground your description, "
|
|
165
|
+
f"do not repeat it):\n\n{context}"
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
data_url = f"data:{mime_type};base64,{image_base64}"
|
|
169
|
+
messages = [
|
|
170
|
+
{
|
|
171
|
+
"role": "user",
|
|
172
|
+
"content": [
|
|
173
|
+
{"type": "text", "text": prompt},
|
|
174
|
+
{"type": "image_url", "image_url": data_url},
|
|
175
|
+
],
|
|
176
|
+
}
|
|
177
|
+
]
|
|
178
|
+
|
|
179
|
+
# Constrained decoding: chat.complete with a json_schema
|
|
180
|
+
# response_format derived from _ImageDescriptionSchema. This is
|
|
181
|
+
# what chat.parse does internally, minus the Pydantic-coercion
|
|
182
|
+
# wrapper that was raising silently and dropping us into an
|
|
183
|
+
# unconstrained fallback. We parse the JSON ourselves so a
|
|
184
|
+
# slightly malformed payload can still yield a usable title.
|
|
185
|
+
optional: dict = {}
|
|
186
|
+
if self._reasoning_effort is not None:
|
|
187
|
+
optional["reasoning_effort"] = self._reasoning_effort
|
|
188
|
+
resp = self._client.chat.complete(
|
|
189
|
+
model=self._model,
|
|
190
|
+
messages=messages,
|
|
191
|
+
response_format=self._response_format,
|
|
192
|
+
timeout_ms=self._timeout_ms,
|
|
193
|
+
**optional,
|
|
194
|
+
)
|
|
195
|
+
content = resp.choices[0].message.content
|
|
196
|
+
if isinstance(content, list):
|
|
197
|
+
# Defensive: chat models occasionally return a list of chunks.
|
|
198
|
+
raw = "".join(
|
|
199
|
+
c.get("text", "") if isinstance(c, dict) else str(c) for c in content
|
|
200
|
+
).strip()
|
|
201
|
+
else:
|
|
202
|
+
raw = (content or "").strip()
|
|
203
|
+
return parse_image_annotation(raw)
|
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: dot-parser
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: Document-to-markdown parser and chunker for RAG pipelines
|
|
5
|
+
Project-URL: Homepage, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
|
|
6
|
+
Project-URL: Repository, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
|
|
7
|
+
Project-URL: Issues, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/issues
|
|
8
|
+
Author-email: Kannon For Deep Tech <louis.letarnec@deepika.ai>
|
|
9
|
+
License-Expression: AGPL-3.0-or-later
|
|
10
|
+
License-File: LICENSE.md
|
|
11
|
+
Keywords: chunking,deepika,markdown,open-toolbox,parser,rag
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
17
|
+
Requires-Python: <3.14,>=3.12
|
|
18
|
+
Requires-Dist: markitdown[all]>=0.1
|
|
19
|
+
Requires-Dist: pillow>=10.0
|
|
20
|
+
Requires-Dist: pydantic>=2
|
|
21
|
+
Requires-Dist: pymupdf4llm>=1.27.2.2
|
|
22
|
+
Requires-Dist: python-docx>=1.2.0
|
|
23
|
+
Requires-Dist: semchunk>=3.0
|
|
24
|
+
Requires-Dist: typing-extensions>=4.16
|
|
25
|
+
Provides-Extra: all
|
|
26
|
+
Requires-Dist: docling>=2.0; extra == 'all'
|
|
27
|
+
Requires-Dist: llama-cloud>=2.4; extra == 'all'
|
|
28
|
+
Requires-Dist: mistralai>=2.0; extra == 'all'
|
|
29
|
+
Provides-Extra: docling
|
|
30
|
+
Requires-Dist: docling>=2.0; extra == 'docling'
|
|
31
|
+
Provides-Extra: llama
|
|
32
|
+
Requires-Dist: llama-cloud>=2.4; extra == 'llama'
|
|
33
|
+
Provides-Extra: mistral
|
|
34
|
+
Requires-Dist: mistralai>=2.0; extra == 'mistral'
|
|
35
|
+
Description-Content-Type: text/markdown
|
|
36
|
+
|
|
37
|
+
# dot-parser
|
|
38
|
+
|
|
39
|
+
[](https://pypi.org/project/dot-parser/)
|
|
40
|
+

|
|
41
|
+
[](LICENSE.md)
|
|
42
|
+
[](https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/pipelines)
|
|
43
|
+
|
|
44
|
+
**Turn documents into clean Markdown, then into retrieval-ready chunks.**
|
|
45
|
+
|
|
46
|
+
```python
|
|
47
|
+
from dot_parser import parse, chunk
|
|
48
|
+
|
|
49
|
+
markdown = parse("report.pdf")
|
|
50
|
+
chunks = chunk(markdown)
|
|
51
|
+
|
|
52
|
+
for c in chunks:
|
|
53
|
+
print(c.section_path, len(c.content))
|
|
54
|
+
# ['# Report', '## Methods', '### Analysis'] 842
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Why dot-parser
|
|
58
|
+
|
|
59
|
+
A RAG pipeline needs two things from a document: faithful text, and chunks that
|
|
60
|
+
keep their place in the document's structure. Most tools give you one or the
|
|
61
|
+
other — parsers stop at raw text, splitters assume you already have Markdown and
|
|
62
|
+
cut it blind to headings.
|
|
63
|
+
|
|
64
|
+
dot-parser does both in one step. It converts PDF, DOCX, PPTX, HTML, XLSX, CSV,
|
|
65
|
+
Markdown and plain text into clean Markdown, then splits that Markdown into
|
|
66
|
+
chunks carrying their heading hierarchy in `section_path` — so a chunk still
|
|
67
|
+
knows it came from *Report › Methods › Analysis*.
|
|
68
|
+
|
|
69
|
+
The default install stays light: heavy backends are optional extras, and you pick
|
|
70
|
+
per document whether to run locally or through a cloud OCR. See
|
|
71
|
+
[docs/DESIGN.md](docs/DESIGN.md) for how it compares to Docling, MarkItDown and
|
|
72
|
+
LangChain splitters.
|
|
73
|
+
|
|
74
|
+
## Features
|
|
75
|
+
|
|
76
|
+
- One `parse()` call for PDF, DOCX, PPTX, HTML, XLSX, CSV, Markdown and text
|
|
77
|
+
- Swappable PDF backends: local and fast, local and layout-aware, or cloud OCR
|
|
78
|
+
- Heading-aware chunking with `section_path` metadata
|
|
79
|
+
- Image extraction with per-image descriptions, from a VLM or from OCR annotations
|
|
80
|
+
- Batch PDF parsing through the Mistral Batch API (50% cheaper)
|
|
81
|
+
- Cost estimation before you send anything
|
|
82
|
+
- Light by default — heavy backends live behind extras
|
|
83
|
+
|
|
84
|
+
## Installation
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
pip install dot-parser
|
|
88
|
+
|
|
89
|
+
# With optional PDF backends (each ~50 MB - 1 GB):
|
|
90
|
+
pip install 'dot-parser[docling]'
|
|
91
|
+
pip install 'dot-parser[mistral]'
|
|
92
|
+
pip install 'dot-parser[llama]'
|
|
93
|
+
pip install 'dot-parser[all]'
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Requires Python 3.12+.
|
|
97
|
+
|
|
98
|
+
## Quick start
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
from dot_parser import parse, chunk
|
|
102
|
+
|
|
103
|
+
markdown: str = parse("report.pdf")
|
|
104
|
+
chunks: list[Chunk] = chunk(markdown)
|
|
105
|
+
|
|
106
|
+
for c in chunks:
|
|
107
|
+
print(c.section_path, c.heading, len(c.content))
|
|
108
|
+
# ["# Report", "## Methods", "### Analysis"], "### Analysis", 842
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
For a different PDF backend:
|
|
112
|
+
|
|
113
|
+
```python
|
|
114
|
+
from dot_parser import parse, parse_pdfs, Mistral, Docling
|
|
115
|
+
|
|
116
|
+
# single PDF, cloud OCR
|
|
117
|
+
md = parse("scanned.pdf", backend=Mistral())
|
|
118
|
+
|
|
119
|
+
# many PDFs, batched (50% off via Mistral Batch API)
|
|
120
|
+
mds = parse_pdfs(["a.pdf", "b.pdf", "c.pdf"], backend=Mistral())
|
|
121
|
+
|
|
122
|
+
# local layout-aware ML pipeline
|
|
123
|
+
md = parse("paper.pdf", backend=Docling())
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
## API
|
|
127
|
+
|
|
128
|
+
### `parse(source, format=None, backend=None) -> str`
|
|
129
|
+
|
|
130
|
+
Converts a document to Markdown.
|
|
131
|
+
|
|
132
|
+
- **source** -- file path (`str` or `Path`) or raw `bytes`
|
|
133
|
+
- **format** -- explicit format hint (e.g. `"pdf"`, `"docx"`). Required when source is `bytes`, otherwise inferred from the file extension.
|
|
134
|
+
- **backend** -- optional PDF backend (`Pymu`, `Docling`, `Mistral`, `Llama`). Defaults to `Pymu()`. Backends only apply to PDF input.
|
|
135
|
+
|
|
136
|
+
Supported formats: `.pdf`, `.md`, `.txt`, `.docx`, `.pptx`, `.html`, `.xhtml`, `.htm`, `.xlsx`, `.csv`
|
|
137
|
+
|
|
138
|
+
Raises `ParseError` on unsupported formats or conversion failures.
|
|
139
|
+
|
|
140
|
+
### `parse_pdfs(sources, backend=None) -> list[str | None]`
|
|
141
|
+
|
|
142
|
+
Converts a list of PDFs to Markdown, in input order. Backends that implement a `parse_pdfs` method (e.g. `Mistral` via the Batch API, 50% off) handle the whole list in a single optimized call. Otherwise falls back to looping `parse()` per source. Failed items are returned as `None` (no exception raised).
|
|
143
|
+
|
|
144
|
+
### `parse_with_images(source, format=None, *, backend=None, vlm=None, annotate_images=True, include_tables=True) -> ParseResult`
|
|
145
|
+
|
|
146
|
+
Converts a `.pdf`, `.docx` or `.pptx` and extracts its images alongside the
|
|
147
|
+
Markdown, keeping `` anchors at each image's position. Each image is
|
|
148
|
+
described either by a VLM or by OCR annotations from the same call. See
|
|
149
|
+
[docs/IMAGE_EXTRACTION.md](docs/IMAGE_EXTRACTION.md) for the two available paths per format.
|
|
150
|
+
|
|
151
|
+
### Backends
|
|
152
|
+
|
|
153
|
+
- **`Pymu(ocr_fallback=True)`** -- pymupdf4llm, default. Fast and CPU-only. The built-in OCR fallback kicks in when an OCR engine (tesseract, rapidocr, paddleocr) is available. Set `ocr_fallback=False` to force the no-OCR code path (matches a runtime image without OCR shipped).
|
|
154
|
+
- **`Docling(force_full_page_ocr=True)`** -- IBM Docling, layout-aware ML pipeline + Tesseract OCR. Robust on image-only and broken-encoding PDFs. Install with `pip install dot-parser[docling]`.
|
|
155
|
+
- **`Mistral(api_key=None, model="mistral-ocr-4-0")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
|
|
156
|
+
- **`Llama(api_key=None, tier="cost_effective")`** -- LlamaCloud parsing API. Reads `LLAMA_CLOUD_API_KEY` from env. Tiers: `fast`, `cost_effective`, `agentic`, `agentic_plus`. Install with `pip install dot-parser[llama]`.
|
|
157
|
+
|
|
158
|
+
#### Mistral OCR enrichments
|
|
159
|
+
|
|
160
|
+
Three optional extras on the `Mistral(...)` constructor, applying to the `*_with_images` methods. All default to off, so existing callers see no change:
|
|
161
|
+
|
|
162
|
+
```python
|
|
163
|
+
from dot_parser import parse_with_images, Mistral
|
|
164
|
+
|
|
165
|
+
result = parse_with_images(
|
|
166
|
+
"spec.pdf",
|
|
167
|
+
backend=Mistral(
|
|
168
|
+
extract_headers_footers=True, # page furniture -> result.pages[i].header/.footer
|
|
169
|
+
confidence_scores="page", # or "word" -> result.pages[i].*_confidence
|
|
170
|
+
extract_captions=True, # figure captions -> image.original_caption
|
|
171
|
+
),
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
for page in result.pages:
|
|
175
|
+
print(page.page, page.average_confidence, page.minimum_confidence)
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
- `extract_headers_footers` moves running headers/footers out of the markdown and into `ParseResult.pages`, so repeated "page 4 of 27" noise stays out of retrieval chunks. It **removes that text from the markdown** -- the only one of the three that changes `result.markdown`.
|
|
179
|
+
- `confidence_scores` surfaces OCR's self-reported quality. **Calibrate any threshold on your own corpus -- absolute cutoffs do not transfer.** Measured over the 39 pages of the benchmark corpus (`native_simple`, `image_only`, `broken_encoding`), `average_confidence` sat at 0.985-0.989 and `minimum_confidence` at 0.23-0.46 on *every* document regardless of quality. `minimum_confidence` is the single worst region on the page, so it is low even on clean pages and is not a page-quality alarm by itself.
|
|
180
|
+
- `extract_captions` fills `ExtractedImage.original_caption` from the figure caption printed beside each image. Needs OCR 4+ **and** `mistralai>=2.8`; on older SDKs it logs a warning and leaves captions `None` rather than failing.
|
|
181
|
+
|
|
182
|
+
If the OCR model rejects any of these parameters, the call is retried once without them -- you get the markdown and images, minus the enrichments.
|
|
183
|
+
|
|
184
|
+
#### Caption-grounded image interpretation
|
|
185
|
+
|
|
186
|
+
Mistral OCR's inline annotations (`annotate_images=True`) are produced from the cropped image alone -- the annotating model sees no caption and no page text, which on domain-specific figures yields confidently wrong descriptions. `interpret_images()` replaces that pass with client-side VLM calls grounded in each figure's own caption and surrounding markdown:
|
|
187
|
+
|
|
188
|
+
```python
|
|
189
|
+
from dot_parser import parse_with_images, interpret_images, Mistral, MistralVLM
|
|
190
|
+
|
|
191
|
+
result = parse_with_images(
|
|
192
|
+
"paper.pdf",
|
|
193
|
+
backend=Mistral(extract_captions=True),
|
|
194
|
+
annotate_images=False, # skip the ungrounded inline pass
|
|
195
|
+
)
|
|
196
|
+
result = interpret_images(result, MistralVLM())
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
Same number of VLM calls as `annotate_images=True`, but each one carries the caption. On a figure-heavy biomechanics paper this turned "magnetic field components" / "fluid flow through a curved pipe" into accurate descriptions of the actual rib diagrams.
|
|
200
|
+
|
|
201
|
+
### `chunk(markdown, max_tokens=None, merge=True) -> list[Chunk]`
|
|
202
|
+
|
|
203
|
+
Splits Markdown into chunks with heading-hierarchy metadata.
|
|
204
|
+
|
|
205
|
+
- **markdown** -- Markdown string (typically output of `parse()`)
|
|
206
|
+
- **max_tokens** -- maximum tokens per chunk. Auto-computed from the 95th percentile of section sizes (clamped 64-512) when `None`.
|
|
207
|
+
- **merge** -- when `True` (default), merges small adjacent sections sharing the same parent heading to reduce fragmentation.
|
|
208
|
+
|
|
209
|
+
The chunking algorithm is described in [docs/DESIGN.md](docs/DESIGN.md).
|
|
210
|
+
|
|
211
|
+
### `Chunk`
|
|
212
|
+
|
|
213
|
+
```python
|
|
214
|
+
@dataclass(frozen=True)
|
|
215
|
+
class Chunk:
|
|
216
|
+
content: str # full text of the chunk (heading + body)
|
|
217
|
+
heading: str | None # the heading line, if any
|
|
218
|
+
section_path: list[str] # hierarchy of headings leading to this chunk
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
### `ParseError`
|
|
222
|
+
|
|
223
|
+
Raised by `parse()` when a document cannot be converted.
|
|
224
|
+
|
|
225
|
+
### Cost estimation
|
|
226
|
+
|
|
227
|
+
`cost_per_1k_pages(backend)` and `estimate_cost(backend, pages, batch=False)`
|
|
228
|
+
report what a cloud backend will cost, without instantiating it or needing an
|
|
229
|
+
API key.
|
|
230
|
+
|
|
231
|
+
## Stability
|
|
232
|
+
|
|
233
|
+
`dot-parser` follows semantic versioning: everything exported from the top-level
|
|
234
|
+
package is covered, anything underscore-prefixed is internal and may change in
|
|
235
|
+
any release. Public names are never removed without a deprecation period.
|
|
236
|
+
|
|
237
|
+
```toml
|
|
238
|
+
dependencies = ["dot-parser>=1.0,<2"]
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
See [docs/VERSIONING.md](docs/VERSIONING.md) for the full policy.
|
|
242
|
+
|
|
243
|
+
## Roadmap
|
|
244
|
+
|
|
245
|
+
- [ ] Table extraction as structured data, not just Markdown
|
|
246
|
+
- [ ] Streaming parse for very large documents
|
|
247
|
+
- [ ] Additional cloud OCR backends
|
|
248
|
+
|
|
249
|
+
## Documentation
|
|
250
|
+
|
|
251
|
+
| Document | Contents |
|
|
252
|
+
|---|---|
|
|
253
|
+
| [docs/DESIGN.md](docs/DESIGN.md) | Design rationale, backend choices, chunking algorithm |
|
|
254
|
+
| [docs/IMAGE_EXTRACTION.md](docs/IMAGE_EXTRACTION.md) | Image-extraction paths per format |
|
|
255
|
+
| [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) | Environment setup, tests, code style |
|
|
256
|
+
| [docs/VERSIONING.md](docs/VERSIONING.md) | Versioning, deprecation policy, how to depend on this package |
|
|
257
|
+
| [docs/PUBLISHING.md](docs/PUBLISHING.md) | Cutting a release |
|
|
258
|
+
| [CHANGELOG.md](CHANGELOG.md) | Release history |
|
|
259
|
+
|
|
260
|
+
## Contributing
|
|
261
|
+
|
|
262
|
+
Contributions are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md) for the DCO
|
|
263
|
+
sign-off requirement, the licensing terms that apply to contributions, and how to
|
|
264
|
+
submit a change.
|
|
265
|
+
|
|
266
|
+
## Licence
|
|
267
|
+
|
|
268
|
+
Copyright (C) 2026 Kannon For Deep Tech (deepika)
|
|
269
|
+
|
|
270
|
+
This software is distributed under the GNU Affero General Public License,
|
|
271
|
+
version 3 or later — see [LICENSE.md](LICENSE.md).
|
|
272
|
+
|
|
273
|
+
A commercial licence is available for use in proprietary environments.
|
|
274
|
+
Contact: louis.letarnec@deepika.ai
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
dot_parser/__init__.py,sha256=oAWCCkbHhoTSbhilZwVM66GBBcGkUq0wwVlZHD1thnQ,1052
|
|
2
|
+
dot_parser/chunking.py,sha256=nvLyCZz8h3EVEsbR4oFu_kE58CLBvWbcYQkwzZD-xKQ,8208
|
|
3
|
+
dot_parser/docx_images.py,sha256=tXBl8OiKazHt_1wbOtlaOTlBXu4gIZ-2zwEcG9xxdTM,26018
|
|
4
|
+
dot_parser/image_utils.py,sha256=26uiRuz6j-ZLWvGshHKbfvRJWQxoh2DfiKLX33k-XHY,6037
|
|
5
|
+
dot_parser/images.py,sha256=OQ64cbn2-nFn9hKjPLM8cO5gBTDzkcBVL0ArFThnbGA,4651
|
|
6
|
+
dot_parser/markdown_utils.py,sha256=KXjLtneqDzPLbiMfND1TEmEHum1oRk0vym0h4qgf9K0,1789
|
|
7
|
+
dot_parser/models.py,sha256=OMiCL7XqLki0iie0K3TrLw0iDTvt0BjeWnopLGYu004,561
|
|
8
|
+
dot_parser/parsers.py,sha256=UurXUTU_Za5DSZat07tQB9fSm5IfSbhLgN2XHICQlmI,12286
|
|
9
|
+
dot_parser/pricing.py,sha256=cjqM-gaZ-GQyGibbUQwWpqIECv33j4OJ3_Shvp33OXc,1684
|
|
10
|
+
dot_parser/tokens.py,sha256=LIERtlBVtN_loTOMUyuDDxuUr_p4q8YSdmxIEI_Uhts,164
|
|
11
|
+
dot_parser/vlms.py,sha256=cACc_Psy8tJggNXPdHXZ3NY9kneg_oCtNmtGYljNh-w,8636
|
|
12
|
+
dot_parser/backends/__init__.py,sha256=nmJciV1nfu8__4Ls37xwh1kvc1yrhpscHoInMaO7c-o,583
|
|
13
|
+
dot_parser/backends/_base.py,sha256=rYmdLAJOoZKpVbm9njGJ-PjjNl8Sp2hi65AiCP_7CW8,2257
|
|
14
|
+
dot_parser/backends/docling.py,sha256=0OoucXeY0AqwRFk1aEbSzP7g9z7s8OuAfUzOz7AHwMw,1976
|
|
15
|
+
dot_parser/backends/llama.py,sha256=5vlGbTr8jJB_ms1ZXp0HGAR59MvM-iNLU67Cq3neHg8,2634
|
|
16
|
+
dot_parser/backends/mistral.py,sha256=cRg_g5szAGtzFNT3mC3KFSfpjKPJJpzBROCWcjxl1_c,25801
|
|
17
|
+
dot_parser/backends/pymu.py,sha256=EIg4JWwyNdjkbKOlzxWLxERmyZAMIUNxulQu-2AsxAw,2253
|
|
18
|
+
dot_parser-2.0.0.dist-info/METADATA,sha256=yvMawlilJPvxb2_VMM_syz5i30mgxudIT2oRPe2B3lk,12199
|
|
19
|
+
dot_parser-2.0.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
20
|
+
dot_parser-2.0.0.dist-info/licenses/LICENSE.md,sha256=ADUqsZhl4juwq34PRTMiBqumpm11s_PMli_dZQjWPqQ,34260
|
|
21
|
+
dot_parser-2.0.0.dist-info/RECORD,,
|