dot-parser 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
dot_parser/pricing.py ADDED
@@ -0,0 +1,58 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ """Per-backend pricing helpers.
5
+
6
+ Exposes ``cost_per_1k_pages`` and ``estimate_cost``. Name-based so callers
7
+ can query pricing without instantiating a backend (no API key needed).
8
+
9
+ Sources: https://mistral.ai/news/ocr-4/ and the LlamaCloud pricing page.
10
+ """
11
+
12
+ from typing import Literal
13
+
14
+ BackendName = Literal["pymu", "docling", "mistral", "llama"]
15
+ LlamaTier = Literal["fast", "cost_effective", "agentic", "agentic_plus"]
16
+
17
+ _MISTRAL_SYNC = 4.0
18
+ _MISTRAL_BATCH = 2.0
19
+
20
+ # LlamaCloud bills in credits; rates below are $/1k pages assuming
21
+ # $0.001/credit (the public list price at the time of writing).
22
+ _LLAMA_PER_TIER: dict[str, float] = {
23
+ "fast": 1.0,
24
+ "cost_effective": 3.0,
25
+ "agentic": 10.0,
26
+ "agentic_plus": 45.0,
27
+ }
28
+
29
+
30
+ def cost_per_1k_pages(
31
+ name: BackendName,
32
+ *,
33
+ tier: LlamaTier = "cost_effective",
34
+ batch: bool = False,
35
+ ) -> float:
36
+ """Return USD cost per 1k pages for a backend.
37
+
38
+ ``tier`` applies only to ``llama``; ``batch=True`` applies only to
39
+ ``mistral`` (Batch API, 50% off sync). Both are ignored otherwise.
40
+ """
41
+ if name in ("pymu", "docling"):
42
+ return 0.0
43
+ if name == "mistral":
44
+ return _MISTRAL_BATCH if batch else _MISTRAL_SYNC
45
+ if name == "llama":
46
+ return _LLAMA_PER_TIER[tier]
47
+ raise ValueError(f"unknown backend: {name}")
48
+
49
+
50
+ def estimate_cost(
51
+ name: BackendName,
52
+ pages: int,
53
+ *,
54
+ tier: LlamaTier = "cost_effective",
55
+ batch: bool = False,
56
+ ) -> float:
57
+ """Return total USD cost for ``pages`` pages on ``name``."""
58
+ return pages * cost_per_1k_pages(name, tier=tier, batch=batch) / 1000
dot_parser/tokens.py ADDED
@@ -0,0 +1,6 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+
5
+ def count_tokens(text: str) -> int:
6
+ return max(1, len(text) // 4)
dot_parser/vlms.py ADDED
@@ -0,0 +1,203 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ """Ready-made `VLM` implementations.
5
+
6
+ `MistralVLM` is the implementation used for the DOCX image-extraction
7
+ path: it sends each image to Mistral's chat-completion API with a
8
+ short, deterministic prompt and returns a description string.
9
+
10
+ Kept separate from the Mistral OCR backend (`backends/mistral.py`)
11
+ because the two use different Mistral endpoints (`ocr.process` vs.
12
+ `chat.complete`) and have independent install/model requirements.
13
+ """
14
+
15
+ import os
16
+
17
+ from pydantic import BaseModel, Field
18
+
19
+ from dot_parser.image_utils import parse_image_annotation
20
+ from dot_parser.images import VLM, ImageDescription
21
+
22
+ # Per-call ceiling for a single image description. The Mistral SDK's own
23
+ # default is 300_000 ms, which is longer than most callers' entire
24
+ # request budget — one stuck image could consume all of it and starve
25
+ # every image behind it. Measured descriptions land at 0.7-8.5 s, so 60 s
26
+ # is roughly 7x the slowest observed call: generous enough never to cut
27
+ # off a legitimately slow one, short enough to fail fast. Pass
28
+ # ``timeout_ms=None`` to restore the SDK default.
29
+ DEFAULT_VLM_TIMEOUT_MS = 60_000
30
+
31
+ _DEFAULT_PROMPT = (
32
+ "Describe this image for a downstream RAG system. Populate the two "
33
+ "fields of the structured response as follows.\n\n"
34
+ "title: a short kebab-case slug (3-8 words, lowercase, "
35
+ "hyphen-separated, no file extension) naming what the image "
36
+ 'depicts (e.g. "system-architecture-diagram", '
37
+ '"exploded-gear-assembly-view", "monthly-revenue-bar-chart").\n\n'
38
+ "description: let image complexity set the length — do not pad "
39
+ "simple images and do not compress complex ones.\n"
40
+ " - Simple imagery (logo, photo, icon, single screenshot): 2-3 "
41
+ "concrete sentences.\n"
42
+ " - Diagrams, schematics, architecture views, flowcharts, "
43
+ "process maps: a structured paragraph that (a) names every "
44
+ "labeled node/block/component, (b) describes each connection/"
45
+ "arrow/flow (source -> target, with relationship label if any), "
46
+ "and (c) transcribes all visible text verbatim.\n"
47
+ " - Charts and plots: state the chart type, axes (with units), "
48
+ "series, and the notable values, trends, or extrema. Transcribe "
49
+ "the title, legend, and axis labels verbatim.\n"
50
+ " - Tables: transcribe the header row verbatim and summarize the "
51
+ "rows; if the table is small (<= 15 rows), transcribe it in full "
52
+ "as markdown inside the description.\n"
53
+ " - Screenshots of UI or code: transcribe all readable text "
54
+ "verbatim and name the visible UI elements or code constructs.\n\n"
55
+ "Always: be concrete, transcribe text exactly as it appears, do "
56
+ "not speculate about anything not visible."
57
+ )
58
+
59
+
60
+ def _language_directive(language: str) -> str:
61
+ """Instruction appended to a description prompt to fix the output language.
62
+
63
+ Only the model's own prose (``description``, and the ``title`` slug)
64
+ is forced into ``language``; text the model transcribes from the image
65
+ must stay in whatever language it appears in, so quoted labels and
66
+ captions are not silently translated.
67
+ """
68
+ return (
69
+ f"\n\nWrite the description and the title slug in {language}. "
70
+ "Text you transcribe verbatim from the image must stay in its "
71
+ "original language — only your own prose description is in "
72
+ f"{language}."
73
+ )
74
+
75
+
76
+ class _ImageDescriptionSchema(BaseModel):
77
+ """Structured-output schema for `chat.parse`.
78
+
79
+ Sending this class as `response_format=` puts the model in
80
+ constrained-decoding mode, so the emitted JSON is guaranteed to be
81
+ syntactically valid (newlines escaped) and to contain both keys
82
+ with string values. This is what prevents the long-description
83
+ failure mode where unescaped newlines broke `json.loads`.
84
+ """
85
+
86
+ title: str = Field(
87
+ description=(
88
+ "Short kebab-case slug, 3-8 words, lowercase, "
89
+ "hyphen-separated, no file extension. "
90
+ 'Example: "system-architecture-diagram".'
91
+ )
92
+ )
93
+ description: str = Field(
94
+ description=(
95
+ "Interpretation of the image content; length scales with "
96
+ "image complexity per the user prompt."
97
+ )
98
+ )
99
+
100
+
101
+ class MistralVLM(VLM):
102
+ """Vision-language model backed by Mistral's chat-completion API.
103
+
104
+ Used by `parse_with_images(format="docx", vlm=MistralVLM())` to
105
+ interpret each embedded image. PDF parsing through the Mistral OCR
106
+ backend does not need this — annotations there come back inline
107
+ from the OCR call itself.
108
+ """
109
+
110
+ def __init__(
111
+ self,
112
+ *,
113
+ api_key: str | None = None,
114
+ model: str = "mistral-medium-2604",
115
+ prompt: str = _DEFAULT_PROMPT,
116
+ language: str | None = None,
117
+ timeout_ms: int | None = DEFAULT_VLM_TIMEOUT_MS,
118
+ reasoning_effort: str | None = None,
119
+ ) -> None:
120
+ try:
121
+ from mistralai.client import Mistral as _Mistral
122
+ from mistralai.extra.utils.response_format import (
123
+ response_format_from_pydantic_model,
124
+ )
125
+ except ImportError as e:
126
+ raise ImportError(
127
+ "MistralVLM requires `mistralai`. Install with: pip install dot-parser[mistral]"
128
+ ) from e
129
+
130
+ key = api_key or os.environ.get("MISTRAL_API_KEY")
131
+ if not key:
132
+ raise ValueError(
133
+ "MistralVLM requires an API key. Set MISTRAL_API_KEY or pass api_key=..."
134
+ )
135
+ self._client = _Mistral(api_key=key)
136
+ self._model = model
137
+ self._timeout_ms = timeout_ms
138
+ # Only sent when set: the accepted values are per-model (reasoning
139
+ # models such as magistral/mistral-small-latest take "none" or
140
+ # "high" and reject "minimal"/"low"/"medium"; non-reasoning models
141
+ # reject the field outright), so an unconditional default would
142
+ # 400 on some models. None keeps the model's own default.
143
+ self._reasoning_effort = reasoning_effort
144
+ # `language`, when set, fixes the output language of the
145
+ # description (e.g. "French"); appended once here so it survives
146
+ # whether or not the caller also overrode `prompt`.
147
+ self._prompt = prompt + (_language_directive(language) if language else "")
148
+ # Precomputed JSON-schema response_format. Sending this on every
149
+ # chat.complete call asks the API for constrained decoding so the
150
+ # model emits a JSON object matching _ImageDescriptionSchema.
151
+ self._response_format = response_format_from_pydantic_model(_ImageDescriptionSchema)
152
+
153
+ def describe_image(
154
+ self,
155
+ image_base64: str,
156
+ *,
157
+ mime_type: str = "image/png",
158
+ context: str | None = None,
159
+ ) -> ImageDescription:
160
+ prompt = self._prompt
161
+ if context:
162
+ prompt = (
163
+ f"{prompt}\n\nThe image appears in a document near the "
164
+ f"following text (use it only to ground your description, "
165
+ f"do not repeat it):\n\n{context}"
166
+ )
167
+
168
+ data_url = f"data:{mime_type};base64,{image_base64}"
169
+ messages = [
170
+ {
171
+ "role": "user",
172
+ "content": [
173
+ {"type": "text", "text": prompt},
174
+ {"type": "image_url", "image_url": data_url},
175
+ ],
176
+ }
177
+ ]
178
+
179
+ # Constrained decoding: chat.complete with a json_schema
180
+ # response_format derived from _ImageDescriptionSchema. This is
181
+ # what chat.parse does internally, minus the Pydantic-coercion
182
+ # wrapper that was raising silently and dropping us into an
183
+ # unconstrained fallback. We parse the JSON ourselves so a
184
+ # slightly malformed payload can still yield a usable title.
185
+ optional: dict = {}
186
+ if self._reasoning_effort is not None:
187
+ optional["reasoning_effort"] = self._reasoning_effort
188
+ resp = self._client.chat.complete(
189
+ model=self._model,
190
+ messages=messages,
191
+ response_format=self._response_format,
192
+ timeout_ms=self._timeout_ms,
193
+ **optional,
194
+ )
195
+ content = resp.choices[0].message.content
196
+ if isinstance(content, list):
197
+ # Defensive: chat models occasionally return a list of chunks.
198
+ raw = "".join(
199
+ c.get("text", "") if isinstance(c, dict) else str(c) for c in content
200
+ ).strip()
201
+ else:
202
+ raw = (content or "").strip()
203
+ return parse_image_annotation(raw)
@@ -0,0 +1,274 @@
1
+ Metadata-Version: 2.5
2
+ Name: dot-parser
3
+ Version: 2.0.0
4
+ Summary: Document-to-markdown parser and chunker for RAG pipelines
5
+ Project-URL: Homepage, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
6
+ Project-URL: Repository, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
7
+ Project-URL: Issues, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/issues
8
+ Author-email: Kannon For Deep Tech <louis.letarnec@deepika.ai>
9
+ License-Expression: AGPL-3.0-or-later
10
+ License-File: LICENSE.md
11
+ Keywords: chunking,deepika,markdown,open-toolbox,parser,rag
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Programming Language :: Python :: 3.13
17
+ Requires-Python: <3.14,>=3.12
18
+ Requires-Dist: markitdown[all]>=0.1
19
+ Requires-Dist: pillow>=10.0
20
+ Requires-Dist: pydantic>=2
21
+ Requires-Dist: pymupdf4llm>=1.27.2.2
22
+ Requires-Dist: python-docx>=1.2.0
23
+ Requires-Dist: semchunk>=3.0
24
+ Requires-Dist: typing-extensions>=4.16
25
+ Provides-Extra: all
26
+ Requires-Dist: docling>=2.0; extra == 'all'
27
+ Requires-Dist: llama-cloud>=2.4; extra == 'all'
28
+ Requires-Dist: mistralai>=2.0; extra == 'all'
29
+ Provides-Extra: docling
30
+ Requires-Dist: docling>=2.0; extra == 'docling'
31
+ Provides-Extra: llama
32
+ Requires-Dist: llama-cloud>=2.4; extra == 'llama'
33
+ Provides-Extra: mistral
34
+ Requires-Dist: mistralai>=2.0; extra == 'mistral'
35
+ Description-Content-Type: text/markdown
36
+
37
+ # dot-parser
38
+
39
+ [![PyPI](https://img.shields.io/pypi/v/dot-parser)](https://pypi.org/project/dot-parser/)
40
+ ![Python Version](https://img.shields.io/badge/python-3.12%2B-blue)
41
+ [![Licence: AGPL v3](https://img.shields.io/badge/licence-AGPL--3.0--or--later-blue)](LICENSE.md)
42
+ [![Pipeline](https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/badges/main/pipeline.svg)](https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/pipelines)
43
+
44
+ **Turn documents into clean Markdown, then into retrieval-ready chunks.**
45
+
46
+ ```python
47
+ from dot_parser import parse, chunk
48
+
49
+ markdown = parse("report.pdf")
50
+ chunks = chunk(markdown)
51
+
52
+ for c in chunks:
53
+ print(c.section_path, len(c.content))
54
+ # ['# Report', '## Methods', '### Analysis'] 842
55
+ ```
56
+
57
+ ## Why dot-parser
58
+
59
+ A RAG pipeline needs two things from a document: faithful text, and chunks that
60
+ keep their place in the document's structure. Most tools give you one or the
61
+ other — parsers stop at raw text, splitters assume you already have Markdown and
62
+ cut it blind to headings.
63
+
64
+ dot-parser does both in one step. It converts PDF, DOCX, PPTX, HTML, XLSX, CSV,
65
+ Markdown and plain text into clean Markdown, then splits that Markdown into
66
+ chunks carrying their heading hierarchy in `section_path` — so a chunk still
67
+ knows it came from *Report › Methods › Analysis*.
68
+
69
+ The default install stays light: heavy backends are optional extras, and you pick
70
+ per document whether to run locally or through a cloud OCR. See
71
+ [docs/DESIGN.md](docs/DESIGN.md) for how it compares to Docling, MarkItDown and
72
+ LangChain splitters.
73
+
74
+ ## Features
75
+
76
+ - One `parse()` call for PDF, DOCX, PPTX, HTML, XLSX, CSV, Markdown and text
77
+ - Swappable PDF backends: local and fast, local and layout-aware, or cloud OCR
78
+ - Heading-aware chunking with `section_path` metadata
79
+ - Image extraction with per-image descriptions, from a VLM or from OCR annotations
80
+ - Batch PDF parsing through the Mistral Batch API (50% cheaper)
81
+ - Cost estimation before you send anything
82
+ - Light by default — heavy backends live behind extras
83
+
84
+ ## Installation
85
+
86
+ ```bash
87
+ pip install dot-parser
88
+
89
+ # With optional PDF backends (each ~50 MB - 1 GB):
90
+ pip install 'dot-parser[docling]'
91
+ pip install 'dot-parser[mistral]'
92
+ pip install 'dot-parser[llama]'
93
+ pip install 'dot-parser[all]'
94
+ ```
95
+
96
+ Requires Python 3.12+.
97
+
98
+ ## Quick start
99
+
100
+ ```python
101
+ from dot_parser import parse, chunk
102
+
103
+ markdown: str = parse("report.pdf")
104
+ chunks: list[Chunk] = chunk(markdown)
105
+
106
+ for c in chunks:
107
+ print(c.section_path, c.heading, len(c.content))
108
+ # ["# Report", "## Methods", "### Analysis"], "### Analysis", 842
109
+ ```
110
+
111
+ For a different PDF backend:
112
+
113
+ ```python
114
+ from dot_parser import parse, parse_pdfs, Mistral, Docling
115
+
116
+ # single PDF, cloud OCR
117
+ md = parse("scanned.pdf", backend=Mistral())
118
+
119
+ # many PDFs, batched (50% off via Mistral Batch API)
120
+ mds = parse_pdfs(["a.pdf", "b.pdf", "c.pdf"], backend=Mistral())
121
+
122
+ # local layout-aware ML pipeline
123
+ md = parse("paper.pdf", backend=Docling())
124
+ ```
125
+
126
+ ## API
127
+
128
+ ### `parse(source, format=None, backend=None) -> str`
129
+
130
+ Converts a document to Markdown.
131
+
132
+ - **source** -- file path (`str` or `Path`) or raw `bytes`
133
+ - **format** -- explicit format hint (e.g. `"pdf"`, `"docx"`). Required when source is `bytes`, otherwise inferred from the file extension.
134
+ - **backend** -- optional PDF backend (`Pymu`, `Docling`, `Mistral`, `Llama`). Defaults to `Pymu()`. Backends only apply to PDF input.
135
+
136
+ Supported formats: `.pdf`, `.md`, `.txt`, `.docx`, `.pptx`, `.html`, `.xhtml`, `.htm`, `.xlsx`, `.csv`
137
+
138
+ Raises `ParseError` on unsupported formats or conversion failures.
139
+
140
+ ### `parse_pdfs(sources, backend=None) -> list[str | None]`
141
+
142
+ Converts a list of PDFs to Markdown, in input order. Backends that implement a `parse_pdfs` method (e.g. `Mistral` via the Batch API, 50% off) handle the whole list in a single optimized call. Otherwise falls back to looping `parse()` per source. Failed items are returned as `None` (no exception raised).
143
+
144
+ ### `parse_with_images(source, format=None, *, backend=None, vlm=None, annotate_images=True, include_tables=True) -> ParseResult`
145
+
146
+ Converts a `.pdf`, `.docx` or `.pptx` and extracts its images alongside the
147
+ Markdown, keeping `![name](name)` anchors at each image's position. Each image is
148
+ described either by a VLM or by OCR annotations from the same call. See
149
+ [docs/IMAGE_EXTRACTION.md](docs/IMAGE_EXTRACTION.md) for the two available paths per format.
150
+
151
+ ### Backends
152
+
153
+ - **`Pymu(ocr_fallback=True)`** -- pymupdf4llm, default. Fast and CPU-only. The built-in OCR fallback kicks in when an OCR engine (tesseract, rapidocr, paddleocr) is available. Set `ocr_fallback=False` to force the no-OCR code path (matches a runtime image without OCR shipped).
154
+ - **`Docling(force_full_page_ocr=True)`** -- IBM Docling, layout-aware ML pipeline + Tesseract OCR. Robust on image-only and broken-encoding PDFs. Install with `pip install dot-parser[docling]`.
155
+ - **`Mistral(api_key=None, model="mistral-ocr-4-0")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
156
+ - **`Llama(api_key=None, tier="cost_effective")`** -- LlamaCloud parsing API. Reads `LLAMA_CLOUD_API_KEY` from env. Tiers: `fast`, `cost_effective`, `agentic`, `agentic_plus`. Install with `pip install dot-parser[llama]`.
157
+
158
+ #### Mistral OCR enrichments
159
+
160
+ Three optional extras on the `Mistral(...)` constructor, applying to the `*_with_images` methods. All default to off, so existing callers see no change:
161
+
162
+ ```python
163
+ from dot_parser import parse_with_images, Mistral
164
+
165
+ result = parse_with_images(
166
+ "spec.pdf",
167
+ backend=Mistral(
168
+ extract_headers_footers=True, # page furniture -> result.pages[i].header/.footer
169
+ confidence_scores="page", # or "word" -> result.pages[i].*_confidence
170
+ extract_captions=True, # figure captions -> image.original_caption
171
+ ),
172
+ )
173
+
174
+ for page in result.pages:
175
+ print(page.page, page.average_confidence, page.minimum_confidence)
176
+ ```
177
+
178
+ - `extract_headers_footers` moves running headers/footers out of the markdown and into `ParseResult.pages`, so repeated "page 4 of 27" noise stays out of retrieval chunks. It **removes that text from the markdown** -- the only one of the three that changes `result.markdown`.
179
+ - `confidence_scores` surfaces OCR's self-reported quality. **Calibrate any threshold on your own corpus -- absolute cutoffs do not transfer.** Measured over the 39 pages of the benchmark corpus (`native_simple`, `image_only`, `broken_encoding`), `average_confidence` sat at 0.985-0.989 and `minimum_confidence` at 0.23-0.46 on *every* document regardless of quality. `minimum_confidence` is the single worst region on the page, so it is low even on clean pages and is not a page-quality alarm by itself.
180
+ - `extract_captions` fills `ExtractedImage.original_caption` from the figure caption printed beside each image. Needs OCR 4+ **and** `mistralai>=2.8`; on older SDKs it logs a warning and leaves captions `None` rather than failing.
181
+
182
+ If the OCR model rejects any of these parameters, the call is retried once without them -- you get the markdown and images, minus the enrichments.
183
+
184
+ #### Caption-grounded image interpretation
185
+
186
+ Mistral OCR's inline annotations (`annotate_images=True`) are produced from the cropped image alone -- the annotating model sees no caption and no page text, which on domain-specific figures yields confidently wrong descriptions. `interpret_images()` replaces that pass with client-side VLM calls grounded in each figure's own caption and surrounding markdown:
187
+
188
+ ```python
189
+ from dot_parser import parse_with_images, interpret_images, Mistral, MistralVLM
190
+
191
+ result = parse_with_images(
192
+ "paper.pdf",
193
+ backend=Mistral(extract_captions=True),
194
+ annotate_images=False, # skip the ungrounded inline pass
195
+ )
196
+ result = interpret_images(result, MistralVLM())
197
+ ```
198
+
199
+ Same number of VLM calls as `annotate_images=True`, but each one carries the caption. On a figure-heavy biomechanics paper this turned "magnetic field components" / "fluid flow through a curved pipe" into accurate descriptions of the actual rib diagrams.
200
+
201
+ ### `chunk(markdown, max_tokens=None, merge=True) -> list[Chunk]`
202
+
203
+ Splits Markdown into chunks with heading-hierarchy metadata.
204
+
205
+ - **markdown** -- Markdown string (typically output of `parse()`)
206
+ - **max_tokens** -- maximum tokens per chunk. Auto-computed from the 95th percentile of section sizes (clamped 64-512) when `None`.
207
+ - **merge** -- when `True` (default), merges small adjacent sections sharing the same parent heading to reduce fragmentation.
208
+
209
+ The chunking algorithm is described in [docs/DESIGN.md](docs/DESIGN.md).
210
+
211
+ ### `Chunk`
212
+
213
+ ```python
214
+ @dataclass(frozen=True)
215
+ class Chunk:
216
+ content: str # full text of the chunk (heading + body)
217
+ heading: str | None # the heading line, if any
218
+ section_path: list[str] # hierarchy of headings leading to this chunk
219
+ ```
220
+
221
+ ### `ParseError`
222
+
223
+ Raised by `parse()` when a document cannot be converted.
224
+
225
+ ### Cost estimation
226
+
227
+ `cost_per_1k_pages(backend)` and `estimate_cost(backend, pages, batch=False)`
228
+ report what a cloud backend will cost, without instantiating it or needing an
229
+ API key.
230
+
231
+ ## Stability
232
+
233
+ `dot-parser` follows semantic versioning: everything exported from the top-level
234
+ package is covered, anything underscore-prefixed is internal and may change in
235
+ any release. Public names are never removed without a deprecation period.
236
+
237
+ ```toml
238
+ dependencies = ["dot-parser>=1.0,<2"]
239
+ ```
240
+
241
+ See [docs/VERSIONING.md](docs/VERSIONING.md) for the full policy.
242
+
243
+ ## Roadmap
244
+
245
+ - [ ] Table extraction as structured data, not just Markdown
246
+ - [ ] Streaming parse for very large documents
247
+ - [ ] Additional cloud OCR backends
248
+
249
+ ## Documentation
250
+
251
+ | Document | Contents |
252
+ |---|---|
253
+ | [docs/DESIGN.md](docs/DESIGN.md) | Design rationale, backend choices, chunking algorithm |
254
+ | [docs/IMAGE_EXTRACTION.md](docs/IMAGE_EXTRACTION.md) | Image-extraction paths per format |
255
+ | [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) | Environment setup, tests, code style |
256
+ | [docs/VERSIONING.md](docs/VERSIONING.md) | Versioning, deprecation policy, how to depend on this package |
257
+ | [docs/PUBLISHING.md](docs/PUBLISHING.md) | Cutting a release |
258
+ | [CHANGELOG.md](CHANGELOG.md) | Release history |
259
+
260
+ ## Contributing
261
+
262
+ Contributions are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md) for the DCO
263
+ sign-off requirement, the licensing terms that apply to contributions, and how to
264
+ submit a change.
265
+
266
+ ## Licence
267
+
268
+ Copyright (C) 2026 Kannon For Deep Tech (deepika)
269
+
270
+ This software is distributed under the GNU Affero General Public License,
271
+ version 3 or later — see [LICENSE.md](LICENSE.md).
272
+
273
+ A commercial licence is available for use in proprietary environments.
274
+ Contact: louis.letarnec@deepika.ai
@@ -0,0 +1,21 @@
1
+ dot_parser/__init__.py,sha256=oAWCCkbHhoTSbhilZwVM66GBBcGkUq0wwVlZHD1thnQ,1052
2
+ dot_parser/chunking.py,sha256=nvLyCZz8h3EVEsbR4oFu_kE58CLBvWbcYQkwzZD-xKQ,8208
3
+ dot_parser/docx_images.py,sha256=tXBl8OiKazHt_1wbOtlaOTlBXu4gIZ-2zwEcG9xxdTM,26018
4
+ dot_parser/image_utils.py,sha256=26uiRuz6j-ZLWvGshHKbfvRJWQxoh2DfiKLX33k-XHY,6037
5
+ dot_parser/images.py,sha256=OQ64cbn2-nFn9hKjPLM8cO5gBTDzkcBVL0ArFThnbGA,4651
6
+ dot_parser/markdown_utils.py,sha256=KXjLtneqDzPLbiMfND1TEmEHum1oRk0vym0h4qgf9K0,1789
7
+ dot_parser/models.py,sha256=OMiCL7XqLki0iie0K3TrLw0iDTvt0BjeWnopLGYu004,561
8
+ dot_parser/parsers.py,sha256=UurXUTU_Za5DSZat07tQB9fSm5IfSbhLgN2XHICQlmI,12286
9
+ dot_parser/pricing.py,sha256=cjqM-gaZ-GQyGibbUQwWpqIECv33j4OJ3_Shvp33OXc,1684
10
+ dot_parser/tokens.py,sha256=LIERtlBVtN_loTOMUyuDDxuUr_p4q8YSdmxIEI_Uhts,164
11
+ dot_parser/vlms.py,sha256=cACc_Psy8tJggNXPdHXZ3NY9kneg_oCtNmtGYljNh-w,8636
12
+ dot_parser/backends/__init__.py,sha256=nmJciV1nfu8__4Ls37xwh1kvc1yrhpscHoInMaO7c-o,583
13
+ dot_parser/backends/_base.py,sha256=rYmdLAJOoZKpVbm9njGJ-PjjNl8Sp2hi65AiCP_7CW8,2257
14
+ dot_parser/backends/docling.py,sha256=0OoucXeY0AqwRFk1aEbSzP7g9z7s8OuAfUzOz7AHwMw,1976
15
+ dot_parser/backends/llama.py,sha256=5vlGbTr8jJB_ms1ZXp0HGAR59MvM-iNLU67Cq3neHg8,2634
16
+ dot_parser/backends/mistral.py,sha256=cRg_g5szAGtzFNT3mC3KFSfpjKPJJpzBROCWcjxl1_c,25801
17
+ dot_parser/backends/pymu.py,sha256=EIg4JWwyNdjkbKOlzxWLxERmyZAMIUNxulQu-2AsxAw,2253
18
+ dot_parser-2.0.0.dist-info/METADATA,sha256=yvMawlilJPvxb2_VMM_syz5i30mgxudIT2oRPe2B3lk,12199
19
+ dot_parser-2.0.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
20
+ dot_parser-2.0.0.dist-info/licenses/LICENSE.md,sha256=ADUqsZhl4juwq34PRTMiBqumpm11s_PMli_dZQjWPqQ,34260
21
+ dot_parser-2.0.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any