dot-parser 2.1.0__tar.gz → 2.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dot_parser-2.1.0 → dot_parser-2.1.1}/CHANGELOG.md +9 -1
- {dot_parser-2.1.0 → dot_parser-2.1.1}/PKG-INFO +2 -2
- {dot_parser-2.1.0 → dot_parser-2.1.1}/README.md +1 -1
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/README.md +1 -1
- {dot_parser-2.1.0 → dot_parser-2.1.1}/docs/IMAGE_EXTRACTION.md +2 -2
- {dot_parser-2.1.0 → dot_parser-2.1.1}/pyproject.toml +1 -1
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/backends/mistral.py +1 -1
- {dot_parser-2.1.0 → dot_parser-2.1.1}/tests/test_images.py +3 -3
- {dot_parser-2.1.0 → dot_parser-2.1.1}/tests/test_mistral_batch_logging.py +1 -1
- {dot_parser-2.1.0 → dot_parser-2.1.1}/uv.lock +1 -1
- {dot_parser-2.1.0 → dot_parser-2.1.1}/.gitignore +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/.gitlab-ci.yml +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/.pre-commit-config.yaml +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/.python-version +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/AUTHORS.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/CONTRIBUTING.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/DCO +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/LICENSE.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/input/broken_encoding.pdf +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/input/image_only.pdf +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/input/native_equations.pdf +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/input/native_simple.pdf +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/input/native_tables.pdf +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_broken_encoding.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_image_only.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_native_equations.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_native_simple.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_native_tables.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_broken_encoding.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_image_only.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_native_equations.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_native_simple.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_native_tables.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_broken_encoding.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_image_only.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_equations.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_simple.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_tables.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_broken_encoding.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_image_only.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_native_equations.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_native_simple.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_native_tables.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_broken_encoding.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_image_only.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_native_equations.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_native_simple.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_native_tables.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/results.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/run.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/docs/DESIGN.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/docs/DEVELOPMENT.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/docs/PUBLISHING.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/docs/VERSIONING.md +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/__init__.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/backends/__init__.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/backends/_base.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/backends/docling.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/backends/llama.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/backends/pymu.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/chunking.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/docx_images.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/docx_markdown.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/image_utils.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/images.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/markdown_utils.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/models.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/parsers.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/pricing.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/tokens.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/vlms.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/tests/test_chunking.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/tests/test_docx_markdown.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/tests/test_parsers.py +0 -0
- {dot_parser-2.1.0 → dot_parser-2.1.1}/tests/test_tokens.py +0 -0
|
@@ -7,6 +7,13 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [2.1.1] - 2026-09-30
|
|
11
|
+
|
|
12
|
+
### Changed
|
|
13
|
+
|
|
14
|
+
- `Mistral` now defaults to `mistral-ocr-4-1`; `mistral-ocr-4-0` retires on
|
|
15
|
+
2026-09-30. Pricing is unchanged.
|
|
16
|
+
|
|
10
17
|
## [2.1.0] - 2026-09-30
|
|
11
18
|
|
|
12
19
|
### Added
|
|
@@ -83,6 +90,7 @@ this release.
|
|
|
83
90
|
- `parse_with_images(backend=...)` accepts DOCX- and PPTX-capable backends in
|
|
84
91
|
its type signature, matching what the function already supported at runtime.
|
|
85
92
|
|
|
86
|
-
[Unreleased]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.1.
|
|
93
|
+
[Unreleased]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.1.1...main
|
|
94
|
+
[2.1.1]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.1.0...v2.1.1
|
|
87
95
|
[2.1.0]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.0.0...v2.1.0
|
|
88
96
|
[2.0.0]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/tags/v2.0.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: dot-parser
|
|
3
|
-
Version: 2.1.
|
|
3
|
+
Version: 2.1.1
|
|
4
4
|
Summary: Document-to-markdown parser and chunker for RAG pipelines
|
|
5
5
|
Project-URL: Homepage, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
|
|
6
6
|
Project-URL: Repository, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
|
|
@@ -154,7 +154,7 @@ described either by a VLM or by OCR annotations from the same call. See
|
|
|
154
154
|
|
|
155
155
|
- **`Pymu(ocr_fallback=True)`** -- pymupdf4llm, default. Fast and CPU-only. The built-in OCR fallback kicks in when an OCR engine (tesseract, rapidocr, paddleocr) is available. Set `ocr_fallback=False` to force the no-OCR code path (matches a runtime image without OCR shipped).
|
|
156
156
|
- **`Docling(force_full_page_ocr=True)`** -- IBM Docling, layout-aware ML pipeline + Tesseract OCR. Robust on image-only and broken-encoding PDFs. Install with `pip install dot-parser[docling]`.
|
|
157
|
-
- **`Mistral(api_key=None, model="mistral-ocr-4-
|
|
157
|
+
- **`Mistral(api_key=None, model="mistral-ocr-4-1")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
|
|
158
158
|
- **`Llama(api_key=None, tier="cost_effective")`** -- LlamaCloud parsing API. Reads `LLAMA_CLOUD_API_KEY` from env. Tiers: `fast`, `cost_effective`, `agentic`, `agentic_plus`. Install with `pip install dot-parser[llama]`.
|
|
159
159
|
|
|
160
160
|
#### Mistral OCR enrichments
|
|
@@ -116,7 +116,7 @@ described either by a VLM or by OCR annotations from the same call. See
|
|
|
116
116
|
|
|
117
117
|
- **`Pymu(ocr_fallback=True)`** -- pymupdf4llm, default. Fast and CPU-only. The built-in OCR fallback kicks in when an OCR engine (tesseract, rapidocr, paddleocr) is available. Set `ocr_fallback=False` to force the no-OCR code path (matches a runtime image without OCR shipped).
|
|
118
118
|
- **`Docling(force_full_page_ocr=True)`** -- IBM Docling, layout-aware ML pipeline + Tesseract OCR. Robust on image-only and broken-encoding PDFs. Install with `pip install dot-parser[docling]`.
|
|
119
|
-
- **`Mistral(api_key=None, model="mistral-ocr-4-
|
|
119
|
+
- **`Mistral(api_key=None, model="mistral-ocr-4-1")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
|
|
120
120
|
- **`Llama(api_key=None, tier="cost_effective")`** -- LlamaCloud parsing API. Reads `LLAMA_CLOUD_API_KEY` from env. Tiers: `fast`, `cost_effective`, `agentic`, `agentic_plus`. Install with `pip install dot-parser[llama]`.
|
|
121
121
|
|
|
122
122
|
#### Mistral OCR enrichments
|
|
@@ -72,5 +72,5 @@ dropped equations, wrong table cells), so we keep it purely manual.
|
|
|
72
72
|
| `pymu` | $0.00 | pymupdf4llm with OCR fallback **disabled** — reproduces le-lab production where no OCR engine is shipped in the runtime image |
|
|
73
73
|
| `pymu_ocr` | $0.00 | pymupdf4llm with built-in OCR fallback **enabled** — uses tesseract (or rapidocr/paddleocr) automatically when text extraction looks suspect |
|
|
74
74
|
| `docling` | $0.00 | layout-aware ML pipeline + Tesseract OCR with `force_full_page_ocr=True` |
|
|
75
|
-
| `mistral` | $4.00 | `mistral-ocr-4-
|
|
75
|
+
| `mistral` | $4.00 | `mistral-ocr-4-1` cloud API. With `--mistral-batch`: $2.00 per 1k pages (50% off) via the Mistral Batch API. |
|
|
76
76
|
| `llama` | $3.00 | LlamaCloud `cost_effective` tier |
|
|
@@ -30,7 +30,7 @@ parse_with_images("doc.docx", backend=Mistral())
|
|
|
30
30
|
|
|
31
31
|
The `.docx` is uploaded to Mistral and `ocr.process` is called with `document_url`
|
|
32
32
|
plus `bbox_annotation_format`: a single call returns markdown, images and
|
|
33
|
-
annotations. Model: `mistral-ocr-
|
|
33
|
+
annotations. Model: `mistral-ocr-4-1`.
|
|
34
34
|
|
|
35
35
|
- **+** One API call, no per-image round trip.
|
|
36
36
|
- **+** Handles natively the formats markitdown cannot pass through (the document is rasterised, so SmartArt / EMF / SVG render correctly).
|
|
@@ -68,7 +68,7 @@ parse_with_images("doc.pdf") # backend=Mistral() implied
|
|
|
68
68
|
parse_with_images("doc.pdf", backend=Mistral()) # explicit, identical
|
|
69
69
|
```
|
|
70
70
|
|
|
71
|
-
Same `bbox_annotation_format` flow as DOCX path B. Model: `mistral-ocr-
|
|
71
|
+
Same `bbox_annotation_format` flow as DOCX path B. Model: `mistral-ocr-4-1`.
|
|
72
72
|
|
|
73
73
|
- **+** One call returns markdown, images and annotations; robust on scanned or broken-encoding PDFs; exact image positions.
|
|
74
74
|
- **−** Billed per page; fixed annotation schema (same caveat as DOCX path B).
|
|
@@ -217,7 +217,7 @@ class Mistral:
|
|
|
217
217
|
self,
|
|
218
218
|
*,
|
|
219
219
|
api_key: str | None = None,
|
|
220
|
-
model: str = "mistral-ocr-4-
|
|
220
|
+
model: str = "mistral-ocr-4-1",
|
|
221
221
|
language: str | None = None,
|
|
222
222
|
extract_headers_footers: bool = False,
|
|
223
223
|
confidence_scores: Literal["page", "word"] | None = None,
|
|
@@ -129,7 +129,7 @@ def _make_mistral_with_fake_client(pages):
|
|
|
129
129
|
constructing the instance manually.
|
|
130
130
|
"""
|
|
131
131
|
instance = Mistral.__new__(Mistral)
|
|
132
|
-
instance._model = "mistral-ocr-4-
|
|
132
|
+
instance._model = "mistral-ocr-4-1"
|
|
133
133
|
instance._language = None
|
|
134
134
|
|
|
135
135
|
uploaded = SimpleNamespace(id="file-xyz")
|
|
@@ -261,7 +261,7 @@ class TestMistralParsePdfWithImages:
|
|
|
261
261
|
]
|
|
262
262
|
|
|
263
263
|
instance = Mistral.__new__(Mistral)
|
|
264
|
-
instance._model = "mistral-ocr-4-
|
|
264
|
+
instance._model = "mistral-ocr-4-1"
|
|
265
265
|
instance._language = None
|
|
266
266
|
uploaded = SimpleNamespace(id="file-xyz")
|
|
267
267
|
signed = SimpleNamespace(url="https://signed.example/doc.pdf")
|
|
@@ -297,7 +297,7 @@ class TestMistralParsePdfWithImages:
|
|
|
297
297
|
def test_retry_failure_propagates(self):
|
|
298
298
|
"""If the no-annotation retry also fails, the exception escapes."""
|
|
299
299
|
instance = Mistral.__new__(Mistral)
|
|
300
|
-
instance._model = "mistral-ocr-4-
|
|
300
|
+
instance._model = "mistral-ocr-4-1"
|
|
301
301
|
instance._language = None
|
|
302
302
|
ocr_calls: list[dict] = []
|
|
303
303
|
|
|
@@ -63,7 +63,7 @@ def _backend(job, downloads: dict[str, bytes]) -> Mistral:
|
|
|
63
63
|
no ``mistralai`` import / real key is needed."""
|
|
64
64
|
m = Mistral.__new__(Mistral)
|
|
65
65
|
m._client = _FakeClient(job, downloads) # ty: ignore[invalid-assignment]
|
|
66
|
-
m._model = "mistral-ocr-4-
|
|
66
|
+
m._model = "mistral-ocr-4-1"
|
|
67
67
|
return m
|
|
68
68
|
|
|
69
69
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_broken_encoding.md
RENAMED
|
File without changes
|
|
File without changes
|
{dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_equations.md
RENAMED
|
File without changes
|
{dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_simple.md
RENAMED
|
File without changes
|
{dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_tables.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|