dot-parser 2.1.0__tar.gz → 2.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. {dot_parser-2.1.0 → dot_parser-2.1.1}/CHANGELOG.md +9 -1
  2. {dot_parser-2.1.0 → dot_parser-2.1.1}/PKG-INFO +2 -2
  3. {dot_parser-2.1.0 → dot_parser-2.1.1}/README.md +1 -1
  4. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/README.md +1 -1
  5. {dot_parser-2.1.0 → dot_parser-2.1.1}/docs/IMAGE_EXTRACTION.md +2 -2
  6. {dot_parser-2.1.0 → dot_parser-2.1.1}/pyproject.toml +1 -1
  7. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/backends/mistral.py +1 -1
  8. {dot_parser-2.1.0 → dot_parser-2.1.1}/tests/test_images.py +3 -3
  9. {dot_parser-2.1.0 → dot_parser-2.1.1}/tests/test_mistral_batch_logging.py +1 -1
  10. {dot_parser-2.1.0 → dot_parser-2.1.1}/uv.lock +1 -1
  11. {dot_parser-2.1.0 → dot_parser-2.1.1}/.gitignore +0 -0
  12. {dot_parser-2.1.0 → dot_parser-2.1.1}/.gitlab-ci.yml +0 -0
  13. {dot_parser-2.1.0 → dot_parser-2.1.1}/.pre-commit-config.yaml +0 -0
  14. {dot_parser-2.1.0 → dot_parser-2.1.1}/.python-version +0 -0
  15. {dot_parser-2.1.0 → dot_parser-2.1.1}/AUTHORS.md +0 -0
  16. {dot_parser-2.1.0 → dot_parser-2.1.1}/CONTRIBUTING.md +0 -0
  17. {dot_parser-2.1.0 → dot_parser-2.1.1}/DCO +0 -0
  18. {dot_parser-2.1.0 → dot_parser-2.1.1}/LICENSE.md +0 -0
  19. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/input/broken_encoding.pdf +0 -0
  20. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/input/image_only.pdf +0 -0
  21. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/input/native_equations.pdf +0 -0
  22. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/input/native_simple.pdf +0 -0
  23. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/input/native_tables.pdf +0 -0
  24. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_broken_encoding.md +0 -0
  25. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_image_only.md +0 -0
  26. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_native_equations.md +0 -0
  27. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_native_simple.md +0 -0
  28. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_native_tables.md +0 -0
  29. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_broken_encoding.md +0 -0
  30. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_image_only.md +0 -0
  31. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_native_equations.md +0 -0
  32. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_native_simple.md +0 -0
  33. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_native_tables.md +0 -0
  34. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_broken_encoding.md +0 -0
  35. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_image_only.md +0 -0
  36. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_equations.md +0 -0
  37. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_simple.md +0 -0
  38. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_tables.md +0 -0
  39. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_broken_encoding.md +0 -0
  40. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_image_only.md +0 -0
  41. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_native_equations.md +0 -0
  42. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_native_simple.md +0 -0
  43. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_native_tables.md +0 -0
  44. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_broken_encoding.md +0 -0
  45. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_image_only.md +0 -0
  46. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_native_equations.md +0 -0
  47. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_native_simple.md +0 -0
  48. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_native_tables.md +0 -0
  49. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/output/results.md +0 -0
  50. {dot_parser-2.1.0 → dot_parser-2.1.1}/benchmark/run.py +0 -0
  51. {dot_parser-2.1.0 → dot_parser-2.1.1}/docs/DESIGN.md +0 -0
  52. {dot_parser-2.1.0 → dot_parser-2.1.1}/docs/DEVELOPMENT.md +0 -0
  53. {dot_parser-2.1.0 → dot_parser-2.1.1}/docs/PUBLISHING.md +0 -0
  54. {dot_parser-2.1.0 → dot_parser-2.1.1}/docs/VERSIONING.md +0 -0
  55. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/__init__.py +0 -0
  56. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/backends/__init__.py +0 -0
  57. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/backends/_base.py +0 -0
  58. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/backends/docling.py +0 -0
  59. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/backends/llama.py +0 -0
  60. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/backends/pymu.py +0 -0
  61. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/chunking.py +0 -0
  62. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/docx_images.py +0 -0
  63. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/docx_markdown.py +0 -0
  64. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/image_utils.py +0 -0
  65. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/images.py +0 -0
  66. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/markdown_utils.py +0 -0
  67. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/models.py +0 -0
  68. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/parsers.py +0 -0
  69. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/pricing.py +0 -0
  70. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/tokens.py +0 -0
  71. {dot_parser-2.1.0 → dot_parser-2.1.1}/src/dot_parser/vlms.py +0 -0
  72. {dot_parser-2.1.0 → dot_parser-2.1.1}/tests/test_chunking.py +0 -0
  73. {dot_parser-2.1.0 → dot_parser-2.1.1}/tests/test_docx_markdown.py +0 -0
  74. {dot_parser-2.1.0 → dot_parser-2.1.1}/tests/test_parsers.py +0 -0
  75. {dot_parser-2.1.0 → dot_parser-2.1.1}/tests/test_tokens.py +0 -0
@@ -7,6 +7,13 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [2.1.1] - 2026-09-30
11
+
12
+ ### Changed
13
+
14
+ - `Mistral` now defaults to `mistral-ocr-4-1`; `mistral-ocr-4-0` retires on
15
+ 2026-09-30. Pricing is unchanged.
16
+
10
17
  ## [2.1.0] - 2026-09-30
11
18
 
12
19
  ### Added
@@ -83,6 +90,7 @@ this release.
83
90
  - `parse_with_images(backend=...)` accepts DOCX- and PPTX-capable backends in
84
91
  its type signature, matching what the function already supported at runtime.
85
92
 
86
- [Unreleased]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.1.0...main
93
+ [Unreleased]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.1.1...main
94
+ [2.1.1]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.1.0...v2.1.1
87
95
  [2.1.0]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.0.0...v2.1.0
88
96
  [2.0.0]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/tags/v2.0.0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: dot-parser
3
- Version: 2.1.0
3
+ Version: 2.1.1
4
4
  Summary: Document-to-markdown parser and chunker for RAG pipelines
5
5
  Project-URL: Homepage, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
6
6
  Project-URL: Repository, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
@@ -154,7 +154,7 @@ described either by a VLM or by OCR annotations from the same call. See
154
154
 
155
155
  - **`Pymu(ocr_fallback=True)`** -- pymupdf4llm, default. Fast and CPU-only. The built-in OCR fallback kicks in when an OCR engine (tesseract, rapidocr, paddleocr) is available. Set `ocr_fallback=False` to force the no-OCR code path (matches a runtime image without OCR shipped).
156
156
  - **`Docling(force_full_page_ocr=True)`** -- IBM Docling, layout-aware ML pipeline + Tesseract OCR. Robust on image-only and broken-encoding PDFs. Install with `pip install dot-parser[docling]`.
157
- - **`Mistral(api_key=None, model="mistral-ocr-4-0")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
157
+ - **`Mistral(api_key=None, model="mistral-ocr-4-1")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
158
158
  - **`Llama(api_key=None, tier="cost_effective")`** -- LlamaCloud parsing API. Reads `LLAMA_CLOUD_API_KEY` from env. Tiers: `fast`, `cost_effective`, `agentic`, `agentic_plus`. Install with `pip install dot-parser[llama]`.
159
159
 
160
160
  #### Mistral OCR enrichments
@@ -116,7 +116,7 @@ described either by a VLM or by OCR annotations from the same call. See
116
116
 
117
117
  - **`Pymu(ocr_fallback=True)`** -- pymupdf4llm, default. Fast and CPU-only. The built-in OCR fallback kicks in when an OCR engine (tesseract, rapidocr, paddleocr) is available. Set `ocr_fallback=False` to force the no-OCR code path (matches a runtime image without OCR shipped).
118
118
  - **`Docling(force_full_page_ocr=True)`** -- IBM Docling, layout-aware ML pipeline + Tesseract OCR. Robust on image-only and broken-encoding PDFs. Install with `pip install dot-parser[docling]`.
119
- - **`Mistral(api_key=None, model="mistral-ocr-4-0")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
119
+ - **`Mistral(api_key=None, model="mistral-ocr-4-1")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
120
120
  - **`Llama(api_key=None, tier="cost_effective")`** -- LlamaCloud parsing API. Reads `LLAMA_CLOUD_API_KEY` from env. Tiers: `fast`, `cost_effective`, `agentic`, `agentic_plus`. Install with `pip install dot-parser[llama]`.
121
121
 
122
122
  #### Mistral OCR enrichments
@@ -72,5 +72,5 @@ dropped equations, wrong table cells), so we keep it purely manual.
72
72
  | `pymu` | $0.00 | pymupdf4llm with OCR fallback **disabled** — reproduces le-lab production where no OCR engine is shipped in the runtime image |
73
73
  | `pymu_ocr` | $0.00 | pymupdf4llm with built-in OCR fallback **enabled** — uses tesseract (or rapidocr/paddleocr) automatically when text extraction looks suspect |
74
74
  | `docling` | $0.00 | layout-aware ML pipeline + Tesseract OCR with `force_full_page_ocr=True` |
75
- | `mistral` | $4.00 | `mistral-ocr-4-0` cloud API. With `--mistral-batch`: $2.00 per 1k pages (50% off) via the Mistral Batch API. |
75
+ | `mistral` | $4.00 | `mistral-ocr-4-1` cloud API. With `--mistral-batch`: $2.00 per 1k pages (50% off) via the Mistral Batch API. |
76
76
  | `llama` | $3.00 | LlamaCloud `cost_effective` tier |
@@ -30,7 +30,7 @@ parse_with_images("doc.docx", backend=Mistral())
30
30
 
31
31
  The `.docx` is uploaded to Mistral and `ocr.process` is called with `document_url`
32
32
  plus `bbox_annotation_format`: a single call returns markdown, images and
33
- annotations. Model: `mistral-ocr-latest`.
33
+ annotations. Model: `mistral-ocr-4-1`.
34
34
 
35
35
  - **+** One API call, no per-image round trip.
36
36
  - **+** Handles natively the formats markitdown cannot pass through (the document is rasterised, so SmartArt / EMF / SVG render correctly).
@@ -68,7 +68,7 @@ parse_with_images("doc.pdf") # backend=Mistral() implied
68
68
  parse_with_images("doc.pdf", backend=Mistral()) # explicit, identical
69
69
  ```
70
70
 
71
- Same `bbox_annotation_format` flow as DOCX path B. Model: `mistral-ocr-latest`.
71
+ Same `bbox_annotation_format` flow as DOCX path B. Model: `mistral-ocr-4-1`.
72
72
 
73
73
  - **+** One call returns markdown, images and annotations; robust on scanned or broken-encoding PDFs; exact image positions.
74
74
  - **−** Billed per page; fixed annotation schema (same caveat as DOCX path B).
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "dot-parser"
3
- version = "2.1.0"
3
+ version = "2.1.1"
4
4
  description = "Document-to-markdown parser and chunker for RAG pipelines"
5
5
  readme = "README.md"
6
6
  authors = [{ name = "Kannon For Deep Tech", email = "louis.letarnec@deepika.ai" }]
@@ -217,7 +217,7 @@ class Mistral:
217
217
  self,
218
218
  *,
219
219
  api_key: str | None = None,
220
- model: str = "mistral-ocr-4-0",
220
+ model: str = "mistral-ocr-4-1",
221
221
  language: str | None = None,
222
222
  extract_headers_footers: bool = False,
223
223
  confidence_scores: Literal["page", "word"] | None = None,
@@ -129,7 +129,7 @@ def _make_mistral_with_fake_client(pages):
129
129
  constructing the instance manually.
130
130
  """
131
131
  instance = Mistral.__new__(Mistral)
132
- instance._model = "mistral-ocr-4-0"
132
+ instance._model = "mistral-ocr-4-1"
133
133
  instance._language = None
134
134
 
135
135
  uploaded = SimpleNamespace(id="file-xyz")
@@ -261,7 +261,7 @@ class TestMistralParsePdfWithImages:
261
261
  ]
262
262
 
263
263
  instance = Mistral.__new__(Mistral)
264
- instance._model = "mistral-ocr-4-0"
264
+ instance._model = "mistral-ocr-4-1"
265
265
  instance._language = None
266
266
  uploaded = SimpleNamespace(id="file-xyz")
267
267
  signed = SimpleNamespace(url="https://signed.example/doc.pdf")
@@ -297,7 +297,7 @@ class TestMistralParsePdfWithImages:
297
297
  def test_retry_failure_propagates(self):
298
298
  """If the no-annotation retry also fails, the exception escapes."""
299
299
  instance = Mistral.__new__(Mistral)
300
- instance._model = "mistral-ocr-4-0"
300
+ instance._model = "mistral-ocr-4-1"
301
301
  instance._language = None
302
302
  ocr_calls: list[dict] = []
303
303
 
@@ -63,7 +63,7 @@ def _backend(job, downloads: dict[str, bytes]) -> Mistral:
63
63
  no ``mistralai`` import / real key is needed."""
64
64
  m = Mistral.__new__(Mistral)
65
65
  m._client = _FakeClient(job, downloads) # ty: ignore[invalid-assignment]
66
- m._model = "mistral-ocr-4-0"
66
+ m._model = "mistral-ocr-4-1"
67
67
  return m
68
68
 
69
69
 
@@ -654,7 +654,7 @@ standard = [
654
654
 
655
655
  [[package]]
656
656
  name = "dot-parser"
657
- version = "2.1.0"
657
+ version = "2.1.1"
658
658
  source = { editable = "." }
659
659
  dependencies = [
660
660
  { name = "beautifulsoup4" },
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes