dot-parser 2.0.0__tar.gz → 2.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. dot_parser-2.1.1/CHANGELOG.md +96 -0
  2. {dot_parser-2.0.0 → dot_parser-2.1.1}/PKG-INFO +5 -3
  3. {dot_parser-2.0.0 → dot_parser-2.1.1}/README.md +2 -2
  4. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/README.md +1 -1
  5. {dot_parser-2.0.0 → dot_parser-2.1.1}/docs/IMAGE_EXTRACTION.md +2 -2
  6. {dot_parser-2.0.0 → dot_parser-2.1.1}/docs/VERSIONING.md +3 -3
  7. {dot_parser-2.0.0 → dot_parser-2.1.1}/pyproject.toml +5 -1
  8. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/__init__.py +4 -1
  9. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/backends/mistral.py +1 -1
  10. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/docx_images.py +16 -1
  11. dot_parser-2.1.1/src/dot_parser/docx_markdown.py +490 -0
  12. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/image_utils.py +46 -1
  13. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/models.py +12 -0
  14. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/parsers.py +27 -2
  15. dot_parser-2.1.1/tests/test_docx_markdown.py +429 -0
  16. {dot_parser-2.0.0 → dot_parser-2.1.1}/tests/test_images.py +91 -3
  17. {dot_parser-2.0.0 → dot_parser-2.1.1}/tests/test_mistral_batch_logging.py +1 -1
  18. {dot_parser-2.0.0 → dot_parser-2.1.1}/tests/test_parsers.py +4 -3
  19. {dot_parser-2.0.0 → dot_parser-2.1.1}/uv.lock +5 -1
  20. dot_parser-2.0.0/CHANGELOG.md +0 -45
  21. {dot_parser-2.0.0 → dot_parser-2.1.1}/.gitignore +0 -0
  22. {dot_parser-2.0.0 → dot_parser-2.1.1}/.gitlab-ci.yml +0 -0
  23. {dot_parser-2.0.0 → dot_parser-2.1.1}/.pre-commit-config.yaml +0 -0
  24. {dot_parser-2.0.0 → dot_parser-2.1.1}/.python-version +0 -0
  25. {dot_parser-2.0.0 → dot_parser-2.1.1}/AUTHORS.md +0 -0
  26. {dot_parser-2.0.0 → dot_parser-2.1.1}/CONTRIBUTING.md +0 -0
  27. {dot_parser-2.0.0 → dot_parser-2.1.1}/DCO +0 -0
  28. {dot_parser-2.0.0 → dot_parser-2.1.1}/LICENSE.md +0 -0
  29. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/input/broken_encoding.pdf +0 -0
  30. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/input/image_only.pdf +0 -0
  31. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/input/native_equations.pdf +0 -0
  32. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/input/native_simple.pdf +0 -0
  33. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/input/native_tables.pdf +0 -0
  34. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_broken_encoding.md +0 -0
  35. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_image_only.md +0 -0
  36. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_native_equations.md +0 -0
  37. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_native_simple.md +0 -0
  38. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_native_tables.md +0 -0
  39. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_broken_encoding.md +0 -0
  40. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_image_only.md +0 -0
  41. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_native_equations.md +0 -0
  42. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_native_simple.md +0 -0
  43. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_native_tables.md +0 -0
  44. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_broken_encoding.md +0 -0
  45. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_image_only.md +0 -0
  46. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_equations.md +0 -0
  47. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_simple.md +0 -0
  48. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_tables.md +0 -0
  49. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_broken_encoding.md +0 -0
  50. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_image_only.md +0 -0
  51. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_native_equations.md +0 -0
  52. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_native_simple.md +0 -0
  53. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_native_tables.md +0 -0
  54. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_broken_encoding.md +0 -0
  55. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_image_only.md +0 -0
  56. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_native_equations.md +0 -0
  57. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_native_simple.md +0 -0
  58. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_native_tables.md +0 -0
  59. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/results.md +0 -0
  60. {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/run.py +0 -0
  61. {dot_parser-2.0.0 → dot_parser-2.1.1}/docs/DESIGN.md +0 -0
  62. {dot_parser-2.0.0 → dot_parser-2.1.1}/docs/DEVELOPMENT.md +0 -0
  63. {dot_parser-2.0.0 → dot_parser-2.1.1}/docs/PUBLISHING.md +0 -0
  64. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/backends/__init__.py +0 -0
  65. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/backends/_base.py +0 -0
  66. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/backends/docling.py +0 -0
  67. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/backends/llama.py +0 -0
  68. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/backends/pymu.py +0 -0
  69. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/chunking.py +0 -0
  70. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/images.py +0 -0
  71. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/markdown_utils.py +0 -0
  72. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/pricing.py +0 -0
  73. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/tokens.py +0 -0
  74. {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/vlms.py +0 -0
  75. {dot_parser-2.0.0 → dot_parser-2.1.1}/tests/test_chunking.py +0 -0
  76. {dot_parser-2.0.0 → dot_parser-2.1.1}/tests/test_tokens.py +0 -0
@@ -0,0 +1,96 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project will be documented in this file.
4
+
5
+ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
6
+ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
+
8
+ ## [Unreleased]
9
+
10
+ ## [2.1.1] - 2026-09-30
11
+
12
+ ### Changed
13
+
14
+ - `Mistral` now defaults to `mistral-ocr-4-1`; `mistral-ocr-4-0` retires on
15
+ 2026-09-30. Pricing is unchanged.
16
+
17
+ ## [2.1.0] - 2026-09-30
18
+
19
+ ### Added
20
+
21
+ - `extract_docx_toc()` returns the table of contents stored in a DOCX as
22
+ `TocEntry(level, text)` items (`TocEntry` is exported), whether Word wrapped
23
+ it in a content control or inserted it as a bare field.
24
+ - `min_image_size=N` on `parse_with_images()`, `parse_docx_with_images()` and
25
+ `parse_pptx_with_images()` drops images whose shorter side is under `N`
26
+ pixels, with their anchors: letters or words pasted as pictures, icons,
27
+ separator lines. On the markitdown + VLM path they are dropped before
28
+ interpretation, so they cost no VLM call. `0` (default) keeps every image;
29
+ images whose size can't be read (EMF, WMF, SVG) are always kept.
30
+
31
+ ### Changed
32
+
33
+ - DOCX conversion (`parse()` and the markitdown + VLM image path) now runs
34
+ through one converter, so a DOCX yields the same text with or without image
35
+ extraction. It keeps markitdown's pipeline and fixes, in the Markdown it
36
+ produces:
37
+ - headings in custom styles derived from `Heading N` or carrying an outline
38
+ level (`Style Titre 2`, `Annexe`...) are emitted as headings instead of
39
+ body text;
40
+ - headings carry Word's automatic numbers (`# 1. Introduction`,
41
+ `## 2.3 Objet`), recomputed from `numbering.xml`; a number already typed
42
+ in the heading is not doubled;
43
+ - `Title` and `Subtitle` paragraphs (cover page) are rendered bold; like
44
+ Word's own TOC, they are not treated as headings;
45
+ - bold inside a heading is dropped (`## Zoom`, not `## **Zoom**`);
46
+ - the stored table of contents and the lists of figures and tables are
47
+ left out: they repeat headings and captions with page numbers, and
48
+ `extract_docx_toc()` exposes the TOC;
49
+ - tables use their first row as the header instead of a blank one, escape
50
+ `|` in cells, stay rectangular across merged cells (a vertically merged
51
+ value is repeated in each row it spans) and flatten nested tables into
52
+ their parent cell.
53
+ - `parse()` drops embedded images from a DOCX instead of emitting
54
+ `![alt](data:image/png;base64...)` anchors with a truncated payload.
55
+ - `parse()` raises `ParseError` on a corrupted DOCX instead of returning its
56
+ raw bytes as text.
57
+ - `mammoth` and `beautifulsoup4` are declared as direct dependencies.
58
+
59
+ ## [2.0.0] - 2026-07-30
60
+
61
+ First public release.
62
+
63
+ ### Added
64
+
65
+ - `parse()` converts PDF, DOCX, PPTX, HTML, XLSX, CSV, Markdown and plain text
66
+ into clean Markdown.
67
+ - Selectable PDF backends: `Pymu` (default, local and fast), `Docling` (local
68
+ layout-aware ML pipeline), `Mistral` (cloud OCR) and `Llama` (LlamaCloud).
69
+ - `parse_with_images()` extracts images alongside the Markdown, with anchors
70
+ kept at their position in the text. Images are described either by a VLM or
71
+ by Mistral OCR annotations in the same call.
72
+ - `parse_pdfs()` parses many PDFs in one batch job through the Mistral Batch
73
+ API, falling back to a per-file loop on backends without batch support.
74
+ - `chunk()` splits Markdown into chunks carrying their heading hierarchy in
75
+ `section_path`.
76
+ - Cost helpers: `cost_per_1k_pages()` and `estimate_cost()`, usable without
77
+ instantiating a backend.
78
+ - Backend protocols (`Backend`, `ImageBackend`, `BatchBackend`,
79
+ `DocxImageBackend`, `PptxImageBackend`) so third-party backends can plug in.
80
+
81
+ ### Changed
82
+
83
+ These matter only if you were installing the package straight from git before
84
+ this release.
85
+
86
+ - `LlamaTier` is a `Literal` of the four accepted tiers rather than a bare
87
+ `str`, and is defined once alongside the pricing table.
88
+ - `zip()` calls over sequences that must line up now use `strict=True`: a length
89
+ mismatch raises instead of silently truncating.
90
+ - `parse_with_images(backend=...)` accepts DOCX- and PPTX-capable backends in
91
+ its type signature, matching what the function already supported at runtime.
92
+
93
+ [Unreleased]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.1.1...main
94
+ [2.1.1]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.1.0...v2.1.1
95
+ [2.1.0]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.0.0...v2.1.0
96
+ [2.0.0]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/tags/v2.0.0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: dot-parser
3
- Version: 2.0.0
3
+ Version: 2.1.1
4
4
  Summary: Document-to-markdown parser and chunker for RAG pipelines
5
5
  Project-URL: Homepage, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
6
6
  Project-URL: Repository, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
@@ -15,6 +15,8 @@ Classifier: Programming Language :: Python :: 3
15
15
  Classifier: Programming Language :: Python :: 3.12
16
16
  Classifier: Programming Language :: Python :: 3.13
17
17
  Requires-Python: <3.14,>=3.12
18
+ Requires-Dist: beautifulsoup4>=4.12
19
+ Requires-Dist: mammoth>=1.11
18
20
  Requires-Dist: markitdown[all]>=0.1
19
21
  Requires-Dist: pillow>=10.0
20
22
  Requires-Dist: pydantic>=2
@@ -152,7 +154,7 @@ described either by a VLM or by OCR annotations from the same call. See
152
154
 
153
155
  - **`Pymu(ocr_fallback=True)`** -- pymupdf4llm, default. Fast and CPU-only. The built-in OCR fallback kicks in when an OCR engine (tesseract, rapidocr, paddleocr) is available. Set `ocr_fallback=False` to force the no-OCR code path (matches a runtime image without OCR shipped).
154
156
  - **`Docling(force_full_page_ocr=True)`** -- IBM Docling, layout-aware ML pipeline + Tesseract OCR. Robust on image-only and broken-encoding PDFs. Install with `pip install dot-parser[docling]`.
155
- - **`Mistral(api_key=None, model="mistral-ocr-4-0")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
157
+ - **`Mistral(api_key=None, model="mistral-ocr-4-1")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
156
158
  - **`Llama(api_key=None, tier="cost_effective")`** -- LlamaCloud parsing API. Reads `LLAMA_CLOUD_API_KEY` from env. Tiers: `fast`, `cost_effective`, `agentic`, `agentic_plus`. Install with `pip install dot-parser[llama]`.
157
159
 
158
160
  #### Mistral OCR enrichments
@@ -235,7 +237,7 @@ package is covered, anything underscore-prefixed is internal and may change in
235
237
  any release. Public names are never removed without a deprecation period.
236
238
 
237
239
  ```toml
238
- dependencies = ["dot-parser>=1.0,<2"]
240
+ dependencies = ["dot-parser>=2.0,<3"]
239
241
  ```
240
242
 
241
243
  See [docs/VERSIONING.md](docs/VERSIONING.md) for the full policy.
@@ -116,7 +116,7 @@ described either by a VLM or by OCR annotations from the same call. See
116
116
 
117
117
  - **`Pymu(ocr_fallback=True)`** -- pymupdf4llm, default. Fast and CPU-only. The built-in OCR fallback kicks in when an OCR engine (tesseract, rapidocr, paddleocr) is available. Set `ocr_fallback=False` to force the no-OCR code path (matches a runtime image without OCR shipped).
118
118
  - **`Docling(force_full_page_ocr=True)`** -- IBM Docling, layout-aware ML pipeline + Tesseract OCR. Robust on image-only and broken-encoding PDFs. Install with `pip install dot-parser[docling]`.
119
- - **`Mistral(api_key=None, model="mistral-ocr-4-0")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
119
+ - **`Mistral(api_key=None, model="mistral-ocr-4-1")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
120
120
  - **`Llama(api_key=None, tier="cost_effective")`** -- LlamaCloud parsing API. Reads `LLAMA_CLOUD_API_KEY` from env. Tiers: `fast`, `cost_effective`, `agentic`, `agentic_plus`. Install with `pip install dot-parser[llama]`.
121
121
 
122
122
  #### Mistral OCR enrichments
@@ -199,7 +199,7 @@ package is covered, anything underscore-prefixed is internal and may change in
199
199
  any release. Public names are never removed without a deprecation period.
200
200
 
201
201
  ```toml
202
- dependencies = ["dot-parser>=1.0,<2"]
202
+ dependencies = ["dot-parser>=2.0,<3"]
203
203
  ```
204
204
 
205
205
  See [docs/VERSIONING.md](docs/VERSIONING.md) for the full policy.
@@ -72,5 +72,5 @@ dropped equations, wrong table cells), so we keep it purely manual.
72
72
  | `pymu` | $0.00 | pymupdf4llm with OCR fallback **disabled** — reproduces le-lab production where no OCR engine is shipped in the runtime image |
73
73
  | `pymu_ocr` | $0.00 | pymupdf4llm with built-in OCR fallback **enabled** — uses tesseract (or rapidocr/paddleocr) automatically when text extraction looks suspect |
74
74
  | `docling` | $0.00 | layout-aware ML pipeline + Tesseract OCR with `force_full_page_ocr=True` |
75
- | `mistral` | $4.00 | `mistral-ocr-4-0` cloud API. With `--mistral-batch`: $2.00 per 1k pages (50% off) via the Mistral Batch API. |
75
+ | `mistral` | $4.00 | `mistral-ocr-4-1` cloud API. With `--mistral-batch`: $2.00 per 1k pages (50% off) via the Mistral Batch API. |
76
76
  | `llama` | $3.00 | LlamaCloud `cost_effective` tier |
@@ -30,7 +30,7 @@ parse_with_images("doc.docx", backend=Mistral())
30
30
 
31
31
  The `.docx` is uploaded to Mistral and `ocr.process` is called with `document_url`
32
32
  plus `bbox_annotation_format`: a single call returns markdown, images and
33
- annotations. Model: `mistral-ocr-latest`.
33
+ annotations. Model: `mistral-ocr-4-1`.
34
34
 
35
35
  - **+** One API call, no per-image round trip.
36
36
  - **+** Handles natively the formats markitdown cannot pass through (the document is rasterised, so SmartArt / EMF / SVG render correctly).
@@ -68,7 +68,7 @@ parse_with_images("doc.pdf") # backend=Mistral() implied
68
68
  parse_with_images("doc.pdf", backend=Mistral()) # explicit, identical
69
69
  ```
70
70
 
71
- Same `bbox_annotation_format` flow as DOCX path B. Model: `mistral-ocr-latest`.
71
+ Same `bbox_annotation_format` flow as DOCX path B. Model: `mistral-ocr-4-1`.
72
72
 
73
73
  - **+** One call returns markdown, images and annotations; robust on scanned or broken-encoding PDFs; exact image positions.
74
74
  - **−** Billed per page; fixed annotation schema (same caveat as DOCX path B).
@@ -1,7 +1,7 @@
1
1
  # Versioning and stability
2
2
 
3
3
  `dot-parser` follows [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
4
- This page says what that actually covers, so `>=1.0,<2` means something concrete.
4
+ This page says what that actually covers, so `>=2.0,<3` means something concrete.
5
5
 
6
6
  ## What is public
7
7
 
@@ -94,11 +94,11 @@ manual. Use them to validate a release without committing to it.
94
94
  ### From PyPI — the supported way
95
95
 
96
96
  ```toml
97
- dependencies = ["dot-parser>=1.0,<2"]
97
+ dependencies = ["dot-parser>=2.0,<3"]
98
98
  ```
99
99
 
100
100
  The upper bound is what makes the guarantees above useful: you get every fix and
101
- feature of the 1.x line, and never an unannounced break.
101
+ feature of the 2.x line, and never an unannounced break.
102
102
 
103
103
  ### From git — development only
104
104
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "dot-parser"
3
- version = "2.0.0"
3
+ version = "2.1.1"
4
4
  description = "Document-to-markdown parser and chunker for RAG pipelines"
5
5
  readme = "README.md"
6
6
  authors = [{ name = "Kannon For Deep Tech", email = "louis.letarnec@deepika.ai" }]
@@ -12,6 +12,10 @@ requires-python = ">=3.12,<3.14"
12
12
  dependencies = [
13
13
  "pymupdf4llm>=1.27.2.2",
14
14
  "markitdown[all]>=0.1",
15
+ # Imported directly by the DOCX converter (docx_markdown.py); markitdown
16
+ # pulls both in too, but only through its optional extras.
17
+ "mammoth>=1.11",
18
+ "beautifulsoup4>=4.12",
15
19
  "semchunk>=3.0",
16
20
  "python-docx>=1.2.0",
17
21
  "pillow>=10.0",
@@ -5,6 +5,7 @@ from importlib.metadata import version
5
5
 
6
6
  from dot_parser.backends import Backend, Docling, ImageBackend, Llama, Mistral, Pymu
7
7
  from dot_parser.chunking import chunk
8
+ from dot_parser.docx_markdown import extract_docx_toc
8
9
  from dot_parser.images import (
9
10
  VLM,
10
11
  ExtractedImage,
@@ -12,7 +13,7 @@ from dot_parser.images import (
12
13
  PageInfo,
13
14
  ParseResult,
14
15
  )
15
- from dot_parser.models import Chunk, ParseError
16
+ from dot_parser.models import Chunk, ParseError, TocEntry
16
17
  from dot_parser.parsers import (
17
18
  interpret_images,
18
19
  parse,
@@ -39,9 +40,11 @@ __all__ = [
39
40
  "ParseError",
40
41
  "ParseResult",
41
42
  "Pymu",
43
+ "TocEntry",
42
44
  "chunk",
43
45
  "cost_per_1k_pages",
44
46
  "estimate_cost",
47
+ "extract_docx_toc",
45
48
  "interpret_images",
46
49
  "parse",
47
50
  "parse_pdfs",
@@ -217,7 +217,7 @@ class Mistral:
217
217
  self,
218
218
  *,
219
219
  api_key: str | None = None,
220
- model: str = "mistral-ocr-4-0",
220
+ model: str = "mistral-ocr-4-1",
221
221
  language: str | None = None,
222
222
  extract_headers_footers: bool = False,
223
223
  confidence_scores: Literal["page", "word"] | None = None,
@@ -34,7 +34,7 @@ import tempfile
34
34
  import time
35
35
  from pathlib import Path
36
36
 
37
- from dot_parser.image_utils import dedupe_titles
37
+ from dot_parser.image_utils import dedupe_titles, drop_small_images
38
38
  from dot_parser.images import (
39
39
  VLM,
40
40
  ExtractedImage,
@@ -523,7 +523,15 @@ def _office_to_markdown_via_markitdown(file_path: str) -> str:
523
523
  first comma (replacing the base64 payload with ``...``) to keep
524
524
  markdown small for the LLM-summarisation use case it was designed
525
525
  for. We need the full payload to extract the image bytes.
526
+
527
+ DOCX goes through :func:`dot_parser.docx_markdown.docx_to_markdown`,
528
+ the same converter :func:`dot_parser.parse` uses, so the text is
529
+ identical with or without image extraction.
526
530
  """
531
+ if Path(file_path).suffix.lower() == ".docx":
532
+ from dot_parser.docx_markdown import docx_to_markdown
533
+
534
+ return docx_to_markdown(file_path, keep_data_uris=True)
527
535
  try:
528
536
  import markitdown
529
537
  except ImportError as e:
@@ -540,6 +548,7 @@ def _parse_office_with_images(
540
548
  suffix: str,
541
549
  include_tables: bool,
542
550
  deadline: float | None = None,
551
+ min_image_size: int = 0,
543
552
  ) -> ParseResult:
544
553
  """Shared markitdown+VLM pipeline for DOCX and PPTX sources."""
545
554
  if isinstance(source, bytes):
@@ -555,6 +564,8 @@ def _parse_office_with_images(
555
564
  raw_md = _office_to_markdown_via_markitdown(str(source))
556
565
 
557
566
  markdown, images = _extract_data_url_images(raw_md)
567
+ # Before interpretation, so dropped images cost no VLM call.
568
+ markdown, images = drop_small_images(markdown, images, min_image_size)
558
569
  if not include_tables:
559
570
  markdown = strip_markdown_tables(markdown)
560
571
  if images:
@@ -574,6 +585,7 @@ def parse_docx_with_images(
574
585
  *,
575
586
  include_tables: bool = True,
576
587
  deadline: float | None = None,
588
+ min_image_size: int = 0,
577
589
  ) -> ParseResult:
578
590
  """Parse a DOCX into markdown + interpreted images.
579
591
 
@@ -599,6 +611,7 @@ def parse_docx_with_images(
599
611
  suffix=".docx",
600
612
  include_tables=include_tables,
601
613
  deadline=deadline,
614
+ min_image_size=min_image_size,
602
615
  )
603
616
 
604
617
 
@@ -608,6 +621,7 @@ def parse_pptx_with_images(
608
621
  *,
609
622
  include_tables: bool = True,
610
623
  deadline: float | None = None,
624
+ min_image_size: int = 0,
611
625
  ) -> ParseResult:
612
626
  """Parse a PPTX into markdown + interpreted embedded pictures.
613
627
 
@@ -630,4 +644,5 @@ def parse_pptx_with_images(
630
644
  suffix=".pptx",
631
645
  include_tables=include_tables,
632
646
  deadline=deadline,
647
+ min_image_size=min_image_size,
633
648
  )