dot-parser 2.0.0__tar.gz → 2.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dot_parser-2.1.1/CHANGELOG.md +96 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/PKG-INFO +5 -3
- {dot_parser-2.0.0 → dot_parser-2.1.1}/README.md +2 -2
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/README.md +1 -1
- {dot_parser-2.0.0 → dot_parser-2.1.1}/docs/IMAGE_EXTRACTION.md +2 -2
- {dot_parser-2.0.0 → dot_parser-2.1.1}/docs/VERSIONING.md +3 -3
- {dot_parser-2.0.0 → dot_parser-2.1.1}/pyproject.toml +5 -1
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/__init__.py +4 -1
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/backends/mistral.py +1 -1
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/docx_images.py +16 -1
- dot_parser-2.1.1/src/dot_parser/docx_markdown.py +490 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/image_utils.py +46 -1
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/models.py +12 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/parsers.py +27 -2
- dot_parser-2.1.1/tests/test_docx_markdown.py +429 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/tests/test_images.py +91 -3
- {dot_parser-2.0.0 → dot_parser-2.1.1}/tests/test_mistral_batch_logging.py +1 -1
- {dot_parser-2.0.0 → dot_parser-2.1.1}/tests/test_parsers.py +4 -3
- {dot_parser-2.0.0 → dot_parser-2.1.1}/uv.lock +5 -1
- dot_parser-2.0.0/CHANGELOG.md +0 -45
- {dot_parser-2.0.0 → dot_parser-2.1.1}/.gitignore +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/.gitlab-ci.yml +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/.pre-commit-config.yaml +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/.python-version +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/AUTHORS.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/CONTRIBUTING.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/DCO +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/LICENSE.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/input/broken_encoding.pdf +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/input/image_only.pdf +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/input/native_equations.pdf +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/input/native_simple.pdf +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/input/native_tables.pdf +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_broken_encoding.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_image_only.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_native_equations.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_native_simple.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_docling_native_tables.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_broken_encoding.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_image_only.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_native_equations.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_native_simple.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_llama_native_tables.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_broken_encoding.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_image_only.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_equations.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_simple.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_mistral_batch_native_tables.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_broken_encoding.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_image_only.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_native_equations.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_native_simple.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_native_tables.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_broken_encoding.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_image_only.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_native_equations.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_native_simple.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/md/out_pymu_ocr_native_tables.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/output/results.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/benchmark/run.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/docs/DESIGN.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/docs/DEVELOPMENT.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/docs/PUBLISHING.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/backends/__init__.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/backends/_base.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/backends/docling.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/backends/llama.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/backends/pymu.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/chunking.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/images.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/markdown_utils.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/pricing.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/tokens.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/src/dot_parser/vlms.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/tests/test_chunking.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.1}/tests/test_tokens.py +0 -0
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project will be documented in this file.
|
|
4
|
+
|
|
5
|
+
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
6
|
+
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
|
+
|
|
8
|
+
## [Unreleased]
|
|
9
|
+
|
|
10
|
+
## [2.1.1] - 2026-09-30
|
|
11
|
+
|
|
12
|
+
### Changed
|
|
13
|
+
|
|
14
|
+
- `Mistral` now defaults to `mistral-ocr-4-1`; `mistral-ocr-4-0` retires on
|
|
15
|
+
2026-09-30. Pricing is unchanged.
|
|
16
|
+
|
|
17
|
+
## [2.1.0] - 2026-09-30
|
|
18
|
+
|
|
19
|
+
### Added
|
|
20
|
+
|
|
21
|
+
- `extract_docx_toc()` returns the table of contents stored in a DOCX as
|
|
22
|
+
`TocEntry(level, text)` items (`TocEntry` is exported), whether Word wrapped
|
|
23
|
+
it in a content control or inserted it as a bare field.
|
|
24
|
+
- `min_image_size=N` on `parse_with_images()`, `parse_docx_with_images()` and
|
|
25
|
+
`parse_pptx_with_images()` drops images whose shorter side is under `N`
|
|
26
|
+
pixels, with their anchors: letters or words pasted as pictures, icons,
|
|
27
|
+
separator lines. On the markitdown + VLM path they are dropped before
|
|
28
|
+
interpretation, so they cost no VLM call. `0` (default) keeps every image;
|
|
29
|
+
images whose size can't be read (EMF, WMF, SVG) are always kept.
|
|
30
|
+
|
|
31
|
+
### Changed
|
|
32
|
+
|
|
33
|
+
- DOCX conversion (`parse()` and the markitdown + VLM image path) now runs
|
|
34
|
+
through one converter, so a DOCX yields the same text with or without image
|
|
35
|
+
extraction. It keeps markitdown's pipeline and fixes, in the Markdown it
|
|
36
|
+
produces:
|
|
37
|
+
- headings in custom styles derived from `Heading N` or carrying an outline
|
|
38
|
+
level (`Style Titre 2`, `Annexe`...) are emitted as headings instead of
|
|
39
|
+
body text;
|
|
40
|
+
- headings carry Word's automatic numbers (`# 1. Introduction`,
|
|
41
|
+
`## 2.3 Objet`), recomputed from `numbering.xml`; a number already typed
|
|
42
|
+
in the heading is not doubled;
|
|
43
|
+
- `Title` and `Subtitle` paragraphs (cover page) are rendered bold; like
|
|
44
|
+
Word's own TOC, they are not treated as headings;
|
|
45
|
+
- bold inside a heading is dropped (`## Zoom`, not `## **Zoom**`);
|
|
46
|
+
- the stored table of contents and the lists of figures and tables are
|
|
47
|
+
left out: they repeat headings and captions with page numbers, and
|
|
48
|
+
`extract_docx_toc()` exposes the TOC;
|
|
49
|
+
- tables use their first row as the header instead of a blank one, escape
|
|
50
|
+
`|` in cells, stay rectangular across merged cells (a vertically merged
|
|
51
|
+
value is repeated in each row it spans) and flatten nested tables into
|
|
52
|
+
their parent cell.
|
|
53
|
+
- `parse()` drops embedded images from a DOCX instead of emitting
|
|
54
|
+
`` anchors with a truncated payload.
|
|
55
|
+
- `parse()` raises `ParseError` on a corrupted DOCX instead of returning its
|
|
56
|
+
raw bytes as text.
|
|
57
|
+
- `mammoth` and `beautifulsoup4` are declared as direct dependencies.
|
|
58
|
+
|
|
59
|
+
## [2.0.0] - 2026-07-30
|
|
60
|
+
|
|
61
|
+
First public release.
|
|
62
|
+
|
|
63
|
+
### Added
|
|
64
|
+
|
|
65
|
+
- `parse()` converts PDF, DOCX, PPTX, HTML, XLSX, CSV, Markdown and plain text
|
|
66
|
+
into clean Markdown.
|
|
67
|
+
- Selectable PDF backends: `Pymu` (default, local and fast), `Docling` (local
|
|
68
|
+
layout-aware ML pipeline), `Mistral` (cloud OCR) and `Llama` (LlamaCloud).
|
|
69
|
+
- `parse_with_images()` extracts images alongside the Markdown, with anchors
|
|
70
|
+
kept at their position in the text. Images are described either by a VLM or
|
|
71
|
+
by Mistral OCR annotations in the same call.
|
|
72
|
+
- `parse_pdfs()` parses many PDFs in one batch job through the Mistral Batch
|
|
73
|
+
API, falling back to a per-file loop on backends without batch support.
|
|
74
|
+
- `chunk()` splits Markdown into chunks carrying their heading hierarchy in
|
|
75
|
+
`section_path`.
|
|
76
|
+
- Cost helpers: `cost_per_1k_pages()` and `estimate_cost()`, usable without
|
|
77
|
+
instantiating a backend.
|
|
78
|
+
- Backend protocols (`Backend`, `ImageBackend`, `BatchBackend`,
|
|
79
|
+
`DocxImageBackend`, `PptxImageBackend`) so third-party backends can plug in.
|
|
80
|
+
|
|
81
|
+
### Changed
|
|
82
|
+
|
|
83
|
+
These matter only if you were installing the package straight from git before
|
|
84
|
+
this release.
|
|
85
|
+
|
|
86
|
+
- `LlamaTier` is a `Literal` of the four accepted tiers rather than a bare
|
|
87
|
+
`str`, and is defined once alongside the pricing table.
|
|
88
|
+
- `zip()` calls over sequences that must line up now use `strict=True`: a length
|
|
89
|
+
mismatch raises instead of silently truncating.
|
|
90
|
+
- `parse_with_images(backend=...)` accepts DOCX- and PPTX-capable backends in
|
|
91
|
+
its type signature, matching what the function already supported at runtime.
|
|
92
|
+
|
|
93
|
+
[Unreleased]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.1.1...main
|
|
94
|
+
[2.1.1]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.1.0...v2.1.1
|
|
95
|
+
[2.1.0]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.0.0...v2.1.0
|
|
96
|
+
[2.0.0]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/tags/v2.0.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: dot-parser
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.1.1
|
|
4
4
|
Summary: Document-to-markdown parser and chunker for RAG pipelines
|
|
5
5
|
Project-URL: Homepage, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
|
|
6
6
|
Project-URL: Repository, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
|
|
@@ -15,6 +15,8 @@ Classifier: Programming Language :: Python :: 3
|
|
|
15
15
|
Classifier: Programming Language :: Python :: 3.12
|
|
16
16
|
Classifier: Programming Language :: Python :: 3.13
|
|
17
17
|
Requires-Python: <3.14,>=3.12
|
|
18
|
+
Requires-Dist: beautifulsoup4>=4.12
|
|
19
|
+
Requires-Dist: mammoth>=1.11
|
|
18
20
|
Requires-Dist: markitdown[all]>=0.1
|
|
19
21
|
Requires-Dist: pillow>=10.0
|
|
20
22
|
Requires-Dist: pydantic>=2
|
|
@@ -152,7 +154,7 @@ described either by a VLM or by OCR annotations from the same call. See
|
|
|
152
154
|
|
|
153
155
|
- **`Pymu(ocr_fallback=True)`** -- pymupdf4llm, default. Fast and CPU-only. The built-in OCR fallback kicks in when an OCR engine (tesseract, rapidocr, paddleocr) is available. Set `ocr_fallback=False` to force the no-OCR code path (matches a runtime image without OCR shipped).
|
|
154
156
|
- **`Docling(force_full_page_ocr=True)`** -- IBM Docling, layout-aware ML pipeline + Tesseract OCR. Robust on image-only and broken-encoding PDFs. Install with `pip install dot-parser[docling]`.
|
|
155
|
-
- **`Mistral(api_key=None, model="mistral-ocr-4-
|
|
157
|
+
- **`Mistral(api_key=None, model="mistral-ocr-4-1")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
|
|
156
158
|
- **`Llama(api_key=None, tier="cost_effective")`** -- LlamaCloud parsing API. Reads `LLAMA_CLOUD_API_KEY` from env. Tiers: `fast`, `cost_effective`, `agentic`, `agentic_plus`. Install with `pip install dot-parser[llama]`.
|
|
157
159
|
|
|
158
160
|
#### Mistral OCR enrichments
|
|
@@ -235,7 +237,7 @@ package is covered, anything underscore-prefixed is internal and may change in
|
|
|
235
237
|
any release. Public names are never removed without a deprecation period.
|
|
236
238
|
|
|
237
239
|
```toml
|
|
238
|
-
dependencies = ["dot-parser>=
|
|
240
|
+
dependencies = ["dot-parser>=2.0,<3"]
|
|
239
241
|
```
|
|
240
242
|
|
|
241
243
|
See [docs/VERSIONING.md](docs/VERSIONING.md) for the full policy.
|
|
@@ -116,7 +116,7 @@ described either by a VLM or by OCR annotations from the same call. See
|
|
|
116
116
|
|
|
117
117
|
- **`Pymu(ocr_fallback=True)`** -- pymupdf4llm, default. Fast and CPU-only. The built-in OCR fallback kicks in when an OCR engine (tesseract, rapidocr, paddleocr) is available. Set `ocr_fallback=False` to force the no-OCR code path (matches a runtime image without OCR shipped).
|
|
118
118
|
- **`Docling(force_full_page_ocr=True)`** -- IBM Docling, layout-aware ML pipeline + Tesseract OCR. Robust on image-only and broken-encoding PDFs. Install with `pip install dot-parser[docling]`.
|
|
119
|
-
- **`Mistral(api_key=None, model="mistral-ocr-4-
|
|
119
|
+
- **`Mistral(api_key=None, model="mistral-ocr-4-1")`** -- Mistral OCR cloud API. Reads `MISTRAL_API_KEY` from env. Implements `parse_pdfs` via the Batch API. Install with `pip install dot-parser[mistral]`.
|
|
120
120
|
- **`Llama(api_key=None, tier="cost_effective")`** -- LlamaCloud parsing API. Reads `LLAMA_CLOUD_API_KEY` from env. Tiers: `fast`, `cost_effective`, `agentic`, `agentic_plus`. Install with `pip install dot-parser[llama]`.
|
|
121
121
|
|
|
122
122
|
#### Mistral OCR enrichments
|
|
@@ -199,7 +199,7 @@ package is covered, anything underscore-prefixed is internal and may change in
|
|
|
199
199
|
any release. Public names are never removed without a deprecation period.
|
|
200
200
|
|
|
201
201
|
```toml
|
|
202
|
-
dependencies = ["dot-parser>=
|
|
202
|
+
dependencies = ["dot-parser>=2.0,<3"]
|
|
203
203
|
```
|
|
204
204
|
|
|
205
205
|
See [docs/VERSIONING.md](docs/VERSIONING.md) for the full policy.
|
|
@@ -72,5 +72,5 @@ dropped equations, wrong table cells), so we keep it purely manual.
|
|
|
72
72
|
| `pymu` | $0.00 | pymupdf4llm with OCR fallback **disabled** — reproduces le-lab production where no OCR engine is shipped in the runtime image |
|
|
73
73
|
| `pymu_ocr` | $0.00 | pymupdf4llm with built-in OCR fallback **enabled** — uses tesseract (or rapidocr/paddleocr) automatically when text extraction looks suspect |
|
|
74
74
|
| `docling` | $0.00 | layout-aware ML pipeline + Tesseract OCR with `force_full_page_ocr=True` |
|
|
75
|
-
| `mistral` | $4.00 | `mistral-ocr-4-
|
|
75
|
+
| `mistral` | $4.00 | `mistral-ocr-4-1` cloud API. With `--mistral-batch`: $2.00 per 1k pages (50% off) via the Mistral Batch API. |
|
|
76
76
|
| `llama` | $3.00 | LlamaCloud `cost_effective` tier |
|
|
@@ -30,7 +30,7 @@ parse_with_images("doc.docx", backend=Mistral())
|
|
|
30
30
|
|
|
31
31
|
The `.docx` is uploaded to Mistral and `ocr.process` is called with `document_url`
|
|
32
32
|
plus `bbox_annotation_format`: a single call returns markdown, images and
|
|
33
|
-
annotations. Model: `mistral-ocr-
|
|
33
|
+
annotations. Model: `mistral-ocr-4-1`.
|
|
34
34
|
|
|
35
35
|
- **+** One API call, no per-image round trip.
|
|
36
36
|
- **+** Handles natively the formats markitdown cannot pass through (the document is rasterised, so SmartArt / EMF / SVG render correctly).
|
|
@@ -68,7 +68,7 @@ parse_with_images("doc.pdf") # backend=Mistral() implied
|
|
|
68
68
|
parse_with_images("doc.pdf", backend=Mistral()) # explicit, identical
|
|
69
69
|
```
|
|
70
70
|
|
|
71
|
-
Same `bbox_annotation_format` flow as DOCX path B. Model: `mistral-ocr-
|
|
71
|
+
Same `bbox_annotation_format` flow as DOCX path B. Model: `mistral-ocr-4-1`.
|
|
72
72
|
|
|
73
73
|
- **+** One call returns markdown, images and annotations; robust on scanned or broken-encoding PDFs; exact image positions.
|
|
74
74
|
- **−** Billed per page; fixed annotation schema (same caveat as DOCX path B).
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Versioning and stability
|
|
2
2
|
|
|
3
3
|
`dot-parser` follows [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
4
|
-
This page says what that actually covers, so `>=
|
|
4
|
+
This page says what that actually covers, so `>=2.0,<3` means something concrete.
|
|
5
5
|
|
|
6
6
|
## What is public
|
|
7
7
|
|
|
@@ -94,11 +94,11 @@ manual. Use them to validate a release without committing to it.
|
|
|
94
94
|
### From PyPI — the supported way
|
|
95
95
|
|
|
96
96
|
```toml
|
|
97
|
-
dependencies = ["dot-parser>=
|
|
97
|
+
dependencies = ["dot-parser>=2.0,<3"]
|
|
98
98
|
```
|
|
99
99
|
|
|
100
100
|
The upper bound is what makes the guarantees above useful: you get every fix and
|
|
101
|
-
feature of the
|
|
101
|
+
feature of the 2.x line, and never an unannounced break.
|
|
102
102
|
|
|
103
103
|
### From git — development only
|
|
104
104
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "dot-parser"
|
|
3
|
-
version = "2.
|
|
3
|
+
version = "2.1.1"
|
|
4
4
|
description = "Document-to-markdown parser and chunker for RAG pipelines"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
authors = [{ name = "Kannon For Deep Tech", email = "louis.letarnec@deepika.ai" }]
|
|
@@ -12,6 +12,10 @@ requires-python = ">=3.12,<3.14"
|
|
|
12
12
|
dependencies = [
|
|
13
13
|
"pymupdf4llm>=1.27.2.2",
|
|
14
14
|
"markitdown[all]>=0.1",
|
|
15
|
+
# Imported directly by the DOCX converter (docx_markdown.py); markitdown
|
|
16
|
+
# pulls both in too, but only through its optional extras.
|
|
17
|
+
"mammoth>=1.11",
|
|
18
|
+
"beautifulsoup4>=4.12",
|
|
15
19
|
"semchunk>=3.0",
|
|
16
20
|
"python-docx>=1.2.0",
|
|
17
21
|
"pillow>=10.0",
|
|
@@ -5,6 +5,7 @@ from importlib.metadata import version
|
|
|
5
5
|
|
|
6
6
|
from dot_parser.backends import Backend, Docling, ImageBackend, Llama, Mistral, Pymu
|
|
7
7
|
from dot_parser.chunking import chunk
|
|
8
|
+
from dot_parser.docx_markdown import extract_docx_toc
|
|
8
9
|
from dot_parser.images import (
|
|
9
10
|
VLM,
|
|
10
11
|
ExtractedImage,
|
|
@@ -12,7 +13,7 @@ from dot_parser.images import (
|
|
|
12
13
|
PageInfo,
|
|
13
14
|
ParseResult,
|
|
14
15
|
)
|
|
15
|
-
from dot_parser.models import Chunk, ParseError
|
|
16
|
+
from dot_parser.models import Chunk, ParseError, TocEntry
|
|
16
17
|
from dot_parser.parsers import (
|
|
17
18
|
interpret_images,
|
|
18
19
|
parse,
|
|
@@ -39,9 +40,11 @@ __all__ = [
|
|
|
39
40
|
"ParseError",
|
|
40
41
|
"ParseResult",
|
|
41
42
|
"Pymu",
|
|
43
|
+
"TocEntry",
|
|
42
44
|
"chunk",
|
|
43
45
|
"cost_per_1k_pages",
|
|
44
46
|
"estimate_cost",
|
|
47
|
+
"extract_docx_toc",
|
|
45
48
|
"interpret_images",
|
|
46
49
|
"parse",
|
|
47
50
|
"parse_pdfs",
|
|
@@ -217,7 +217,7 @@ class Mistral:
|
|
|
217
217
|
self,
|
|
218
218
|
*,
|
|
219
219
|
api_key: str | None = None,
|
|
220
|
-
model: str = "mistral-ocr-4-
|
|
220
|
+
model: str = "mistral-ocr-4-1",
|
|
221
221
|
language: str | None = None,
|
|
222
222
|
extract_headers_footers: bool = False,
|
|
223
223
|
confidence_scores: Literal["page", "word"] | None = None,
|
|
@@ -34,7 +34,7 @@ import tempfile
|
|
|
34
34
|
import time
|
|
35
35
|
from pathlib import Path
|
|
36
36
|
|
|
37
|
-
from dot_parser.image_utils import dedupe_titles
|
|
37
|
+
from dot_parser.image_utils import dedupe_titles, drop_small_images
|
|
38
38
|
from dot_parser.images import (
|
|
39
39
|
VLM,
|
|
40
40
|
ExtractedImage,
|
|
@@ -523,7 +523,15 @@ def _office_to_markdown_via_markitdown(file_path: str) -> str:
|
|
|
523
523
|
first comma (replacing the base64 payload with ``...``) to keep
|
|
524
524
|
markdown small for the LLM-summarisation use case it was designed
|
|
525
525
|
for. We need the full payload to extract the image bytes.
|
|
526
|
+
|
|
527
|
+
DOCX goes through :func:`dot_parser.docx_markdown.docx_to_markdown`,
|
|
528
|
+
the same converter :func:`dot_parser.parse` uses, so the text is
|
|
529
|
+
identical with or without image extraction.
|
|
526
530
|
"""
|
|
531
|
+
if Path(file_path).suffix.lower() == ".docx":
|
|
532
|
+
from dot_parser.docx_markdown import docx_to_markdown
|
|
533
|
+
|
|
534
|
+
return docx_to_markdown(file_path, keep_data_uris=True)
|
|
527
535
|
try:
|
|
528
536
|
import markitdown
|
|
529
537
|
except ImportError as e:
|
|
@@ -540,6 +548,7 @@ def _parse_office_with_images(
|
|
|
540
548
|
suffix: str,
|
|
541
549
|
include_tables: bool,
|
|
542
550
|
deadline: float | None = None,
|
|
551
|
+
min_image_size: int = 0,
|
|
543
552
|
) -> ParseResult:
|
|
544
553
|
"""Shared markitdown+VLM pipeline for DOCX and PPTX sources."""
|
|
545
554
|
if isinstance(source, bytes):
|
|
@@ -555,6 +564,8 @@ def _parse_office_with_images(
|
|
|
555
564
|
raw_md = _office_to_markdown_via_markitdown(str(source))
|
|
556
565
|
|
|
557
566
|
markdown, images = _extract_data_url_images(raw_md)
|
|
567
|
+
# Before interpretation, so dropped images cost no VLM call.
|
|
568
|
+
markdown, images = drop_small_images(markdown, images, min_image_size)
|
|
558
569
|
if not include_tables:
|
|
559
570
|
markdown = strip_markdown_tables(markdown)
|
|
560
571
|
if images:
|
|
@@ -574,6 +585,7 @@ def parse_docx_with_images(
|
|
|
574
585
|
*,
|
|
575
586
|
include_tables: bool = True,
|
|
576
587
|
deadline: float | None = None,
|
|
588
|
+
min_image_size: int = 0,
|
|
577
589
|
) -> ParseResult:
|
|
578
590
|
"""Parse a DOCX into markdown + interpreted images.
|
|
579
591
|
|
|
@@ -599,6 +611,7 @@ def parse_docx_with_images(
|
|
|
599
611
|
suffix=".docx",
|
|
600
612
|
include_tables=include_tables,
|
|
601
613
|
deadline=deadline,
|
|
614
|
+
min_image_size=min_image_size,
|
|
602
615
|
)
|
|
603
616
|
|
|
604
617
|
|
|
@@ -608,6 +621,7 @@ def parse_pptx_with_images(
|
|
|
608
621
|
*,
|
|
609
622
|
include_tables: bool = True,
|
|
610
623
|
deadline: float | None = None,
|
|
624
|
+
min_image_size: int = 0,
|
|
611
625
|
) -> ParseResult:
|
|
612
626
|
"""Parse a PPTX into markdown + interpreted embedded pictures.
|
|
613
627
|
|
|
@@ -630,4 +644,5 @@ def parse_pptx_with_images(
|
|
|
630
644
|
suffix=".pptx",
|
|
631
645
|
include_tables=include_tables,
|
|
632
646
|
deadline=deadline,
|
|
647
|
+
min_image_size=min_image_size,
|
|
633
648
|
)
|