dot-parser 2.0.0__tar.gz → 2.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dot_parser-2.1.0/CHANGELOG.md +88 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/PKG-INFO +4 -2
- {dot_parser-2.0.0 → dot_parser-2.1.0}/README.md +1 -1
- {dot_parser-2.0.0 → dot_parser-2.1.0}/docs/VERSIONING.md +3 -3
- {dot_parser-2.0.0 → dot_parser-2.1.0}/pyproject.toml +5 -1
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/__init__.py +4 -1
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/docx_images.py +16 -1
- dot_parser-2.1.0/src/dot_parser/docx_markdown.py +490 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/image_utils.py +46 -1
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/models.py +12 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/parsers.py +27 -2
- dot_parser-2.1.0/tests/test_docx_markdown.py +429 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/tests/test_images.py +88 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/tests/test_parsers.py +4 -3
- {dot_parser-2.0.0 → dot_parser-2.1.0}/uv.lock +5 -1
- dot_parser-2.0.0/CHANGELOG.md +0 -45
- {dot_parser-2.0.0 → dot_parser-2.1.0}/.gitignore +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/.gitlab-ci.yml +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/.pre-commit-config.yaml +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/.python-version +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/AUTHORS.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/CONTRIBUTING.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/DCO +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/LICENSE.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/README.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/input/broken_encoding.pdf +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/input/image_only.pdf +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/input/native_equations.pdf +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/input/native_simple.pdf +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/input/native_tables.pdf +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_docling_broken_encoding.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_docling_image_only.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_docling_native_equations.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_docling_native_simple.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_docling_native_tables.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_llama_broken_encoding.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_llama_image_only.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_llama_native_equations.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_llama_native_simple.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_llama_native_tables.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_mistral_batch_broken_encoding.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_mistral_batch_image_only.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_mistral_batch_native_equations.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_mistral_batch_native_simple.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_mistral_batch_native_tables.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_broken_encoding.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_image_only.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_native_equations.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_native_simple.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_native_tables.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_ocr_broken_encoding.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_ocr_image_only.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_ocr_native_equations.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_ocr_native_simple.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_ocr_native_tables.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/results.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/run.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/docs/DESIGN.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/docs/DEVELOPMENT.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/docs/IMAGE_EXTRACTION.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/docs/PUBLISHING.md +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/backends/__init__.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/backends/_base.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/backends/docling.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/backends/llama.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/backends/mistral.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/backends/pymu.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/chunking.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/images.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/markdown_utils.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/pricing.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/tokens.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/vlms.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/tests/test_chunking.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/tests/test_mistral_batch_logging.py +0 -0
- {dot_parser-2.0.0 → dot_parser-2.1.0}/tests/test_tokens.py +0 -0
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project will be documented in this file.
|
|
4
|
+
|
|
5
|
+
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
6
|
+
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
|
+
|
|
8
|
+
## [Unreleased]
|
|
9
|
+
|
|
10
|
+
## [2.1.0] - 2026-09-30
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
|
|
14
|
+
- `extract_docx_toc()` returns the table of contents stored in a DOCX as
|
|
15
|
+
`TocEntry(level, text)` items (`TocEntry` is exported), whether Word wrapped
|
|
16
|
+
it in a content control or inserted it as a bare field.
|
|
17
|
+
- `min_image_size=N` on `parse_with_images()`, `parse_docx_with_images()` and
|
|
18
|
+
`parse_pptx_with_images()` drops images whose shorter side is under `N`
|
|
19
|
+
pixels, with their anchors: letters or words pasted as pictures, icons,
|
|
20
|
+
separator lines. On the markitdown + VLM path they are dropped before
|
|
21
|
+
interpretation, so they cost no VLM call. `0` (default) keeps every image;
|
|
22
|
+
images whose size can't be read (EMF, WMF, SVG) are always kept.
|
|
23
|
+
|
|
24
|
+
### Changed
|
|
25
|
+
|
|
26
|
+
- DOCX conversion (`parse()` and the markitdown + VLM image path) now runs
|
|
27
|
+
through one converter, so a DOCX yields the same text with or without image
|
|
28
|
+
extraction. It keeps markitdown's pipeline and fixes, in the Markdown it
|
|
29
|
+
produces:
|
|
30
|
+
- headings in custom styles derived from `Heading N` or carrying an outline
|
|
31
|
+
level (`Style Titre 2`, `Annexe`...) are emitted as headings instead of
|
|
32
|
+
body text;
|
|
33
|
+
- headings carry Word's automatic numbers (`# 1. Introduction`,
|
|
34
|
+
`## 2.3 Objet`), recomputed from `numbering.xml`; a number already typed
|
|
35
|
+
in the heading is not doubled;
|
|
36
|
+
- `Title` and `Subtitle` paragraphs (cover page) are rendered bold; like
|
|
37
|
+
Word's own TOC, they are not treated as headings;
|
|
38
|
+
- bold inside a heading is dropped (`## Zoom`, not `## **Zoom**`);
|
|
39
|
+
- the stored table of contents and the lists of figures and tables are
|
|
40
|
+
left out: they repeat headings and captions with page numbers, and
|
|
41
|
+
`extract_docx_toc()` exposes the TOC;
|
|
42
|
+
- tables use their first row as the header instead of a blank one, escape
|
|
43
|
+
`|` in cells, stay rectangular across merged cells (a vertically merged
|
|
44
|
+
value is repeated in each row it spans) and flatten nested tables into
|
|
45
|
+
their parent cell.
|
|
46
|
+
- `parse()` drops embedded images from a DOCX instead of emitting
|
|
47
|
+
`` anchors with a truncated payload.
|
|
48
|
+
- `parse()` raises `ParseError` on a corrupted DOCX instead of returning its
|
|
49
|
+
raw bytes as text.
|
|
50
|
+
- `mammoth` and `beautifulsoup4` are declared as direct dependencies.
|
|
51
|
+
|
|
52
|
+
## [2.0.0] - 2026-07-30
|
|
53
|
+
|
|
54
|
+
First public release.
|
|
55
|
+
|
|
56
|
+
### Added
|
|
57
|
+
|
|
58
|
+
- `parse()` converts PDF, DOCX, PPTX, HTML, XLSX, CSV, Markdown and plain text
|
|
59
|
+
into clean Markdown.
|
|
60
|
+
- Selectable PDF backends: `Pymu` (default, local and fast), `Docling` (local
|
|
61
|
+
layout-aware ML pipeline), `Mistral` (cloud OCR) and `Llama` (LlamaCloud).
|
|
62
|
+
- `parse_with_images()` extracts images alongside the Markdown, with anchors
|
|
63
|
+
kept at their position in the text. Images are described either by a VLM or
|
|
64
|
+
by Mistral OCR annotations in the same call.
|
|
65
|
+
- `parse_pdfs()` parses many PDFs in one batch job through the Mistral Batch
|
|
66
|
+
API, falling back to a per-file loop on backends without batch support.
|
|
67
|
+
- `chunk()` splits Markdown into chunks carrying their heading hierarchy in
|
|
68
|
+
`section_path`.
|
|
69
|
+
- Cost helpers: `cost_per_1k_pages()` and `estimate_cost()`, usable without
|
|
70
|
+
instantiating a backend.
|
|
71
|
+
- Backend protocols (`Backend`, `ImageBackend`, `BatchBackend`,
|
|
72
|
+
`DocxImageBackend`, `PptxImageBackend`) so third-party backends can plug in.
|
|
73
|
+
|
|
74
|
+
### Changed
|
|
75
|
+
|
|
76
|
+
These matter only if you were installing the package straight from git before
|
|
77
|
+
this release.
|
|
78
|
+
|
|
79
|
+
- `LlamaTier` is a `Literal` of the four accepted tiers rather than a bare
|
|
80
|
+
`str`, and is defined once alongside the pricing table.
|
|
81
|
+
- `zip()` calls over sequences that must line up now use `strict=True`: a length
|
|
82
|
+
mismatch raises instead of silently truncating.
|
|
83
|
+
- `parse_with_images(backend=...)` accepts DOCX- and PPTX-capable backends in
|
|
84
|
+
its type signature, matching what the function already supported at runtime.
|
|
85
|
+
|
|
86
|
+
[Unreleased]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.1.0...main
|
|
87
|
+
[2.1.0]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.0.0...v2.1.0
|
|
88
|
+
[2.0.0]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/tags/v2.0.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: dot-parser
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.1.0
|
|
4
4
|
Summary: Document-to-markdown parser and chunker for RAG pipelines
|
|
5
5
|
Project-URL: Homepage, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
|
|
6
6
|
Project-URL: Repository, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
|
|
@@ -15,6 +15,8 @@ Classifier: Programming Language :: Python :: 3
|
|
|
15
15
|
Classifier: Programming Language :: Python :: 3.12
|
|
16
16
|
Classifier: Programming Language :: Python :: 3.13
|
|
17
17
|
Requires-Python: <3.14,>=3.12
|
|
18
|
+
Requires-Dist: beautifulsoup4>=4.12
|
|
19
|
+
Requires-Dist: mammoth>=1.11
|
|
18
20
|
Requires-Dist: markitdown[all]>=0.1
|
|
19
21
|
Requires-Dist: pillow>=10.0
|
|
20
22
|
Requires-Dist: pydantic>=2
|
|
@@ -235,7 +237,7 @@ package is covered, anything underscore-prefixed is internal and may change in
|
|
|
235
237
|
any release. Public names are never removed without a deprecation period.
|
|
236
238
|
|
|
237
239
|
```toml
|
|
238
|
-
dependencies = ["dot-parser>=
|
|
240
|
+
dependencies = ["dot-parser>=2.0,<3"]
|
|
239
241
|
```
|
|
240
242
|
|
|
241
243
|
See [docs/VERSIONING.md](docs/VERSIONING.md) for the full policy.
|
|
@@ -199,7 +199,7 @@ package is covered, anything underscore-prefixed is internal and may change in
|
|
|
199
199
|
any release. Public names are never removed without a deprecation period.
|
|
200
200
|
|
|
201
201
|
```toml
|
|
202
|
-
dependencies = ["dot-parser>=
|
|
202
|
+
dependencies = ["dot-parser>=2.0,<3"]
|
|
203
203
|
```
|
|
204
204
|
|
|
205
205
|
See [docs/VERSIONING.md](docs/VERSIONING.md) for the full policy.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Versioning and stability
|
|
2
2
|
|
|
3
3
|
`dot-parser` follows [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
4
|
-
This page says what that actually covers, so `>=
|
|
4
|
+
This page says what that actually covers, so `>=2.0,<3` means something concrete.
|
|
5
5
|
|
|
6
6
|
## What is public
|
|
7
7
|
|
|
@@ -94,11 +94,11 @@ manual. Use them to validate a release without committing to it.
|
|
|
94
94
|
### From PyPI — the supported way
|
|
95
95
|
|
|
96
96
|
```toml
|
|
97
|
-
dependencies = ["dot-parser>=
|
|
97
|
+
dependencies = ["dot-parser>=2.0,<3"]
|
|
98
98
|
```
|
|
99
99
|
|
|
100
100
|
The upper bound is what makes the guarantees above useful: you get every fix and
|
|
101
|
-
feature of the
|
|
101
|
+
feature of the 2.x line, and never an unannounced break.
|
|
102
102
|
|
|
103
103
|
### From git — development only
|
|
104
104
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "dot-parser"
|
|
3
|
-
version = "2.
|
|
3
|
+
version = "2.1.0"
|
|
4
4
|
description = "Document-to-markdown parser and chunker for RAG pipelines"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
authors = [{ name = "Kannon For Deep Tech", email = "louis.letarnec@deepika.ai" }]
|
|
@@ -12,6 +12,10 @@ requires-python = ">=3.12,<3.14"
|
|
|
12
12
|
dependencies = [
|
|
13
13
|
"pymupdf4llm>=1.27.2.2",
|
|
14
14
|
"markitdown[all]>=0.1",
|
|
15
|
+
# Imported directly by the DOCX converter (docx_markdown.py); markitdown
|
|
16
|
+
# pulls both in too, but only through its optional extras.
|
|
17
|
+
"mammoth>=1.11",
|
|
18
|
+
"beautifulsoup4>=4.12",
|
|
15
19
|
"semchunk>=3.0",
|
|
16
20
|
"python-docx>=1.2.0",
|
|
17
21
|
"pillow>=10.0",
|
|
@@ -5,6 +5,7 @@ from importlib.metadata import version
|
|
|
5
5
|
|
|
6
6
|
from dot_parser.backends import Backend, Docling, ImageBackend, Llama, Mistral, Pymu
|
|
7
7
|
from dot_parser.chunking import chunk
|
|
8
|
+
from dot_parser.docx_markdown import extract_docx_toc
|
|
8
9
|
from dot_parser.images import (
|
|
9
10
|
VLM,
|
|
10
11
|
ExtractedImage,
|
|
@@ -12,7 +13,7 @@ from dot_parser.images import (
|
|
|
12
13
|
PageInfo,
|
|
13
14
|
ParseResult,
|
|
14
15
|
)
|
|
15
|
-
from dot_parser.models import Chunk, ParseError
|
|
16
|
+
from dot_parser.models import Chunk, ParseError, TocEntry
|
|
16
17
|
from dot_parser.parsers import (
|
|
17
18
|
interpret_images,
|
|
18
19
|
parse,
|
|
@@ -39,9 +40,11 @@ __all__ = [
|
|
|
39
40
|
"ParseError",
|
|
40
41
|
"ParseResult",
|
|
41
42
|
"Pymu",
|
|
43
|
+
"TocEntry",
|
|
42
44
|
"chunk",
|
|
43
45
|
"cost_per_1k_pages",
|
|
44
46
|
"estimate_cost",
|
|
47
|
+
"extract_docx_toc",
|
|
45
48
|
"interpret_images",
|
|
46
49
|
"parse",
|
|
47
50
|
"parse_pdfs",
|
|
@@ -34,7 +34,7 @@ import tempfile
|
|
|
34
34
|
import time
|
|
35
35
|
from pathlib import Path
|
|
36
36
|
|
|
37
|
-
from dot_parser.image_utils import dedupe_titles
|
|
37
|
+
from dot_parser.image_utils import dedupe_titles, drop_small_images
|
|
38
38
|
from dot_parser.images import (
|
|
39
39
|
VLM,
|
|
40
40
|
ExtractedImage,
|
|
@@ -523,7 +523,15 @@ def _office_to_markdown_via_markitdown(file_path: str) -> str:
|
|
|
523
523
|
first comma (replacing the base64 payload with ``...``) to keep
|
|
524
524
|
markdown small for the LLM-summarisation use case it was designed
|
|
525
525
|
for. We need the full payload to extract the image bytes.
|
|
526
|
+
|
|
527
|
+
DOCX goes through :func:`dot_parser.docx_markdown.docx_to_markdown`,
|
|
528
|
+
the same converter :func:`dot_parser.parse` uses, so the text is
|
|
529
|
+
identical with or without image extraction.
|
|
526
530
|
"""
|
|
531
|
+
if Path(file_path).suffix.lower() == ".docx":
|
|
532
|
+
from dot_parser.docx_markdown import docx_to_markdown
|
|
533
|
+
|
|
534
|
+
return docx_to_markdown(file_path, keep_data_uris=True)
|
|
527
535
|
try:
|
|
528
536
|
import markitdown
|
|
529
537
|
except ImportError as e:
|
|
@@ -540,6 +548,7 @@ def _parse_office_with_images(
|
|
|
540
548
|
suffix: str,
|
|
541
549
|
include_tables: bool,
|
|
542
550
|
deadline: float | None = None,
|
|
551
|
+
min_image_size: int = 0,
|
|
543
552
|
) -> ParseResult:
|
|
544
553
|
"""Shared markitdown+VLM pipeline for DOCX and PPTX sources."""
|
|
545
554
|
if isinstance(source, bytes):
|
|
@@ -555,6 +564,8 @@ def _parse_office_with_images(
|
|
|
555
564
|
raw_md = _office_to_markdown_via_markitdown(str(source))
|
|
556
565
|
|
|
557
566
|
markdown, images = _extract_data_url_images(raw_md)
|
|
567
|
+
# Before interpretation, so dropped images cost no VLM call.
|
|
568
|
+
markdown, images = drop_small_images(markdown, images, min_image_size)
|
|
558
569
|
if not include_tables:
|
|
559
570
|
markdown = strip_markdown_tables(markdown)
|
|
560
571
|
if images:
|
|
@@ -574,6 +585,7 @@ def parse_docx_with_images(
|
|
|
574
585
|
*,
|
|
575
586
|
include_tables: bool = True,
|
|
576
587
|
deadline: float | None = None,
|
|
588
|
+
min_image_size: int = 0,
|
|
577
589
|
) -> ParseResult:
|
|
578
590
|
"""Parse a DOCX into markdown + interpreted images.
|
|
579
591
|
|
|
@@ -599,6 +611,7 @@ def parse_docx_with_images(
|
|
|
599
611
|
suffix=".docx",
|
|
600
612
|
include_tables=include_tables,
|
|
601
613
|
deadline=deadline,
|
|
614
|
+
min_image_size=min_image_size,
|
|
602
615
|
)
|
|
603
616
|
|
|
604
617
|
|
|
@@ -608,6 +621,7 @@ def parse_pptx_with_images(
|
|
|
608
621
|
*,
|
|
609
622
|
include_tables: bool = True,
|
|
610
623
|
deadline: float | None = None,
|
|
624
|
+
min_image_size: int = 0,
|
|
611
625
|
) -> ParseResult:
|
|
612
626
|
"""Parse a PPTX into markdown + interpreted embedded pictures.
|
|
613
627
|
|
|
@@ -630,4 +644,5 @@ def parse_pptx_with_images(
|
|
|
630
644
|
suffix=".pptx",
|
|
631
645
|
include_tables=include_tables,
|
|
632
646
|
deadline=deadline,
|
|
647
|
+
min_image_size=min_image_size,
|
|
633
648
|
)
|