dot-parser 2.0.0__tar.gz → 2.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. dot_parser-2.1.0/CHANGELOG.md +88 -0
  2. {dot_parser-2.0.0 → dot_parser-2.1.0}/PKG-INFO +4 -2
  3. {dot_parser-2.0.0 → dot_parser-2.1.0}/README.md +1 -1
  4. {dot_parser-2.0.0 → dot_parser-2.1.0}/docs/VERSIONING.md +3 -3
  5. {dot_parser-2.0.0 → dot_parser-2.1.0}/pyproject.toml +5 -1
  6. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/__init__.py +4 -1
  7. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/docx_images.py +16 -1
  8. dot_parser-2.1.0/src/dot_parser/docx_markdown.py +490 -0
  9. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/image_utils.py +46 -1
  10. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/models.py +12 -0
  11. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/parsers.py +27 -2
  12. dot_parser-2.1.0/tests/test_docx_markdown.py +429 -0
  13. {dot_parser-2.0.0 → dot_parser-2.1.0}/tests/test_images.py +88 -0
  14. {dot_parser-2.0.0 → dot_parser-2.1.0}/tests/test_parsers.py +4 -3
  15. {dot_parser-2.0.0 → dot_parser-2.1.0}/uv.lock +5 -1
  16. dot_parser-2.0.0/CHANGELOG.md +0 -45
  17. {dot_parser-2.0.0 → dot_parser-2.1.0}/.gitignore +0 -0
  18. {dot_parser-2.0.0 → dot_parser-2.1.0}/.gitlab-ci.yml +0 -0
  19. {dot_parser-2.0.0 → dot_parser-2.1.0}/.pre-commit-config.yaml +0 -0
  20. {dot_parser-2.0.0 → dot_parser-2.1.0}/.python-version +0 -0
  21. {dot_parser-2.0.0 → dot_parser-2.1.0}/AUTHORS.md +0 -0
  22. {dot_parser-2.0.0 → dot_parser-2.1.0}/CONTRIBUTING.md +0 -0
  23. {dot_parser-2.0.0 → dot_parser-2.1.0}/DCO +0 -0
  24. {dot_parser-2.0.0 → dot_parser-2.1.0}/LICENSE.md +0 -0
  25. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/README.md +0 -0
  26. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/input/broken_encoding.pdf +0 -0
  27. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/input/image_only.pdf +0 -0
  28. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/input/native_equations.pdf +0 -0
  29. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/input/native_simple.pdf +0 -0
  30. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/input/native_tables.pdf +0 -0
  31. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_docling_broken_encoding.md +0 -0
  32. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_docling_image_only.md +0 -0
  33. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_docling_native_equations.md +0 -0
  34. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_docling_native_simple.md +0 -0
  35. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_docling_native_tables.md +0 -0
  36. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_llama_broken_encoding.md +0 -0
  37. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_llama_image_only.md +0 -0
  38. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_llama_native_equations.md +0 -0
  39. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_llama_native_simple.md +0 -0
  40. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_llama_native_tables.md +0 -0
  41. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_mistral_batch_broken_encoding.md +0 -0
  42. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_mistral_batch_image_only.md +0 -0
  43. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_mistral_batch_native_equations.md +0 -0
  44. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_mistral_batch_native_simple.md +0 -0
  45. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_mistral_batch_native_tables.md +0 -0
  46. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_broken_encoding.md +0 -0
  47. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_image_only.md +0 -0
  48. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_native_equations.md +0 -0
  49. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_native_simple.md +0 -0
  50. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_native_tables.md +0 -0
  51. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_ocr_broken_encoding.md +0 -0
  52. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_ocr_image_only.md +0 -0
  53. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_ocr_native_equations.md +0 -0
  54. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_ocr_native_simple.md +0 -0
  55. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/md/out_pymu_ocr_native_tables.md +0 -0
  56. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/output/results.md +0 -0
  57. {dot_parser-2.0.0 → dot_parser-2.1.0}/benchmark/run.py +0 -0
  58. {dot_parser-2.0.0 → dot_parser-2.1.0}/docs/DESIGN.md +0 -0
  59. {dot_parser-2.0.0 → dot_parser-2.1.0}/docs/DEVELOPMENT.md +0 -0
  60. {dot_parser-2.0.0 → dot_parser-2.1.0}/docs/IMAGE_EXTRACTION.md +0 -0
  61. {dot_parser-2.0.0 → dot_parser-2.1.0}/docs/PUBLISHING.md +0 -0
  62. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/backends/__init__.py +0 -0
  63. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/backends/_base.py +0 -0
  64. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/backends/docling.py +0 -0
  65. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/backends/llama.py +0 -0
  66. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/backends/mistral.py +0 -0
  67. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/backends/pymu.py +0 -0
  68. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/chunking.py +0 -0
  69. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/images.py +0 -0
  70. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/markdown_utils.py +0 -0
  71. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/pricing.py +0 -0
  72. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/tokens.py +0 -0
  73. {dot_parser-2.0.0 → dot_parser-2.1.0}/src/dot_parser/vlms.py +0 -0
  74. {dot_parser-2.0.0 → dot_parser-2.1.0}/tests/test_chunking.py +0 -0
  75. {dot_parser-2.0.0 → dot_parser-2.1.0}/tests/test_mistral_batch_logging.py +0 -0
  76. {dot_parser-2.0.0 → dot_parser-2.1.0}/tests/test_tokens.py +0 -0
@@ -0,0 +1,88 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project will be documented in this file.
4
+
5
+ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
6
+ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
+
8
+ ## [Unreleased]
9
+
10
+ ## [2.1.0] - 2026-09-30
11
+
12
+ ### Added
13
+
14
+ - `extract_docx_toc()` returns the table of contents stored in a DOCX as
15
+ `TocEntry(level, text)` items (`TocEntry` is exported), whether Word wrapped
16
+ it in a content control or inserted it as a bare field.
17
+ - `min_image_size=N` on `parse_with_images()`, `parse_docx_with_images()` and
18
+ `parse_pptx_with_images()` drops images whose shorter side is under `N`
19
+ pixels, with their anchors: letters or words pasted as pictures, icons,
20
+ separator lines. On the markitdown + VLM path they are dropped before
21
+ interpretation, so they cost no VLM call. `0` (default) keeps every image;
22
+ images whose size can't be read (EMF, WMF, SVG) are always kept.
23
+
24
+ ### Changed
25
+
26
+ - DOCX conversion (`parse()` and the markitdown + VLM image path) now runs
27
+ through one converter, so a DOCX yields the same text with or without image
28
+ extraction. It keeps markitdown's pipeline and fixes, in the Markdown it
29
+ produces:
30
+ - headings in custom styles derived from `Heading N` or carrying an outline
31
+ level (`Style Titre 2`, `Annexe`...) are emitted as headings instead of
32
+ body text;
33
+ - headings carry Word's automatic numbers (`# 1. Introduction`,
34
+ `## 2.3 Objet`), recomputed from `numbering.xml`; a number already typed
35
+ in the heading is not doubled;
36
+ - `Title` and `Subtitle` paragraphs (cover page) are rendered bold; like
37
+ Word's own TOC, they are not treated as headings;
38
+ - bold inside a heading is dropped (`## Zoom`, not `## **Zoom**`);
39
+ - the stored table of contents and the lists of figures and tables are
40
+ left out: they repeat headings and captions with page numbers, and
41
+ `extract_docx_toc()` exposes the TOC;
42
+ - tables use their first row as the header instead of a blank one, escape
43
+ `|` in cells, stay rectangular across merged cells (a vertically merged
44
+ value is repeated in each row it spans) and flatten nested tables into
45
+ their parent cell.
46
+ - `parse()` drops embedded images from a DOCX instead of emitting
47
+ `![alt](data:image/png;base64...)` anchors with a truncated payload.
48
+ - `parse()` raises `ParseError` on a corrupted DOCX instead of returning its
49
+ raw bytes as text.
50
+ - `mammoth` and `beautifulsoup4` are declared as direct dependencies.
51
+
52
+ ## [2.0.0] - 2026-07-30
53
+
54
+ First public release.
55
+
56
+ ### Added
57
+
58
+ - `parse()` converts PDF, DOCX, PPTX, HTML, XLSX, CSV, Markdown and plain text
59
+ into clean Markdown.
60
+ - Selectable PDF backends: `Pymu` (default, local and fast), `Docling` (local
61
+ layout-aware ML pipeline), `Mistral` (cloud OCR) and `Llama` (LlamaCloud).
62
+ - `parse_with_images()` extracts images alongside the Markdown, with anchors
63
+ kept at their position in the text. Images are described either by a VLM or
64
+ by Mistral OCR annotations in the same call.
65
+ - `parse_pdfs()` parses many PDFs in one batch job through the Mistral Batch
66
+ API, falling back to a per-file loop on backends without batch support.
67
+ - `chunk()` splits Markdown into chunks carrying their heading hierarchy in
68
+ `section_path`.
69
+ - Cost helpers: `cost_per_1k_pages()` and `estimate_cost()`, usable without
70
+ instantiating a backend.
71
+ - Backend protocols (`Backend`, `ImageBackend`, `BatchBackend`,
72
+ `DocxImageBackend`, `PptxImageBackend`) so third-party backends can plug in.
73
+
74
+ ### Changed
75
+
76
+ These matter only if you were installing the package straight from git before
77
+ this release.
78
+
79
+ - `LlamaTier` is a `Literal` of the four accepted tiers rather than a bare
80
+ `str`, and is defined once alongside the pricing table.
81
+ - `zip()` calls over sequences that must line up now use `strict=True`: a length
82
+ mismatch raises instead of silently truncating.
83
+ - `parse_with_images(backend=...)` accepts DOCX- and PPTX-capable backends in
84
+ its type signature, matching what the function already supported at runtime.
85
+
86
+ [Unreleased]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.1.0...main
87
+ [2.1.0]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/compare/v2.0.0...v2.1.0
88
+ [2.0.0]: https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser/-/tags/v2.0.0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: dot-parser
3
- Version: 2.0.0
3
+ Version: 2.1.0
4
4
  Summary: Document-to-markdown parser and chunker for RAG pipelines
5
5
  Project-URL: Homepage, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
6
6
  Project-URL: Repository, https://gitlab.com/deepika6190303/deepika-open-toolbox/dot-parser
@@ -15,6 +15,8 @@ Classifier: Programming Language :: Python :: 3
15
15
  Classifier: Programming Language :: Python :: 3.12
16
16
  Classifier: Programming Language :: Python :: 3.13
17
17
  Requires-Python: <3.14,>=3.12
18
+ Requires-Dist: beautifulsoup4>=4.12
19
+ Requires-Dist: mammoth>=1.11
18
20
  Requires-Dist: markitdown[all]>=0.1
19
21
  Requires-Dist: pillow>=10.0
20
22
  Requires-Dist: pydantic>=2
@@ -235,7 +237,7 @@ package is covered, anything underscore-prefixed is internal and may change in
235
237
  any release. Public names are never removed without a deprecation period.
236
238
 
237
239
  ```toml
238
- dependencies = ["dot-parser>=1.0,<2"]
240
+ dependencies = ["dot-parser>=2.0,<3"]
239
241
  ```
240
242
 
241
243
  See [docs/VERSIONING.md](docs/VERSIONING.md) for the full policy.
@@ -199,7 +199,7 @@ package is covered, anything underscore-prefixed is internal and may change in
199
199
  any release. Public names are never removed without a deprecation period.
200
200
 
201
201
  ```toml
202
- dependencies = ["dot-parser>=1.0,<2"]
202
+ dependencies = ["dot-parser>=2.0,<3"]
203
203
  ```
204
204
 
205
205
  See [docs/VERSIONING.md](docs/VERSIONING.md) for the full policy.
@@ -1,7 +1,7 @@
1
1
  # Versioning and stability
2
2
 
3
3
  `dot-parser` follows [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
4
- This page says what that actually covers, so `>=1.0,<2` means something concrete.
4
+ This page says what that actually covers, so `>=2.0,<3` means something concrete.
5
5
 
6
6
  ## What is public
7
7
 
@@ -94,11 +94,11 @@ manual. Use them to validate a release without committing to it.
94
94
  ### From PyPI — the supported way
95
95
 
96
96
  ```toml
97
- dependencies = ["dot-parser>=1.0,<2"]
97
+ dependencies = ["dot-parser>=2.0,<3"]
98
98
  ```
99
99
 
100
100
  The upper bound is what makes the guarantees above useful: you get every fix and
101
- feature of the 1.x line, and never an unannounced break.
101
+ feature of the 2.x line, and never an unannounced break.
102
102
 
103
103
  ### From git — development only
104
104
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "dot-parser"
3
- version = "2.0.0"
3
+ version = "2.1.0"
4
4
  description = "Document-to-markdown parser and chunker for RAG pipelines"
5
5
  readme = "README.md"
6
6
  authors = [{ name = "Kannon For Deep Tech", email = "louis.letarnec@deepika.ai" }]
@@ -12,6 +12,10 @@ requires-python = ">=3.12,<3.14"
12
12
  dependencies = [
13
13
  "pymupdf4llm>=1.27.2.2",
14
14
  "markitdown[all]>=0.1",
15
+ # Imported directly by the DOCX converter (docx_markdown.py); markitdown
16
+ # pulls both in too, but only through its optional extras.
17
+ "mammoth>=1.11",
18
+ "beautifulsoup4>=4.12",
15
19
  "semchunk>=3.0",
16
20
  "python-docx>=1.2.0",
17
21
  "pillow>=10.0",
@@ -5,6 +5,7 @@ from importlib.metadata import version
5
5
 
6
6
  from dot_parser.backends import Backend, Docling, ImageBackend, Llama, Mistral, Pymu
7
7
  from dot_parser.chunking import chunk
8
+ from dot_parser.docx_markdown import extract_docx_toc
8
9
  from dot_parser.images import (
9
10
  VLM,
10
11
  ExtractedImage,
@@ -12,7 +13,7 @@ from dot_parser.images import (
12
13
  PageInfo,
13
14
  ParseResult,
14
15
  )
15
- from dot_parser.models import Chunk, ParseError
16
+ from dot_parser.models import Chunk, ParseError, TocEntry
16
17
  from dot_parser.parsers import (
17
18
  interpret_images,
18
19
  parse,
@@ -39,9 +40,11 @@ __all__ = [
39
40
  "ParseError",
40
41
  "ParseResult",
41
42
  "Pymu",
43
+ "TocEntry",
42
44
  "chunk",
43
45
  "cost_per_1k_pages",
44
46
  "estimate_cost",
47
+ "extract_docx_toc",
45
48
  "interpret_images",
46
49
  "parse",
47
50
  "parse_pdfs",
@@ -34,7 +34,7 @@ import tempfile
34
34
  import time
35
35
  from pathlib import Path
36
36
 
37
- from dot_parser.image_utils import dedupe_titles
37
+ from dot_parser.image_utils import dedupe_titles, drop_small_images
38
38
  from dot_parser.images import (
39
39
  VLM,
40
40
  ExtractedImage,
@@ -523,7 +523,15 @@ def _office_to_markdown_via_markitdown(file_path: str) -> str:
523
523
  first comma (replacing the base64 payload with ``...``) to keep
524
524
  markdown small for the LLM-summarisation use case it was designed
525
525
  for. We need the full payload to extract the image bytes.
526
+
527
+ DOCX goes through :func:`dot_parser.docx_markdown.docx_to_markdown`,
528
+ the same converter :func:`dot_parser.parse` uses, so the text is
529
+ identical with or without image extraction.
526
530
  """
531
+ if Path(file_path).suffix.lower() == ".docx":
532
+ from dot_parser.docx_markdown import docx_to_markdown
533
+
534
+ return docx_to_markdown(file_path, keep_data_uris=True)
527
535
  try:
528
536
  import markitdown
529
537
  except ImportError as e:
@@ -540,6 +548,7 @@ def _parse_office_with_images(
540
548
  suffix: str,
541
549
  include_tables: bool,
542
550
  deadline: float | None = None,
551
+ min_image_size: int = 0,
543
552
  ) -> ParseResult:
544
553
  """Shared markitdown+VLM pipeline for DOCX and PPTX sources."""
545
554
  if isinstance(source, bytes):
@@ -555,6 +564,8 @@ def _parse_office_with_images(
555
564
  raw_md = _office_to_markdown_via_markitdown(str(source))
556
565
 
557
566
  markdown, images = _extract_data_url_images(raw_md)
567
+ # Before interpretation, so dropped images cost no VLM call.
568
+ markdown, images = drop_small_images(markdown, images, min_image_size)
558
569
  if not include_tables:
559
570
  markdown = strip_markdown_tables(markdown)
560
571
  if images:
@@ -574,6 +585,7 @@ def parse_docx_with_images(
574
585
  *,
575
586
  include_tables: bool = True,
576
587
  deadline: float | None = None,
588
+ min_image_size: int = 0,
577
589
  ) -> ParseResult:
578
590
  """Parse a DOCX into markdown + interpreted images.
579
591
 
@@ -599,6 +611,7 @@ def parse_docx_with_images(
599
611
  suffix=".docx",
600
612
  include_tables=include_tables,
601
613
  deadline=deadline,
614
+ min_image_size=min_image_size,
602
615
  )
603
616
 
604
617
 
@@ -608,6 +621,7 @@ def parse_pptx_with_images(
608
621
  *,
609
622
  include_tables: bool = True,
610
623
  deadline: float | None = None,
624
+ min_image_size: int = 0,
611
625
  ) -> ParseResult:
612
626
  """Parse a PPTX into markdown + interpreted embedded pictures.
613
627
 
@@ -630,4 +644,5 @@ def parse_pptx_with_images(
630
644
  suffix=".pptx",
631
645
  include_tables=include_tables,
632
646
  deadline=deadline,
647
+ min_image_size=min_image_size,
633
648
  )