dot-parser 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dot_parser/__init__.py +49 -0
- dot_parser/backends/__init__.py +26 -0
- dot_parser/backends/_base.py +79 -0
- dot_parser/backends/docling.py +59 -0
- dot_parser/backends/llama.py +76 -0
- dot_parser/backends/mistral.py +622 -0
- dot_parser/backends/pymu.py +56 -0
- dot_parser/chunking.py +257 -0
- dot_parser/docx_images.py +633 -0
- dot_parser/image_utils.py +152 -0
- dot_parser/images.py +125 -0
- dot_parser/markdown_utils.py +48 -0
- dot_parser/models.py +22 -0
- dot_parser/parsers.py +325 -0
- dot_parser/pricing.py +58 -0
- dot_parser/tokens.py +6 -0
- dot_parser/vlms.py +203 -0
- dot_parser-2.0.0.dist-info/METADATA +274 -0
- dot_parser-2.0.0.dist-info/RECORD +21 -0
- dot_parser-2.0.0.dist-info/WHEEL +4 -0
- dot_parser-2.0.0.dist-info/licenses/LICENSE.md +660 -0
dot_parser/__init__.py
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
from importlib.metadata import version
|
|
5
|
+
|
|
6
|
+
from dot_parser.backends import Backend, Docling, ImageBackend, Llama, Mistral, Pymu
|
|
7
|
+
from dot_parser.chunking import chunk
|
|
8
|
+
from dot_parser.images import (
|
|
9
|
+
VLM,
|
|
10
|
+
ExtractedImage,
|
|
11
|
+
ImageDescription,
|
|
12
|
+
PageInfo,
|
|
13
|
+
ParseResult,
|
|
14
|
+
)
|
|
15
|
+
from dot_parser.models import Chunk, ParseError
|
|
16
|
+
from dot_parser.parsers import (
|
|
17
|
+
interpret_images,
|
|
18
|
+
parse,
|
|
19
|
+
parse_pdfs,
|
|
20
|
+
parse_with_images,
|
|
21
|
+
)
|
|
22
|
+
from dot_parser.pricing import cost_per_1k_pages, estimate_cost
|
|
23
|
+
from dot_parser.vlms import MistralVLM
|
|
24
|
+
|
|
25
|
+
__version__ = version("dot-parser")
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"VLM",
|
|
29
|
+
"Backend",
|
|
30
|
+
"Chunk",
|
|
31
|
+
"Docling",
|
|
32
|
+
"ExtractedImage",
|
|
33
|
+
"ImageBackend",
|
|
34
|
+
"ImageDescription",
|
|
35
|
+
"Llama",
|
|
36
|
+
"Mistral",
|
|
37
|
+
"MistralVLM",
|
|
38
|
+
"PageInfo",
|
|
39
|
+
"ParseError",
|
|
40
|
+
"ParseResult",
|
|
41
|
+
"Pymu",
|
|
42
|
+
"chunk",
|
|
43
|
+
"cost_per_1k_pages",
|
|
44
|
+
"estimate_cost",
|
|
45
|
+
"interpret_images",
|
|
46
|
+
"parse",
|
|
47
|
+
"parse_pdfs",
|
|
48
|
+
"parse_with_images",
|
|
49
|
+
]
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
from dot_parser.backends._base import (
|
|
5
|
+
Backend,
|
|
6
|
+
BatchBackend,
|
|
7
|
+
DocxImageBackend,
|
|
8
|
+
ImageBackend,
|
|
9
|
+
PptxImageBackend,
|
|
10
|
+
)
|
|
11
|
+
from dot_parser.backends.docling import Docling
|
|
12
|
+
from dot_parser.backends.llama import Llama
|
|
13
|
+
from dot_parser.backends.mistral import Mistral
|
|
14
|
+
from dot_parser.backends.pymu import Pymu
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"Backend",
|
|
18
|
+
"BatchBackend",
|
|
19
|
+
"Docling",
|
|
20
|
+
"DocxImageBackend",
|
|
21
|
+
"ImageBackend",
|
|
22
|
+
"Llama",
|
|
23
|
+
"Mistral",
|
|
24
|
+
"PptxImageBackend",
|
|
25
|
+
"Pymu",
|
|
26
|
+
]
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Protocol, runtime_checkable
|
|
6
|
+
|
|
7
|
+
from dot_parser.images import ParseResult
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@runtime_checkable
|
|
11
|
+
class Backend(Protocol):
|
|
12
|
+
"""A PDF parsing backend that produces Markdown."""
|
|
13
|
+
|
|
14
|
+
def parse_pdf(self, source: str | Path | bytes) -> str: ...
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@runtime_checkable
|
|
18
|
+
class BatchBackend(Protocol):
|
|
19
|
+
"""A PDF backend that parses a whole list in one optimized call.
|
|
20
|
+
|
|
21
|
+
Backends implementing this protocol (e.g. ``Mistral`` via the Batch API)
|
|
22
|
+
take over ``parse_pdfs()`` entirely; the others fall back to a per-file
|
|
23
|
+
loop. Returns one Markdown string per input, in order, with ``None`` for
|
|
24
|
+
items that failed.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
def parse_pdfs(self, sources: list[str | Path | bytes]) -> list[str | None]: ...
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@runtime_checkable
|
|
31
|
+
class ImageBackend(Protocol):
|
|
32
|
+
"""A PDF backend that can also extract and annotate images.
|
|
33
|
+
|
|
34
|
+
Backends that implement this protocol are eligible for the
|
|
35
|
+
`parse_with_images()` PDF path. The returned markdown must contain
|
|
36
|
+
`` anchors at the position of each image, matching the
|
|
37
|
+
`ExtractedImage.name` of the corresponding image record.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
def parse_pdf_with_images(
|
|
41
|
+
self,
|
|
42
|
+
source: str | Path | bytes,
|
|
43
|
+
*,
|
|
44
|
+
annotate_images: bool = True,
|
|
45
|
+
include_tables: bool = True,
|
|
46
|
+
) -> ParseResult: ...
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@runtime_checkable
|
|
50
|
+
class DocxImageBackend(Protocol):
|
|
51
|
+
"""A backend that parses DOCX end-to-end through OCR.
|
|
52
|
+
|
|
53
|
+
When one is passed to `parse_with_images()`, the DOCX goes through the
|
|
54
|
+
backend directly instead of the markitdown + per-image VLM path.
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
def parse_docx_with_images(
|
|
58
|
+
self,
|
|
59
|
+
source: str | Path | bytes,
|
|
60
|
+
*,
|
|
61
|
+
annotate_images: bool = True,
|
|
62
|
+
include_tables: bool = True,
|
|
63
|
+
) -> ParseResult: ...
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@runtime_checkable
|
|
67
|
+
class PptxImageBackend(Protocol):
|
|
68
|
+
"""A backend that parses PPTX end-to-end through OCR.
|
|
69
|
+
|
|
70
|
+
PPTX counterpart of :class:`DocxImageBackend`.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
def parse_pptx_with_images(
|
|
74
|
+
self,
|
|
75
|
+
source: str | Path | bytes,
|
|
76
|
+
*,
|
|
77
|
+
annotate_images: bool = True,
|
|
78
|
+
include_tables: bool = True,
|
|
79
|
+
) -> ParseResult: ...
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
import tempfile
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class Docling:
|
|
9
|
+
"""Docling backend (IBM, MIT licence).
|
|
10
|
+
|
|
11
|
+
Layout-aware ML pipeline (DocLayNet) + table parser (TableFormer) + OCR.
|
|
12
|
+
Robust on image-only and broken-encoding PDFs when ``force_full_page_ocr``
|
|
13
|
+
is enabled.
|
|
14
|
+
|
|
15
|
+
Install with::
|
|
16
|
+
|
|
17
|
+
pip install dot-parser[docling]
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
def __init__(
|
|
21
|
+
self,
|
|
22
|
+
*,
|
|
23
|
+
force_full_page_ocr: bool = True,
|
|
24
|
+
ocr_lang: tuple[str, ...] = ("eng", "fra"),
|
|
25
|
+
) -> None:
|
|
26
|
+
try:
|
|
27
|
+
from docling.datamodel.base_models import InputFormat
|
|
28
|
+
from docling.datamodel.pipeline_options import (
|
|
29
|
+
PdfPipelineOptions,
|
|
30
|
+
TesseractCliOcrOptions,
|
|
31
|
+
)
|
|
32
|
+
from docling.document_converter import DocumentConverter, PdfFormatOption
|
|
33
|
+
except ImportError as e:
|
|
34
|
+
raise ImportError(
|
|
35
|
+
"Docling backend requires `docling`. Install with: pip install dot-parser[docling]"
|
|
36
|
+
) from e
|
|
37
|
+
|
|
38
|
+
opts = PdfPipelineOptions()
|
|
39
|
+
opts.do_ocr = True
|
|
40
|
+
opts.ocr_options = TesseractCliOcrOptions(lang=list(ocr_lang))
|
|
41
|
+
opts.ocr_options.force_full_page_ocr = force_full_page_ocr
|
|
42
|
+
|
|
43
|
+
self._converter = DocumentConverter(
|
|
44
|
+
format_options={InputFormat.PDF: PdfFormatOption(pipeline_options=opts)}
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
def parse_pdf(self, source: str | Path | bytes) -> str:
|
|
48
|
+
if isinstance(source, bytes):
|
|
49
|
+
with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as tmp:
|
|
50
|
+
tmp.write(source)
|
|
51
|
+
tmp.flush()
|
|
52
|
+
path = tmp.name
|
|
53
|
+
try:
|
|
54
|
+
result = self._converter.convert(path)
|
|
55
|
+
finally:
|
|
56
|
+
Path(path).unlink(missing_ok=True)
|
|
57
|
+
else:
|
|
58
|
+
result = self._converter.convert(str(source))
|
|
59
|
+
return result.document.export_to_markdown()
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: Kannon For Deep Tech
|
|
2
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
3
|
+
|
|
4
|
+
import os
|
|
5
|
+
import tempfile
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
# Tiers exposed by the LlamaCloud parsing API. Single definition, shared
|
|
9
|
+
# with the pricing table so the two can't drift:
|
|
10
|
+
# fast — basic text extraction, no OCR (1 credit/page)
|
|
11
|
+
# cost_effective — OCR-capable, balanced quality (3 credits/page) — default
|
|
12
|
+
# agentic — agent-based, higher quality (10 credits/page)
|
|
13
|
+
# agentic_plus — premium (45 credits/page)
|
|
14
|
+
from dot_parser.pricing import LlamaTier
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class Llama:
|
|
18
|
+
"""LlamaParse / LlamaCloud backend.
|
|
19
|
+
|
|
20
|
+
Calls the LlamaCloud parsing API (``client.parsing.parse``). Robust on
|
|
21
|
+
image-only PDFs, but drops centered block equations (only inline math
|
|
22
|
+
is preserved at ``cost_effective`` tier).
|
|
23
|
+
|
|
24
|
+
Requires a LlamaCloud API key. Reads ``LLAMA_CLOUD_API_KEY`` (or the
|
|
25
|
+
fallback ``LLAMA_API_KEY``) from environment by default, or accepts it
|
|
26
|
+
as the ``api_key`` argument.
|
|
27
|
+
|
|
28
|
+
Install with::
|
|
29
|
+
|
|
30
|
+
pip install dot-parser[llama]
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
def __init__(
|
|
34
|
+
self,
|
|
35
|
+
*,
|
|
36
|
+
api_key: str | None = None,
|
|
37
|
+
tier: LlamaTier = "cost_effective",
|
|
38
|
+
version: str = "latest",
|
|
39
|
+
) -> None:
|
|
40
|
+
try:
|
|
41
|
+
from llama_cloud import LlamaCloud
|
|
42
|
+
except ImportError as e:
|
|
43
|
+
raise ImportError(
|
|
44
|
+
"Llama backend requires `llama-cloud`. Install with: pip install dot-parser[llama]"
|
|
45
|
+
) from e
|
|
46
|
+
|
|
47
|
+
key = api_key or os.environ.get("LLAMA_CLOUD_API_KEY") or os.environ.get("LLAMA_API_KEY")
|
|
48
|
+
if not key:
|
|
49
|
+
raise ValueError(
|
|
50
|
+
"Llama backend requires an API key. Set LLAMA_CLOUD_API_KEY or pass api_key=..."
|
|
51
|
+
)
|
|
52
|
+
self._client = LlamaCloud(api_key=key)
|
|
53
|
+
self._tier = tier
|
|
54
|
+
self._version = version
|
|
55
|
+
|
|
56
|
+
def parse_pdf(self, source: str | Path | bytes) -> str:
|
|
57
|
+
if isinstance(source, bytes):
|
|
58
|
+
with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as tmp:
|
|
59
|
+
tmp.write(source)
|
|
60
|
+
tmp.flush()
|
|
61
|
+
path = Path(tmp.name)
|
|
62
|
+
try:
|
|
63
|
+
return self._parse(path)
|
|
64
|
+
finally:
|
|
65
|
+
path.unlink(missing_ok=True)
|
|
66
|
+
return self._parse(Path(source))
|
|
67
|
+
|
|
68
|
+
def _parse(self, path: Path) -> str:
|
|
69
|
+
with path.open("rb") as fh:
|
|
70
|
+
result = self._client.parsing.parse(
|
|
71
|
+
upload_file=fh,
|
|
72
|
+
tier=self._tier,
|
|
73
|
+
version=self._version,
|
|
74
|
+
expand=["markdown_full"],
|
|
75
|
+
)
|
|
76
|
+
return result.markdown_full or ""
|