dot-parser 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
dot_parser/__init__.py ADDED
@@ -0,0 +1,49 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ from importlib.metadata import version
5
+
6
+ from dot_parser.backends import Backend, Docling, ImageBackend, Llama, Mistral, Pymu
7
+ from dot_parser.chunking import chunk
8
+ from dot_parser.images import (
9
+ VLM,
10
+ ExtractedImage,
11
+ ImageDescription,
12
+ PageInfo,
13
+ ParseResult,
14
+ )
15
+ from dot_parser.models import Chunk, ParseError
16
+ from dot_parser.parsers import (
17
+ interpret_images,
18
+ parse,
19
+ parse_pdfs,
20
+ parse_with_images,
21
+ )
22
+ from dot_parser.pricing import cost_per_1k_pages, estimate_cost
23
+ from dot_parser.vlms import MistralVLM
24
+
25
+ __version__ = version("dot-parser")
26
+
27
+ __all__ = [
28
+ "VLM",
29
+ "Backend",
30
+ "Chunk",
31
+ "Docling",
32
+ "ExtractedImage",
33
+ "ImageBackend",
34
+ "ImageDescription",
35
+ "Llama",
36
+ "Mistral",
37
+ "MistralVLM",
38
+ "PageInfo",
39
+ "ParseError",
40
+ "ParseResult",
41
+ "Pymu",
42
+ "chunk",
43
+ "cost_per_1k_pages",
44
+ "estimate_cost",
45
+ "interpret_images",
46
+ "parse",
47
+ "parse_pdfs",
48
+ "parse_with_images",
49
+ ]
@@ -0,0 +1,26 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ from dot_parser.backends._base import (
5
+ Backend,
6
+ BatchBackend,
7
+ DocxImageBackend,
8
+ ImageBackend,
9
+ PptxImageBackend,
10
+ )
11
+ from dot_parser.backends.docling import Docling
12
+ from dot_parser.backends.llama import Llama
13
+ from dot_parser.backends.mistral import Mistral
14
+ from dot_parser.backends.pymu import Pymu
15
+
16
+ __all__ = [
17
+ "Backend",
18
+ "BatchBackend",
19
+ "Docling",
20
+ "DocxImageBackend",
21
+ "ImageBackend",
22
+ "Llama",
23
+ "Mistral",
24
+ "PptxImageBackend",
25
+ "Pymu",
26
+ ]
@@ -0,0 +1,79 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ from pathlib import Path
5
+ from typing import Protocol, runtime_checkable
6
+
7
+ from dot_parser.images import ParseResult
8
+
9
+
10
+ @runtime_checkable
11
+ class Backend(Protocol):
12
+ """A PDF parsing backend that produces Markdown."""
13
+
14
+ def parse_pdf(self, source: str | Path | bytes) -> str: ...
15
+
16
+
17
+ @runtime_checkable
18
+ class BatchBackend(Protocol):
19
+ """A PDF backend that parses a whole list in one optimized call.
20
+
21
+ Backends implementing this protocol (e.g. ``Mistral`` via the Batch API)
22
+ take over ``parse_pdfs()`` entirely; the others fall back to a per-file
23
+ loop. Returns one Markdown string per input, in order, with ``None`` for
24
+ items that failed.
25
+ """
26
+
27
+ def parse_pdfs(self, sources: list[str | Path | bytes]) -> list[str | None]: ...
28
+
29
+
30
+ @runtime_checkable
31
+ class ImageBackend(Protocol):
32
+ """A PDF backend that can also extract and annotate images.
33
+
34
+ Backends that implement this protocol are eligible for the
35
+ `parse_with_images()` PDF path. The returned markdown must contain
36
+ `![name](name)` anchors at the position of each image, matching the
37
+ `ExtractedImage.name` of the corresponding image record.
38
+ """
39
+
40
+ def parse_pdf_with_images(
41
+ self,
42
+ source: str | Path | bytes,
43
+ *,
44
+ annotate_images: bool = True,
45
+ include_tables: bool = True,
46
+ ) -> ParseResult: ...
47
+
48
+
49
+ @runtime_checkable
50
+ class DocxImageBackend(Protocol):
51
+ """A backend that parses DOCX end-to-end through OCR.
52
+
53
+ When one is passed to `parse_with_images()`, the DOCX goes through the
54
+ backend directly instead of the markitdown + per-image VLM path.
55
+ """
56
+
57
+ def parse_docx_with_images(
58
+ self,
59
+ source: str | Path | bytes,
60
+ *,
61
+ annotate_images: bool = True,
62
+ include_tables: bool = True,
63
+ ) -> ParseResult: ...
64
+
65
+
66
+ @runtime_checkable
67
+ class PptxImageBackend(Protocol):
68
+ """A backend that parses PPTX end-to-end through OCR.
69
+
70
+ PPTX counterpart of :class:`DocxImageBackend`.
71
+ """
72
+
73
+ def parse_pptx_with_images(
74
+ self,
75
+ source: str | Path | bytes,
76
+ *,
77
+ annotate_images: bool = True,
78
+ include_tables: bool = True,
79
+ ) -> ParseResult: ...
@@ -0,0 +1,59 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ import tempfile
5
+ from pathlib import Path
6
+
7
+
8
+ class Docling:
9
+ """Docling backend (IBM, MIT licence).
10
+
11
+ Layout-aware ML pipeline (DocLayNet) + table parser (TableFormer) + OCR.
12
+ Robust on image-only and broken-encoding PDFs when ``force_full_page_ocr``
13
+ is enabled.
14
+
15
+ Install with::
16
+
17
+ pip install dot-parser[docling]
18
+ """
19
+
20
+ def __init__(
21
+ self,
22
+ *,
23
+ force_full_page_ocr: bool = True,
24
+ ocr_lang: tuple[str, ...] = ("eng", "fra"),
25
+ ) -> None:
26
+ try:
27
+ from docling.datamodel.base_models import InputFormat
28
+ from docling.datamodel.pipeline_options import (
29
+ PdfPipelineOptions,
30
+ TesseractCliOcrOptions,
31
+ )
32
+ from docling.document_converter import DocumentConverter, PdfFormatOption
33
+ except ImportError as e:
34
+ raise ImportError(
35
+ "Docling backend requires `docling`. Install with: pip install dot-parser[docling]"
36
+ ) from e
37
+
38
+ opts = PdfPipelineOptions()
39
+ opts.do_ocr = True
40
+ opts.ocr_options = TesseractCliOcrOptions(lang=list(ocr_lang))
41
+ opts.ocr_options.force_full_page_ocr = force_full_page_ocr
42
+
43
+ self._converter = DocumentConverter(
44
+ format_options={InputFormat.PDF: PdfFormatOption(pipeline_options=opts)}
45
+ )
46
+
47
+ def parse_pdf(self, source: str | Path | bytes) -> str:
48
+ if isinstance(source, bytes):
49
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as tmp:
50
+ tmp.write(source)
51
+ tmp.flush()
52
+ path = tmp.name
53
+ try:
54
+ result = self._converter.convert(path)
55
+ finally:
56
+ Path(path).unlink(missing_ok=True)
57
+ else:
58
+ result = self._converter.convert(str(source))
59
+ return result.document.export_to_markdown()
@@ -0,0 +1,76 @@
1
+ # SPDX-FileCopyrightText: Kannon For Deep Tech
2
+ # SPDX-License-Identifier: AGPL-3.0-or-later
3
+
4
+ import os
5
+ import tempfile
6
+ from pathlib import Path
7
+
8
+ # Tiers exposed by the LlamaCloud parsing API. Single definition, shared
9
+ # with the pricing table so the two can't drift:
10
+ # fast — basic text extraction, no OCR (1 credit/page)
11
+ # cost_effective — OCR-capable, balanced quality (3 credits/page) — default
12
+ # agentic — agent-based, higher quality (10 credits/page)
13
+ # agentic_plus — premium (45 credits/page)
14
+ from dot_parser.pricing import LlamaTier
15
+
16
+
17
+ class Llama:
18
+ """LlamaParse / LlamaCloud backend.
19
+
20
+ Calls the LlamaCloud parsing API (``client.parsing.parse``). Robust on
21
+ image-only PDFs, but drops centered block equations (only inline math
22
+ is preserved at ``cost_effective`` tier).
23
+
24
+ Requires a LlamaCloud API key. Reads ``LLAMA_CLOUD_API_KEY`` (or the
25
+ fallback ``LLAMA_API_KEY``) from environment by default, or accepts it
26
+ as the ``api_key`` argument.
27
+
28
+ Install with::
29
+
30
+ pip install dot-parser[llama]
31
+ """
32
+
33
+ def __init__(
34
+ self,
35
+ *,
36
+ api_key: str | None = None,
37
+ tier: LlamaTier = "cost_effective",
38
+ version: str = "latest",
39
+ ) -> None:
40
+ try:
41
+ from llama_cloud import LlamaCloud
42
+ except ImportError as e:
43
+ raise ImportError(
44
+ "Llama backend requires `llama-cloud`. Install with: pip install dot-parser[llama]"
45
+ ) from e
46
+
47
+ key = api_key or os.environ.get("LLAMA_CLOUD_API_KEY") or os.environ.get("LLAMA_API_KEY")
48
+ if not key:
49
+ raise ValueError(
50
+ "Llama backend requires an API key. Set LLAMA_CLOUD_API_KEY or pass api_key=..."
51
+ )
52
+ self._client = LlamaCloud(api_key=key)
53
+ self._tier = tier
54
+ self._version = version
55
+
56
+ def parse_pdf(self, source: str | Path | bytes) -> str:
57
+ if isinstance(source, bytes):
58
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as tmp:
59
+ tmp.write(source)
60
+ tmp.flush()
61
+ path = Path(tmp.name)
62
+ try:
63
+ return self._parse(path)
64
+ finally:
65
+ path.unlink(missing_ok=True)
66
+ return self._parse(Path(source))
67
+
68
+ def _parse(self, path: Path) -> str:
69
+ with path.open("rb") as fh:
70
+ result = self._client.parsing.parse(
71
+ upload_file=fh,
72
+ tier=self._tier,
73
+ version=self._version,
74
+ expand=["markdown_full"],
75
+ )
76
+ return result.markdown_full or ""