vera-ingest-docling 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,16 @@
1
+ .venv/
2
+ __pycache__/
3
+ .pytest_cache/
4
+ *.pyc
5
+ *.vera
6
+ dist/
7
+ dist-electron/
8
+ release/
9
+ build/
10
+ node_modules/
11
+ *.egg-info/
12
+ examples/*.pdf
13
+ site/
14
+ .env
15
+ .env.local
16
+ .env.*.local
@@ -0,0 +1,100 @@
1
+ Metadata-Version: 2.5
2
+ Name: vera-ingest-docling
3
+ Version: 0.3.0
4
+ Summary: Optional Docling HybridChunker ingest pipeline for VERA
5
+ Project-URL: Homepage, https://github.com/dkylewillis/vera
6
+ Project-URL: Repository, https://github.com/dkylewillis/vera
7
+ Project-URL: Documentation, https://dkylewillis.github.io/vera/packages/vera-ingest-docling/
8
+ Author: Kyle Willis
9
+ License: Apache-2.0
10
+ Keywords: chunking,docling,pdf,semantic-search,vera
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: License :: OSI Approved :: Apache Software License
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Topic :: Text Processing :: Indexing
15
+ Requires-Python: >=3.10
16
+ Requires-Dist: docling[rapidocr]<2.119,>=2.118
17
+ Requires-Dist: vera-ingest>=0.3.0
18
+ Description-Content-Type: text/markdown
19
+
20
+ # vera-ingest-docling
21
+
22
+ Optional Docling ingest pipeline for VERA. Registers the `docling` provider
23
+ (default variant `hybrid`) under the `vera.ingest_pipelines` entry-point group.
24
+
25
+ The pipeline uses Docling's `DocumentConverter` for layout-aware PDF parsing and
26
+ `HybridChunker` for token-aware chunks. Readable chunk text is stored for
27
+ keyword search; contextualized text from `HybridChunker.contextualize()` is used
28
+ for embeddings.
29
+
30
+ ## Install
31
+
32
+ ```bash
33
+ python -m pip install "vera-ingest-docling>=0.3.0"
34
+ ```
35
+
36
+ From a repository checkout with uv (workspace `.venv` for CLI and tests):
37
+
38
+ ```bash
39
+ uv sync --extra docling
40
+ ```
41
+
42
+ Non-desktop users can install the CLI extra:
43
+
44
+ ```bash
45
+ pip install "vera-cli[docling]>=0.3.0"
46
+ ```
47
+
48
+ Python 3.10 or newer is required. The package depends on Docling's `rapidocr`
49
+ extra so RapidOCR and `onnxruntime` are installed for OCR. RapidOCR ONNX
50
+ weights come with that extra; first conversion may still download Docling
51
+ layout models (about 380 MB: Heron ONNX + TableFormer accurate). Set
52
+ `DOCLING_ARTIFACTS_PATH` to a local cache (or prefetch layout models offline)
53
+ for air-gapped runs. Incomplete caches are not treated as ready. Hub progress
54
+ is visible on CLI stderr.
55
+ This extra is not bundled in the 0.3.0 desktop installer and is not listed in
56
+ Convert.
57
+
58
+ ## Usage
59
+
60
+ ```bash
61
+ vera convert "manual.pdf" "manual.vera" --parser docling
62
+ # or explicitly:
63
+ vera convert "manual.pdf" "manual.vera" --parser docling:hybrid
64
+ ```
65
+
66
+ ```python
67
+ from vera_ingest import convert
68
+
69
+ convert("manual.pdf", "manual.vera", parser="docling")
70
+ ```
71
+
72
+ ## Notes
73
+
74
+ - Defaults: `chunk_size=500` whitespace tokens (not LLM subword tokens),
75
+ `ocr_mode=auto`, `ocr_language=en`,
76
+ `pdf_backend=docling_parse`. Docling does **not** advertise `overlap` or
77
+ `ocr_dpi`, so those legacy convert/CLI aliases are not forwarded. The
78
+ Tesseract `--ocr-language` alias is also not forwarded; Docling keeps `en`.
79
+ - Prefer `pipeline_options=` / `--pipeline-option KEY=VALUE` for provider-owned
80
+ settings; `--chunk-size` and `--ocr*` remain compatibility aliases for
81
+ pipelines that accept them.
82
+ - OCR modes map to Docling/RapidOCR: `off`, `auto` (default), and `force`
83
+ (full-page OCR). `ocr_language` expects a RapidOCR-native code (`en`, `fr`,
84
+ `cyrillic`, ...) — Tesseract-style codes such as `eng` are **not**
85
+ translated. The shared `--ocr-language` CLI default (`eng`) is not
86
+ forwarded; pass `--pipeline-option ocr_language=fr` (or another RapidOCR
87
+ code) when you need a non-default language.
88
+ - Torch model compilation is disabled so Windows does not need MSVC `cl.exe`.
89
+ - Picture crops are stored as figure attachments and linked onto a nearby
90
+ same-page chunk so search `--figures` can return them. Docling's
91
+ HybridChunker omits pictures from chunk text.
92
+ - On page-level memory errors (`bad_alloc`), VERA retries failed pages then
93
+ falls back to whole-document `pypdfium2`, then page-batch `pypdfium2` if
94
+ that still raises. Force the backend with
95
+ `--pipeline-option pdf_backend=pypdfium2`. Conversion rejects only when
96
+ recovery is exhausted. Failures include the underlying exception and print
97
+ it on sidecar stderr.
98
+
99
+ See the [vera-ingest-docling documentation](https://dkylewillis.github.io/vera/packages/vera-ingest-docling/)
100
+ and [conversion guide](https://github.com/dkylewillis/vera/blob/main/docs/conversion.md).
@@ -0,0 +1,81 @@
1
+ # vera-ingest-docling
2
+
3
+ Optional Docling ingest pipeline for VERA. Registers the `docling` provider
4
+ (default variant `hybrid`) under the `vera.ingest_pipelines` entry-point group.
5
+
6
+ The pipeline uses Docling's `DocumentConverter` for layout-aware PDF parsing and
7
+ `HybridChunker` for token-aware chunks. Readable chunk text is stored for
8
+ keyword search; contextualized text from `HybridChunker.contextualize()` is used
9
+ for embeddings.
10
+
11
+ ## Install
12
+
13
+ ```bash
14
+ python -m pip install "vera-ingest-docling>=0.3.0"
15
+ ```
16
+
17
+ From a repository checkout with uv (workspace `.venv` for CLI and tests):
18
+
19
+ ```bash
20
+ uv sync --extra docling
21
+ ```
22
+
23
+ Non-desktop users can install the CLI extra:
24
+
25
+ ```bash
26
+ pip install "vera-cli[docling]>=0.3.0"
27
+ ```
28
+
29
+ Python 3.10 or newer is required. The package depends on Docling's `rapidocr`
30
+ extra so RapidOCR and `onnxruntime` are installed for OCR. RapidOCR ONNX
31
+ weights come with that extra; first conversion may still download Docling
32
+ layout models (about 380 MB: Heron ONNX + TableFormer accurate). Set
33
+ `DOCLING_ARTIFACTS_PATH` to a local cache (or prefetch layout models offline)
34
+ for air-gapped runs. Incomplete caches are not treated as ready. Hub progress
35
+ is visible on CLI stderr.
36
+ This extra is not bundled in the 0.3.0 desktop installer and is not listed in
37
+ Convert.
38
+
39
+ ## Usage
40
+
41
+ ```bash
42
+ vera convert "manual.pdf" "manual.vera" --parser docling
43
+ # or explicitly:
44
+ vera convert "manual.pdf" "manual.vera" --parser docling:hybrid
45
+ ```
46
+
47
+ ```python
48
+ from vera_ingest import convert
49
+
50
+ convert("manual.pdf", "manual.vera", parser="docling")
51
+ ```
52
+
53
+ ## Notes
54
+
55
+ - Defaults: `chunk_size=500` whitespace tokens (not LLM subword tokens),
56
+ `ocr_mode=auto`, `ocr_language=en`,
57
+ `pdf_backend=docling_parse`. Docling does **not** advertise `overlap` or
58
+ `ocr_dpi`, so those legacy convert/CLI aliases are not forwarded. The
59
+ Tesseract `--ocr-language` alias is also not forwarded; Docling keeps `en`.
60
+ - Prefer `pipeline_options=` / `--pipeline-option KEY=VALUE` for provider-owned
61
+ settings; `--chunk-size` and `--ocr*` remain compatibility aliases for
62
+ pipelines that accept them.
63
+ - OCR modes map to Docling/RapidOCR: `off`, `auto` (default), and `force`
64
+ (full-page OCR). `ocr_language` expects a RapidOCR-native code (`en`, `fr`,
65
+ `cyrillic`, ...) — Tesseract-style codes such as `eng` are **not**
66
+ translated. The shared `--ocr-language` CLI default (`eng`) is not
67
+ forwarded; pass `--pipeline-option ocr_language=fr` (or another RapidOCR
68
+ code) when you need a non-default language.
69
+ - Torch model compilation is disabled so Windows does not need MSVC `cl.exe`.
70
+ - Picture crops are stored as figure attachments and linked onto a nearby
71
+ same-page chunk so search `--figures` can return them. Docling's
72
+ HybridChunker omits pictures from chunk text.
73
+ - On page-level memory errors (`bad_alloc`), VERA retries failed pages then
74
+ falls back to whole-document `pypdfium2`, then page-batch `pypdfium2` if
75
+ that still raises. Force the backend with
76
+ `--pipeline-option pdf_backend=pypdfium2`. Conversion rejects only when
77
+ recovery is exhausted. Failures include the underlying exception and print
78
+ it on sidecar stderr.
79
+
80
+ See the [vera-ingest-docling documentation](https://dkylewillis.github.io/vera/packages/vera-ingest-docling/)
81
+ and [conversion guide](https://github.com/dkylewillis/vera/blob/main/docs/conversion.md).
@@ -0,0 +1,38 @@
1
+ [project]
2
+ name = "vera-ingest-docling"
3
+ version = "0.3.0"
4
+ description = "Optional Docling HybridChunker ingest pipeline for VERA"
5
+ readme = "README.md"
6
+ requires-python = ">=3.10"
7
+ dependencies = [
8
+ "vera-ingest>=0.3.0",
9
+ # rapidocr extra pulls RapidOCR + onnxruntime (required for Docling OCR).
10
+ "docling[rapidocr]>=2.118,<2.119",
11
+ ]
12
+ authors = [{name = "Kyle Willis"}]
13
+ license = {text = "Apache-2.0"}
14
+ keywords = ["docling", "pdf", "chunking", "semantic-search", "vera"]
15
+ classifiers = [
16
+ "Development Status :: 3 - Alpha",
17
+ "License :: OSI Approved :: Apache Software License",
18
+ "Programming Language :: Python :: 3",
19
+ "Topic :: Text Processing :: Indexing",
20
+ ]
21
+
22
+ [project.urls]
23
+ Homepage = "https://github.com/dkylewillis/vera"
24
+ Repository = "https://github.com/dkylewillis/vera"
25
+ Documentation = "https://dkylewillis.github.io/vera/packages/vera-ingest-docling/"
26
+
27
+ [project.entry-points."vera.ingest_pipelines"]
28
+ docling = "vera_ingest_docling:create_pipeline"
29
+
30
+ [project.entry-points."vera.ingest_pipeline_descriptors"]
31
+ docling = "vera_ingest_docling:create_descriptor"
32
+
33
+ [build-system]
34
+ requires = ["hatchling"]
35
+ build-backend = "hatchling.build"
36
+
37
+ [tool.hatch.build.targets.wheel]
38
+ packages = ["src/vera_ingest_docling"]
@@ -0,0 +1,73 @@
1
+ """Optional Docling HybridChunker ingest pipeline for VERA."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from vera_ingest.descriptors import PipelineDescriptor
8
+ from vera_ingest.pipeline import (
9
+ IngestPipeline,
10
+ UnknownIngestPipelineError,
11
+ register_ingest_pipeline,
12
+ register_ingest_pipeline_descriptor,
13
+ )
14
+
15
+ from .options import DoclingOptions, _docling_runtime_available, describe_pipeline
16
+
17
+ __all__ = [
18
+ "DoclingHybridPipeline",
19
+ "DoclingOptions",
20
+ "create_descriptor",
21
+ "create_pipeline",
22
+ "describe_pipeline",
23
+ "ensure_registered",
24
+ ]
25
+
26
+
27
+ def __getattr__(name: str) -> Any:
28
+ if name == "DoclingHybridPipeline":
29
+ from .pipeline import DoclingHybridPipeline
30
+
31
+ return DoclingHybridPipeline
32
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
33
+
34
+
35
+ def create_pipeline(variant: str = "hybrid") -> IngestPipeline:
36
+ """Entry-point factory for ``vera.ingest_pipelines`` provider ``docling``."""
37
+ normalized = (variant or "hybrid").strip().lower()
38
+ if normalized not in {"", "hybrid"}:
39
+ raise UnknownIngestPipelineError(
40
+ f"Unknown Docling pipeline variant {variant!r}; use 'docling' or 'docling:hybrid'."
41
+ )
42
+ if not _docling_runtime_available():
43
+ raise UnknownIngestPipelineError(
44
+ "Docling is not installed in this environment. "
45
+ "Install with: python -m pip install 'vera-ingest-docling>=0.3.0'"
46
+ )
47
+ from vera_ingest.timing import timed_step
48
+
49
+ with timed_step("import_docling_pipeline"):
50
+ from .pipeline import DoclingHybridPipeline
51
+
52
+ return DoclingHybridPipeline()
53
+
54
+
55
+ def create_descriptor(variant: str = "hybrid") -> PipelineDescriptor:
56
+ """Entry-point factory for ``vera.ingest_pipeline_descriptors``."""
57
+ try:
58
+ return describe_pipeline(variant)
59
+ except ValueError as exc:
60
+ raise UnknownIngestPipelineError(str(exc)) from exc
61
+
62
+
63
+ def ensure_registered(*, replace: bool = True) -> None:
64
+ """Register the ``docling`` pipeline without relying on package metadata.
65
+
66
+ Entry-point discovery fails in PyInstaller freezes and PYTHONPATH-only
67
+ source runs that never install ``vera-ingest-docling`` dist-info.
68
+ """
69
+ register_ingest_pipeline("docling", create_pipeline, replace=replace)
70
+ register_ingest_pipeline_descriptor("docling", create_descriptor, replace=replace)
71
+
72
+
73
+ ensure_registered()