vera-ingest-docling 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vera_ingest_docling-0.3.0/.gitignore +16 -0
- vera_ingest_docling-0.3.0/PKG-INFO +100 -0
- vera_ingest_docling-0.3.0/README.md +81 -0
- vera_ingest_docling-0.3.0/pyproject.toml +38 -0
- vera_ingest_docling-0.3.0/src/vera_ingest_docling/__init__.py +73 -0
- vera_ingest_docling-0.3.0/src/vera_ingest_docling/converter.py +483 -0
- vera_ingest_docling-0.3.0/src/vera_ingest_docling/mapping.py +429 -0
- vera_ingest_docling-0.3.0/src/vera_ingest_docling/options.py +147 -0
- vera_ingest_docling-0.3.0/src/vera_ingest_docling/pipeline.py +152 -0
- vera_ingest_docling-0.3.0/src/vera_ingest_docling/recovery.py +712 -0
- vera_ingest_docling-0.3.0/tests/test_docling_descriptor.py +45 -0
- vera_ingest_docling-0.3.0/tests/test_docling_model_cache.py +186 -0
- vera_ingest_docling-0.3.0/tests/test_docling_pipeline.py +1470 -0
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: vera-ingest-docling
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Optional Docling HybridChunker ingest pipeline for VERA
|
|
5
|
+
Project-URL: Homepage, https://github.com/dkylewillis/vera
|
|
6
|
+
Project-URL: Repository, https://github.com/dkylewillis/vera
|
|
7
|
+
Project-URL: Documentation, https://dkylewillis.github.io/vera/packages/vera-ingest-docling/
|
|
8
|
+
Author: Kyle Willis
|
|
9
|
+
License: Apache-2.0
|
|
10
|
+
Keywords: chunking,docling,pdf,semantic-search,vera
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Text Processing :: Indexing
|
|
15
|
+
Requires-Python: >=3.10
|
|
16
|
+
Requires-Dist: docling[rapidocr]<2.119,>=2.118
|
|
17
|
+
Requires-Dist: vera-ingest>=0.3.0
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
|
|
20
|
+
# vera-ingest-docling
|
|
21
|
+
|
|
22
|
+
Optional Docling ingest pipeline for VERA. Registers the `docling` provider
|
|
23
|
+
(default variant `hybrid`) under the `vera.ingest_pipelines` entry-point group.
|
|
24
|
+
|
|
25
|
+
The pipeline uses Docling's `DocumentConverter` for layout-aware PDF parsing and
|
|
26
|
+
`HybridChunker` for token-aware chunks. Readable chunk text is stored for
|
|
27
|
+
keyword search; contextualized text from `HybridChunker.contextualize()` is used
|
|
28
|
+
for embeddings.
|
|
29
|
+
|
|
30
|
+
## Install
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
python -m pip install "vera-ingest-docling>=0.3.0"
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
From a repository checkout with uv (workspace `.venv` for CLI and tests):
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
uv sync --extra docling
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Non-desktop users can install the CLI extra:
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
pip install "vera-cli[docling]>=0.3.0"
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Python 3.10 or newer is required. The package depends on Docling's `rapidocr`
|
|
49
|
+
extra so RapidOCR and `onnxruntime` are installed for OCR. RapidOCR ONNX
|
|
50
|
+
weights come with that extra; first conversion may still download Docling
|
|
51
|
+
layout models (about 380 MB: Heron ONNX + TableFormer accurate). Set
|
|
52
|
+
`DOCLING_ARTIFACTS_PATH` to a local cache (or prefetch layout models offline)
|
|
53
|
+
for air-gapped runs. Incomplete caches are not treated as ready. Hub progress
|
|
54
|
+
is visible on CLI stderr.
|
|
55
|
+
This extra is not bundled in the 0.3.0 desktop installer and is not listed in
|
|
56
|
+
Convert.
|
|
57
|
+
|
|
58
|
+
## Usage
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
vera convert "manual.pdf" "manual.vera" --parser docling
|
|
62
|
+
# or explicitly:
|
|
63
|
+
vera convert "manual.pdf" "manual.vera" --parser docling:hybrid
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
```python
|
|
67
|
+
from vera_ingest import convert
|
|
68
|
+
|
|
69
|
+
convert("manual.pdf", "manual.vera", parser="docling")
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## Notes
|
|
73
|
+
|
|
74
|
+
- Defaults: `chunk_size=500` whitespace tokens (not LLM subword tokens),
|
|
75
|
+
`ocr_mode=auto`, `ocr_language=en`,
|
|
76
|
+
`pdf_backend=docling_parse`. Docling does **not** advertise `overlap` or
|
|
77
|
+
`ocr_dpi`, so those legacy convert/CLI aliases are not forwarded. The
|
|
78
|
+
Tesseract `--ocr-language` alias is also not forwarded; Docling keeps `en`.
|
|
79
|
+
- Prefer `pipeline_options=` / `--pipeline-option KEY=VALUE` for provider-owned
|
|
80
|
+
settings; `--chunk-size` and `--ocr*` remain compatibility aliases for
|
|
81
|
+
pipelines that accept them.
|
|
82
|
+
- OCR modes map to Docling/RapidOCR: `off`, `auto` (default), and `force`
|
|
83
|
+
(full-page OCR). `ocr_language` expects a RapidOCR-native code (`en`, `fr`,
|
|
84
|
+
`cyrillic`, ...) — Tesseract-style codes such as `eng` are **not**
|
|
85
|
+
translated. The shared `--ocr-language` CLI default (`eng`) is not
|
|
86
|
+
forwarded; pass `--pipeline-option ocr_language=fr` (or another RapidOCR
|
|
87
|
+
code) when you need a non-default language.
|
|
88
|
+
- Torch model compilation is disabled so Windows does not need MSVC `cl.exe`.
|
|
89
|
+
- Picture crops are stored as figure attachments and linked onto a nearby
|
|
90
|
+
same-page chunk so search `--figures` can return them. Docling's
|
|
91
|
+
HybridChunker omits pictures from chunk text.
|
|
92
|
+
- On page-level memory errors (`bad_alloc`), VERA retries failed pages then
|
|
93
|
+
falls back to whole-document `pypdfium2`, then page-batch `pypdfium2` if
|
|
94
|
+
that still raises. Force the backend with
|
|
95
|
+
`--pipeline-option pdf_backend=pypdfium2`. Conversion rejects only when
|
|
96
|
+
recovery is exhausted. Failures include the underlying exception and print
|
|
97
|
+
it on sidecar stderr.
|
|
98
|
+
|
|
99
|
+
See the [vera-ingest-docling documentation](https://dkylewillis.github.io/vera/packages/vera-ingest-docling/)
|
|
100
|
+
and [conversion guide](https://github.com/dkylewillis/vera/blob/main/docs/conversion.md).
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
# vera-ingest-docling
|
|
2
|
+
|
|
3
|
+
Optional Docling ingest pipeline for VERA. Registers the `docling` provider
|
|
4
|
+
(default variant `hybrid`) under the `vera.ingest_pipelines` entry-point group.
|
|
5
|
+
|
|
6
|
+
The pipeline uses Docling's `DocumentConverter` for layout-aware PDF parsing and
|
|
7
|
+
`HybridChunker` for token-aware chunks. Readable chunk text is stored for
|
|
8
|
+
keyword search; contextualized text from `HybridChunker.contextualize()` is used
|
|
9
|
+
for embeddings.
|
|
10
|
+
|
|
11
|
+
## Install
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
python -m pip install "vera-ingest-docling>=0.3.0"
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
From a repository checkout with uv (workspace `.venv` for CLI and tests):
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
uv sync --extra docling
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Non-desktop users can install the CLI extra:
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pip install "vera-cli[docling]>=0.3.0"
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
Python 3.10 or newer is required. The package depends on Docling's `rapidocr`
|
|
30
|
+
extra so RapidOCR and `onnxruntime` are installed for OCR. RapidOCR ONNX
|
|
31
|
+
weights come with that extra; first conversion may still download Docling
|
|
32
|
+
layout models (about 380 MB: Heron ONNX + TableFormer accurate). Set
|
|
33
|
+
`DOCLING_ARTIFACTS_PATH` to a local cache (or prefetch layout models offline)
|
|
34
|
+
for air-gapped runs. Incomplete caches are not treated as ready. Hub progress
|
|
35
|
+
is visible on CLI stderr.
|
|
36
|
+
This extra is not bundled in the 0.3.0 desktop installer and is not listed in
|
|
37
|
+
Convert.
|
|
38
|
+
|
|
39
|
+
## Usage
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
vera convert "manual.pdf" "manual.vera" --parser docling
|
|
43
|
+
# or explicitly:
|
|
44
|
+
vera convert "manual.pdf" "manual.vera" --parser docling:hybrid
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
from vera_ingest import convert
|
|
49
|
+
|
|
50
|
+
convert("manual.pdf", "manual.vera", parser="docling")
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## Notes
|
|
54
|
+
|
|
55
|
+
- Defaults: `chunk_size=500` whitespace tokens (not LLM subword tokens),
|
|
56
|
+
`ocr_mode=auto`, `ocr_language=en`,
|
|
57
|
+
`pdf_backend=docling_parse`. Docling does **not** advertise `overlap` or
|
|
58
|
+
`ocr_dpi`, so those legacy convert/CLI aliases are not forwarded. The
|
|
59
|
+
Tesseract `--ocr-language` alias is also not forwarded; Docling keeps `en`.
|
|
60
|
+
- Prefer `pipeline_options=` / `--pipeline-option KEY=VALUE` for provider-owned
|
|
61
|
+
settings; `--chunk-size` and `--ocr*` remain compatibility aliases for
|
|
62
|
+
pipelines that accept them.
|
|
63
|
+
- OCR modes map to Docling/RapidOCR: `off`, `auto` (default), and `force`
|
|
64
|
+
(full-page OCR). `ocr_language` expects a RapidOCR-native code (`en`, `fr`,
|
|
65
|
+
`cyrillic`, ...) — Tesseract-style codes such as `eng` are **not**
|
|
66
|
+
translated. The shared `--ocr-language` CLI default (`eng`) is not
|
|
67
|
+
forwarded; pass `--pipeline-option ocr_language=fr` (or another RapidOCR
|
|
68
|
+
code) when you need a non-default language.
|
|
69
|
+
- Torch model compilation is disabled so Windows does not need MSVC `cl.exe`.
|
|
70
|
+
- Picture crops are stored as figure attachments and linked onto a nearby
|
|
71
|
+
same-page chunk so search `--figures` can return them. Docling's
|
|
72
|
+
HybridChunker omits pictures from chunk text.
|
|
73
|
+
- On page-level memory errors (`bad_alloc`), VERA retries failed pages then
|
|
74
|
+
falls back to whole-document `pypdfium2`, then page-batch `pypdfium2` if
|
|
75
|
+
that still raises. Force the backend with
|
|
76
|
+
`--pipeline-option pdf_backend=pypdfium2`. Conversion rejects only when
|
|
77
|
+
recovery is exhausted. Failures include the underlying exception and print
|
|
78
|
+
it on sidecar stderr.
|
|
79
|
+
|
|
80
|
+
See the [vera-ingest-docling documentation](https://dkylewillis.github.io/vera/packages/vera-ingest-docling/)
|
|
81
|
+
and [conversion guide](https://github.com/dkylewillis/vera/blob/main/docs/conversion.md).
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "vera-ingest-docling"
|
|
3
|
+
version = "0.3.0"
|
|
4
|
+
description = "Optional Docling HybridChunker ingest pipeline for VERA"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
dependencies = [
|
|
8
|
+
"vera-ingest>=0.3.0",
|
|
9
|
+
# rapidocr extra pulls RapidOCR + onnxruntime (required for Docling OCR).
|
|
10
|
+
"docling[rapidocr]>=2.118,<2.119",
|
|
11
|
+
]
|
|
12
|
+
authors = [{name = "Kyle Willis"}]
|
|
13
|
+
license = {text = "Apache-2.0"}
|
|
14
|
+
keywords = ["docling", "pdf", "chunking", "semantic-search", "vera"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 3 - Alpha",
|
|
17
|
+
"License :: OSI Approved :: Apache Software License",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Topic :: Text Processing :: Indexing",
|
|
20
|
+
]
|
|
21
|
+
|
|
22
|
+
[project.urls]
|
|
23
|
+
Homepage = "https://github.com/dkylewillis/vera"
|
|
24
|
+
Repository = "https://github.com/dkylewillis/vera"
|
|
25
|
+
Documentation = "https://dkylewillis.github.io/vera/packages/vera-ingest-docling/"
|
|
26
|
+
|
|
27
|
+
[project.entry-points."vera.ingest_pipelines"]
|
|
28
|
+
docling = "vera_ingest_docling:create_pipeline"
|
|
29
|
+
|
|
30
|
+
[project.entry-points."vera.ingest_pipeline_descriptors"]
|
|
31
|
+
docling = "vera_ingest_docling:create_descriptor"
|
|
32
|
+
|
|
33
|
+
[build-system]
|
|
34
|
+
requires = ["hatchling"]
|
|
35
|
+
build-backend = "hatchling.build"
|
|
36
|
+
|
|
37
|
+
[tool.hatch.build.targets.wheel]
|
|
38
|
+
packages = ["src/vera_ingest_docling"]
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Optional Docling HybridChunker ingest pipeline for VERA."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from vera_ingest.descriptors import PipelineDescriptor
|
|
8
|
+
from vera_ingest.pipeline import (
|
|
9
|
+
IngestPipeline,
|
|
10
|
+
UnknownIngestPipelineError,
|
|
11
|
+
register_ingest_pipeline,
|
|
12
|
+
register_ingest_pipeline_descriptor,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
from .options import DoclingOptions, _docling_runtime_available, describe_pipeline
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"DoclingHybridPipeline",
|
|
19
|
+
"DoclingOptions",
|
|
20
|
+
"create_descriptor",
|
|
21
|
+
"create_pipeline",
|
|
22
|
+
"describe_pipeline",
|
|
23
|
+
"ensure_registered",
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def __getattr__(name: str) -> Any:
|
|
28
|
+
if name == "DoclingHybridPipeline":
|
|
29
|
+
from .pipeline import DoclingHybridPipeline
|
|
30
|
+
|
|
31
|
+
return DoclingHybridPipeline
|
|
32
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def create_pipeline(variant: str = "hybrid") -> IngestPipeline:
|
|
36
|
+
"""Entry-point factory for ``vera.ingest_pipelines`` provider ``docling``."""
|
|
37
|
+
normalized = (variant or "hybrid").strip().lower()
|
|
38
|
+
if normalized not in {"", "hybrid"}:
|
|
39
|
+
raise UnknownIngestPipelineError(
|
|
40
|
+
f"Unknown Docling pipeline variant {variant!r}; use 'docling' or 'docling:hybrid'."
|
|
41
|
+
)
|
|
42
|
+
if not _docling_runtime_available():
|
|
43
|
+
raise UnknownIngestPipelineError(
|
|
44
|
+
"Docling is not installed in this environment. "
|
|
45
|
+
"Install with: python -m pip install 'vera-ingest-docling>=0.3.0'"
|
|
46
|
+
)
|
|
47
|
+
from vera_ingest.timing import timed_step
|
|
48
|
+
|
|
49
|
+
with timed_step("import_docling_pipeline"):
|
|
50
|
+
from .pipeline import DoclingHybridPipeline
|
|
51
|
+
|
|
52
|
+
return DoclingHybridPipeline()
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def create_descriptor(variant: str = "hybrid") -> PipelineDescriptor:
|
|
56
|
+
"""Entry-point factory for ``vera.ingest_pipeline_descriptors``."""
|
|
57
|
+
try:
|
|
58
|
+
return describe_pipeline(variant)
|
|
59
|
+
except ValueError as exc:
|
|
60
|
+
raise UnknownIngestPipelineError(str(exc)) from exc
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def ensure_registered(*, replace: bool = True) -> None:
|
|
64
|
+
"""Register the ``docling`` pipeline without relying on package metadata.
|
|
65
|
+
|
|
66
|
+
Entry-point discovery fails in PyInstaller freezes and PYTHONPATH-only
|
|
67
|
+
source runs that never install ``vera-ingest-docling`` dist-info.
|
|
68
|
+
"""
|
|
69
|
+
register_ingest_pipeline("docling", create_pipeline, replace=replace)
|
|
70
|
+
register_ingest_pipeline_descriptor("docling", create_descriptor, replace=replace)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
ensure_registered()
|