docling-pp-doc-layout 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docling_pp_doc_layout/__init__.py +3 -0
- docling_pp_doc_layout/label_mapping.py +34 -0
- docling_pp_doc_layout/model.py +225 -0
- docling_pp_doc_layout/options.py +46 -0
- docling_pp_doc_layout/plugin.py +12 -0
- docling_pp_doc_layout/py.typed +0 -0
- docling_pp_doc_layout-0.1.0.dist-info/METADATA +154 -0
- docling_pp_doc_layout-0.1.0.dist-info/RECORD +11 -0
- docling_pp_doc_layout-0.1.0.dist-info/WHEEL +4 -0
- docling_pp_doc_layout-0.1.0.dist-info/entry_points.txt +2 -0
- docling_pp_doc_layout-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Mapping from PP-DocLayout-V3 label names to docling DocItemLabel values.
|
|
2
|
+
|
|
3
|
+
Every label produced here must exist in
|
|
4
|
+
``docling.utils.layout_postprocessor.LayoutPostprocessor.CONFIDENCE_THRESHOLDS``
|
|
5
|
+
so that the postprocessor can apply confidence filtering without a ``KeyError``.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from docling_core.types.doc import DocItemLabel
|
|
11
|
+
|
|
12
|
+
LABEL_MAP: dict[str, DocItemLabel] = {
|
|
13
|
+
"abstract": DocItemLabel.TEXT,
|
|
14
|
+
"algorithm": DocItemLabel.CODE,
|
|
15
|
+
"aside_text": DocItemLabel.TEXT,
|
|
16
|
+
"chart": DocItemLabel.PICTURE,
|
|
17
|
+
"content": DocItemLabel.TEXT,
|
|
18
|
+
"doc_title": DocItemLabel.TITLE,
|
|
19
|
+
"figure_title": DocItemLabel.CAPTION,
|
|
20
|
+
"footer": DocItemLabel.PAGE_FOOTER,
|
|
21
|
+
"footnote": DocItemLabel.FOOTNOTE,
|
|
22
|
+
"formula": DocItemLabel.FORMULA,
|
|
23
|
+
"formula_number": DocItemLabel.TEXT,
|
|
24
|
+
"header": DocItemLabel.PAGE_HEADER,
|
|
25
|
+
"image": DocItemLabel.PICTURE,
|
|
26
|
+
"number": DocItemLabel.TEXT,
|
|
27
|
+
"paragraph_title": DocItemLabel.SECTION_HEADER,
|
|
28
|
+
"reference": DocItemLabel.TEXT,
|
|
29
|
+
"reference_content": DocItemLabel.TEXT,
|
|
30
|
+
"seal": DocItemLabel.PICTURE,
|
|
31
|
+
"table": DocItemLabel.TABLE,
|
|
32
|
+
"text": DocItemLabel.TEXT,
|
|
33
|
+
"vision_footnote": DocItemLabel.FOOTNOTE,
|
|
34
|
+
}
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
"""PP-DocLayout-V3 layout model for the docling standard pipeline.
|
|
2
|
+
|
|
3
|
+
Runs PaddlePaddle PP-DocLayout-V3 locally via HuggingFace ``transformers``
|
|
4
|
+
to detect document layout elements and returns ``LayoutPrediction`` objects
|
|
5
|
+
that docling merges with its standard-pipeline output.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import logging
|
|
11
|
+
import warnings
|
|
12
|
+
from typing import TYPE_CHECKING
|
|
13
|
+
|
|
14
|
+
import numpy as np
|
|
15
|
+
import torch
|
|
16
|
+
from docling.datamodel.base_models import BoundingBox, Cluster, LayoutPrediction, Page
|
|
17
|
+
from docling.models.base_layout_model import BaseLayoutModel
|
|
18
|
+
from docling.utils.accelerator_utils import decide_device
|
|
19
|
+
from docling.utils.layout_postprocessor import LayoutPostprocessor
|
|
20
|
+
from docling.utils.profiling import TimeRecorder
|
|
21
|
+
from docling_core.types.doc import DocItemLabel
|
|
22
|
+
from transformers import AutoImageProcessor, AutoModelForObjectDetection
|
|
23
|
+
|
|
24
|
+
from docling_pp_doc_layout.label_mapping import LABEL_MAP
|
|
25
|
+
from docling_pp_doc_layout.options import PPDocLayoutV3Options
|
|
26
|
+
|
|
27
|
+
if TYPE_CHECKING:
|
|
28
|
+
from collections.abc import Sequence
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
|
|
31
|
+
from docling.datamodel.accelerator_options import AcceleratorOptions
|
|
32
|
+
from docling.datamodel.document import ConversionResult
|
|
33
|
+
from docling.datamodel.pipeline_options import BaseLayoutOptions
|
|
34
|
+
from PIL import Image
|
|
35
|
+
|
|
36
|
+
logger = logging.getLogger(__name__)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class PPDocLayoutV3Model(BaseLayoutModel):
|
|
40
|
+
"""Layout engine using PP-DocLayout-V3 via HuggingFace transformers."""
|
|
41
|
+
|
|
42
|
+
def __init__(
|
|
43
|
+
self,
|
|
44
|
+
artifacts_path: Path | None,
|
|
45
|
+
accelerator_options: AcceleratorOptions,
|
|
46
|
+
options: PPDocLayoutV3Options,
|
|
47
|
+
*,
|
|
48
|
+
enable_remote_services: bool = False, # noqa: ARG002
|
|
49
|
+
) -> None:
|
|
50
|
+
self.options = options
|
|
51
|
+
self.artifacts_path = artifacts_path
|
|
52
|
+
self.accelerator_options = accelerator_options
|
|
53
|
+
|
|
54
|
+
self._device = decide_device(accelerator_options.device)
|
|
55
|
+
logger.info(
|
|
56
|
+
"Loading PP-DocLayout-V3 model %s on device=%s",
|
|
57
|
+
options.model_name,
|
|
58
|
+
self._device,
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
self._image_processor = AutoImageProcessor.from_pretrained(
|
|
62
|
+
options.model_name,
|
|
63
|
+
)
|
|
64
|
+
self._model = AutoModelForObjectDetection.from_pretrained(
|
|
65
|
+
options.model_name,
|
|
66
|
+
).to(self._device)
|
|
67
|
+
self._model.eval()
|
|
68
|
+
|
|
69
|
+
self._id2label: dict[int, str] = self._model.config.id2label
|
|
70
|
+
logger.info("PP-DocLayout-V3 model loaded successfully")
|
|
71
|
+
|
|
72
|
+
@classmethod
|
|
73
|
+
def get_options_type(cls) -> type[BaseLayoutOptions]:
|
|
74
|
+
"""Return the options class for this layout model."""
|
|
75
|
+
return PPDocLayoutV3Options
|
|
76
|
+
|
|
77
|
+
def _run_inference(
|
|
78
|
+
self,
|
|
79
|
+
images: list[Image.Image],
|
|
80
|
+
) -> list[list[dict]]:
|
|
81
|
+
"""Run PP-DocLayout-V3 on a batch of PIL images.
|
|
82
|
+
|
|
83
|
+
Returns a list (per image) of lists of detection dicts with keys
|
|
84
|
+
``label``, ``confidence``, ``l``, ``t``, ``r``, ``b``.
|
|
85
|
+
"""
|
|
86
|
+
inputs = self._image_processor(images=images, return_tensors="pt")
|
|
87
|
+
inputs = {k: v.to(self._device) for k, v in inputs.items()}
|
|
88
|
+
|
|
89
|
+
with torch.no_grad():
|
|
90
|
+
outputs = self._model(**inputs)
|
|
91
|
+
|
|
92
|
+
target_sizes = [img.size[::-1] for img in images] # (height, width)
|
|
93
|
+
results = self._image_processor.post_process_object_detection(
|
|
94
|
+
outputs,
|
|
95
|
+
target_sizes=target_sizes,
|
|
96
|
+
threshold=self.options.confidence_threshold,
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
batch_detections: list[list[dict]] = []
|
|
100
|
+
for result in results:
|
|
101
|
+
detections: list[dict] = []
|
|
102
|
+
|
|
103
|
+
polys = result.get("polygons") or result.get("polygon_points")
|
|
104
|
+
if polys is None:
|
|
105
|
+
polys = [None] * len(result["scores"])
|
|
106
|
+
|
|
107
|
+
for score, label_id, box, poly in zip(
|
|
108
|
+
result["scores"],
|
|
109
|
+
result["labels"],
|
|
110
|
+
result["boxes"],
|
|
111
|
+
polys,
|
|
112
|
+
strict=True,
|
|
113
|
+
):
|
|
114
|
+
raw_label = self._id2label.get(label_id.item(), "text")
|
|
115
|
+
doc_label = LABEL_MAP.get(raw_label, DocItemLabel.TEXT)
|
|
116
|
+
|
|
117
|
+
if poly is not None and len(poly) > 0:
|
|
118
|
+
# Flatten or handle nested points to extract min/max
|
|
119
|
+
if isinstance(poly[0], int | float):
|
|
120
|
+
xs = poly[0::2]
|
|
121
|
+
ys = poly[1::2]
|
|
122
|
+
else:
|
|
123
|
+
xs = [pt[0] for pt in poly]
|
|
124
|
+
ys = [pt[1] for pt in poly]
|
|
125
|
+
x_min, x_max = min(xs), max(xs)
|
|
126
|
+
y_min, y_max = min(ys), max(ys)
|
|
127
|
+
else:
|
|
128
|
+
x_min, y_min, x_max, y_max = box.tolist()
|
|
129
|
+
|
|
130
|
+
detections.append({
|
|
131
|
+
"label": doc_label,
|
|
132
|
+
"confidence": score.item(),
|
|
133
|
+
"l": x_min,
|
|
134
|
+
"t": y_min,
|
|
135
|
+
"r": x_max,
|
|
136
|
+
"b": y_max,
|
|
137
|
+
})
|
|
138
|
+
batch_detections.append(detections)
|
|
139
|
+
|
|
140
|
+
return batch_detections
|
|
141
|
+
|
|
142
|
+
def predict_layout(
|
|
143
|
+
self,
|
|
144
|
+
conv_res: ConversionResult,
|
|
145
|
+
pages: Sequence[Page],
|
|
146
|
+
) -> Sequence[LayoutPrediction]:
|
|
147
|
+
"""Detect layout regions for a batch of document pages."""
|
|
148
|
+
pages = list(pages)
|
|
149
|
+
|
|
150
|
+
valid_pages: list[Page] = []
|
|
151
|
+
valid_images: list[Image.Image] = []
|
|
152
|
+
is_page_valid: list[bool] = []
|
|
153
|
+
|
|
154
|
+
for page in pages:
|
|
155
|
+
if page._backend is None or not page._backend.is_valid(): # noqa: SLF001
|
|
156
|
+
is_page_valid.append(False)
|
|
157
|
+
continue
|
|
158
|
+
if page.size is None:
|
|
159
|
+
is_page_valid.append(False)
|
|
160
|
+
continue
|
|
161
|
+
page_image = page.get_image(scale=1.0)
|
|
162
|
+
if page_image is None:
|
|
163
|
+
is_page_valid.append(False)
|
|
164
|
+
continue
|
|
165
|
+
|
|
166
|
+
valid_pages.append(page)
|
|
167
|
+
valid_images.append(page_image)
|
|
168
|
+
is_page_valid.append(True)
|
|
169
|
+
|
|
170
|
+
batch_detections: list[list[dict]] = []
|
|
171
|
+
if valid_images:
|
|
172
|
+
with TimeRecorder(conv_res, "layout"):
|
|
173
|
+
bs = self.options.batch_size
|
|
174
|
+
for i in range(0, len(valid_images), bs):
|
|
175
|
+
batch = valid_images[i : i + bs]
|
|
176
|
+
batch_detections.extend(self._run_inference(batch))
|
|
177
|
+
|
|
178
|
+
layout_predictions: list[LayoutPrediction] = []
|
|
179
|
+
valid_idx = 0
|
|
180
|
+
|
|
181
|
+
for idx, page in enumerate(pages):
|
|
182
|
+
if not is_page_valid[idx]:
|
|
183
|
+
existing = page.predictions.layout or LayoutPrediction()
|
|
184
|
+
layout_predictions.append(existing)
|
|
185
|
+
continue
|
|
186
|
+
|
|
187
|
+
detections = batch_detections[valid_idx]
|
|
188
|
+
valid_idx += 1
|
|
189
|
+
|
|
190
|
+
clusters: list[Cluster] = []
|
|
191
|
+
for ix, det in enumerate(detections):
|
|
192
|
+
cluster = Cluster(
|
|
193
|
+
id=ix,
|
|
194
|
+
label=det["label"],
|
|
195
|
+
confidence=det["confidence"],
|
|
196
|
+
bbox=BoundingBox(
|
|
197
|
+
l=det["l"],
|
|
198
|
+
t=det["t"],
|
|
199
|
+
r=det["r"],
|
|
200
|
+
b=det["b"],
|
|
201
|
+
),
|
|
202
|
+
cells=[],
|
|
203
|
+
)
|
|
204
|
+
clusters.append(cluster)
|
|
205
|
+
|
|
206
|
+
processed_clusters, processed_cells = LayoutPostprocessor(page, clusters, self.options).postprocess()
|
|
207
|
+
|
|
208
|
+
with warnings.catch_warnings():
|
|
209
|
+
warnings.filterwarnings(
|
|
210
|
+
"ignore",
|
|
211
|
+
"Mean of empty slice|invalid value encountered in scalar divide",
|
|
212
|
+
RuntimeWarning,
|
|
213
|
+
"numpy",
|
|
214
|
+
)
|
|
215
|
+
conv_res.confidence.pages[page.page_no].layout_score = float(
|
|
216
|
+
np.mean([c.confidence for c in processed_clusters])
|
|
217
|
+
)
|
|
218
|
+
conv_res.confidence.pages[page.page_no].ocr_score = float(
|
|
219
|
+
np.mean([c.confidence for c in processed_cells if c.from_ocr])
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
prediction = LayoutPrediction(clusters=processed_clusters)
|
|
223
|
+
layout_predictions.append(prediction)
|
|
224
|
+
|
|
225
|
+
return layout_predictions
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""Configuration model for the PP-DocLayout-V3 layout engine."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Annotated, ClassVar, Literal
|
|
6
|
+
|
|
7
|
+
from docling.datamodel.pipeline_options import LayoutOptions
|
|
8
|
+
from pydantic import ConfigDict, Field
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class PPDocLayoutV3Options(LayoutOptions):
|
|
12
|
+
"""Options for the PP-DocLayout-V3 layout detection engine.
|
|
13
|
+
|
|
14
|
+
Uses a HuggingFace-hosted PP-DocLayout-V3 model to detect document
|
|
15
|
+
layout elements (text, tables, figures, headers, etc.) in page images.
|
|
16
|
+
|
|
17
|
+
Attributes:
|
|
18
|
+
model_name: HuggingFace model repository ID.
|
|
19
|
+
confidence_threshold: Minimum confidence score for detections.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
kind: ClassVar[Literal["ppdoclayout-v3"]] = "ppdoclayout-v3"
|
|
23
|
+
|
|
24
|
+
model_name: Annotated[
|
|
25
|
+
str,
|
|
26
|
+
Field(description="HuggingFace model repository ID for PP-DocLayout-V3."),
|
|
27
|
+
] = "PaddlePaddle/PP-DocLayoutV3_safetensors"
|
|
28
|
+
|
|
29
|
+
confidence_threshold: Annotated[
|
|
30
|
+
float,
|
|
31
|
+
Field(
|
|
32
|
+
ge=0.0,
|
|
33
|
+
le=1.0,
|
|
34
|
+
description="Minimum confidence score to keep a detection.",
|
|
35
|
+
),
|
|
36
|
+
] = 0.5
|
|
37
|
+
|
|
38
|
+
batch_size: Annotated[
|
|
39
|
+
int,
|
|
40
|
+
Field(
|
|
41
|
+
gt=0,
|
|
42
|
+
description="Batch size for layout inference.",
|
|
43
|
+
),
|
|
44
|
+
] = 8
|
|
45
|
+
|
|
46
|
+
model_config = ConfigDict(extra="forbid")
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Docling plugin entry point registering the PP-DocLayout-V3 layout engine."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from docling_pp_doc_layout.model import PPDocLayoutV3Model
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def layout_engines() -> dict[str, Any]:
|
|
11
|
+
"""Return layout engine classes provided by this plugin."""
|
|
12
|
+
return {"layout_engines": [PPDocLayoutV3Model]}
|
|
File without changes
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: docling-pp-doc-layout
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A Docling plugin for PaddlePaddle PP-DocLayout-V3 model document layout detection.
|
|
5
|
+
Project-URL: Homepage, https://github.com/DCC-BS/docling-pp-doc-layout
|
|
6
|
+
Project-URL: Repository, https://github.com/DCC-BS/docling-pp-doc-layout
|
|
7
|
+
Project-URL: Issues, https://github.com/DCC-BS/docling-pp-doc-layout/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/DCC-BS/docling-pp-doc-layout/releases
|
|
9
|
+
Author-email: Yanick Schraner <yanick.schraner@bs.ch>, Tobias Bollinger <tobias.bollinger@bs.ch>
|
|
10
|
+
License: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Typing :: Typed
|
|
18
|
+
Requires-Python: >=3.13
|
|
19
|
+
Requires-Dist: docling>=2.73
|
|
20
|
+
Requires-Dist: torch
|
|
21
|
+
Requires-Dist: transformers>=5.1.0
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
|
|
24
|
+
# docling-pp-doc-layout
|
|
25
|
+
|
|
26
|
+
A [Docling](https://github.com/docling-project/docling) plugin that provides document layout detection using the PaddlePaddle PP-DocLayout-V3 model.
|
|
27
|
+
|
|
28
|
+
This plugin seamlessly integrates with Docling's standard pipeline to replace the default layout models with [PP-DocLayout-V3](https://huggingface.co/PaddlePaddle/PP-DocLayoutV3), enabling high-accuracy, instance segmentation-based layout analysis with polygon bounding box support, properly processed in optimized batches for enterprise scalability.
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
<p align="center">
|
|
33
|
+
<a href="https://github.com/DCC-BS/docling-pp-doc-layout">GitHub</a>
|
|
34
|
+
|
|
|
35
|
+
<a href="https://pypi.org/project/docling-pp-doc-layout/">PyPI</a>
|
|
36
|
+
</p>
|
|
37
|
+
|
|
38
|
+
---
|
|
39
|
+
|
|
40
|
+
[](https://pypi.org/project/docling-pp-doc-layout/)
|
|
41
|
+
[](https://pypi.org/project/docling-pp-doc-layout/)
|
|
42
|
+
[](https://github.com/DCC-BS/docling-pp-doc-layout/blob/main/LICENSE)
|
|
43
|
+
[](https://github.com/DCC-BS/docling-pp-doc-layout/actions/workflows/main.yml)
|
|
44
|
+
[](https://github.com/astral-sh/ruff)
|
|
45
|
+
[](https://codecov.io/gh/DCC-BS/docling-pp-doc-layout)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
## Overview
|
|
49
|
+
|
|
50
|
+
`docling-pp-doc-layout` provides the `PPDocLayoutV3Model` layout engine for Docling. It automatically registers itself into Docling's plugin system upon installation. When configured in a Docling `DocumentConverter`, it intercepts page images, batches them, and infers document structural elements (text, tables, figures, headers, etc.) using HuggingFace's transformers library.
|
|
51
|
+
|
|
52
|
+
Key Features:
|
|
53
|
+
- **High Accuracy Layout Parsing**: Uses the RT-DETR instance segmentation framework.
|
|
54
|
+
- **Polygon Conversion**: Gracefully flattens complex polygon masks to Docling-compatible bounding boxes.
|
|
55
|
+
- **Enterprise Scalability**: Configurable batch sizing avoids out-of-memory (OOM) errors on large documents.
|
|
56
|
+
|
|
57
|
+
## Architecture & Integration
|
|
58
|
+
|
|
59
|
+
When you install this package, Docling discovers it automatically through standard Python package entry points.
|
|
60
|
+
|
|
61
|
+
```mermaid
|
|
62
|
+
flowchart TD
|
|
63
|
+
A[Docling DocumentConverter] --> B[PdfPipeline]
|
|
64
|
+
|
|
65
|
+
subgraph Plugin System
|
|
66
|
+
C[Docling PluginManager] -.->|Discovers via entry-points| D[docling-pp-doc-layout]
|
|
67
|
+
D -.->|Registers| E[PPDocLayoutV3Model]
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
B -->|Initialization| C
|
|
71
|
+
B -->|Predict Layout Pages| E
|
|
72
|
+
E -->|Batched Tensors| F[HuggingFace AutoModel]
|
|
73
|
+
F -->|Raw Polygons / Boxes| E
|
|
74
|
+
E -->|Post-processed Clusters & BoundingBoxes| B
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
## Requirements
|
|
78
|
+
|
|
79
|
+
- Python 3.13+
|
|
80
|
+
- `docling>=2.73`
|
|
81
|
+
- `transformers>=5.1.0`
|
|
82
|
+
- `torch`
|
|
83
|
+
|
|
84
|
+
## Installation
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
# with uv (recommended)
|
|
88
|
+
uv add docling-pp-doc-layout
|
|
89
|
+
|
|
90
|
+
# with pip
|
|
91
|
+
pip install docling-pp-doc-layout
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
## Usage
|
|
95
|
+
|
|
96
|
+
Using `docling-pp-doc-layout` is exactly like configuring standard Docling options.
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
from docling.document_converter import DocumentConverter, PdfFormatOption
|
|
100
|
+
from docling.datamodel.pipeline_options import PdfPipelineOptions
|
|
101
|
+
from docling_pp_doc_layout.options import PPDocLayoutV3Options
|
|
102
|
+
|
|
103
|
+
# 1. Define Pipeline Options
|
|
104
|
+
pipeline_options = PdfPipelineOptions()
|
|
105
|
+
|
|
106
|
+
# 2. Configure our custom PPDocLayoutV3Options
|
|
107
|
+
pipeline_options.layout_options = PPDocLayoutV3Options(
|
|
108
|
+
batch_size=8, # Tweak for GPU VRAM usage
|
|
109
|
+
confidence_threshold=0.5, # Filter low-confidence detections
|
|
110
|
+
model_name="PaddlePaddle/PP-DocLayoutV3_safetensors" # Target HuggingFace model repo
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
# 3. Create the converter
|
|
114
|
+
converter = DocumentConverter(
|
|
115
|
+
format_options={
|
|
116
|
+
"pdf": PdfFormatOption(pipeline_options=pipeline_options)
|
|
117
|
+
}
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
# 4. Convert Document
|
|
121
|
+
result = converter.convert("path/to/your/document.pdf")
|
|
122
|
+
print("Converted elements:", len(result.document.elements))
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
## Configuration Options
|
|
126
|
+
|
|
127
|
+
The `PPDocLayoutV3Options` dataclass gives you full control over the engine:
|
|
128
|
+
|
|
129
|
+
| Parameter | Type | Default | Description |
|
|
130
|
+
|-------------------------|---------|---------|-------------|
|
|
131
|
+
| `batch_size` | `int` | 8 | How many pages to process per single step. Decrease to lower memory usage; Increase to speed up processing of large documents. |
|
|
132
|
+
| `confidence_threshold` | `float` | 0.5 | The minimum confidence score (0.0 - 1.0) required to keep a layout detection cluster. |
|
|
133
|
+
| `model_name` | `str` | `"PaddlePaddle/PP-DocLayoutV3_safetensors"` | HuggingFace repository ID. Allows overriding if you host your local copy or a fine-tuned version. |
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
## Development
|
|
137
|
+
|
|
138
|
+
If you wish to contribute or modify the plugin locally:
|
|
139
|
+
|
|
140
|
+
```bash
|
|
141
|
+
git clone https://github.com/DCC-BS/docling-pp-doc-layout.git
|
|
142
|
+
cd docling-pp-doc-layout
|
|
143
|
+
|
|
144
|
+
# Install dependencies and pre-commit hooks
|
|
145
|
+
make install
|
|
146
|
+
|
|
147
|
+
# Run checks (ruff, ty) and tests (pytest)
|
|
148
|
+
make check
|
|
149
|
+
make test
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
## License
|
|
153
|
+
|
|
154
|
+
[MIT](LICENSE) © DCC Data Competence Center
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
docling_pp_doc_layout/__init__.py,sha256=5oubMsuaB2fR-gWT6HCmKZN6GGntqFHq4CpbODzCJws,112
|
|
2
|
+
docling_pp_doc_layout/label_mapping.py,sha256=jXBBwfYIWLOtQW5qdtm1HYdV0Is7THFQLu450Xg-vZA,1207
|
|
3
|
+
docling_pp_doc_layout/model.py,sha256=FSvZg5aNpNI5eQn59h2grtlOcDjZIjBAOYlkX20dGHY,8110
|
|
4
|
+
docling_pp_doc_layout/options.py,sha256=oN6xtSP6fZaWRhtSqXWB28HfpOuBgAC7Gf8H4ExebXs,1302
|
|
5
|
+
docling_pp_doc_layout/plugin.py,sha256=ByMR-eUTYtZ0bjc0DPa1I_VlSb8PLmElDgTvBPYTztk,358
|
|
6
|
+
docling_pp_doc_layout/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
7
|
+
docling_pp_doc_layout-0.1.0.dist-info/METADATA,sha256=vT7CeQv9gO48AWEjV1xN5zZiP-duWXsEydpAShUTJdM,6203
|
|
8
|
+
docling_pp_doc_layout-0.1.0.dist-info/WHEEL,sha256=WLgqFyCfm_KASv4WHyYy0P3pM_m7J5L9k2skdKLirC8,87
|
|
9
|
+
docling_pp_doc_layout-0.1.0.dist-info/entry_points.txt,sha256=GWD1I-zcU5f4ZRLrUVBSb0VpeUcB5o5Q6_xJv78vnzQ,77
|
|
10
|
+
docling_pp_doc_layout-0.1.0.dist-info/licenses/LICENSE,sha256=APpdFXym3_5C78pFAfnlLvmP7umPBWYpQ2MKidnoRuw,1091
|
|
11
|
+
docling_pp_doc_layout-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Data Competence Center Basel-Stadt
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|