docling-pp-doc-layout 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ """A Docling plugin for PaddlePaddle PP-DocLayout-V3 model document layout detection."""
2
+
3
+ __version__ = "0.1.0"
@@ -0,0 +1,34 @@
1
+ """Mapping from PP-DocLayout-V3 label names to docling DocItemLabel values.
2
+
3
+ Every label produced here must exist in
4
+ ``docling.utils.layout_postprocessor.LayoutPostprocessor.CONFIDENCE_THRESHOLDS``
5
+ so that the postprocessor can apply confidence filtering without a ``KeyError``.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from docling_core.types.doc import DocItemLabel
11
+
12
+ LABEL_MAP: dict[str, DocItemLabel] = {
13
+ "abstract": DocItemLabel.TEXT,
14
+ "algorithm": DocItemLabel.CODE,
15
+ "aside_text": DocItemLabel.TEXT,
16
+ "chart": DocItemLabel.PICTURE,
17
+ "content": DocItemLabel.TEXT,
18
+ "doc_title": DocItemLabel.TITLE,
19
+ "figure_title": DocItemLabel.CAPTION,
20
+ "footer": DocItemLabel.PAGE_FOOTER,
21
+ "footnote": DocItemLabel.FOOTNOTE,
22
+ "formula": DocItemLabel.FORMULA,
23
+ "formula_number": DocItemLabel.TEXT,
24
+ "header": DocItemLabel.PAGE_HEADER,
25
+ "image": DocItemLabel.PICTURE,
26
+ "number": DocItemLabel.TEXT,
27
+ "paragraph_title": DocItemLabel.SECTION_HEADER,
28
+ "reference": DocItemLabel.TEXT,
29
+ "reference_content": DocItemLabel.TEXT,
30
+ "seal": DocItemLabel.PICTURE,
31
+ "table": DocItemLabel.TABLE,
32
+ "text": DocItemLabel.TEXT,
33
+ "vision_footnote": DocItemLabel.FOOTNOTE,
34
+ }
@@ -0,0 +1,225 @@
1
+ """PP-DocLayout-V3 layout model for the docling standard pipeline.
2
+
3
+ Runs PaddlePaddle PP-DocLayout-V3 locally via HuggingFace ``transformers``
4
+ to detect document layout elements and returns ``LayoutPrediction`` objects
5
+ that docling merges with its standard-pipeline output.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import logging
11
+ import warnings
12
+ from typing import TYPE_CHECKING
13
+
14
+ import numpy as np
15
+ import torch
16
+ from docling.datamodel.base_models import BoundingBox, Cluster, LayoutPrediction, Page
17
+ from docling.models.base_layout_model import BaseLayoutModel
18
+ from docling.utils.accelerator_utils import decide_device
19
+ from docling.utils.layout_postprocessor import LayoutPostprocessor
20
+ from docling.utils.profiling import TimeRecorder
21
+ from docling_core.types.doc import DocItemLabel
22
+ from transformers import AutoImageProcessor, AutoModelForObjectDetection
23
+
24
+ from docling_pp_doc_layout.label_mapping import LABEL_MAP
25
+ from docling_pp_doc_layout.options import PPDocLayoutV3Options
26
+
27
+ if TYPE_CHECKING:
28
+ from collections.abc import Sequence
29
+ from pathlib import Path
30
+
31
+ from docling.datamodel.accelerator_options import AcceleratorOptions
32
+ from docling.datamodel.document import ConversionResult
33
+ from docling.datamodel.pipeline_options import BaseLayoutOptions
34
+ from PIL import Image
35
+
36
+ logger = logging.getLogger(__name__)
37
+
38
+
39
+ class PPDocLayoutV3Model(BaseLayoutModel):
40
+ """Layout engine using PP-DocLayout-V3 via HuggingFace transformers."""
41
+
42
+ def __init__(
43
+ self,
44
+ artifacts_path: Path | None,
45
+ accelerator_options: AcceleratorOptions,
46
+ options: PPDocLayoutV3Options,
47
+ *,
48
+ enable_remote_services: bool = False, # noqa: ARG002
49
+ ) -> None:
50
+ self.options = options
51
+ self.artifacts_path = artifacts_path
52
+ self.accelerator_options = accelerator_options
53
+
54
+ self._device = decide_device(accelerator_options.device)
55
+ logger.info(
56
+ "Loading PP-DocLayout-V3 model %s on device=%s",
57
+ options.model_name,
58
+ self._device,
59
+ )
60
+
61
+ self._image_processor = AutoImageProcessor.from_pretrained(
62
+ options.model_name,
63
+ )
64
+ self._model = AutoModelForObjectDetection.from_pretrained(
65
+ options.model_name,
66
+ ).to(self._device)
67
+ self._model.eval()
68
+
69
+ self._id2label: dict[int, str] = self._model.config.id2label
70
+ logger.info("PP-DocLayout-V3 model loaded successfully")
71
+
72
+ @classmethod
73
+ def get_options_type(cls) -> type[BaseLayoutOptions]:
74
+ """Return the options class for this layout model."""
75
+ return PPDocLayoutV3Options
76
+
77
+ def _run_inference(
78
+ self,
79
+ images: list[Image.Image],
80
+ ) -> list[list[dict]]:
81
+ """Run PP-DocLayout-V3 on a batch of PIL images.
82
+
83
+ Returns a list (per image) of lists of detection dicts with keys
84
+ ``label``, ``confidence``, ``l``, ``t``, ``r``, ``b``.
85
+ """
86
+ inputs = self._image_processor(images=images, return_tensors="pt")
87
+ inputs = {k: v.to(self._device) for k, v in inputs.items()}
88
+
89
+ with torch.no_grad():
90
+ outputs = self._model(**inputs)
91
+
92
+ target_sizes = [img.size[::-1] for img in images] # (height, width)
93
+ results = self._image_processor.post_process_object_detection(
94
+ outputs,
95
+ target_sizes=target_sizes,
96
+ threshold=self.options.confidence_threshold,
97
+ )
98
+
99
+ batch_detections: list[list[dict]] = []
100
+ for result in results:
101
+ detections: list[dict] = []
102
+
103
+ polys = result.get("polygons") or result.get("polygon_points")
104
+ if polys is None:
105
+ polys = [None] * len(result["scores"])
106
+
107
+ for score, label_id, box, poly in zip(
108
+ result["scores"],
109
+ result["labels"],
110
+ result["boxes"],
111
+ polys,
112
+ strict=True,
113
+ ):
114
+ raw_label = self._id2label.get(label_id.item(), "text")
115
+ doc_label = LABEL_MAP.get(raw_label, DocItemLabel.TEXT)
116
+
117
+ if poly is not None and len(poly) > 0:
118
+ # Flatten or handle nested points to extract min/max
119
+ if isinstance(poly[0], int | float):
120
+ xs = poly[0::2]
121
+ ys = poly[1::2]
122
+ else:
123
+ xs = [pt[0] for pt in poly]
124
+ ys = [pt[1] for pt in poly]
125
+ x_min, x_max = min(xs), max(xs)
126
+ y_min, y_max = min(ys), max(ys)
127
+ else:
128
+ x_min, y_min, x_max, y_max = box.tolist()
129
+
130
+ detections.append({
131
+ "label": doc_label,
132
+ "confidence": score.item(),
133
+ "l": x_min,
134
+ "t": y_min,
135
+ "r": x_max,
136
+ "b": y_max,
137
+ })
138
+ batch_detections.append(detections)
139
+
140
+ return batch_detections
141
+
142
+ def predict_layout(
143
+ self,
144
+ conv_res: ConversionResult,
145
+ pages: Sequence[Page],
146
+ ) -> Sequence[LayoutPrediction]:
147
+ """Detect layout regions for a batch of document pages."""
148
+ pages = list(pages)
149
+
150
+ valid_pages: list[Page] = []
151
+ valid_images: list[Image.Image] = []
152
+ is_page_valid: list[bool] = []
153
+
154
+ for page in pages:
155
+ if page._backend is None or not page._backend.is_valid(): # noqa: SLF001
156
+ is_page_valid.append(False)
157
+ continue
158
+ if page.size is None:
159
+ is_page_valid.append(False)
160
+ continue
161
+ page_image = page.get_image(scale=1.0)
162
+ if page_image is None:
163
+ is_page_valid.append(False)
164
+ continue
165
+
166
+ valid_pages.append(page)
167
+ valid_images.append(page_image)
168
+ is_page_valid.append(True)
169
+
170
+ batch_detections: list[list[dict]] = []
171
+ if valid_images:
172
+ with TimeRecorder(conv_res, "layout"):
173
+ bs = self.options.batch_size
174
+ for i in range(0, len(valid_images), bs):
175
+ batch = valid_images[i : i + bs]
176
+ batch_detections.extend(self._run_inference(batch))
177
+
178
+ layout_predictions: list[LayoutPrediction] = []
179
+ valid_idx = 0
180
+
181
+ for idx, page in enumerate(pages):
182
+ if not is_page_valid[idx]:
183
+ existing = page.predictions.layout or LayoutPrediction()
184
+ layout_predictions.append(existing)
185
+ continue
186
+
187
+ detections = batch_detections[valid_idx]
188
+ valid_idx += 1
189
+
190
+ clusters: list[Cluster] = []
191
+ for ix, det in enumerate(detections):
192
+ cluster = Cluster(
193
+ id=ix,
194
+ label=det["label"],
195
+ confidence=det["confidence"],
196
+ bbox=BoundingBox(
197
+ l=det["l"],
198
+ t=det["t"],
199
+ r=det["r"],
200
+ b=det["b"],
201
+ ),
202
+ cells=[],
203
+ )
204
+ clusters.append(cluster)
205
+
206
+ processed_clusters, processed_cells = LayoutPostprocessor(page, clusters, self.options).postprocess()
207
+
208
+ with warnings.catch_warnings():
209
+ warnings.filterwarnings(
210
+ "ignore",
211
+ "Mean of empty slice|invalid value encountered in scalar divide",
212
+ RuntimeWarning,
213
+ "numpy",
214
+ )
215
+ conv_res.confidence.pages[page.page_no].layout_score = float(
216
+ np.mean([c.confidence for c in processed_clusters])
217
+ )
218
+ conv_res.confidence.pages[page.page_no].ocr_score = float(
219
+ np.mean([c.confidence for c in processed_cells if c.from_ocr])
220
+ )
221
+
222
+ prediction = LayoutPrediction(clusters=processed_clusters)
223
+ layout_predictions.append(prediction)
224
+
225
+ return layout_predictions
@@ -0,0 +1,46 @@
1
+ """Configuration model for the PP-DocLayout-V3 layout engine."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Annotated, ClassVar, Literal
6
+
7
+ from docling.datamodel.pipeline_options import LayoutOptions
8
+ from pydantic import ConfigDict, Field
9
+
10
+
11
+ class PPDocLayoutV3Options(LayoutOptions):
12
+ """Options for the PP-DocLayout-V3 layout detection engine.
13
+
14
+ Uses a HuggingFace-hosted PP-DocLayout-V3 model to detect document
15
+ layout elements (text, tables, figures, headers, etc.) in page images.
16
+
17
+ Attributes:
18
+ model_name: HuggingFace model repository ID.
19
+ confidence_threshold: Minimum confidence score for detections.
20
+ """
21
+
22
+ kind: ClassVar[Literal["ppdoclayout-v3"]] = "ppdoclayout-v3"
23
+
24
+ model_name: Annotated[
25
+ str,
26
+ Field(description="HuggingFace model repository ID for PP-DocLayout-V3."),
27
+ ] = "PaddlePaddle/PP-DocLayoutV3_safetensors"
28
+
29
+ confidence_threshold: Annotated[
30
+ float,
31
+ Field(
32
+ ge=0.0,
33
+ le=1.0,
34
+ description="Minimum confidence score to keep a detection.",
35
+ ),
36
+ ] = 0.5
37
+
38
+ batch_size: Annotated[
39
+ int,
40
+ Field(
41
+ gt=0,
42
+ description="Batch size for layout inference.",
43
+ ),
44
+ ] = 8
45
+
46
+ model_config = ConfigDict(extra="forbid")
@@ -0,0 +1,12 @@
1
+ """Docling plugin entry point registering the PP-DocLayout-V3 layout engine."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from docling_pp_doc_layout.model import PPDocLayoutV3Model
8
+
9
+
10
+ def layout_engines() -> dict[str, Any]:
11
+ """Return layout engine classes provided by this plugin."""
12
+ return {"layout_engines": [PPDocLayoutV3Model]}
File without changes
@@ -0,0 +1,154 @@
1
+ Metadata-Version: 2.4
2
+ Name: docling-pp-doc-layout
3
+ Version: 0.1.0
4
+ Summary: A Docling plugin for PaddlePaddle PP-DocLayout-V3 model document layout detection.
5
+ Project-URL: Homepage, https://github.com/DCC-BS/docling-pp-doc-layout
6
+ Project-URL: Repository, https://github.com/DCC-BS/docling-pp-doc-layout
7
+ Project-URL: Issues, https://github.com/DCC-BS/docling-pp-doc-layout/issues
8
+ Project-URL: Changelog, https://github.com/DCC-BS/docling-pp-doc-layout/releases
9
+ Author-email: Yanick Schraner <yanick.schraner@bs.ch>, Tobias Bollinger <tobias.bollinger@bs.ch>
10
+ License: MIT
11
+ License-File: LICENSE
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Typing :: Typed
18
+ Requires-Python: >=3.13
19
+ Requires-Dist: docling>=2.73
20
+ Requires-Dist: torch
21
+ Requires-Dist: transformers>=5.1.0
22
+ Description-Content-Type: text/markdown
23
+
24
+ # docling-pp-doc-layout
25
+
26
+ A [Docling](https://github.com/docling-project/docling) plugin that provides document layout detection using the PaddlePaddle PP-DocLayout-V3 model.
27
+
28
+ This plugin seamlessly integrates with Docling's standard pipeline to replace the default layout models with [PP-DocLayout-V3](https://huggingface.co/PaddlePaddle/PP-DocLayoutV3), enabling high-accuracy, instance segmentation-based layout analysis with polygon bounding box support, properly processed in optimized batches for enterprise scalability.
29
+
30
+ ---
31
+
32
+ <p align="center">
33
+ <a href="https://github.com/DCC-BS/docling-pp-doc-layout">GitHub</a>
34
+ &nbsp;|&nbsp;
35
+ <a href="https://pypi.org/project/docling-pp-doc-layout/">PyPI</a>
36
+ </p>
37
+
38
+ ---
39
+
40
+ [![PyPI version](https://img.shields.io/pypi/v/docling-pp-doc-layout.svg)](https://pypi.org/project/docling-pp-doc-layout/)
41
+ [![Python versions](https://img.shields.io/pypi/pyversions/docling-pp-doc-layout.svg)](https://pypi.org/project/docling-pp-doc-layout/)
42
+ [![License](https://img.shields.io/github/license/DCC-BS/docling-pp-doc-layout)](https://github.com/DCC-BS/docling-pp-doc-layout/blob/main/LICENSE)
43
+ [![CI](https://github.com/DCC-BS/docling-pp-doc-layout/actions/workflows/main.yml/badge.svg)](https://github.com/DCC-BS/docling-pp-doc-layout/actions/workflows/main.yml)
44
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
45
+ [![Coverage](https://codecov.io/gh/DCC-BS/docling-pp-doc-layout/graph/badge.svg)](https://codecov.io/gh/DCC-BS/docling-pp-doc-layout)
46
+
47
+
48
+ ## Overview
49
+
50
+ `docling-pp-doc-layout` provides the `PPDocLayoutV3Model` layout engine for Docling. It automatically registers itself into Docling's plugin system upon installation. When configured in a Docling `DocumentConverter`, it intercepts page images, batches them, and infers document structural elements (text, tables, figures, headers, etc.) using HuggingFace's transformers library.
51
+
52
+ Key Features:
53
+ - **High Accuracy Layout Parsing**: Uses the RT-DETR instance segmentation framework.
54
+ - **Polygon Conversion**: Gracefully flattens complex polygon masks to Docling-compatible bounding boxes.
55
+ - **Enterprise Scalability**: Configurable batch sizing avoids out-of-memory (OOM) errors on large documents.
56
+
57
+ ## Architecture & Integration
58
+
59
+ When you install this package, Docling discovers it automatically through standard Python package entry points.
60
+
61
+ ```mermaid
62
+ flowchart TD
63
+ A[Docling DocumentConverter] --> B[PdfPipeline]
64
+
65
+ subgraph Plugin System
66
+ C[Docling PluginManager] -.->|Discovers via entry-points| D[docling-pp-doc-layout]
67
+ D -.->|Registers| E[PPDocLayoutV3Model]
68
+ end
69
+
70
+ B -->|Initialization| C
71
+ B -->|Predict Layout Pages| E
72
+ E -->|Batched Tensors| F[HuggingFace AutoModel]
73
+ F -->|Raw Polygons / Boxes| E
74
+ E -->|Post-processed Clusters & BoundingBoxes| B
75
+ ```
76
+
77
+ ## Requirements
78
+
79
+ - Python 3.13+
80
+ - `docling>=2.73`
81
+ - `transformers>=5.1.0`
82
+ - `torch`
83
+
84
+ ## Installation
85
+
86
+ ```bash
87
+ # with uv (recommended)
88
+ uv add docling-pp-doc-layout
89
+
90
+ # with pip
91
+ pip install docling-pp-doc-layout
92
+ ```
93
+
94
+ ## Usage
95
+
96
+ Using `docling-pp-doc-layout` is exactly like configuring standard Docling options.
97
+
98
+ ```python
99
+ from docling.document_converter import DocumentConverter, PdfFormatOption
100
+ from docling.datamodel.pipeline_options import PdfPipelineOptions
101
+ from docling_pp_doc_layout.options import PPDocLayoutV3Options
102
+
103
+ # 1. Define Pipeline Options
104
+ pipeline_options = PdfPipelineOptions()
105
+
106
+ # 2. Configure our custom PPDocLayoutV3Options
107
+ pipeline_options.layout_options = PPDocLayoutV3Options(
108
+ batch_size=8, # Tweak for GPU VRAM usage
109
+ confidence_threshold=0.5, # Filter low-confidence detections
110
+ model_name="PaddlePaddle/PP-DocLayoutV3_safetensors" # Target HuggingFace model repo
111
+ )
112
+
113
+ # 3. Create the converter
114
+ converter = DocumentConverter(
115
+ format_options={
116
+ "pdf": PdfFormatOption(pipeline_options=pipeline_options)
117
+ }
118
+ )
119
+
120
+ # 4. Convert Document
121
+ result = converter.convert("path/to/your/document.pdf")
122
+ print("Converted elements:", len(result.document.elements))
123
+ ```
124
+
125
+ ## Configuration Options
126
+
127
+ The `PPDocLayoutV3Options` dataclass gives you full control over the engine:
128
+
129
+ | Parameter | Type | Default | Description |
130
+ |-------------------------|---------|---------|-------------|
131
+ | `batch_size` | `int` | 8 | How many pages to process per single step. Decrease to lower memory usage; Increase to speed up processing of large documents. |
132
+ | `confidence_threshold` | `float` | 0.5 | The minimum confidence score (0.0 - 1.0) required to keep a layout detection cluster. |
133
+ | `model_name` | `str` | `"PaddlePaddle/PP-DocLayoutV3_safetensors"` | HuggingFace repository ID. Allows overriding if you host your local copy or a fine-tuned version. |
134
+
135
+
136
+ ## Development
137
+
138
+ If you wish to contribute or modify the plugin locally:
139
+
140
+ ```bash
141
+ git clone https://github.com/DCC-BS/docling-pp-doc-layout.git
142
+ cd docling-pp-doc-layout
143
+
144
+ # Install dependencies and pre-commit hooks
145
+ make install
146
+
147
+ # Run checks (ruff, ty) and tests (pytest)
148
+ make check
149
+ make test
150
+ ```
151
+
152
+ ## License
153
+
154
+ [MIT](LICENSE) © DCC Data Competence Center
@@ -0,0 +1,11 @@
1
+ docling_pp_doc_layout/__init__.py,sha256=5oubMsuaB2fR-gWT6HCmKZN6GGntqFHq4CpbODzCJws,112
2
+ docling_pp_doc_layout/label_mapping.py,sha256=jXBBwfYIWLOtQW5qdtm1HYdV0Is7THFQLu450Xg-vZA,1207
3
+ docling_pp_doc_layout/model.py,sha256=FSvZg5aNpNI5eQn59h2grtlOcDjZIjBAOYlkX20dGHY,8110
4
+ docling_pp_doc_layout/options.py,sha256=oN6xtSP6fZaWRhtSqXWB28HfpOuBgAC7Gf8H4ExebXs,1302
5
+ docling_pp_doc_layout/plugin.py,sha256=ByMR-eUTYtZ0bjc0DPa1I_VlSb8PLmElDgTvBPYTztk,358
6
+ docling_pp_doc_layout/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
7
+ docling_pp_doc_layout-0.1.0.dist-info/METADATA,sha256=vT7CeQv9gO48AWEjV1xN5zZiP-duWXsEydpAShUTJdM,6203
8
+ docling_pp_doc_layout-0.1.0.dist-info/WHEEL,sha256=WLgqFyCfm_KASv4WHyYy0P3pM_m7J5L9k2skdKLirC8,87
9
+ docling_pp_doc_layout-0.1.0.dist-info/entry_points.txt,sha256=GWD1I-zcU5f4ZRLrUVBSb0VpeUcB5o5Q6_xJv78vnzQ,77
10
+ docling_pp_doc_layout-0.1.0.dist-info/licenses/LICENSE,sha256=APpdFXym3_5C78pFAfnlLvmP7umPBWYpQ2MKidnoRuw,1091
11
+ docling_pp_doc_layout-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.28.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [docling.plugins]
2
+ layout_model = docling_pp_doc_layout.plugin:layout_engines
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Data Competence Center Basel-Stadt
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.