docling-pp-ocrv6 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docling_pp_ocrv6/__init__.py +8 -0
- docling_pp_ocrv6/model.py +275 -0
- docling_pp_ocrv6/options.py +91 -0
- docling_pp_ocrv6/plugin.py +8 -0
- docling_pp_ocrv6/py.typed +0 -0
- docling_pp_ocrv6-0.1.0.dist-info/METADATA +114 -0
- docling_pp_ocrv6-0.1.0.dist-info/RECORD +10 -0
- docling_pp_ocrv6-0.1.0.dist-info/WHEEL +4 -0
- docling_pp_ocrv6-0.1.0.dist-info/entry_points.txt +2 -0
- docling_pp_ocrv6-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
"""PP-OCRv6 OCR engine for the docling standard pipeline.
|
|
2
|
+
|
|
3
|
+
Runs PaddlePaddle PP-OCRv6 detection and recognition ONNX models locally via
|
|
4
|
+
RapidOCR (onnxruntime) and returns the recognised text as ``TextCell`` objects
|
|
5
|
+
that docling merges with its standard-pipeline output.
|
|
6
|
+
|
|
7
|
+
The detection and recognition ONNX models are downloaded from HuggingFace on
|
|
8
|
+
first use and cached under docling's model cache. The recognition character
|
|
9
|
+
dictionary is extracted from the recognition model's ``inference.yml``. Angle
|
|
10
|
+
classification uses RapidOCR's bundled cls model unless an explicit path is
|
|
11
|
+
given.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import logging
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import TYPE_CHECKING, Any, cast
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
import yaml
|
|
22
|
+
from docling.datamodel.accelerator_options import AcceleratorDevice
|
|
23
|
+
from docling.datamodel.settings import settings
|
|
24
|
+
from docling.models.base_ocr_model import BaseOcrModel
|
|
25
|
+
from docling.utils.accelerator_utils import decide_device
|
|
26
|
+
from docling.utils.profiling import TimeRecorder
|
|
27
|
+
from docling_core.types.doc import BoundingBox, CoordOrigin
|
|
28
|
+
from docling_core.types.doc.page import BoundingRectangle, TextCell
|
|
29
|
+
|
|
30
|
+
from docling_pp_ocrv6.options import PPOCRv6Options
|
|
31
|
+
|
|
32
|
+
if TYPE_CHECKING:
|
|
33
|
+
from collections.abc import Iterable
|
|
34
|
+
|
|
35
|
+
from docling.datamodel.accelerator_options import AcceleratorOptions
|
|
36
|
+
from docling.datamodel.base_models import Page
|
|
37
|
+
from docling.datamodel.document import ConversionResult
|
|
38
|
+
from docling.datamodel.pipeline_options import OcrOptions
|
|
39
|
+
|
|
40
|
+
logger = logging.getLogger(__name__)
|
|
41
|
+
|
|
42
|
+
_ONNX_FILE = "inference.onnx"
|
|
43
|
+
_CONFIG_FILE = "inference.yml"
|
|
44
|
+
_REC_KEYS_FILE = "ppocrv6_keys.txt"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class PPOCRv6Model(BaseOcrModel):
|
|
48
|
+
"""OCR engine running PP-OCRv6 ONNX models through RapidOCR."""
|
|
49
|
+
|
|
50
|
+
_model_repo_folder = "PPOCRv6"
|
|
51
|
+
|
|
52
|
+
def __init__(
|
|
53
|
+
self,
|
|
54
|
+
enabled: bool, # noqa: FBT001
|
|
55
|
+
artifacts_path: Path | None,
|
|
56
|
+
options: PPOCRv6Options,
|
|
57
|
+
accelerator_options: AcceleratorOptions,
|
|
58
|
+
) -> None:
|
|
59
|
+
"""Initialise the OCR engine, downloading models on first use when enabled."""
|
|
60
|
+
super().__init__(
|
|
61
|
+
enabled=enabled,
|
|
62
|
+
artifacts_path=artifacts_path,
|
|
63
|
+
options=options,
|
|
64
|
+
accelerator_options=accelerator_options,
|
|
65
|
+
)
|
|
66
|
+
self.options: PPOCRv6Options
|
|
67
|
+
self.scale = 3 # multiplier for 72 dpi == 216 dpi.
|
|
68
|
+
|
|
69
|
+
if not self.enabled:
|
|
70
|
+
return
|
|
71
|
+
|
|
72
|
+
try:
|
|
73
|
+
from rapidocr import EngineType, RapidOCR # noqa: PLC0415
|
|
74
|
+
except ImportError as err:
|
|
75
|
+
msg = (
|
|
76
|
+
"RapidOCR is not installed. Install it via "
|
|
77
|
+
"`pip install rapidocr onnxruntime` (or `onnxruntime-gpu` for CUDA) "
|
|
78
|
+
"to use the PP-OCRv6 OCR engine."
|
|
79
|
+
)
|
|
80
|
+
raise ImportError(msg) from err
|
|
81
|
+
|
|
82
|
+
device = decide_device(accelerator_options.device)
|
|
83
|
+
use_cuda = str(AcceleratorDevice.CUDA.value).lower() in device
|
|
84
|
+
use_dml = accelerator_options.device == AcceleratorDevice.AUTO
|
|
85
|
+
gpu_id = int(device.split(":")[1]) if (use_cuda and ":" in device) else 0
|
|
86
|
+
|
|
87
|
+
det_path, rec_path, rec_keys_path, cls_path = self._resolve_models()
|
|
88
|
+
logger.info(
|
|
89
|
+
"Loading PP-OCRv6 (device=%s, cuda=%s): det=%s rec=%s",
|
|
90
|
+
device,
|
|
91
|
+
use_cuda,
|
|
92
|
+
det_path,
|
|
93
|
+
rec_path,
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
params: dict = {
|
|
97
|
+
"Global.text_score": self.options.text_score,
|
|
98
|
+
"EngineConfig.onnxruntime.intra_op_num_threads": accelerator_options.num_threads,
|
|
99
|
+
"Det.model_path": str(det_path),
|
|
100
|
+
"Det.engine_type": EngineType.ONNXRUNTIME,
|
|
101
|
+
"Det.use_cuda": use_cuda,
|
|
102
|
+
"Det.use_dml": use_dml,
|
|
103
|
+
"Rec.model_path": str(rec_path),
|
|
104
|
+
"Rec.rec_keys_path": str(rec_keys_path),
|
|
105
|
+
"Rec.engine_type": EngineType.ONNXRUNTIME,
|
|
106
|
+
"Rec.use_cuda": use_cuda,
|
|
107
|
+
"Rec.use_dml": use_dml,
|
|
108
|
+
"Cls.engine_type": EngineType.ONNXRUNTIME,
|
|
109
|
+
"Cls.use_cuda": use_cuda,
|
|
110
|
+
"Cls.use_dml": use_dml,
|
|
111
|
+
"EngineConfig.onnxruntime.use_cuda": use_cuda,
|
|
112
|
+
"EngineConfig.onnxruntime.cuda_ep_cfg.device_id": gpu_id,
|
|
113
|
+
}
|
|
114
|
+
# When no explicit cls model is given, let RapidOCR use its bundled cls model.
|
|
115
|
+
if cls_path is not None:
|
|
116
|
+
params["Cls.model_path"] = str(cls_path)
|
|
117
|
+
|
|
118
|
+
if self.options.rapidocr_params:
|
|
119
|
+
params.update(self.options.rapidocr_params)
|
|
120
|
+
|
|
121
|
+
self.reader = RapidOCR(params=params)
|
|
122
|
+
|
|
123
|
+
def _resolve_models(self) -> tuple[Path, Path, Path, Path | None]:
|
|
124
|
+
"""Resolve detection, recognition, rec-keys and (optional) cls model paths.
|
|
125
|
+
|
|
126
|
+
Explicit option paths win; otherwise models are downloaded from
|
|
127
|
+
HuggingFace and cached. The recognition character dictionary is
|
|
128
|
+
extracted from the recognition model's ``inference.yml`` when not
|
|
129
|
+
provided explicitly.
|
|
130
|
+
"""
|
|
131
|
+
local_dir = self.download_models(
|
|
132
|
+
det_repo=self.options.det_repo,
|
|
133
|
+
rec_repo=self.options.rec_repo,
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
det_path = Path(self.options.det_model_path) if self.options.det_model_path else local_dir / "det" / _ONNX_FILE
|
|
137
|
+
rec_path = Path(self.options.rec_model_path) if self.options.rec_model_path else local_dir / "rec" / _ONNX_FILE
|
|
138
|
+
|
|
139
|
+
if self.options.rec_keys_path:
|
|
140
|
+
rec_keys_path = Path(self.options.rec_keys_path)
|
|
141
|
+
else:
|
|
142
|
+
rec_keys_path = self._ensure_rec_keys(local_dir / "rec")
|
|
143
|
+
|
|
144
|
+
cls_path = Path(self.options.cls_model_path) if self.options.cls_model_path else None
|
|
145
|
+
|
|
146
|
+
for path in (det_path, rec_path, rec_keys_path):
|
|
147
|
+
if not path.exists():
|
|
148
|
+
logger.warning("PP-OCRv6 model path does not exist: %s", path)
|
|
149
|
+
|
|
150
|
+
return det_path, rec_path, rec_keys_path, cls_path
|
|
151
|
+
|
|
152
|
+
@staticmethod
|
|
153
|
+
def _ensure_rec_keys(rec_dir: Path) -> Path:
|
|
154
|
+
"""Extract the recognition character dictionary into a RapidOCR keys file.
|
|
155
|
+
|
|
156
|
+
PaddleX exports embed the dictionary as ``PostProcess.character_dict``
|
|
157
|
+
inside ``inference.yml``; RapidOCR expects a plain text file with one
|
|
158
|
+
character per line.
|
|
159
|
+
"""
|
|
160
|
+
keys_path = rec_dir / _REC_KEYS_FILE
|
|
161
|
+
if keys_path.exists():
|
|
162
|
+
return keys_path
|
|
163
|
+
|
|
164
|
+
config = yaml.safe_load((rec_dir / _CONFIG_FILE).read_text(encoding="utf-8"))
|
|
165
|
+
chars = config.get("PostProcess", {}).get("character_dict")
|
|
166
|
+
if not chars:
|
|
167
|
+
msg = f"No 'PostProcess.character_dict' found in {rec_dir / _CONFIG_FILE}"
|
|
168
|
+
raise ValueError(msg)
|
|
169
|
+
|
|
170
|
+
keys_path.write_text("\n".join(chars) + "\n", encoding="utf-8")
|
|
171
|
+
logger.info("Wrote PP-OCRv6 recognition dictionary (%d entries) to %s", len(chars), keys_path)
|
|
172
|
+
return keys_path
|
|
173
|
+
|
|
174
|
+
@staticmethod
|
|
175
|
+
def download_models(
|
|
176
|
+
det_repo: str = "PaddlePaddle/PP-OCRv6_medium_det_onnx",
|
|
177
|
+
rec_repo: str = "PaddlePaddle/PP-OCRv6_medium_rec_onnx",
|
|
178
|
+
local_dir: Path | None = None,
|
|
179
|
+
force: bool = False, # noqa: FBT001, FBT002
|
|
180
|
+
) -> Path:
|
|
181
|
+
"""Download the PP-OCRv6 detection and recognition ONNX models from HuggingFace.
|
|
182
|
+
|
|
183
|
+
Returns the local directory containing ``det/`` and ``rec/`` sub-folders.
|
|
184
|
+
Pre-fetching at image-build time avoids blocking the first request on the
|
|
185
|
+
download. Pass ``force=True`` to re-download even when a cached copy exists
|
|
186
|
+
(e.g. to repair a corrupted file).
|
|
187
|
+
"""
|
|
188
|
+
from huggingface_hub import hf_hub_download # noqa: PLC0415
|
|
189
|
+
|
|
190
|
+
if local_dir is None:
|
|
191
|
+
local_dir = settings.cache_dir / "models" / PPOCRv6Model._model_repo_folder
|
|
192
|
+
|
|
193
|
+
det_dir = local_dir / "det"
|
|
194
|
+
rec_dir = local_dir / "rec"
|
|
195
|
+
det_dir.mkdir(parents=True, exist_ok=True)
|
|
196
|
+
rec_dir.mkdir(parents=True, exist_ok=True)
|
|
197
|
+
|
|
198
|
+
hf_hub_download(det_repo, _ONNX_FILE, local_dir=det_dir, force_download=force)
|
|
199
|
+
hf_hub_download(rec_repo, _ONNX_FILE, local_dir=rec_dir, force_download=force)
|
|
200
|
+
hf_hub_download(rec_repo, _CONFIG_FILE, local_dir=rec_dir, force_download=force)
|
|
201
|
+
|
|
202
|
+
return local_dir
|
|
203
|
+
|
|
204
|
+
def __call__(self, conv_res: ConversionResult, page_batch: Iterable[Page]) -> Iterable[Page]:
|
|
205
|
+
"""Run OCR on each page crop and yield pages with recognised text cells."""
|
|
206
|
+
if not self.enabled:
|
|
207
|
+
yield from page_batch
|
|
208
|
+
return
|
|
209
|
+
|
|
210
|
+
for page in page_batch:
|
|
211
|
+
if page._backend is None or not page._backend.is_valid(): # noqa: SLF001
|
|
212
|
+
yield page
|
|
213
|
+
continue
|
|
214
|
+
|
|
215
|
+
with TimeRecorder(conv_res, "ocr"):
|
|
216
|
+
ocr_rects = self.get_ocr_rects(page)
|
|
217
|
+
all_ocr_cells: list[TextCell] = []
|
|
218
|
+
cell_idx = 0
|
|
219
|
+
|
|
220
|
+
for ocr_rect in ocr_rects:
|
|
221
|
+
if ocr_rect.area() == 0:
|
|
222
|
+
continue
|
|
223
|
+
high_res_image = page._backend.get_page_image(scale=self.scale, cropbox=ocr_rect) # noqa: SLF001
|
|
224
|
+
im = np.array(high_res_image)
|
|
225
|
+
# cast: RapidOCR's return type is a union of stage-specific outputs
|
|
226
|
+
result = cast(
|
|
227
|
+
"Any",
|
|
228
|
+
self.reader(
|
|
229
|
+
im,
|
|
230
|
+
use_det=self.options.use_det,
|
|
231
|
+
use_cls=self.options.use_cls,
|
|
232
|
+
use_rec=self.options.use_rec,
|
|
233
|
+
),
|
|
234
|
+
)
|
|
235
|
+
# a disabled stage means the matching attribute is absent
|
|
236
|
+
boxes = getattr(result, "boxes", None) if result is not None else None
|
|
237
|
+
txts = getattr(result, "txts", None)
|
|
238
|
+
scores = getattr(result, "scores", None)
|
|
239
|
+
if boxes is None or txts is None or scores is None:
|
|
240
|
+
continue
|
|
241
|
+
|
|
242
|
+
for box, text, score in zip(boxes.tolist(), txts, scores, strict=False):
|
|
243
|
+
all_ocr_cells.append(
|
|
244
|
+
TextCell(
|
|
245
|
+
index=cell_idx,
|
|
246
|
+
text=text,
|
|
247
|
+
orig=text,
|
|
248
|
+
confidence=score,
|
|
249
|
+
from_ocr=True,
|
|
250
|
+
rect=BoundingRectangle.from_bounding_box(
|
|
251
|
+
BoundingBox.from_tuple(
|
|
252
|
+
coord=(
|
|
253
|
+
(box[0][0] / self.scale) + ocr_rect.l,
|
|
254
|
+
(box[0][1] / self.scale) + ocr_rect.t,
|
|
255
|
+
(box[2][0] / self.scale) + ocr_rect.l,
|
|
256
|
+
(box[2][1] / self.scale) + ocr_rect.t,
|
|
257
|
+
),
|
|
258
|
+
origin=CoordOrigin.TOPLEFT,
|
|
259
|
+
)
|
|
260
|
+
),
|
|
261
|
+
)
|
|
262
|
+
)
|
|
263
|
+
cell_idx += 1
|
|
264
|
+
|
|
265
|
+
self.post_process_cells(all_ocr_cells, page)
|
|
266
|
+
|
|
267
|
+
if settings.debug.visualize_ocr:
|
|
268
|
+
self.draw_ocr_rects_and_cells(conv_res, page, ocr_rects)
|
|
269
|
+
|
|
270
|
+
yield page
|
|
271
|
+
|
|
272
|
+
@classmethod
|
|
273
|
+
def get_options_type(cls) -> type[OcrOptions]:
|
|
274
|
+
"""Return the options class for this OCR engine."""
|
|
275
|
+
return PPOCRv6Options
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Configuration model for the PP-OCRv6 OCR engine."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
from typing import ClassVar, Literal
|
|
7
|
+
|
|
8
|
+
from docling.datamodel.pipeline_options import OcrOptions
|
|
9
|
+
from pydantic import ConfigDict, Field
|
|
10
|
+
|
|
11
|
+
_DEFAULT_DET_REPO = "PaddlePaddle/PP-OCRv6_medium_det_onnx"
|
|
12
|
+
_DEFAULT_REC_REPO = "PaddlePaddle/PP-OCRv6_medium_rec_onnx"
|
|
13
|
+
|
|
14
|
+
# PP-OCRv6 medium is a single multilingual model covering Latin-script
|
|
15
|
+
# European languages (and many more); ``lang`` is only a passthrough hint to
|
|
16
|
+
# docling and does not switch models. The default leads with German and
|
|
17
|
+
# includes the common European languages.
|
|
18
|
+
_DEFAULT_LANGS = "de,en,fr,it,es,nl,pt,pl,sv,da,fi,nb,cs,ro,hu"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _env_bool(name: str, default: bool) -> bool: # noqa: FBT001
|
|
22
|
+
raw = os.environ.get(name)
|
|
23
|
+
if raw is None:
|
|
24
|
+
return default
|
|
25
|
+
return raw.strip().lower() in {"1", "true", "yes", "on"}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class PPOCRv6Options(OcrOptions):
|
|
29
|
+
"""Options for the PP-OCRv6 OCR engine.
|
|
30
|
+
|
|
31
|
+
The engine runs PaddlePaddle's PP-OCRv6 detection and recognition ONNX
|
|
32
|
+
models locally through RapidOCR (onnxruntime). Models are downloaded from
|
|
33
|
+
HuggingFace on first use and cached. GPU is used automatically when the
|
|
34
|
+
docling accelerator device resolves to CUDA and ``onnxruntime-gpu`` is
|
|
35
|
+
installed.
|
|
36
|
+
|
|
37
|
+
All options fall back to environment variables when not set explicitly,
|
|
38
|
+
allowing configuration without code changes (e.g. in Docker / Compose
|
|
39
|
+
deployments).
|
|
40
|
+
|
|
41
|
+
Attributes:
|
|
42
|
+
lang: Language hint list (passed to docling). PP-OCRv6 medium is a
|
|
43
|
+
single multilingual model, so this does not switch models; it
|
|
44
|
+
recognises its supported languages automatically. Defaults to a
|
|
45
|
+
German-led set of common European languages. Falls back to the
|
|
46
|
+
``PPOCRV6_LANG`` env var as a comma-separated string.
|
|
47
|
+
text_score: Minimum recognition confidence; lower-scoring cells are
|
|
48
|
+
dropped by RapidOCR. Falls back to ``PPOCRV6_TEXT_SCORE``.
|
|
49
|
+
use_det: Run the text-detection stage. Falls back to ``PPOCRV6_USE_DET``.
|
|
50
|
+
use_cls: Run the angle-classification stage (RapidOCR's bundled cls
|
|
51
|
+
model). Falls back to ``PPOCRV6_USE_CLS``.
|
|
52
|
+
use_rec: Run the text-recognition stage. Falls back to
|
|
53
|
+
``PPOCRV6_USE_REC``.
|
|
54
|
+
det_repo: HuggingFace repo id for the detection ONNX model.
|
|
55
|
+
Falls back to ``PPOCRV6_DET_REPO``.
|
|
56
|
+
rec_repo: HuggingFace repo id for the recognition ONNX model.
|
|
57
|
+
Falls back to ``PPOCRV6_REC_REPO``.
|
|
58
|
+
det_model_path: Explicit local path to a detection ONNX file. When set,
|
|
59
|
+
no detection model is downloaded. Falls back to ``PPOCRV6_DET_MODEL_PATH``.
|
|
60
|
+
rec_model_path: Explicit local path to a recognition ONNX file. When set,
|
|
61
|
+
no recognition model is downloaded. Falls back to ``PPOCRV6_REC_MODEL_PATH``.
|
|
62
|
+
rec_keys_path: Explicit local path to the recognition character
|
|
63
|
+
dictionary (one character per line). When unset it is extracted
|
|
64
|
+
from the recognition model's ``inference.yml``. Falls back to
|
|
65
|
+
``PPOCRV6_REC_KEYS_PATH``.
|
|
66
|
+
cls_model_path: Explicit local path to an angle-classification ONNX
|
|
67
|
+
file. When unset, RapidOCR's bundled cls model is used. Falls back
|
|
68
|
+
to ``PPOCRV6_CLS_MODEL_PATH``.
|
|
69
|
+
rapidocr_params: Extra RapidOCR ``params`` overrides merged on top of
|
|
70
|
+
the engine defaults.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
kind: ClassVar[Literal["pp-ocrv6"]] = "pp-ocrv6"
|
|
74
|
+
|
|
75
|
+
lang: list[str] = Field(default_factory=lambda: os.environ.get("PPOCRV6_LANG", _DEFAULT_LANGS).split(","))
|
|
76
|
+
text_score: float = Field(default_factory=lambda: float(os.environ.get("PPOCRV6_TEXT_SCORE", "0.5")))
|
|
77
|
+
use_det: bool = Field(default_factory=lambda: _env_bool("PPOCRV6_USE_DET", default=True))
|
|
78
|
+
use_cls: bool = Field(default_factory=lambda: _env_bool("PPOCRV6_USE_CLS", default=True))
|
|
79
|
+
use_rec: bool = Field(default_factory=lambda: _env_bool("PPOCRV6_USE_REC", default=True))
|
|
80
|
+
|
|
81
|
+
det_repo: str = Field(default_factory=lambda: os.environ.get("PPOCRV6_DET_REPO", _DEFAULT_DET_REPO))
|
|
82
|
+
rec_repo: str = Field(default_factory=lambda: os.environ.get("PPOCRV6_REC_REPO", _DEFAULT_REC_REPO))
|
|
83
|
+
|
|
84
|
+
det_model_path: str | None = Field(default_factory=lambda: os.environ.get("PPOCRV6_DET_MODEL_PATH") or None)
|
|
85
|
+
rec_model_path: str | None = Field(default_factory=lambda: os.environ.get("PPOCRV6_REC_MODEL_PATH") or None)
|
|
86
|
+
rec_keys_path: str | None = Field(default_factory=lambda: os.environ.get("PPOCRV6_REC_KEYS_PATH") or None)
|
|
87
|
+
cls_model_path: str | None = Field(default_factory=lambda: os.environ.get("PPOCRV6_CLS_MODEL_PATH") or None)
|
|
88
|
+
|
|
89
|
+
rapidocr_params: dict = Field(default_factory=dict)
|
|
90
|
+
|
|
91
|
+
model_config = ConfigDict(extra="forbid")
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""Docling plugin entry point registering the PP-OCRv6 OCR engine."""
|
|
2
|
+
|
|
3
|
+
from docling_pp_ocrv6.model import PPOCRv6Model
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def ocr_engines() -> dict[str, list[type[PPOCRv6Model]]]:
|
|
7
|
+
"""Return the OCR engine classes provided by this plugin."""
|
|
8
|
+
return {"ocr_engines": [PPOCRv6Model]}
|
|
File without changes
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: docling-pp-ocrv6
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A docling OCR plugin for PaddlePaddle PP-OCRv6 (ONNX via RapidOCR)
|
|
5
|
+
Project-URL: Homepage, https://github.com/DCC-BS/docling-pp-ocrv6
|
|
6
|
+
Project-URL: Repository, https://github.com/DCC-BS/docling-pp-ocrv6
|
|
7
|
+
Project-URL: Issues, https://github.com/DCC-BS/docling-pp-ocrv6/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/DCC-BS/docling-pp-ocrv6/releases
|
|
9
|
+
Author-email: Yanick Schraner <yanick.schraner@bs.ch>, Tobias Bollinger <tobias.bollinger@bs.ch>
|
|
10
|
+
License: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Typing :: Typed
|
|
18
|
+
Requires-Python: >=3.12
|
|
19
|
+
Requires-Dist: docling>=2.73
|
|
20
|
+
Requires-Dist: huggingface-hub>=1.3.0
|
|
21
|
+
Requires-Dist: numpy
|
|
22
|
+
Requires-Dist: pillow>=10.0
|
|
23
|
+
Requires-Dist: pyyaml>=6.0
|
|
24
|
+
Requires-Dist: rapidocr>=3.0
|
|
25
|
+
Provides-Extra: cpu
|
|
26
|
+
Requires-Dist: onnxruntime>=1.17; extra == 'cpu'
|
|
27
|
+
Provides-Extra: gpu
|
|
28
|
+
Requires-Dist: onnxruntime-gpu>=1.17; extra == 'gpu'
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
|
|
31
|
+
# docling-pp-ocrv6
|
|
32
|
+
|
|
33
|
+
A [Docling](https://github.com/docling-project/docling) OCR plugin for
|
|
34
|
+
PaddlePaddle's **PP-OCRv6** models. It runs the PP-OCRv6 detection and
|
|
35
|
+
recognition **ONNX** checkpoints locally through
|
|
36
|
+
[RapidOCR](https://github.com/RapidAI/RapidOCR) (onnxruntime), so OCR happens
|
|
37
|
+
inside the docling worker — no external service required.
|
|
38
|
+
|
|
39
|
+
GPU acceleration is automatic when the docling accelerator device resolves to
|
|
40
|
+
CUDA and `onnxruntime-gpu` is installed.
|
|
41
|
+
|
|
42
|
+
## Installation
|
|
43
|
+
|
|
44
|
+
Pick exactly one onnxruntime extra — installing both the CPU and GPU wheels at
|
|
45
|
+
once is unsupported and prevents the CUDA provider from registering:
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
pip install "docling-pp-ocrv6[cpu]" # CPU (onnxruntime)
|
|
49
|
+
pip install "docling-pp-ocrv6[gpu]" # CUDA (onnxruntime-gpu)
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
The detection and recognition ONNX models are downloaded from HuggingFace on
|
|
53
|
+
first use and cached under docling's model cache. To pre-fetch them (e.g. at
|
|
54
|
+
container build time):
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
from docling_pp_ocrv6 import PPOCRv6Model
|
|
58
|
+
PPOCRv6Model.download_models()
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## Usage
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
from docling.document_converter import DocumentConverter, PdfFormatOption
|
|
65
|
+
from docling.datamodel.base_models import InputFormat
|
|
66
|
+
from docling.datamodel.pipeline_options import PdfPipelineOptions
|
|
67
|
+
from docling_pp_ocrv6 import PPOCRv6Options
|
|
68
|
+
|
|
69
|
+
pipeline_options = PdfPipelineOptions(do_ocr=True)
|
|
70
|
+
pipeline_options.ocr_options = PPOCRv6Options()
|
|
71
|
+
|
|
72
|
+
converter = DocumentConverter(
|
|
73
|
+
format_options={InputFormat.PDF: PdfFormatOption(pipeline_options=pipeline_options)}
|
|
74
|
+
)
|
|
75
|
+
result = converter.convert("scanned.pdf")
|
|
76
|
+
print(result.document.export_to_markdown())
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
With **docling-serve**, request the engine by its `kind`:
|
|
80
|
+
|
|
81
|
+
```json
|
|
82
|
+
{ "options": { "ocr": true, "ocr_engine": "pp-ocrv6" } }
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
(`DOCLING_SERVE_ALLOW_EXTERNAL_PLUGINS=true` must be set for the plugin to load.)
|
|
86
|
+
|
|
87
|
+
## Configuration
|
|
88
|
+
|
|
89
|
+
All options are settable via `PPOCRv6Options(...)` or environment variables:
|
|
90
|
+
|
|
91
|
+
| Option | Env var | Default |
|
|
92
|
+
| --- | --- | --- |
|
|
93
|
+
| `lang` | `PPOCRV6_LANG` | `de,en,fr,it,es,nl,pt,...` (German-led European set) |
|
|
94
|
+
| `text_score` | `PPOCRV6_TEXT_SCORE` | `0.5` |
|
|
95
|
+
| `use_det` / `use_cls` / `use_rec` | `PPOCRV6_USE_DET` / `_CLS` / `_REC` | `true` |
|
|
96
|
+
| `det_repo` | `PPOCRV6_DET_REPO` | `PaddlePaddle/PP-OCRv6_medium_det_onnx` |
|
|
97
|
+
| `rec_repo` | `PPOCRV6_REC_REPO` | `PaddlePaddle/PP-OCRv6_medium_rec_onnx` |
|
|
98
|
+
| `det_model_path` / `rec_model_path` / `rec_keys_path` / `cls_model_path` | `PPOCRV6_*_MODEL_PATH` / `_KEYS_PATH` | auto |
|
|
99
|
+
|
|
100
|
+
The recognition character dictionary is extracted automatically from the
|
|
101
|
+
recognition model's `inference.yml`. Angle classification uses RapidOCR's
|
|
102
|
+
bundled cls model unless `cls_model_path` is set.
|
|
103
|
+
|
|
104
|
+
## Development
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
make install # uv sync + pre-commit
|
|
108
|
+
make check # ruff lint + format check + ty type check
|
|
109
|
+
make test # pytest with coverage
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
## License
|
|
113
|
+
|
|
114
|
+
MIT
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
docling_pp_ocrv6/__init__.py,sha256=o_T8rpqV13O-qtFZussbl34WXDt1tgCbQONMtzeH7is,224
|
|
2
|
+
docling_pp_ocrv6/model.py,sha256=Ra4s-FucBduaChpOaeH86--Vf9BmUqIyedwr2e1bPj8,11418
|
|
3
|
+
docling_pp_ocrv6/options.py,sha256=ZKLXIxIxz-GjzsvHaS8H9DGY_s2V00fUY4oK2RMIkkI,4767
|
|
4
|
+
docling_pp_ocrv6/plugin.py,sha256=x0u-TlioVNeAg6GRy13vajLyHbfCN8ltc21rBDpRrTQ,287
|
|
5
|
+
docling_pp_ocrv6/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
6
|
+
docling_pp_ocrv6-0.1.0.dist-info/METADATA,sha256=RvAj8fCMzXk8Ulo9CkXl5btDPY8ufn-7o2OkTsB3iLE,4049
|
|
7
|
+
docling_pp_ocrv6-0.1.0.dist-info/WHEEL,sha256=mffPy8wBnZQn2VnJUU5jE99KsxaSfiyMHV9Yt0aLVxs,87
|
|
8
|
+
docling_pp_ocrv6-0.1.0.dist-info/entry_points.txt,sha256=my-zYx4Xr5Z68incYws4EyF9zMXDnC5wbnnFiUHuweU,53
|
|
9
|
+
docling_pp_ocrv6-0.1.0.dist-info/licenses/LICENSE,sha256=APpdFXym3_5C78pFAfnlLvmP7umPBWYpQ2MKidnoRuw,1091
|
|
10
|
+
docling_pp_ocrv6-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Data Competence Center Basel-Stadt
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|