docling-pp-ocrv6 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,8 @@
1
+ """A docling OCR plugin for PaddlePaddle PP-OCRv6."""
2
+
3
+ __version__ = "0.1.0"
4
+
5
+ from docling_pp_ocrv6.model import PPOCRv6Model
6
+ from docling_pp_ocrv6.options import PPOCRv6Options
7
+
8
+ __all__ = ["PPOCRv6Model", "PPOCRv6Options"]
@@ -0,0 +1,275 @@
1
+ """PP-OCRv6 OCR engine for the docling standard pipeline.
2
+
3
+ Runs PaddlePaddle PP-OCRv6 detection and recognition ONNX models locally via
4
+ RapidOCR (onnxruntime) and returns the recognised text as ``TextCell`` objects
5
+ that docling merges with its standard-pipeline output.
6
+
7
+ The detection and recognition ONNX models are downloaded from HuggingFace on
8
+ first use and cached under docling's model cache. The recognition character
9
+ dictionary is extracted from the recognition model's ``inference.yml``. Angle
10
+ classification uses RapidOCR's bundled cls model unless an explicit path is
11
+ given.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import logging
17
+ from pathlib import Path
18
+ from typing import TYPE_CHECKING, Any, cast
19
+
20
+ import numpy as np
21
+ import yaml
22
+ from docling.datamodel.accelerator_options import AcceleratorDevice
23
+ from docling.datamodel.settings import settings
24
+ from docling.models.base_ocr_model import BaseOcrModel
25
+ from docling.utils.accelerator_utils import decide_device
26
+ from docling.utils.profiling import TimeRecorder
27
+ from docling_core.types.doc import BoundingBox, CoordOrigin
28
+ from docling_core.types.doc.page import BoundingRectangle, TextCell
29
+
30
+ from docling_pp_ocrv6.options import PPOCRv6Options
31
+
32
+ if TYPE_CHECKING:
33
+ from collections.abc import Iterable
34
+
35
+ from docling.datamodel.accelerator_options import AcceleratorOptions
36
+ from docling.datamodel.base_models import Page
37
+ from docling.datamodel.document import ConversionResult
38
+ from docling.datamodel.pipeline_options import OcrOptions
39
+
40
+ logger = logging.getLogger(__name__)
41
+
42
+ _ONNX_FILE = "inference.onnx"
43
+ _CONFIG_FILE = "inference.yml"
44
+ _REC_KEYS_FILE = "ppocrv6_keys.txt"
45
+
46
+
47
+ class PPOCRv6Model(BaseOcrModel):
48
+ """OCR engine running PP-OCRv6 ONNX models through RapidOCR."""
49
+
50
+ _model_repo_folder = "PPOCRv6"
51
+
52
+ def __init__(
53
+ self,
54
+ enabled: bool, # noqa: FBT001
55
+ artifacts_path: Path | None,
56
+ options: PPOCRv6Options,
57
+ accelerator_options: AcceleratorOptions,
58
+ ) -> None:
59
+ """Initialise the OCR engine, downloading models on first use when enabled."""
60
+ super().__init__(
61
+ enabled=enabled,
62
+ artifacts_path=artifacts_path,
63
+ options=options,
64
+ accelerator_options=accelerator_options,
65
+ )
66
+ self.options: PPOCRv6Options
67
+ self.scale = 3 # multiplier for 72 dpi == 216 dpi.
68
+
69
+ if not self.enabled:
70
+ return
71
+
72
+ try:
73
+ from rapidocr import EngineType, RapidOCR # noqa: PLC0415
74
+ except ImportError as err:
75
+ msg = (
76
+ "RapidOCR is not installed. Install it via "
77
+ "`pip install rapidocr onnxruntime` (or `onnxruntime-gpu` for CUDA) "
78
+ "to use the PP-OCRv6 OCR engine."
79
+ )
80
+ raise ImportError(msg) from err
81
+
82
+ device = decide_device(accelerator_options.device)
83
+ use_cuda = str(AcceleratorDevice.CUDA.value).lower() in device
84
+ use_dml = accelerator_options.device == AcceleratorDevice.AUTO
85
+ gpu_id = int(device.split(":")[1]) if (use_cuda and ":" in device) else 0
86
+
87
+ det_path, rec_path, rec_keys_path, cls_path = self._resolve_models()
88
+ logger.info(
89
+ "Loading PP-OCRv6 (device=%s, cuda=%s): det=%s rec=%s",
90
+ device,
91
+ use_cuda,
92
+ det_path,
93
+ rec_path,
94
+ )
95
+
96
+ params: dict = {
97
+ "Global.text_score": self.options.text_score,
98
+ "EngineConfig.onnxruntime.intra_op_num_threads": accelerator_options.num_threads,
99
+ "Det.model_path": str(det_path),
100
+ "Det.engine_type": EngineType.ONNXRUNTIME,
101
+ "Det.use_cuda": use_cuda,
102
+ "Det.use_dml": use_dml,
103
+ "Rec.model_path": str(rec_path),
104
+ "Rec.rec_keys_path": str(rec_keys_path),
105
+ "Rec.engine_type": EngineType.ONNXRUNTIME,
106
+ "Rec.use_cuda": use_cuda,
107
+ "Rec.use_dml": use_dml,
108
+ "Cls.engine_type": EngineType.ONNXRUNTIME,
109
+ "Cls.use_cuda": use_cuda,
110
+ "Cls.use_dml": use_dml,
111
+ "EngineConfig.onnxruntime.use_cuda": use_cuda,
112
+ "EngineConfig.onnxruntime.cuda_ep_cfg.device_id": gpu_id,
113
+ }
114
+ # When no explicit cls model is given, let RapidOCR use its bundled cls model.
115
+ if cls_path is not None:
116
+ params["Cls.model_path"] = str(cls_path)
117
+
118
+ if self.options.rapidocr_params:
119
+ params.update(self.options.rapidocr_params)
120
+
121
+ self.reader = RapidOCR(params=params)
122
+
123
+ def _resolve_models(self) -> tuple[Path, Path, Path, Path | None]:
124
+ """Resolve detection, recognition, rec-keys and (optional) cls model paths.
125
+
126
+ Explicit option paths win; otherwise models are downloaded from
127
+ HuggingFace and cached. The recognition character dictionary is
128
+ extracted from the recognition model's ``inference.yml`` when not
129
+ provided explicitly.
130
+ """
131
+ local_dir = self.download_models(
132
+ det_repo=self.options.det_repo,
133
+ rec_repo=self.options.rec_repo,
134
+ )
135
+
136
+ det_path = Path(self.options.det_model_path) if self.options.det_model_path else local_dir / "det" / _ONNX_FILE
137
+ rec_path = Path(self.options.rec_model_path) if self.options.rec_model_path else local_dir / "rec" / _ONNX_FILE
138
+
139
+ if self.options.rec_keys_path:
140
+ rec_keys_path = Path(self.options.rec_keys_path)
141
+ else:
142
+ rec_keys_path = self._ensure_rec_keys(local_dir / "rec")
143
+
144
+ cls_path = Path(self.options.cls_model_path) if self.options.cls_model_path else None
145
+
146
+ for path in (det_path, rec_path, rec_keys_path):
147
+ if not path.exists():
148
+ logger.warning("PP-OCRv6 model path does not exist: %s", path)
149
+
150
+ return det_path, rec_path, rec_keys_path, cls_path
151
+
152
+ @staticmethod
153
+ def _ensure_rec_keys(rec_dir: Path) -> Path:
154
+ """Extract the recognition character dictionary into a RapidOCR keys file.
155
+
156
+ PaddleX exports embed the dictionary as ``PostProcess.character_dict``
157
+ inside ``inference.yml``; RapidOCR expects a plain text file with one
158
+ character per line.
159
+ """
160
+ keys_path = rec_dir / _REC_KEYS_FILE
161
+ if keys_path.exists():
162
+ return keys_path
163
+
164
+ config = yaml.safe_load((rec_dir / _CONFIG_FILE).read_text(encoding="utf-8"))
165
+ chars = config.get("PostProcess", {}).get("character_dict")
166
+ if not chars:
167
+ msg = f"No 'PostProcess.character_dict' found in {rec_dir / _CONFIG_FILE}"
168
+ raise ValueError(msg)
169
+
170
+ keys_path.write_text("\n".join(chars) + "\n", encoding="utf-8")
171
+ logger.info("Wrote PP-OCRv6 recognition dictionary (%d entries) to %s", len(chars), keys_path)
172
+ return keys_path
173
+
174
+ @staticmethod
175
+ def download_models(
176
+ det_repo: str = "PaddlePaddle/PP-OCRv6_medium_det_onnx",
177
+ rec_repo: str = "PaddlePaddle/PP-OCRv6_medium_rec_onnx",
178
+ local_dir: Path | None = None,
179
+ force: bool = False, # noqa: FBT001, FBT002
180
+ ) -> Path:
181
+ """Download the PP-OCRv6 detection and recognition ONNX models from HuggingFace.
182
+
183
+ Returns the local directory containing ``det/`` and ``rec/`` sub-folders.
184
+ Pre-fetching at image-build time avoids blocking the first request on the
185
+ download. Pass ``force=True`` to re-download even when a cached copy exists
186
+ (e.g. to repair a corrupted file).
187
+ """
188
+ from huggingface_hub import hf_hub_download # noqa: PLC0415
189
+
190
+ if local_dir is None:
191
+ local_dir = settings.cache_dir / "models" / PPOCRv6Model._model_repo_folder
192
+
193
+ det_dir = local_dir / "det"
194
+ rec_dir = local_dir / "rec"
195
+ det_dir.mkdir(parents=True, exist_ok=True)
196
+ rec_dir.mkdir(parents=True, exist_ok=True)
197
+
198
+ hf_hub_download(det_repo, _ONNX_FILE, local_dir=det_dir, force_download=force)
199
+ hf_hub_download(rec_repo, _ONNX_FILE, local_dir=rec_dir, force_download=force)
200
+ hf_hub_download(rec_repo, _CONFIG_FILE, local_dir=rec_dir, force_download=force)
201
+
202
+ return local_dir
203
+
204
+ def __call__(self, conv_res: ConversionResult, page_batch: Iterable[Page]) -> Iterable[Page]:
205
+ """Run OCR on each page crop and yield pages with recognised text cells."""
206
+ if not self.enabled:
207
+ yield from page_batch
208
+ return
209
+
210
+ for page in page_batch:
211
+ if page._backend is None or not page._backend.is_valid(): # noqa: SLF001
212
+ yield page
213
+ continue
214
+
215
+ with TimeRecorder(conv_res, "ocr"):
216
+ ocr_rects = self.get_ocr_rects(page)
217
+ all_ocr_cells: list[TextCell] = []
218
+ cell_idx = 0
219
+
220
+ for ocr_rect in ocr_rects:
221
+ if ocr_rect.area() == 0:
222
+ continue
223
+ high_res_image = page._backend.get_page_image(scale=self.scale, cropbox=ocr_rect) # noqa: SLF001
224
+ im = np.array(high_res_image)
225
+ # cast: RapidOCR's return type is a union of stage-specific outputs
226
+ result = cast(
227
+ "Any",
228
+ self.reader(
229
+ im,
230
+ use_det=self.options.use_det,
231
+ use_cls=self.options.use_cls,
232
+ use_rec=self.options.use_rec,
233
+ ),
234
+ )
235
+ # a disabled stage means the matching attribute is absent
236
+ boxes = getattr(result, "boxes", None) if result is not None else None
237
+ txts = getattr(result, "txts", None)
238
+ scores = getattr(result, "scores", None)
239
+ if boxes is None or txts is None or scores is None:
240
+ continue
241
+
242
+ for box, text, score in zip(boxes.tolist(), txts, scores, strict=False):
243
+ all_ocr_cells.append(
244
+ TextCell(
245
+ index=cell_idx,
246
+ text=text,
247
+ orig=text,
248
+ confidence=score,
249
+ from_ocr=True,
250
+ rect=BoundingRectangle.from_bounding_box(
251
+ BoundingBox.from_tuple(
252
+ coord=(
253
+ (box[0][0] / self.scale) + ocr_rect.l,
254
+ (box[0][1] / self.scale) + ocr_rect.t,
255
+ (box[2][0] / self.scale) + ocr_rect.l,
256
+ (box[2][1] / self.scale) + ocr_rect.t,
257
+ ),
258
+ origin=CoordOrigin.TOPLEFT,
259
+ )
260
+ ),
261
+ )
262
+ )
263
+ cell_idx += 1
264
+
265
+ self.post_process_cells(all_ocr_cells, page)
266
+
267
+ if settings.debug.visualize_ocr:
268
+ self.draw_ocr_rects_and_cells(conv_res, page, ocr_rects)
269
+
270
+ yield page
271
+
272
+ @classmethod
273
+ def get_options_type(cls) -> type[OcrOptions]:
274
+ """Return the options class for this OCR engine."""
275
+ return PPOCRv6Options
@@ -0,0 +1,91 @@
1
+ """Configuration model for the PP-OCRv6 OCR engine."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ from typing import ClassVar, Literal
7
+
8
+ from docling.datamodel.pipeline_options import OcrOptions
9
+ from pydantic import ConfigDict, Field
10
+
11
+ _DEFAULT_DET_REPO = "PaddlePaddle/PP-OCRv6_medium_det_onnx"
12
+ _DEFAULT_REC_REPO = "PaddlePaddle/PP-OCRv6_medium_rec_onnx"
13
+
14
+ # PP-OCRv6 medium is a single multilingual model covering Latin-script
15
+ # European languages (and many more); ``lang`` is only a passthrough hint to
16
+ # docling and does not switch models. The default leads with German and
17
+ # includes the common European languages.
18
+ _DEFAULT_LANGS = "de,en,fr,it,es,nl,pt,pl,sv,da,fi,nb,cs,ro,hu"
19
+
20
+
21
+ def _env_bool(name: str, default: bool) -> bool: # noqa: FBT001
22
+ raw = os.environ.get(name)
23
+ if raw is None:
24
+ return default
25
+ return raw.strip().lower() in {"1", "true", "yes", "on"}
26
+
27
+
28
+ class PPOCRv6Options(OcrOptions):
29
+ """Options for the PP-OCRv6 OCR engine.
30
+
31
+ The engine runs PaddlePaddle's PP-OCRv6 detection and recognition ONNX
32
+ models locally through RapidOCR (onnxruntime). Models are downloaded from
33
+ HuggingFace on first use and cached. GPU is used automatically when the
34
+ docling accelerator device resolves to CUDA and ``onnxruntime-gpu`` is
35
+ installed.
36
+
37
+ All options fall back to environment variables when not set explicitly,
38
+ allowing configuration without code changes (e.g. in Docker / Compose
39
+ deployments).
40
+
41
+ Attributes:
42
+ lang: Language hint list (passed to docling). PP-OCRv6 medium is a
43
+ single multilingual model, so this does not switch models; it
44
+ recognises its supported languages automatically. Defaults to a
45
+ German-led set of common European languages. Falls back to the
46
+ ``PPOCRV6_LANG`` env var as a comma-separated string.
47
+ text_score: Minimum recognition confidence; lower-scoring cells are
48
+ dropped by RapidOCR. Falls back to ``PPOCRV6_TEXT_SCORE``.
49
+ use_det: Run the text-detection stage. Falls back to ``PPOCRV6_USE_DET``.
50
+ use_cls: Run the angle-classification stage (RapidOCR's bundled cls
51
+ model). Falls back to ``PPOCRV6_USE_CLS``.
52
+ use_rec: Run the text-recognition stage. Falls back to
53
+ ``PPOCRV6_USE_REC``.
54
+ det_repo: HuggingFace repo id for the detection ONNX model.
55
+ Falls back to ``PPOCRV6_DET_REPO``.
56
+ rec_repo: HuggingFace repo id for the recognition ONNX model.
57
+ Falls back to ``PPOCRV6_REC_REPO``.
58
+ det_model_path: Explicit local path to a detection ONNX file. When set,
59
+ no detection model is downloaded. Falls back to ``PPOCRV6_DET_MODEL_PATH``.
60
+ rec_model_path: Explicit local path to a recognition ONNX file. When set,
61
+ no recognition model is downloaded. Falls back to ``PPOCRV6_REC_MODEL_PATH``.
62
+ rec_keys_path: Explicit local path to the recognition character
63
+ dictionary (one character per line). When unset it is extracted
64
+ from the recognition model's ``inference.yml``. Falls back to
65
+ ``PPOCRV6_REC_KEYS_PATH``.
66
+ cls_model_path: Explicit local path to an angle-classification ONNX
67
+ file. When unset, RapidOCR's bundled cls model is used. Falls back
68
+ to ``PPOCRV6_CLS_MODEL_PATH``.
69
+ rapidocr_params: Extra RapidOCR ``params`` overrides merged on top of
70
+ the engine defaults.
71
+ """
72
+
73
+ kind: ClassVar[Literal["pp-ocrv6"]] = "pp-ocrv6"
74
+
75
+ lang: list[str] = Field(default_factory=lambda: os.environ.get("PPOCRV6_LANG", _DEFAULT_LANGS).split(","))
76
+ text_score: float = Field(default_factory=lambda: float(os.environ.get("PPOCRV6_TEXT_SCORE", "0.5")))
77
+ use_det: bool = Field(default_factory=lambda: _env_bool("PPOCRV6_USE_DET", default=True))
78
+ use_cls: bool = Field(default_factory=lambda: _env_bool("PPOCRV6_USE_CLS", default=True))
79
+ use_rec: bool = Field(default_factory=lambda: _env_bool("PPOCRV6_USE_REC", default=True))
80
+
81
+ det_repo: str = Field(default_factory=lambda: os.environ.get("PPOCRV6_DET_REPO", _DEFAULT_DET_REPO))
82
+ rec_repo: str = Field(default_factory=lambda: os.environ.get("PPOCRV6_REC_REPO", _DEFAULT_REC_REPO))
83
+
84
+ det_model_path: str | None = Field(default_factory=lambda: os.environ.get("PPOCRV6_DET_MODEL_PATH") or None)
85
+ rec_model_path: str | None = Field(default_factory=lambda: os.environ.get("PPOCRV6_REC_MODEL_PATH") or None)
86
+ rec_keys_path: str | None = Field(default_factory=lambda: os.environ.get("PPOCRV6_REC_KEYS_PATH") or None)
87
+ cls_model_path: str | None = Field(default_factory=lambda: os.environ.get("PPOCRV6_CLS_MODEL_PATH") or None)
88
+
89
+ rapidocr_params: dict = Field(default_factory=dict)
90
+
91
+ model_config = ConfigDict(extra="forbid")
@@ -0,0 +1,8 @@
1
+ """Docling plugin entry point registering the PP-OCRv6 OCR engine."""
2
+
3
+ from docling_pp_ocrv6.model import PPOCRv6Model
4
+
5
+
6
+ def ocr_engines() -> dict[str, list[type[PPOCRv6Model]]]:
7
+ """Return the OCR engine classes provided by this plugin."""
8
+ return {"ocr_engines": [PPOCRv6Model]}
File without changes
@@ -0,0 +1,114 @@
1
+ Metadata-Version: 2.4
2
+ Name: docling-pp-ocrv6
3
+ Version: 0.1.0
4
+ Summary: A docling OCR plugin for PaddlePaddle PP-OCRv6 (ONNX via RapidOCR)
5
+ Project-URL: Homepage, https://github.com/DCC-BS/docling-pp-ocrv6
6
+ Project-URL: Repository, https://github.com/DCC-BS/docling-pp-ocrv6
7
+ Project-URL: Issues, https://github.com/DCC-BS/docling-pp-ocrv6/issues
8
+ Project-URL: Changelog, https://github.com/DCC-BS/docling-pp-ocrv6/releases
9
+ Author-email: Yanick Schraner <yanick.schraner@bs.ch>, Tobias Bollinger <tobias.bollinger@bs.ch>
10
+ License: MIT
11
+ License-File: LICENSE
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Typing :: Typed
18
+ Requires-Python: >=3.12
19
+ Requires-Dist: docling>=2.73
20
+ Requires-Dist: huggingface-hub>=1.3.0
21
+ Requires-Dist: numpy
22
+ Requires-Dist: pillow>=10.0
23
+ Requires-Dist: pyyaml>=6.0
24
+ Requires-Dist: rapidocr>=3.0
25
+ Provides-Extra: cpu
26
+ Requires-Dist: onnxruntime>=1.17; extra == 'cpu'
27
+ Provides-Extra: gpu
28
+ Requires-Dist: onnxruntime-gpu>=1.17; extra == 'gpu'
29
+ Description-Content-Type: text/markdown
30
+
31
+ # docling-pp-ocrv6
32
+
33
+ A [Docling](https://github.com/docling-project/docling) OCR plugin for
34
+ PaddlePaddle's **PP-OCRv6** models. It runs the PP-OCRv6 detection and
35
+ recognition **ONNX** checkpoints locally through
36
+ [RapidOCR](https://github.com/RapidAI/RapidOCR) (onnxruntime), so OCR happens
37
+ inside the docling worker — no external service required.
38
+
39
+ GPU acceleration is automatic when the docling accelerator device resolves to
40
+ CUDA and `onnxruntime-gpu` is installed.
41
+
42
+ ## Installation
43
+
44
+ Pick exactly one onnxruntime extra — installing both the CPU and GPU wheels at
45
+ once is unsupported and prevents the CUDA provider from registering:
46
+
47
+ ```bash
48
+ pip install "docling-pp-ocrv6[cpu]" # CPU (onnxruntime)
49
+ pip install "docling-pp-ocrv6[gpu]" # CUDA (onnxruntime-gpu)
50
+ ```
51
+
52
+ The detection and recognition ONNX models are downloaded from HuggingFace on
53
+ first use and cached under docling's model cache. To pre-fetch them (e.g. at
54
+ container build time):
55
+
56
+ ```python
57
+ from docling_pp_ocrv6 import PPOCRv6Model
58
+ PPOCRv6Model.download_models()
59
+ ```
60
+
61
+ ## Usage
62
+
63
+ ```python
64
+ from docling.document_converter import DocumentConverter, PdfFormatOption
65
+ from docling.datamodel.base_models import InputFormat
66
+ from docling.datamodel.pipeline_options import PdfPipelineOptions
67
+ from docling_pp_ocrv6 import PPOCRv6Options
68
+
69
+ pipeline_options = PdfPipelineOptions(do_ocr=True)
70
+ pipeline_options.ocr_options = PPOCRv6Options()
71
+
72
+ converter = DocumentConverter(
73
+ format_options={InputFormat.PDF: PdfFormatOption(pipeline_options=pipeline_options)}
74
+ )
75
+ result = converter.convert("scanned.pdf")
76
+ print(result.document.export_to_markdown())
77
+ ```
78
+
79
+ With **docling-serve**, request the engine by its `kind`:
80
+
81
+ ```json
82
+ { "options": { "ocr": true, "ocr_engine": "pp-ocrv6" } }
83
+ ```
84
+
85
+ (`DOCLING_SERVE_ALLOW_EXTERNAL_PLUGINS=true` must be set for the plugin to load.)
86
+
87
+ ## Configuration
88
+
89
+ All options are settable via `PPOCRv6Options(...)` or environment variables:
90
+
91
+ | Option | Env var | Default |
92
+ | --- | --- | --- |
93
+ | `lang` | `PPOCRV6_LANG` | `de,en,fr,it,es,nl,pt,...` (German-led European set) |
94
+ | `text_score` | `PPOCRV6_TEXT_SCORE` | `0.5` |
95
+ | `use_det` / `use_cls` / `use_rec` | `PPOCRV6_USE_DET` / `_CLS` / `_REC` | `true` |
96
+ | `det_repo` | `PPOCRV6_DET_REPO` | `PaddlePaddle/PP-OCRv6_medium_det_onnx` |
97
+ | `rec_repo` | `PPOCRV6_REC_REPO` | `PaddlePaddle/PP-OCRv6_medium_rec_onnx` |
98
+ | `det_model_path` / `rec_model_path` / `rec_keys_path` / `cls_model_path` | `PPOCRV6_*_MODEL_PATH` / `_KEYS_PATH` | auto |
99
+
100
+ The recognition character dictionary is extracted automatically from the
101
+ recognition model's `inference.yml`. Angle classification uses RapidOCR's
102
+ bundled cls model unless `cls_model_path` is set.
103
+
104
+ ## Development
105
+
106
+ ```bash
107
+ make install # uv sync + pre-commit
108
+ make check # ruff lint + format check + ty type check
109
+ make test # pytest with coverage
110
+ ```
111
+
112
+ ## License
113
+
114
+ MIT
@@ -0,0 +1,10 @@
1
+ docling_pp_ocrv6/__init__.py,sha256=o_T8rpqV13O-qtFZussbl34WXDt1tgCbQONMtzeH7is,224
2
+ docling_pp_ocrv6/model.py,sha256=Ra4s-FucBduaChpOaeH86--Vf9BmUqIyedwr2e1bPj8,11418
3
+ docling_pp_ocrv6/options.py,sha256=ZKLXIxIxz-GjzsvHaS8H9DGY_s2V00fUY4oK2RMIkkI,4767
4
+ docling_pp_ocrv6/plugin.py,sha256=x0u-TlioVNeAg6GRy13vajLyHbfCN8ltc21rBDpRrTQ,287
5
+ docling_pp_ocrv6/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
6
+ docling_pp_ocrv6-0.1.0.dist-info/METADATA,sha256=RvAj8fCMzXk8Ulo9CkXl5btDPY8ufn-7o2OkTsB3iLE,4049
7
+ docling_pp_ocrv6-0.1.0.dist-info/WHEEL,sha256=mffPy8wBnZQn2VnJUU5jE99KsxaSfiyMHV9Yt0aLVxs,87
8
+ docling_pp_ocrv6-0.1.0.dist-info/entry_points.txt,sha256=my-zYx4Xr5Z68incYws4EyF9zMXDnC5wbnnFiUHuweU,53
9
+ docling_pp_ocrv6-0.1.0.dist-info/licenses/LICENSE,sha256=APpdFXym3_5C78pFAfnlLvmP7umPBWYpQ2MKidnoRuw,1091
10
+ docling_pp_ocrv6-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.30.1
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [docling]
2
+ docling_pp_ocrv6 = docling_pp_ocrv6.plugin
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Data Competence Center Basel-Stadt
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.