docling-pk 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docling_pk-1.0.0/LICENSE +21 -0
- docling_pk-1.0.0/PKG-INFO +274 -0
- docling_pk-1.0.0/README.md +225 -0
- docling_pk-1.0.0/pyproject.toml +103 -0
- docling_pk-1.0.0/setup.cfg +4 -0
- docling_pk-1.0.0/src/docling_pk/__init__.py +28 -0
- docling_pk-1.0.0/src/docling_pk/cli.py +197 -0
- docling_pk-1.0.0/src/docling_pk/errors.py +22 -0
- docling_pk-1.0.0/src/docling_pk/extractor.py +281 -0
- docling_pk-1.0.0/src/docling_pk/layout.py +146 -0
- docling_pk-1.0.0/src/docling_pk/ocr/__init__.py +5 -0
- docling_pk-1.0.0/src/docling_pk/ocr/base.py +130 -0
- docling_pk-1.0.0/src/docling_pk/ocr/easyocr_backend.py +84 -0
- docling_pk-1.0.0/src/docling_pk/ocr/rapidocr_backend.py +54 -0
- docling_pk-1.0.0/src/docling_pk/parsers/__init__.py +1 -0
- docling_pk-1.0.0/src/docling_pk/parsers/base.py +156 -0
- docling_pk-1.0.0/src/docling_pk/parsers/certificate.py +552 -0
- docling_pk-1.0.0/src/docling_pk/parsers/cnic.py +618 -0
- docling_pk-1.0.0/src/docling_pk/parsers/degree.py +217 -0
- docling_pk-1.0.0/src/docling_pk/pipeline.py +274 -0
- docling_pk-1.0.0/src/docling_pk/py.typed +0 -0
- docling_pk-1.0.0/src/docling_pk/schema.py +211 -0
- docling_pk-1.0.0/src/docling_pk/urdu.py +294 -0
- docling_pk-1.0.0/src/docling_pk/validation/__init__.py +1 -0
- docling_pk-1.0.0/src/docling_pk/validation/cnic.py +125 -0
- docling_pk-1.0.0/src/docling_pk/validation/dates.py +136 -0
- docling_pk-1.0.0/src/docling_pk/validation/marks.py +135 -0
- docling_pk-1.0.0/src/docling_pk/validation/text.py +76 -0
- docling_pk-1.0.0/src/docling_pk/vision/__init__.py +1 -0
- docling_pk-1.0.0/src/docling_pk/vision/geometry.py +103 -0
- docling_pk-1.0.0/src/docling_pk/vision/image_io.py +138 -0
- docling_pk-1.0.0/src/docling_pk/vision/quality.py +98 -0
- docling_pk-1.0.0/src/docling_pk.egg-info/PKG-INFO +274 -0
- docling_pk-1.0.0/src/docling_pk.egg-info/SOURCES.txt +43 -0
- docling_pk-1.0.0/src/docling_pk.egg-info/dependency_links.txt +1 -0
- docling_pk-1.0.0/src/docling_pk.egg-info/entry_points.txt +2 -0
- docling_pk-1.0.0/src/docling_pk.egg-info/requires.txt +24 -0
- docling_pk-1.0.0/src/docling_pk.egg-info/top_level.txt +1 -0
- docling_pk-1.0.0/tests/test_certificate_parser.py +239 -0
- docling_pk-1.0.0/tests/test_cnic_parser.py +230 -0
- docling_pk-1.0.0/tests/test_degree_urdu_layout.py +150 -0
- docling_pk-1.0.0/tests/test_integration_ocr.py +164 -0
- docling_pk-1.0.0/tests/test_pipeline_extractor_cli.py +435 -0
- docling_pk-1.0.0/tests/test_schema_and_io.py +241 -0
- docling_pk-1.0.0/tests/test_validation.py +161 -0
docling_pk-1.0.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Abdul Moiz Muhammad, INFERENCE Lab
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: docling-pk
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Structured data extraction from Pakistani identity and education documents (CNIC, Matric, Intermediate, degrees).
|
|
5
|
+
Author: Abdul Moiz Muhammad, INFERENCE Lab
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/Inference-LAB/docling-pk
|
|
8
|
+
Project-URL: Repository, https://github.com/Inference-LAB/docling-pk
|
|
9
|
+
Project-URL: Issues, https://github.com/Inference-LAB/docling-pk/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/Inference-LAB/docling-pk/blob/main/CHANGELOG.md
|
|
11
|
+
Keywords: ocr,cnic,pakistan,kyc,document-ai,information-extraction,easyocr,urdu
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Image Recognition
|
|
23
|
+
Classifier: Topic :: Text Processing
|
|
24
|
+
Classifier: Typing :: Typed
|
|
25
|
+
Requires-Python: >=3.9
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
License-File: LICENSE
|
|
28
|
+
Requires-Dist: numpy>=1.21
|
|
29
|
+
Requires-Dist: opencv-python>=4.5
|
|
30
|
+
Requires-Dist: rapidocr>=3.0
|
|
31
|
+
Requires-Dist: onnxruntime>=1.16
|
|
32
|
+
Requires-Dist: rapidfuzz>=3.0
|
|
33
|
+
Requires-Dist: typer>=0.9
|
|
34
|
+
Requires-Dist: Pillow>=9.0
|
|
35
|
+
Provides-Extra: urdu
|
|
36
|
+
Requires-Dist: easyocr>=1.7; extra == "urdu"
|
|
37
|
+
Provides-Extra: pdf
|
|
38
|
+
Requires-Dist: pymupdf>=1.23; extra == "pdf"
|
|
39
|
+
Provides-Extra: all
|
|
40
|
+
Requires-Dist: docling-pk[pdf,urdu]; extra == "all"
|
|
41
|
+
Provides-Extra: dev
|
|
42
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
43
|
+
Requires-Dist: pytest-cov>=4; extra == "dev"
|
|
44
|
+
Requires-Dist: ruff>=0.5; extra == "dev"
|
|
45
|
+
Requires-Dist: mypy>=1.8; extra == "dev"
|
|
46
|
+
Requires-Dist: build>=1.0; extra == "dev"
|
|
47
|
+
Requires-Dist: twine>=5; extra == "dev"
|
|
48
|
+
Dynamic: license-file
|
|
49
|
+
|
|
50
|
+
# docling-pk
|
|
51
|
+
|
|
52
|
+
**Structured data extraction from Pakistani identity and education documents.**
|
|
53
|
+
Give it a photo of a CNIC, a Matric or Intermediate certificate, or a university
|
|
54
|
+
degree / transcript, and get back typed fields with confidence scores, plus a
|
|
55
|
+
reason for every field it could not read.
|
|
56
|
+
|
|
57
|
+
[](https://github.com/Inference-LAB/docling-pk/actions/workflows/ci.yml)
|
|
58
|
+

|
|
59
|
+

|
|
60
|
+
|
|
61
|
+
```python
|
|
62
|
+
from docling_pk import extract
|
|
63
|
+
|
|
64
|
+
result = extract("cnic_front.jpg", document_type="cnic")
|
|
65
|
+
print(result.to_json())
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
```json
|
|
69
|
+
{
|
|
70
|
+
"document_type": "cnic",
|
|
71
|
+
"fields": {
|
|
72
|
+
"name": {"value": "Muhammad Ali", "confidence": 0.99, "status": "ok"},
|
|
73
|
+
"father_name": {"value": "Muhammad Akram", "confidence": 0.99, "status": "ok"},
|
|
74
|
+
"gender": {"value": "M", "confidence": 1.0, "status": "ok"},
|
|
75
|
+
"country_of_stay":{"value": "Pakistan", "confidence": 1.0, "status": "ok"},
|
|
76
|
+
"cnic_number": {"value": "35202-1234567-9", "confidence": 1.0, "status": "ok"},
|
|
77
|
+
"date_of_birth": {"value": "01-01-1995", "confidence": 1.0, "status": "ok"},
|
|
78
|
+
"date_of_issue": {"value": "15-03-2020", "confidence": 1.0, "status": "ok"},
|
|
79
|
+
"date_of_expiry": {"value": "15-03-2030", "confidence": 1.0, "status": "ok"},
|
|
80
|
+
"address": {"value": null, "confidence": 0.0, "status": "not_applicable",
|
|
81
|
+
"reason": "printed on the back of the card; pass the back image too"},
|
|
82
|
+
"permanent_address": {"value": null, "confidence": 0.0, "status": "not_applicable", "reason": "..."}
|
|
83
|
+
},
|
|
84
|
+
"confidence": 0.99,
|
|
85
|
+
"warnings": [],
|
|
86
|
+
"metadata": {"ocr_backend": "rapidocr", "sides": ["front"], "pages": [{"rotation_degrees": 90, "skew_degrees": 0.0}]}
|
|
87
|
+
}
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
## Why it is built this way
|
|
91
|
+
|
|
92
|
+
- **It does not return a wrong value as `ok`.** On the synthetic benchmark, every
|
|
93
|
+
field reported with status `ok` was correct; on 16 real photos, 99.2% were.
|
|
94
|
+
When a field cannot be read reliably, `value` is `None` and the result says
|
|
95
|
+
why (`not_found`, `invalid`, `low_confidence`, `not_applicable`). A blank
|
|
96
|
+
goes to a human. A wrong CNIC digit would quietly change someone's identity.
|
|
97
|
+
- **Cross-checks, not just OCR.** CNIC numbers and dates are re-read from a
|
|
98
|
+
tight crop by a second OCR engine. The three CNIC dates must satisfy
|
|
99
|
+
birth < issue < expiry. The CNIC's last digit must match the printed gender
|
|
100
|
+
(odd = male). Marks are checked against the TOTAL row, the subject rows'
|
|
101
|
+
sum, and the "marks in words" line.
|
|
102
|
+
- **Real phone photos.** It detects and fixes 90/180/270-degree rotation and
|
|
103
|
+
skew, filters security-paper micro-text, and warns about blur, glare and low
|
|
104
|
+
resolution instead of rejecting the image.
|
|
105
|
+
- **Light install.** The default engine is RapidOCR (PP-OCR models on ONNX
|
|
106
|
+
Runtime): about 370 MB installed, CPU-only, around 2.5 s per document. No
|
|
107
|
+
PyTorch is needed unless you want the Urdu address.
|
|
108
|
+
|
|
109
|
+
## Installation
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
pip install docling-pk # CNIC front, certificates, degrees (CPU, ~370 MB)
|
|
113
|
+
pip install "docling-pk[urdu]" # + Urdu address on CNIC backs, + second-engine verification (EasyOCR/PyTorch)
|
|
114
|
+
pip install "docling-pk[pdf]" # + PDF input (PyMuPDF)
|
|
115
|
+
pip install "docling-pk[all]"
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
Python 3.9 to 3.12 on Linux, Windows and macOS. Models are bundled with or
|
|
119
|
+
downloaded by the OCR engines on first use; nothing is sent to any server.
|
|
120
|
+
On minimal Linux images (e.g. `python:3.x-slim` in Docker), OpenCV needs two
|
|
121
|
+
system libraries: `apt-get install -y libgl1 libglib2.0-0`.
|
|
122
|
+
|
|
123
|
+
## Usage
|
|
124
|
+
|
|
125
|
+
### Python
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
from docling_pk import extract
|
|
129
|
+
|
|
130
|
+
# CNIC: pass front and back together to also get the address
|
|
131
|
+
r = extract(["cnic_front.jpg", "cnic_back.jpg"], document_type="cnic")
|
|
132
|
+
r.fields["cnic_number"].value # '35202-1234567-9'
|
|
133
|
+
r.fields["date_of_birth"].as_date() # datetime.date(1995, 1, 1)
|
|
134
|
+
r["fields"]["name"]["value"] # dict-style access mirrors the JSON
|
|
135
|
+
|
|
136
|
+
# Board certificate: subject-wise marks come back as a table
|
|
137
|
+
r = extract("matric.jpg", document_type="matric")
|
|
138
|
+
r.fields["total_marks"].value # '787'
|
|
139
|
+
r.tables["subjects"][0] # {'subject': 'ENGLISH (COMPULSORY)', 'max_marks': 150, 'obtained_marks': 118}
|
|
140
|
+
|
|
141
|
+
# Let it figure out the document type
|
|
142
|
+
r = extract("unknown_scan.png") # document_type="auto"
|
|
143
|
+
r.document_type # 'intermediate'
|
|
144
|
+
|
|
145
|
+
# Inputs: path, bytes, NumPy array (BGR), PIL image, PDF, or a list of these
|
|
146
|
+
r = extract(open("card.jpg", "rb").read(), "cnic")
|
|
147
|
+
|
|
148
|
+
# Acting only on fields that passed every check
|
|
149
|
+
trusted = {k: f.value for k, f in r.fields.items() if f.status == "ok"}
|
|
150
|
+
needs_review = {k: f.reason for k, f in r.fields.items() if f.status != "ok"}
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
Options: `backend="auto" | "rapidocr" | "easyocr"` (or your own
|
|
154
|
+
`OCRBackend`), `verify=True` (second-engine re-reads when EasyOCR is
|
|
155
|
+
installed), `urdu=True`, `gpu=None` (auto), `auto_rotate=True`,
|
|
156
|
+
`include_raw_text=True`.
|
|
157
|
+
|
|
158
|
+
### Command line
|
|
159
|
+
|
|
160
|
+
```bash
|
|
161
|
+
docling-pk extract cnic_front.jpg cnic_back.jpg --type cnic # JSON to stdout
|
|
162
|
+
docling-pk extract certificate.jpg --type auto --output table # readable table
|
|
163
|
+
docling-pk extract transcript.pdf --type degree --out result.json
|
|
164
|
+
docling-pk batch ./scans --type matric --out results.jsonl # a folder, one JSON line per file
|
|
165
|
+
docling-pk info # engines installed, GPU
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
Exit codes: `0` success, `1` processed but nothing could be extracted, `2` bad
|
|
169
|
+
input (missing file, unreadable image, unknown type).
|
|
170
|
+
|
|
171
|
+
## Supported documents
|
|
172
|
+
|
|
173
|
+
| `document_type` | Fields | Verified on real samples |
|
|
174
|
+
|---|---|---|
|
|
175
|
+
| `cnic` | `name`, `father_name`, `gender`, `country_of_stay`, `cnic_number`, `date_of_birth`, `date_of_issue`, `date_of_expiry`, `address`, `permanent_address` | Current green CNIC / SNIC, front and back |
|
|
176
|
+
| `matric` (`ssc`) | `student_name`, `father_name`, `date_of_birth`, `roll_number`, `registration_number`, `serial_number`, `certificate_number`, `board`, `exam`, `year`, `session`, `group`, `institute`, `grade`, `total_marks`, `max_marks`, plus `tables["subjects"]` | FBISE |
|
|
177
|
+
| `intermediate` (`hssc`, `fsc`, `inter`) | Same as matric | FBISE |
|
|
178
|
+
| `degree` (`transcript`) | `student_name`, `father_name`, `degree`, `institution`, `campus`, `registration_number`, `serial_number`, `cgpa`, `division`, `date_of_birth`, `date_of_issue`, `year` | COMSATS degree and transcript |
|
|
179
|
+
|
|
180
|
+
Every field key is always present for its document type. Matric and
|
|
181
|
+
intermediate are confirmed against the certificate title, which wins if it
|
|
182
|
+
disagrees with `document_type` (a warning says so). Other boards (BISE
|
|
183
|
+
Lahore, Rawalpindi, Karachi, Peshawar...) are detected and parsed with the
|
|
184
|
+
same label vocabulary, but come with a warning because their layouts have
|
|
185
|
+
not been verified on real samples yet.
|
|
186
|
+
|
|
187
|
+
## Accuracy
|
|
188
|
+
|
|
189
|
+
Exact-match field accuracy. *Precision* is the share of returned values that
|
|
190
|
+
are correct; *ok-precision* is the same restricted to fields with status
|
|
191
|
+
`ok`, which is the number that matters if your system auto-accepts them.
|
|
192
|
+
|
|
193
|
+
| Dataset | Accuracy | Precision | ok-precision | Marks rows | Time / doc |
|
|
194
|
+
|---|---|---|---|---|---|
|
|
195
|
+
| 16 real photos, original v0.1 pipeline | 34.9% | n/a | n/a | 0 / 30 | n/a |
|
|
196
|
+
| 16 real photos, v1 (`pip install docling-pk`) | **94.4%** | 97.1% | 98.5% | 30 / 30 | 3.7 s |
|
|
197
|
+
| 16 real photos, v1 with the `urdu` extra | **94.4%** | 96.4% | **99.2%** | 30 / 30 | 3.7 s |
|
|
198
|
+
| 23 synthetic fixtures (committed) | **96.2%** | 100% | **100%** | 53 / 53 | 2.5 s (CPU) |
|
|
199
|
+
|
|
200
|
+
By document type on real photos: CNIC 91.2%, Matric 96.4%, Intermediate
|
|
201
|
+
100%, Degree 100%, Transcript 90%. Every date and gender on the real CNICs was
|
|
202
|
+
correct. The misses: the Urdu address (both cards), three fields on a
|
|
203
|
+
motion-blurred photo (left blank, plus a misspelled name flagged
|
|
204
|
+
`low_confidence`), one father's name on a small soft-focus photo (left
|
|
205
|
+
blank), one registration number with an I/l confusion (flagged
|
|
206
|
+
`low_confidence`) and one dropped space in a school name. Full methodology
|
|
207
|
+
and per-field numbers:
|
|
208
|
+
[docs/benchmark.md](https://github.com/Inference-LAB/docling-pk/blob/main/docs/benchmark.md). The real photos contain personal data
|
|
209
|
+
and are not published; only aggregate counts are
|
|
210
|
+
([benchmarks/results/real_samples_summary.json](https://github.com/Inference-LAB/docling-pk/blob/main/benchmarks/results/real_samples_summary.json)).
|
|
211
|
+
|
|
212
|
+
Reproduce the public benchmark:
|
|
213
|
+
|
|
214
|
+
```bash
|
|
215
|
+
python benchmarks/run_benchmark.py tests/fixtures/synthetic/labels.json
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
## Known limitations
|
|
219
|
+
|
|
220
|
+
- **Urdu address (CNIC back) is experimental.** Nastaliq OCR with EasyOCR
|
|
221
|
+
averages about 49% character error on real cards, even after Urdu
|
|
222
|
+
normalization and lexicon correction of address words and district names.
|
|
223
|
+
It is always returned as `low_confidence`. UTRNet, the strongest open Urdu
|
|
224
|
+
recognizer we evaluated, is licensed CC BY-NC-SA (non-commercial), so it
|
|
225
|
+
cannot ship in an MIT library.
|
|
226
|
+
- **Handwritten fields, heavily damaged documents, and stamps covering more
|
|
227
|
+
than ~30% of a field** are out of scope (per the v1 brief). They return
|
|
228
|
+
`None` with a reason; they do not crash.
|
|
229
|
+
- **Old (pre-2012) Urdu-only CNICs**: numbers and dates are read; Urdu-only
|
|
230
|
+
names are not.
|
|
231
|
+
- **Board layouts other than FBISE** are parsed best-effort and flagged.
|
|
232
|
+
- **Severe motion blur** loses fields. The result is a blur warning and blank
|
|
233
|
+
fields, never invented values.
|
|
234
|
+
- **No NADRA verification.** docling-pk reads documents; it does not check
|
|
235
|
+
that they are genuine.
|
|
236
|
+
|
|
237
|
+
## Privacy
|
|
238
|
+
|
|
239
|
+
Processing is fully local. `DocumentResult.to_dict()` and the CLI leave out
|
|
240
|
+
the raw OCR text by default, because it can contain personal data that is not
|
|
241
|
+
part of any extracted field (`include_raw=True` / `--raw` to include it). The
|
|
242
|
+
repository's test images are synthetic and stamped "SYNTHETIC SPECIMEN".
|
|
243
|
+
|
|
244
|
+
## How it works
|
|
245
|
+
|
|
246
|
+
```
|
|
247
|
+
image/PDF ─► load (Unicode paths, EXIF, PDF pages) ─► quality checks (blur, glare, exposure)
|
|
248
|
+
─► orientation (0/90/180/270 by Latin-text score × box shape) ─► deskew from text-line angles
|
|
249
|
+
─► OCR with positions (RapidOCR / EasyOCR) ─► micro-text filter (+ ink-only re-read if needed)
|
|
250
|
+
─► per-type parser: labels found by fuzzy match, values by position
|
|
251
|
+
─► validation & cross-checks (second engine, CNIC structure, date order, marks sums)
|
|
252
|
+
─► DocumentResult (value, confidence, status, reason per field)
|
|
253
|
+
```
|
|
254
|
+
|
|
255
|
+
Details and the decisions behind them, including the measurements that
|
|
256
|
+
replaced the brief's EasyOCR default: [docs/design_doc.md](https://github.com/Inference-LAB/docling-pk/blob/main/docs/design_doc.md).
|
|
257
|
+
|
|
258
|
+
## Development
|
|
259
|
+
|
|
260
|
+
```bash
|
|
261
|
+
pip install -e ".[dev,pdf]"
|
|
262
|
+
pytest -m "not ocr" # 211 fast tests, no models needed
|
|
263
|
+
pytest # + end-to-end OCR tests on synthetic fixtures
|
|
264
|
+
ruff check . && ruff format --check . && mypy src
|
|
265
|
+
python benchmarks/synthetic/generate.py # regenerate the synthetic fixtures
|
|
266
|
+
```
|
|
267
|
+
|
|
268
|
+
See [CONTRIBUTING.md](https://github.com/Inference-LAB/docling-pk/blob/main/CONTRIBUTING.md). Releases are published to PyPI from
|
|
269
|
+
version tags by GitHub Actions (trusted publishing).
|
|
270
|
+
|
|
271
|
+
## License
|
|
272
|
+
|
|
273
|
+
MIT. See [LICENSE](https://github.com/Inference-LAB/docling-pk/blob/main/LICENSE). Built at [INFERENCE Lab](https://inference-lab.org)
|
|
274
|
+
as part of Engineering Fellowship Cohort 01.
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
# docling-pk
|
|
2
|
+
|
|
3
|
+
**Structured data extraction from Pakistani identity and education documents.**
|
|
4
|
+
Give it a photo of a CNIC, a Matric or Intermediate certificate, or a university
|
|
5
|
+
degree / transcript, and get back typed fields with confidence scores, plus a
|
|
6
|
+
reason for every field it could not read.
|
|
7
|
+
|
|
8
|
+
[](https://github.com/Inference-LAB/docling-pk/actions/workflows/ci.yml)
|
|
9
|
+

|
|
10
|
+

|
|
11
|
+
|
|
12
|
+
```python
|
|
13
|
+
from docling_pk import extract
|
|
14
|
+
|
|
15
|
+
result = extract("cnic_front.jpg", document_type="cnic")
|
|
16
|
+
print(result.to_json())
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
```json
|
|
20
|
+
{
|
|
21
|
+
"document_type": "cnic",
|
|
22
|
+
"fields": {
|
|
23
|
+
"name": {"value": "Muhammad Ali", "confidence": 0.99, "status": "ok"},
|
|
24
|
+
"father_name": {"value": "Muhammad Akram", "confidence": 0.99, "status": "ok"},
|
|
25
|
+
"gender": {"value": "M", "confidence": 1.0, "status": "ok"},
|
|
26
|
+
"country_of_stay":{"value": "Pakistan", "confidence": 1.0, "status": "ok"},
|
|
27
|
+
"cnic_number": {"value": "35202-1234567-9", "confidence": 1.0, "status": "ok"},
|
|
28
|
+
"date_of_birth": {"value": "01-01-1995", "confidence": 1.0, "status": "ok"},
|
|
29
|
+
"date_of_issue": {"value": "15-03-2020", "confidence": 1.0, "status": "ok"},
|
|
30
|
+
"date_of_expiry": {"value": "15-03-2030", "confidence": 1.0, "status": "ok"},
|
|
31
|
+
"address": {"value": null, "confidence": 0.0, "status": "not_applicable",
|
|
32
|
+
"reason": "printed on the back of the card; pass the back image too"},
|
|
33
|
+
"permanent_address": {"value": null, "confidence": 0.0, "status": "not_applicable", "reason": "..."}
|
|
34
|
+
},
|
|
35
|
+
"confidence": 0.99,
|
|
36
|
+
"warnings": [],
|
|
37
|
+
"metadata": {"ocr_backend": "rapidocr", "sides": ["front"], "pages": [{"rotation_degrees": 90, "skew_degrees": 0.0}]}
|
|
38
|
+
}
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Why it is built this way
|
|
42
|
+
|
|
43
|
+
- **It does not return a wrong value as `ok`.** On the synthetic benchmark, every
|
|
44
|
+
field reported with status `ok` was correct; on 16 real photos, 99.2% were.
|
|
45
|
+
When a field cannot be read reliably, `value` is `None` and the result says
|
|
46
|
+
why (`not_found`, `invalid`, `low_confidence`, `not_applicable`). A blank
|
|
47
|
+
goes to a human. A wrong CNIC digit would quietly change someone's identity.
|
|
48
|
+
- **Cross-checks, not just OCR.** CNIC numbers and dates are re-read from a
|
|
49
|
+
tight crop by a second OCR engine. The three CNIC dates must satisfy
|
|
50
|
+
birth < issue < expiry. The CNIC's last digit must match the printed gender
|
|
51
|
+
(odd = male). Marks are checked against the TOTAL row, the subject rows'
|
|
52
|
+
sum, and the "marks in words" line.
|
|
53
|
+
- **Real phone photos.** It detects and fixes 90/180/270-degree rotation and
|
|
54
|
+
skew, filters security-paper micro-text, and warns about blur, glare and low
|
|
55
|
+
resolution instead of rejecting the image.
|
|
56
|
+
- **Light install.** The default engine is RapidOCR (PP-OCR models on ONNX
|
|
57
|
+
Runtime): about 370 MB installed, CPU-only, around 2.5 s per document. No
|
|
58
|
+
PyTorch is needed unless you want the Urdu address.
|
|
59
|
+
|
|
60
|
+
## Installation
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
pip install docling-pk # CNIC front, certificates, degrees (CPU, ~370 MB)
|
|
64
|
+
pip install "docling-pk[urdu]" # + Urdu address on CNIC backs, + second-engine verification (EasyOCR/PyTorch)
|
|
65
|
+
pip install "docling-pk[pdf]" # + PDF input (PyMuPDF)
|
|
66
|
+
pip install "docling-pk[all]"
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Python 3.9 to 3.12 on Linux, Windows and macOS. Models are bundled with or
|
|
70
|
+
downloaded by the OCR engines on first use; nothing is sent to any server.
|
|
71
|
+
On minimal Linux images (e.g. `python:3.x-slim` in Docker), OpenCV needs two
|
|
72
|
+
system libraries: `apt-get install -y libgl1 libglib2.0-0`.
|
|
73
|
+
|
|
74
|
+
## Usage
|
|
75
|
+
|
|
76
|
+
### Python
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
from docling_pk import extract
|
|
80
|
+
|
|
81
|
+
# CNIC: pass front and back together to also get the address
|
|
82
|
+
r = extract(["cnic_front.jpg", "cnic_back.jpg"], document_type="cnic")
|
|
83
|
+
r.fields["cnic_number"].value # '35202-1234567-9'
|
|
84
|
+
r.fields["date_of_birth"].as_date() # datetime.date(1995, 1, 1)
|
|
85
|
+
r["fields"]["name"]["value"] # dict-style access mirrors the JSON
|
|
86
|
+
|
|
87
|
+
# Board certificate: subject-wise marks come back as a table
|
|
88
|
+
r = extract("matric.jpg", document_type="matric")
|
|
89
|
+
r.fields["total_marks"].value # '787'
|
|
90
|
+
r.tables["subjects"][0] # {'subject': 'ENGLISH (COMPULSORY)', 'max_marks': 150, 'obtained_marks': 118}
|
|
91
|
+
|
|
92
|
+
# Let it figure out the document type
|
|
93
|
+
r = extract("unknown_scan.png") # document_type="auto"
|
|
94
|
+
r.document_type # 'intermediate'
|
|
95
|
+
|
|
96
|
+
# Inputs: path, bytes, NumPy array (BGR), PIL image, PDF, or a list of these
|
|
97
|
+
r = extract(open("card.jpg", "rb").read(), "cnic")
|
|
98
|
+
|
|
99
|
+
# Acting only on fields that passed every check
|
|
100
|
+
trusted = {k: f.value for k, f in r.fields.items() if f.status == "ok"}
|
|
101
|
+
needs_review = {k: f.reason for k, f in r.fields.items() if f.status != "ok"}
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Options: `backend="auto" | "rapidocr" | "easyocr"` (or your own
|
|
105
|
+
`OCRBackend`), `verify=True` (second-engine re-reads when EasyOCR is
|
|
106
|
+
installed), `urdu=True`, `gpu=None` (auto), `auto_rotate=True`,
|
|
107
|
+
`include_raw_text=True`.
|
|
108
|
+
|
|
109
|
+
### Command line
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
docling-pk extract cnic_front.jpg cnic_back.jpg --type cnic # JSON to stdout
|
|
113
|
+
docling-pk extract certificate.jpg --type auto --output table # readable table
|
|
114
|
+
docling-pk extract transcript.pdf --type degree --out result.json
|
|
115
|
+
docling-pk batch ./scans --type matric --out results.jsonl # a folder, one JSON line per file
|
|
116
|
+
docling-pk info # engines installed, GPU
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
Exit codes: `0` success, `1` processed but nothing could be extracted, `2` bad
|
|
120
|
+
input (missing file, unreadable image, unknown type).
|
|
121
|
+
|
|
122
|
+
## Supported documents
|
|
123
|
+
|
|
124
|
+
| `document_type` | Fields | Verified on real samples |
|
|
125
|
+
|---|---|---|
|
|
126
|
+
| `cnic` | `name`, `father_name`, `gender`, `country_of_stay`, `cnic_number`, `date_of_birth`, `date_of_issue`, `date_of_expiry`, `address`, `permanent_address` | Current green CNIC / SNIC, front and back |
|
|
127
|
+
| `matric` (`ssc`) | `student_name`, `father_name`, `date_of_birth`, `roll_number`, `registration_number`, `serial_number`, `certificate_number`, `board`, `exam`, `year`, `session`, `group`, `institute`, `grade`, `total_marks`, `max_marks`, plus `tables["subjects"]` | FBISE |
|
|
128
|
+
| `intermediate` (`hssc`, `fsc`, `inter`) | Same as matric | FBISE |
|
|
129
|
+
| `degree` (`transcript`) | `student_name`, `father_name`, `degree`, `institution`, `campus`, `registration_number`, `serial_number`, `cgpa`, `division`, `date_of_birth`, `date_of_issue`, `year` | COMSATS degree and transcript |
|
|
130
|
+
|
|
131
|
+
Every field key is always present for its document type. Matric and
|
|
132
|
+
intermediate are confirmed against the certificate title, which wins if it
|
|
133
|
+
disagrees with `document_type` (a warning says so). Other boards (BISE
|
|
134
|
+
Lahore, Rawalpindi, Karachi, Peshawar...) are detected and parsed with the
|
|
135
|
+
same label vocabulary, but come with a warning because their layouts have
|
|
136
|
+
not been verified on real samples yet.
|
|
137
|
+
|
|
138
|
+
## Accuracy
|
|
139
|
+
|
|
140
|
+
Exact-match field accuracy. *Precision* is the share of returned values that
|
|
141
|
+
are correct; *ok-precision* is the same restricted to fields with status
|
|
142
|
+
`ok`, which is the number that matters if your system auto-accepts them.
|
|
143
|
+
|
|
144
|
+
| Dataset | Accuracy | Precision | ok-precision | Marks rows | Time / doc |
|
|
145
|
+
|---|---|---|---|---|---|
|
|
146
|
+
| 16 real photos, original v0.1 pipeline | 34.9% | n/a | n/a | 0 / 30 | n/a |
|
|
147
|
+
| 16 real photos, v1 (`pip install docling-pk`) | **94.4%** | 97.1% | 98.5% | 30 / 30 | 3.7 s |
|
|
148
|
+
| 16 real photos, v1 with the `urdu` extra | **94.4%** | 96.4% | **99.2%** | 30 / 30 | 3.7 s |
|
|
149
|
+
| 23 synthetic fixtures (committed) | **96.2%** | 100% | **100%** | 53 / 53 | 2.5 s (CPU) |
|
|
150
|
+
|
|
151
|
+
By document type on real photos: CNIC 91.2%, Matric 96.4%, Intermediate
|
|
152
|
+
100%, Degree 100%, Transcript 90%. Every date and gender on the real CNICs was
|
|
153
|
+
correct. The misses: the Urdu address (both cards), three fields on a
|
|
154
|
+
motion-blurred photo (left blank, plus a misspelled name flagged
|
|
155
|
+
`low_confidence`), one father's name on a small soft-focus photo (left
|
|
156
|
+
blank), one registration number with an I/l confusion (flagged
|
|
157
|
+
`low_confidence`) and one dropped space in a school name. Full methodology
|
|
158
|
+
and per-field numbers:
|
|
159
|
+
[docs/benchmark.md](https://github.com/Inference-LAB/docling-pk/blob/main/docs/benchmark.md). The real photos contain personal data
|
|
160
|
+
and are not published; only aggregate counts are
|
|
161
|
+
([benchmarks/results/real_samples_summary.json](https://github.com/Inference-LAB/docling-pk/blob/main/benchmarks/results/real_samples_summary.json)).
|
|
162
|
+
|
|
163
|
+
Reproduce the public benchmark:
|
|
164
|
+
|
|
165
|
+
```bash
|
|
166
|
+
python benchmarks/run_benchmark.py tests/fixtures/synthetic/labels.json
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
## Known limitations
|
|
170
|
+
|
|
171
|
+
- **Urdu address (CNIC back) is experimental.** Nastaliq OCR with EasyOCR
|
|
172
|
+
averages about 49% character error on real cards, even after Urdu
|
|
173
|
+
normalization and lexicon correction of address words and district names.
|
|
174
|
+
It is always returned as `low_confidence`. UTRNet, the strongest open Urdu
|
|
175
|
+
recognizer we evaluated, is licensed CC BY-NC-SA (non-commercial), so it
|
|
176
|
+
cannot ship in an MIT library.
|
|
177
|
+
- **Handwritten fields, heavily damaged documents, and stamps covering more
|
|
178
|
+
than ~30% of a field** are out of scope (per the v1 brief). They return
|
|
179
|
+
`None` with a reason; they do not crash.
|
|
180
|
+
- **Old (pre-2012) Urdu-only CNICs**: numbers and dates are read; Urdu-only
|
|
181
|
+
names are not.
|
|
182
|
+
- **Board layouts other than FBISE** are parsed best-effort and flagged.
|
|
183
|
+
- **Severe motion blur** loses fields. The result is a blur warning and blank
|
|
184
|
+
fields, never invented values.
|
|
185
|
+
- **No NADRA verification.** docling-pk reads documents; it does not check
|
|
186
|
+
that they are genuine.
|
|
187
|
+
|
|
188
|
+
## Privacy
|
|
189
|
+
|
|
190
|
+
Processing is fully local. `DocumentResult.to_dict()` and the CLI leave out
|
|
191
|
+
the raw OCR text by default, because it can contain personal data that is not
|
|
192
|
+
part of any extracted field (`include_raw=True` / `--raw` to include it). The
|
|
193
|
+
repository's test images are synthetic and stamped "SYNTHETIC SPECIMEN".
|
|
194
|
+
|
|
195
|
+
## How it works
|
|
196
|
+
|
|
197
|
+
```
|
|
198
|
+
image/PDF ─► load (Unicode paths, EXIF, PDF pages) ─► quality checks (blur, glare, exposure)
|
|
199
|
+
─► orientation (0/90/180/270 by Latin-text score × box shape) ─► deskew from text-line angles
|
|
200
|
+
─► OCR with positions (RapidOCR / EasyOCR) ─► micro-text filter (+ ink-only re-read if needed)
|
|
201
|
+
─► per-type parser: labels found by fuzzy match, values by position
|
|
202
|
+
─► validation & cross-checks (second engine, CNIC structure, date order, marks sums)
|
|
203
|
+
─► DocumentResult (value, confidence, status, reason per field)
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
Details and the decisions behind them, including the measurements that
|
|
207
|
+
replaced the brief's EasyOCR default: [docs/design_doc.md](https://github.com/Inference-LAB/docling-pk/blob/main/docs/design_doc.md).
|
|
208
|
+
|
|
209
|
+
## Development
|
|
210
|
+
|
|
211
|
+
```bash
|
|
212
|
+
pip install -e ".[dev,pdf]"
|
|
213
|
+
pytest -m "not ocr" # 211 fast tests, no models needed
|
|
214
|
+
pytest # + end-to-end OCR tests on synthetic fixtures
|
|
215
|
+
ruff check . && ruff format --check . && mypy src
|
|
216
|
+
python benchmarks/synthetic/generate.py # regenerate the synthetic fixtures
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
See [CONTRIBUTING.md](https://github.com/Inference-LAB/docling-pk/blob/main/CONTRIBUTING.md). Releases are published to PyPI from
|
|
220
|
+
version tags by GitHub Actions (trusted publishing).
|
|
221
|
+
|
|
222
|
+
## License
|
|
223
|
+
|
|
224
|
+
MIT. See [LICENSE](https://github.com/Inference-LAB/docling-pk/blob/main/LICENSE). Built at [INFERENCE Lab](https://inference-lab.org)
|
|
225
|
+
as part of Engineering Fellowship Cohort 01.
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "docling-pk"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "Structured data extraction from Pakistani identity and education documents (CNIC, Matric, Intermediate, degrees)."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Abdul Moiz Muhammad" }, { name = "INFERENCE Lab" }]
|
|
14
|
+
keywords = ["ocr", "cnic", "pakistan", "kyc", "document-ai", "information-extraction", "easyocr", "urdu"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 4 - Beta",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"Intended Audience :: Science/Research",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
22
|
+
"Programming Language :: Python :: 3.9",
|
|
23
|
+
"Programming Language :: Python :: 3.10",
|
|
24
|
+
"Programming Language :: Python :: 3.11",
|
|
25
|
+
"Programming Language :: Python :: 3.12",
|
|
26
|
+
"Topic :: Scientific/Engineering :: Image Recognition",
|
|
27
|
+
"Topic :: Text Processing",
|
|
28
|
+
"Typing :: Typed",
|
|
29
|
+
]
|
|
30
|
+
dependencies = [
|
|
31
|
+
"numpy>=1.21",
|
|
32
|
+
"opencv-python>=4.5", # same distribution RapidOCR requires, so only one cv2 is installed
|
|
33
|
+
"rapidocr>=3.0",
|
|
34
|
+
"onnxruntime>=1.16",
|
|
35
|
+
"rapidfuzz>=3.0",
|
|
36
|
+
"typer>=0.9",
|
|
37
|
+
"Pillow>=9.0",
|
|
38
|
+
]
|
|
39
|
+
|
|
40
|
+
[project.optional-dependencies]
|
|
41
|
+
# EasyOCR (PyTorch) adds Urdu address reading and an independent second
|
|
42
|
+
# engine that cross-checks CNIC numbers, dates and IDs.
|
|
43
|
+
urdu = ["easyocr>=1.7"]
|
|
44
|
+
pdf = ["pymupdf>=1.23"]
|
|
45
|
+
all = ["docling-pk[urdu,pdf]"]
|
|
46
|
+
dev = [
|
|
47
|
+
"pytest>=7",
|
|
48
|
+
"pytest-cov>=4",
|
|
49
|
+
"ruff>=0.5",
|
|
50
|
+
"mypy>=1.8",
|
|
51
|
+
"build>=1.0",
|
|
52
|
+
"twine>=5",
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
[project.scripts]
|
|
56
|
+
docling-pk = "docling_pk.cli:app"
|
|
57
|
+
|
|
58
|
+
[project.urls]
|
|
59
|
+
Homepage = "https://github.com/Inference-LAB/docling-pk"
|
|
60
|
+
Repository = "https://github.com/Inference-LAB/docling-pk"
|
|
61
|
+
Issues = "https://github.com/Inference-LAB/docling-pk/issues"
|
|
62
|
+
Changelog = "https://github.com/Inference-LAB/docling-pk/blob/main/CHANGELOG.md"
|
|
63
|
+
|
|
64
|
+
[tool.setuptools.dynamic]
|
|
65
|
+
version = { attr = "docling_pk.__version__" }
|
|
66
|
+
|
|
67
|
+
[tool.setuptools.packages.find]
|
|
68
|
+
where = ["src"]
|
|
69
|
+
|
|
70
|
+
[tool.setuptools.package-data]
|
|
71
|
+
docling_pk = ["py.typed"]
|
|
72
|
+
|
|
73
|
+
[tool.pytest.ini_options]
|
|
74
|
+
testpaths = ["tests"]
|
|
75
|
+
addopts = "-ra --strict-markers"
|
|
76
|
+
markers = [
|
|
77
|
+
"ocr: needs the EasyOCR models (downloaded on first run); deselect with -m 'not ocr'",
|
|
78
|
+
]
|
|
79
|
+
|
|
80
|
+
[tool.coverage.run]
|
|
81
|
+
source = ["docling_pk"]
|
|
82
|
+
branch = true
|
|
83
|
+
|
|
84
|
+
[tool.coverage.report]
|
|
85
|
+
show_missing = true
|
|
86
|
+
skip_covered = false
|
|
87
|
+
exclude_lines = ["pragma: no cover", "if __name__ == .__main__.:", "raise NotImplementedError"]
|
|
88
|
+
|
|
89
|
+
[tool.ruff]
|
|
90
|
+
line-length = 120
|
|
91
|
+
target-version = "py39"
|
|
92
|
+
src = ["src", "tests"]
|
|
93
|
+
|
|
94
|
+
[tool.ruff.lint]
|
|
95
|
+
select = ["E", "F", "W", "I", "B", "UP", "SIM"]
|
|
96
|
+
# UP045/UP007: keep Optional[X]; Typer evaluates CLI annotations at runtime and
|
|
97
|
+
# "X | None" fails there on Python 3.9.
|
|
98
|
+
ignore = ["E501", "UP006", "UP007", "UP035", "UP045", "B008", "SIM108"]
|
|
99
|
+
|
|
100
|
+
[tool.mypy]
|
|
101
|
+
python_version = "3.10"
|
|
102
|
+
ignore_missing_imports = true
|
|
103
|
+
warn_unused_ignores = true
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""
|
|
2
|
+
docling-pk: structured data extraction from Pakistani identity and education
|
|
3
|
+
documents.
|
|
4
|
+
|
|
5
|
+
Example:
|
|
6
|
+
>>> from docling_pk import extract
|
|
7
|
+
>>> result = extract("cnic.jpg", document_type="cnic") # doctest: +SKIP
|
|
8
|
+
>>> result.fields["cnic_number"].value # doctest: +SKIP
|
|
9
|
+
'35202-1234567-9'
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
__version__ = "1.0.0"
|
|
13
|
+
|
|
14
|
+
from docling_pk.errors import DoclingPKError, ImageLoadError, UnsupportedDocumentTypeError # noqa: E402
|
|
15
|
+
from docling_pk.extractor import SUPPORTED_TYPES, extract # noqa: E402
|
|
16
|
+
from docling_pk.schema import DocumentResult, FieldResult, FieldStatus # noqa: E402
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"extract",
|
|
20
|
+
"SUPPORTED_TYPES",
|
|
21
|
+
"DocumentResult",
|
|
22
|
+
"FieldResult",
|
|
23
|
+
"FieldStatus",
|
|
24
|
+
"DoclingPKError",
|
|
25
|
+
"ImageLoadError",
|
|
26
|
+
"UnsupportedDocumentTypeError",
|
|
27
|
+
"__version__",
|
|
28
|
+
]
|