docling-pk 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. docling_pk-1.0.0/LICENSE +21 -0
  2. docling_pk-1.0.0/PKG-INFO +274 -0
  3. docling_pk-1.0.0/README.md +225 -0
  4. docling_pk-1.0.0/pyproject.toml +103 -0
  5. docling_pk-1.0.0/setup.cfg +4 -0
  6. docling_pk-1.0.0/src/docling_pk/__init__.py +28 -0
  7. docling_pk-1.0.0/src/docling_pk/cli.py +197 -0
  8. docling_pk-1.0.0/src/docling_pk/errors.py +22 -0
  9. docling_pk-1.0.0/src/docling_pk/extractor.py +281 -0
  10. docling_pk-1.0.0/src/docling_pk/layout.py +146 -0
  11. docling_pk-1.0.0/src/docling_pk/ocr/__init__.py +5 -0
  12. docling_pk-1.0.0/src/docling_pk/ocr/base.py +130 -0
  13. docling_pk-1.0.0/src/docling_pk/ocr/easyocr_backend.py +84 -0
  14. docling_pk-1.0.0/src/docling_pk/ocr/rapidocr_backend.py +54 -0
  15. docling_pk-1.0.0/src/docling_pk/parsers/__init__.py +1 -0
  16. docling_pk-1.0.0/src/docling_pk/parsers/base.py +156 -0
  17. docling_pk-1.0.0/src/docling_pk/parsers/certificate.py +552 -0
  18. docling_pk-1.0.0/src/docling_pk/parsers/cnic.py +618 -0
  19. docling_pk-1.0.0/src/docling_pk/parsers/degree.py +217 -0
  20. docling_pk-1.0.0/src/docling_pk/pipeline.py +274 -0
  21. docling_pk-1.0.0/src/docling_pk/py.typed +0 -0
  22. docling_pk-1.0.0/src/docling_pk/schema.py +211 -0
  23. docling_pk-1.0.0/src/docling_pk/urdu.py +294 -0
  24. docling_pk-1.0.0/src/docling_pk/validation/__init__.py +1 -0
  25. docling_pk-1.0.0/src/docling_pk/validation/cnic.py +125 -0
  26. docling_pk-1.0.0/src/docling_pk/validation/dates.py +136 -0
  27. docling_pk-1.0.0/src/docling_pk/validation/marks.py +135 -0
  28. docling_pk-1.0.0/src/docling_pk/validation/text.py +76 -0
  29. docling_pk-1.0.0/src/docling_pk/vision/__init__.py +1 -0
  30. docling_pk-1.0.0/src/docling_pk/vision/geometry.py +103 -0
  31. docling_pk-1.0.0/src/docling_pk/vision/image_io.py +138 -0
  32. docling_pk-1.0.0/src/docling_pk/vision/quality.py +98 -0
  33. docling_pk-1.0.0/src/docling_pk.egg-info/PKG-INFO +274 -0
  34. docling_pk-1.0.0/src/docling_pk.egg-info/SOURCES.txt +43 -0
  35. docling_pk-1.0.0/src/docling_pk.egg-info/dependency_links.txt +1 -0
  36. docling_pk-1.0.0/src/docling_pk.egg-info/entry_points.txt +2 -0
  37. docling_pk-1.0.0/src/docling_pk.egg-info/requires.txt +24 -0
  38. docling_pk-1.0.0/src/docling_pk.egg-info/top_level.txt +1 -0
  39. docling_pk-1.0.0/tests/test_certificate_parser.py +239 -0
  40. docling_pk-1.0.0/tests/test_cnic_parser.py +230 -0
  41. docling_pk-1.0.0/tests/test_degree_urdu_layout.py +150 -0
  42. docling_pk-1.0.0/tests/test_integration_ocr.py +164 -0
  43. docling_pk-1.0.0/tests/test_pipeline_extractor_cli.py +435 -0
  44. docling_pk-1.0.0/tests/test_schema_and_io.py +241 -0
  45. docling_pk-1.0.0/tests/test_validation.py +161 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Abdul Moiz Muhammad, INFERENCE Lab
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,274 @@
1
+ Metadata-Version: 2.4
2
+ Name: docling-pk
3
+ Version: 1.0.0
4
+ Summary: Structured data extraction from Pakistani identity and education documents (CNIC, Matric, Intermediate, degrees).
5
+ Author: Abdul Moiz Muhammad, INFERENCE Lab
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/Inference-LAB/docling-pk
8
+ Project-URL: Repository, https://github.com/Inference-LAB/docling-pk
9
+ Project-URL: Issues, https://github.com/Inference-LAB/docling-pk/issues
10
+ Project-URL: Changelog, https://github.com/Inference-LAB/docling-pk/blob/main/CHANGELOG.md
11
+ Keywords: ocr,cnic,pakistan,kyc,document-ai,information-extraction,easyocr,urdu
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3 :: Only
18
+ Classifier: Programming Language :: Python :: 3.9
19
+ Classifier: Programming Language :: Python :: 3.10
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Topic :: Scientific/Engineering :: Image Recognition
23
+ Classifier: Topic :: Text Processing
24
+ Classifier: Typing :: Typed
25
+ Requires-Python: >=3.9
26
+ Description-Content-Type: text/markdown
27
+ License-File: LICENSE
28
+ Requires-Dist: numpy>=1.21
29
+ Requires-Dist: opencv-python>=4.5
30
+ Requires-Dist: rapidocr>=3.0
31
+ Requires-Dist: onnxruntime>=1.16
32
+ Requires-Dist: rapidfuzz>=3.0
33
+ Requires-Dist: typer>=0.9
34
+ Requires-Dist: Pillow>=9.0
35
+ Provides-Extra: urdu
36
+ Requires-Dist: easyocr>=1.7; extra == "urdu"
37
+ Provides-Extra: pdf
38
+ Requires-Dist: pymupdf>=1.23; extra == "pdf"
39
+ Provides-Extra: all
40
+ Requires-Dist: docling-pk[pdf,urdu]; extra == "all"
41
+ Provides-Extra: dev
42
+ Requires-Dist: pytest>=7; extra == "dev"
43
+ Requires-Dist: pytest-cov>=4; extra == "dev"
44
+ Requires-Dist: ruff>=0.5; extra == "dev"
45
+ Requires-Dist: mypy>=1.8; extra == "dev"
46
+ Requires-Dist: build>=1.0; extra == "dev"
47
+ Requires-Dist: twine>=5; extra == "dev"
48
+ Dynamic: license-file
49
+
50
+ # docling-pk
51
+
52
+ **Structured data extraction from Pakistani identity and education documents.**
53
+ Give it a photo of a CNIC, a Matric or Intermediate certificate, or a university
54
+ degree / transcript, and get back typed fields with confidence scores, plus a
55
+ reason for every field it could not read.
56
+
57
+ [![CI](https://github.com/Inference-LAB/docling-pk/actions/workflows/ci.yml/badge.svg)](https://github.com/Inference-LAB/docling-pk/actions/workflows/ci.yml)
58
+ ![Python](https://img.shields.io/badge/python-3.9%20%7C%203.10%20%7C%203.11%20%7C%203.12-blue)
59
+ ![License](https://img.shields.io/badge/license-MIT-green)
60
+
61
+ ```python
62
+ from docling_pk import extract
63
+
64
+ result = extract("cnic_front.jpg", document_type="cnic")
65
+ print(result.to_json())
66
+ ```
67
+
68
+ ```json
69
+ {
70
+ "document_type": "cnic",
71
+ "fields": {
72
+ "name": {"value": "Muhammad Ali", "confidence": 0.99, "status": "ok"},
73
+ "father_name": {"value": "Muhammad Akram", "confidence": 0.99, "status": "ok"},
74
+ "gender": {"value": "M", "confidence": 1.0, "status": "ok"},
75
+ "country_of_stay":{"value": "Pakistan", "confidence": 1.0, "status": "ok"},
76
+ "cnic_number": {"value": "35202-1234567-9", "confidence": 1.0, "status": "ok"},
77
+ "date_of_birth": {"value": "01-01-1995", "confidence": 1.0, "status": "ok"},
78
+ "date_of_issue": {"value": "15-03-2020", "confidence": 1.0, "status": "ok"},
79
+ "date_of_expiry": {"value": "15-03-2030", "confidence": 1.0, "status": "ok"},
80
+ "address": {"value": null, "confidence": 0.0, "status": "not_applicable",
81
+ "reason": "printed on the back of the card; pass the back image too"},
82
+ "permanent_address": {"value": null, "confidence": 0.0, "status": "not_applicable", "reason": "..."}
83
+ },
84
+ "confidence": 0.99,
85
+ "warnings": [],
86
+ "metadata": {"ocr_backend": "rapidocr", "sides": ["front"], "pages": [{"rotation_degrees": 90, "skew_degrees": 0.0}]}
87
+ }
88
+ ```
89
+
90
+ ## Why it is built this way
91
+
92
+ - **It does not return a wrong value as `ok`.** On the synthetic benchmark, every
93
+ field reported with status `ok` was correct; on 16 real photos, 99.2% were.
94
+ When a field cannot be read reliably, `value` is `None` and the result says
95
+ why (`not_found`, `invalid`, `low_confidence`, `not_applicable`). A blank
96
+ goes to a human. A wrong CNIC digit would quietly change someone's identity.
97
+ - **Cross-checks, not just OCR.** CNIC numbers and dates are re-read from a
98
+ tight crop by a second OCR engine. The three CNIC dates must satisfy
99
+ birth < issue < expiry. The CNIC's last digit must match the printed gender
100
+ (odd = male). Marks are checked against the TOTAL row, the subject rows'
101
+ sum, and the "marks in words" line.
102
+ - **Real phone photos.** It detects and fixes 90/180/270-degree rotation and
103
+ skew, filters security-paper micro-text, and warns about blur, glare and low
104
+ resolution instead of rejecting the image.
105
+ - **Light install.** The default engine is RapidOCR (PP-OCR models on ONNX
106
+ Runtime): about 370 MB installed, CPU-only, around 2.5 s per document. No
107
+ PyTorch is needed unless you want the Urdu address.
108
+
109
+ ## Installation
110
+
111
+ ```bash
112
+ pip install docling-pk # CNIC front, certificates, degrees (CPU, ~370 MB)
113
+ pip install "docling-pk[urdu]" # + Urdu address on CNIC backs, + second-engine verification (EasyOCR/PyTorch)
114
+ pip install "docling-pk[pdf]" # + PDF input (PyMuPDF)
115
+ pip install "docling-pk[all]"
116
+ ```
117
+
118
+ Python 3.9 to 3.12 on Linux, Windows and macOS. Models are bundled with or
119
+ downloaded by the OCR engines on first use; nothing is sent to any server.
120
+ On minimal Linux images (e.g. `python:3.x-slim` in Docker), OpenCV needs two
121
+ system libraries: `apt-get install -y libgl1 libglib2.0-0`.
122
+
123
+ ## Usage
124
+
125
+ ### Python
126
+
127
+ ```python
128
+ from docling_pk import extract
129
+
130
+ # CNIC: pass front and back together to also get the address
131
+ r = extract(["cnic_front.jpg", "cnic_back.jpg"], document_type="cnic")
132
+ r.fields["cnic_number"].value # '35202-1234567-9'
133
+ r.fields["date_of_birth"].as_date() # datetime.date(1995, 1, 1)
134
+ r["fields"]["name"]["value"] # dict-style access mirrors the JSON
135
+
136
+ # Board certificate: subject-wise marks come back as a table
137
+ r = extract("matric.jpg", document_type="matric")
138
+ r.fields["total_marks"].value # '787'
139
+ r.tables["subjects"][0] # {'subject': 'ENGLISH (COMPULSORY)', 'max_marks': 150, 'obtained_marks': 118}
140
+
141
+ # Let it figure out the document type
142
+ r = extract("unknown_scan.png") # document_type="auto"
143
+ r.document_type # 'intermediate'
144
+
145
+ # Inputs: path, bytes, NumPy array (BGR), PIL image, PDF, or a list of these
146
+ r = extract(open("card.jpg", "rb").read(), "cnic")
147
+
148
+ # Acting only on fields that passed every check
149
+ trusted = {k: f.value for k, f in r.fields.items() if f.status == "ok"}
150
+ needs_review = {k: f.reason for k, f in r.fields.items() if f.status != "ok"}
151
+ ```
152
+
153
+ Options: `backend="auto" | "rapidocr" | "easyocr"` (or your own
154
+ `OCRBackend`), `verify=True` (second-engine re-reads when EasyOCR is
155
+ installed), `urdu=True`, `gpu=None` (auto), `auto_rotate=True`,
156
+ `include_raw_text=True`.
157
+
158
+ ### Command line
159
+
160
+ ```bash
161
+ docling-pk extract cnic_front.jpg cnic_back.jpg --type cnic # JSON to stdout
162
+ docling-pk extract certificate.jpg --type auto --output table # readable table
163
+ docling-pk extract transcript.pdf --type degree --out result.json
164
+ docling-pk batch ./scans --type matric --out results.jsonl # a folder, one JSON line per file
165
+ docling-pk info # engines installed, GPU
166
+ ```
167
+
168
+ Exit codes: `0` success, `1` processed but nothing could be extracted, `2` bad
169
+ input (missing file, unreadable image, unknown type).
170
+
171
+ ## Supported documents
172
+
173
+ | `document_type` | Fields | Verified on real samples |
174
+ |---|---|---|
175
+ | `cnic` | `name`, `father_name`, `gender`, `country_of_stay`, `cnic_number`, `date_of_birth`, `date_of_issue`, `date_of_expiry`, `address`, `permanent_address` | Current green CNIC / SNIC, front and back |
176
+ | `matric` (`ssc`) | `student_name`, `father_name`, `date_of_birth`, `roll_number`, `registration_number`, `serial_number`, `certificate_number`, `board`, `exam`, `year`, `session`, `group`, `institute`, `grade`, `total_marks`, `max_marks`, plus `tables["subjects"]` | FBISE |
177
+ | `intermediate` (`hssc`, `fsc`, `inter`) | Same as matric | FBISE |
178
+ | `degree` (`transcript`) | `student_name`, `father_name`, `degree`, `institution`, `campus`, `registration_number`, `serial_number`, `cgpa`, `division`, `date_of_birth`, `date_of_issue`, `year` | COMSATS degree and transcript |
179
+
180
+ Every field key is always present for its document type. Matric and
181
+ intermediate are confirmed against the certificate title, which wins if it
182
+ disagrees with `document_type` (a warning says so). Other boards (BISE
183
+ Lahore, Rawalpindi, Karachi, Peshawar...) are detected and parsed with the
184
+ same label vocabulary, but come with a warning because their layouts have
185
+ not been verified on real samples yet.
186
+
187
+ ## Accuracy
188
+
189
+ Exact-match field accuracy. *Precision* is the share of returned values that
190
+ are correct; *ok-precision* is the same restricted to fields with status
191
+ `ok`, which is the number that matters if your system auto-accepts them.
192
+
193
+ | Dataset | Accuracy | Precision | ok-precision | Marks rows | Time / doc |
194
+ |---|---|---|---|---|---|
195
+ | 16 real photos, original v0.1 pipeline | 34.9% | n/a | n/a | 0 / 30 | n/a |
196
+ | 16 real photos, v1 (`pip install docling-pk`) | **94.4%** | 97.1% | 98.5% | 30 / 30 | 3.7 s |
197
+ | 16 real photos, v1 with the `urdu` extra | **94.4%** | 96.4% | **99.2%** | 30 / 30 | 3.7 s |
198
+ | 23 synthetic fixtures (committed) | **96.2%** | 100% | **100%** | 53 / 53 | 2.5 s (CPU) |
199
+
200
+ By document type on real photos: CNIC 91.2%, Matric 96.4%, Intermediate
201
+ 100%, Degree 100%, Transcript 90%. Every date and gender on the real CNICs was
202
+ correct. The misses: the Urdu address (both cards), three fields on a
203
+ motion-blurred photo (left blank, plus a misspelled name flagged
204
+ `low_confidence`), one father's name on a small soft-focus photo (left
205
+ blank), one registration number with an I/l confusion (flagged
206
+ `low_confidence`) and one dropped space in a school name. Full methodology
207
+ and per-field numbers:
208
+ [docs/benchmark.md](https://github.com/Inference-LAB/docling-pk/blob/main/docs/benchmark.md). The real photos contain personal data
209
+ and are not published; only aggregate counts are
210
+ ([benchmarks/results/real_samples_summary.json](https://github.com/Inference-LAB/docling-pk/blob/main/benchmarks/results/real_samples_summary.json)).
211
+
212
+ Reproduce the public benchmark:
213
+
214
+ ```bash
215
+ python benchmarks/run_benchmark.py tests/fixtures/synthetic/labels.json
216
+ ```
217
+
218
+ ## Known limitations
219
+
220
+ - **Urdu address (CNIC back) is experimental.** Nastaliq OCR with EasyOCR
221
+ averages about 49% character error on real cards, even after Urdu
222
+ normalization and lexicon correction of address words and district names.
223
+ It is always returned as `low_confidence`. UTRNet, the strongest open Urdu
224
+ recognizer we evaluated, is licensed CC BY-NC-SA (non-commercial), so it
225
+ cannot ship in an MIT library.
226
+ - **Handwritten fields, heavily damaged documents, and stamps covering more
227
+ than ~30% of a field** are out of scope (per the v1 brief). They return
228
+ `None` with a reason; they do not crash.
229
+ - **Old (pre-2012) Urdu-only CNICs**: numbers and dates are read; Urdu-only
230
+ names are not.
231
+ - **Board layouts other than FBISE** are parsed best-effort and flagged.
232
+ - **Severe motion blur** loses fields. The result is a blur warning and blank
233
+ fields, never invented values.
234
+ - **No NADRA verification.** docling-pk reads documents; it does not check
235
+ that they are genuine.
236
+
237
+ ## Privacy
238
+
239
+ Processing is fully local. `DocumentResult.to_dict()` and the CLI leave out
240
+ the raw OCR text by default, because it can contain personal data that is not
241
+ part of any extracted field (`include_raw=True` / `--raw` to include it). The
242
+ repository's test images are synthetic and stamped "SYNTHETIC SPECIMEN".
243
+
244
+ ## How it works
245
+
246
+ ```
247
+ image/PDF ─► load (Unicode paths, EXIF, PDF pages) ─► quality checks (blur, glare, exposure)
248
+ ─► orientation (0/90/180/270 by Latin-text score × box shape) ─► deskew from text-line angles
249
+ ─► OCR with positions (RapidOCR / EasyOCR) ─► micro-text filter (+ ink-only re-read if needed)
250
+ ─► per-type parser: labels found by fuzzy match, values by position
251
+ ─► validation & cross-checks (second engine, CNIC structure, date order, marks sums)
252
+ ─► DocumentResult (value, confidence, status, reason per field)
253
+ ```
254
+
255
+ Details and the decisions behind them, including the measurements that
256
+ replaced the brief's EasyOCR default: [docs/design_doc.md](https://github.com/Inference-LAB/docling-pk/blob/main/docs/design_doc.md).
257
+
258
+ ## Development
259
+
260
+ ```bash
261
+ pip install -e ".[dev,pdf]"
262
+ pytest -m "not ocr" # 211 fast tests, no models needed
263
+ pytest # + end-to-end OCR tests on synthetic fixtures
264
+ ruff check . && ruff format --check . && mypy src
265
+ python benchmarks/synthetic/generate.py # regenerate the synthetic fixtures
266
+ ```
267
+
268
+ See [CONTRIBUTING.md](https://github.com/Inference-LAB/docling-pk/blob/main/CONTRIBUTING.md). Releases are published to PyPI from
269
+ version tags by GitHub Actions (trusted publishing).
270
+
271
+ ## License
272
+
273
+ MIT. See [LICENSE](https://github.com/Inference-LAB/docling-pk/blob/main/LICENSE). Built at [INFERENCE Lab](https://inference-lab.org)
274
+ as part of Engineering Fellowship Cohort 01.
@@ -0,0 +1,225 @@
1
+ # docling-pk
2
+
3
+ **Structured data extraction from Pakistani identity and education documents.**
4
+ Give it a photo of a CNIC, a Matric or Intermediate certificate, or a university
5
+ degree / transcript, and get back typed fields with confidence scores, plus a
6
+ reason for every field it could not read.
7
+
8
+ [![CI](https://github.com/Inference-LAB/docling-pk/actions/workflows/ci.yml/badge.svg)](https://github.com/Inference-LAB/docling-pk/actions/workflows/ci.yml)
9
+ ![Python](https://img.shields.io/badge/python-3.9%20%7C%203.10%20%7C%203.11%20%7C%203.12-blue)
10
+ ![License](https://img.shields.io/badge/license-MIT-green)
11
+
12
+ ```python
13
+ from docling_pk import extract
14
+
15
+ result = extract("cnic_front.jpg", document_type="cnic")
16
+ print(result.to_json())
17
+ ```
18
+
19
+ ```json
20
+ {
21
+ "document_type": "cnic",
22
+ "fields": {
23
+ "name": {"value": "Muhammad Ali", "confidence": 0.99, "status": "ok"},
24
+ "father_name": {"value": "Muhammad Akram", "confidence": 0.99, "status": "ok"},
25
+ "gender": {"value": "M", "confidence": 1.0, "status": "ok"},
26
+ "country_of_stay":{"value": "Pakistan", "confidence": 1.0, "status": "ok"},
27
+ "cnic_number": {"value": "35202-1234567-9", "confidence": 1.0, "status": "ok"},
28
+ "date_of_birth": {"value": "01-01-1995", "confidence": 1.0, "status": "ok"},
29
+ "date_of_issue": {"value": "15-03-2020", "confidence": 1.0, "status": "ok"},
30
+ "date_of_expiry": {"value": "15-03-2030", "confidence": 1.0, "status": "ok"},
31
+ "address": {"value": null, "confidence": 0.0, "status": "not_applicable",
32
+ "reason": "printed on the back of the card; pass the back image too"},
33
+ "permanent_address": {"value": null, "confidence": 0.0, "status": "not_applicable", "reason": "..."}
34
+ },
35
+ "confidence": 0.99,
36
+ "warnings": [],
37
+ "metadata": {"ocr_backend": "rapidocr", "sides": ["front"], "pages": [{"rotation_degrees": 90, "skew_degrees": 0.0}]}
38
+ }
39
+ ```
40
+
41
+ ## Why it is built this way
42
+
43
+ - **It does not return a wrong value as `ok`.** On the synthetic benchmark, every
44
+ field reported with status `ok` was correct; on 16 real photos, 99.2% were.
45
+ When a field cannot be read reliably, `value` is `None` and the result says
46
+ why (`not_found`, `invalid`, `low_confidence`, `not_applicable`). A blank
47
+ goes to a human. A wrong CNIC digit would quietly change someone's identity.
48
+ - **Cross-checks, not just OCR.** CNIC numbers and dates are re-read from a
49
+ tight crop by a second OCR engine. The three CNIC dates must satisfy
50
+ birth < issue < expiry. The CNIC's last digit must match the printed gender
51
+ (odd = male). Marks are checked against the TOTAL row, the subject rows'
52
+ sum, and the "marks in words" line.
53
+ - **Real phone photos.** It detects and fixes 90/180/270-degree rotation and
54
+ skew, filters security-paper micro-text, and warns about blur, glare and low
55
+ resolution instead of rejecting the image.
56
+ - **Light install.** The default engine is RapidOCR (PP-OCR models on ONNX
57
+ Runtime): about 370 MB installed, CPU-only, around 2.5 s per document. No
58
+ PyTorch is needed unless you want the Urdu address.
59
+
60
+ ## Installation
61
+
62
+ ```bash
63
+ pip install docling-pk # CNIC front, certificates, degrees (CPU, ~370 MB)
64
+ pip install "docling-pk[urdu]" # + Urdu address on CNIC backs, + second-engine verification (EasyOCR/PyTorch)
65
+ pip install "docling-pk[pdf]" # + PDF input (PyMuPDF)
66
+ pip install "docling-pk[all]"
67
+ ```
68
+
69
+ Python 3.9 to 3.12 on Linux, Windows and macOS. Models are bundled with or
70
+ downloaded by the OCR engines on first use; nothing is sent to any server.
71
+ On minimal Linux images (e.g. `python:3.x-slim` in Docker), OpenCV needs two
72
+ system libraries: `apt-get install -y libgl1 libglib2.0-0`.
73
+
74
+ ## Usage
75
+
76
+ ### Python
77
+
78
+ ```python
79
+ from docling_pk import extract
80
+
81
+ # CNIC: pass front and back together to also get the address
82
+ r = extract(["cnic_front.jpg", "cnic_back.jpg"], document_type="cnic")
83
+ r.fields["cnic_number"].value # '35202-1234567-9'
84
+ r.fields["date_of_birth"].as_date() # datetime.date(1995, 1, 1)
85
+ r["fields"]["name"]["value"] # dict-style access mirrors the JSON
86
+
87
+ # Board certificate: subject-wise marks come back as a table
88
+ r = extract("matric.jpg", document_type="matric")
89
+ r.fields["total_marks"].value # '787'
90
+ r.tables["subjects"][0] # {'subject': 'ENGLISH (COMPULSORY)', 'max_marks': 150, 'obtained_marks': 118}
91
+
92
+ # Let it figure out the document type
93
+ r = extract("unknown_scan.png") # document_type="auto"
94
+ r.document_type # 'intermediate'
95
+
96
+ # Inputs: path, bytes, NumPy array (BGR), PIL image, PDF, or a list of these
97
+ r = extract(open("card.jpg", "rb").read(), "cnic")
98
+
99
+ # Acting only on fields that passed every check
100
+ trusted = {k: f.value for k, f in r.fields.items() if f.status == "ok"}
101
+ needs_review = {k: f.reason for k, f in r.fields.items() if f.status != "ok"}
102
+ ```
103
+
104
+ Options: `backend="auto" | "rapidocr" | "easyocr"` (or your own
105
+ `OCRBackend`), `verify=True` (second-engine re-reads when EasyOCR is
106
+ installed), `urdu=True`, `gpu=None` (auto), `auto_rotate=True`,
107
+ `include_raw_text=True`.
108
+
109
+ ### Command line
110
+
111
+ ```bash
112
+ docling-pk extract cnic_front.jpg cnic_back.jpg --type cnic # JSON to stdout
113
+ docling-pk extract certificate.jpg --type auto --output table # readable table
114
+ docling-pk extract transcript.pdf --type degree --out result.json
115
+ docling-pk batch ./scans --type matric --out results.jsonl # a folder, one JSON line per file
116
+ docling-pk info # engines installed, GPU
117
+ ```
118
+
119
+ Exit codes: `0` success, `1` processed but nothing could be extracted, `2` bad
120
+ input (missing file, unreadable image, unknown type).
121
+
122
+ ## Supported documents
123
+
124
+ | `document_type` | Fields | Verified on real samples |
125
+ |---|---|---|
126
+ | `cnic` | `name`, `father_name`, `gender`, `country_of_stay`, `cnic_number`, `date_of_birth`, `date_of_issue`, `date_of_expiry`, `address`, `permanent_address` | Current green CNIC / SNIC, front and back |
127
+ | `matric` (`ssc`) | `student_name`, `father_name`, `date_of_birth`, `roll_number`, `registration_number`, `serial_number`, `certificate_number`, `board`, `exam`, `year`, `session`, `group`, `institute`, `grade`, `total_marks`, `max_marks`, plus `tables["subjects"]` | FBISE |
128
+ | `intermediate` (`hssc`, `fsc`, `inter`) | Same as matric | FBISE |
129
+ | `degree` (`transcript`) | `student_name`, `father_name`, `degree`, `institution`, `campus`, `registration_number`, `serial_number`, `cgpa`, `division`, `date_of_birth`, `date_of_issue`, `year` | COMSATS degree and transcript |
130
+
131
+ Every field key is always present for its document type. Matric and
132
+ intermediate are confirmed against the certificate title, which wins if it
133
+ disagrees with `document_type` (a warning says so). Other boards (BISE
134
+ Lahore, Rawalpindi, Karachi, Peshawar...) are detected and parsed with the
135
+ same label vocabulary, but come with a warning because their layouts have
136
+ not been verified on real samples yet.
137
+
138
+ ## Accuracy
139
+
140
+ Exact-match field accuracy. *Precision* is the share of returned values that
141
+ are correct; *ok-precision* is the same restricted to fields with status
142
+ `ok`, which is the number that matters if your system auto-accepts them.
143
+
144
+ | Dataset | Accuracy | Precision | ok-precision | Marks rows | Time / doc |
145
+ |---|---|---|---|---|---|
146
+ | 16 real photos, original v0.1 pipeline | 34.9% | n/a | n/a | 0 / 30 | n/a |
147
+ | 16 real photos, v1 (`pip install docling-pk`) | **94.4%** | 97.1% | 98.5% | 30 / 30 | 3.7 s |
148
+ | 16 real photos, v1 with the `urdu` extra | **94.4%** | 96.4% | **99.2%** | 30 / 30 | 3.7 s |
149
+ | 23 synthetic fixtures (committed) | **96.2%** | 100% | **100%** | 53 / 53 | 2.5 s (CPU) |
150
+
151
+ By document type on real photos: CNIC 91.2%, Matric 96.4%, Intermediate
152
+ 100%, Degree 100%, Transcript 90%. Every date and gender on the real CNICs was
153
+ correct. The misses: the Urdu address (both cards), three fields on a
154
+ motion-blurred photo (left blank, plus a misspelled name flagged
155
+ `low_confidence`), one father's name on a small soft-focus photo (left
156
+ blank), one registration number with an I/l confusion (flagged
157
+ `low_confidence`) and one dropped space in a school name. Full methodology
158
+ and per-field numbers:
159
+ [docs/benchmark.md](https://github.com/Inference-LAB/docling-pk/blob/main/docs/benchmark.md). The real photos contain personal data
160
+ and are not published; only aggregate counts are
161
+ ([benchmarks/results/real_samples_summary.json](https://github.com/Inference-LAB/docling-pk/blob/main/benchmarks/results/real_samples_summary.json)).
162
+
163
+ Reproduce the public benchmark:
164
+
165
+ ```bash
166
+ python benchmarks/run_benchmark.py tests/fixtures/synthetic/labels.json
167
+ ```
168
+
169
+ ## Known limitations
170
+
171
+ - **Urdu address (CNIC back) is experimental.** Nastaliq OCR with EasyOCR
172
+ averages about 49% character error on real cards, even after Urdu
173
+ normalization and lexicon correction of address words and district names.
174
+ It is always returned as `low_confidence`. UTRNet, the strongest open Urdu
175
+ recognizer we evaluated, is licensed CC BY-NC-SA (non-commercial), so it
176
+ cannot ship in an MIT library.
177
+ - **Handwritten fields, heavily damaged documents, and stamps covering more
178
+ than ~30% of a field** are out of scope (per the v1 brief). They return
179
+ `None` with a reason; they do not crash.
180
+ - **Old (pre-2012) Urdu-only CNICs**: numbers and dates are read; Urdu-only
181
+ names are not.
182
+ - **Board layouts other than FBISE** are parsed best-effort and flagged.
183
+ - **Severe motion blur** loses fields. The result is a blur warning and blank
184
+ fields, never invented values.
185
+ - **No NADRA verification.** docling-pk reads documents; it does not check
186
+ that they are genuine.
187
+
188
+ ## Privacy
189
+
190
+ Processing is fully local. `DocumentResult.to_dict()` and the CLI leave out
191
+ the raw OCR text by default, because it can contain personal data that is not
192
+ part of any extracted field (`include_raw=True` / `--raw` to include it). The
193
+ repository's test images are synthetic and stamped "SYNTHETIC SPECIMEN".
194
+
195
+ ## How it works
196
+
197
+ ```
198
+ image/PDF ─► load (Unicode paths, EXIF, PDF pages) ─► quality checks (blur, glare, exposure)
199
+ ─► orientation (0/90/180/270 by Latin-text score × box shape) ─► deskew from text-line angles
200
+ ─► OCR with positions (RapidOCR / EasyOCR) ─► micro-text filter (+ ink-only re-read if needed)
201
+ ─► per-type parser: labels found by fuzzy match, values by position
202
+ ─► validation & cross-checks (second engine, CNIC structure, date order, marks sums)
203
+ ─► DocumentResult (value, confidence, status, reason per field)
204
+ ```
205
+
206
+ Details and the decisions behind them, including the measurements that
207
+ replaced the brief's EasyOCR default: [docs/design_doc.md](https://github.com/Inference-LAB/docling-pk/blob/main/docs/design_doc.md).
208
+
209
+ ## Development
210
+
211
+ ```bash
212
+ pip install -e ".[dev,pdf]"
213
+ pytest -m "not ocr" # 211 fast tests, no models needed
214
+ pytest # + end-to-end OCR tests on synthetic fixtures
215
+ ruff check . && ruff format --check . && mypy src
216
+ python benchmarks/synthetic/generate.py # regenerate the synthetic fixtures
217
+ ```
218
+
219
+ See [CONTRIBUTING.md](https://github.com/Inference-LAB/docling-pk/blob/main/CONTRIBUTING.md). Releases are published to PyPI from
220
+ version tags by GitHub Actions (trusted publishing).
221
+
222
+ ## License
223
+
224
+ MIT. See [LICENSE](https://github.com/Inference-LAB/docling-pk/blob/main/LICENSE). Built at [INFERENCE Lab](https://inference-lab.org)
225
+ as part of Engineering Fellowship Cohort 01.
@@ -0,0 +1,103 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "docling-pk"
7
+ dynamic = ["version"]
8
+ description = "Structured data extraction from Pakistani identity and education documents (CNIC, Matric, Intermediate, degrees)."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Abdul Moiz Muhammad" }, { name = "INFERENCE Lab" }]
14
+ keywords = ["ocr", "cnic", "pakistan", "kyc", "document-ai", "information-extraction", "easyocr", "urdu"]
15
+ classifiers = [
16
+ "Development Status :: 4 - Beta",
17
+ "Intended Audience :: Developers",
18
+ "Intended Audience :: Science/Research",
19
+ "Operating System :: OS Independent",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3 :: Only",
22
+ "Programming Language :: Python :: 3.9",
23
+ "Programming Language :: Python :: 3.10",
24
+ "Programming Language :: Python :: 3.11",
25
+ "Programming Language :: Python :: 3.12",
26
+ "Topic :: Scientific/Engineering :: Image Recognition",
27
+ "Topic :: Text Processing",
28
+ "Typing :: Typed",
29
+ ]
30
+ dependencies = [
31
+ "numpy>=1.21",
32
+ "opencv-python>=4.5", # same distribution RapidOCR requires, so only one cv2 is installed
33
+ "rapidocr>=3.0",
34
+ "onnxruntime>=1.16",
35
+ "rapidfuzz>=3.0",
36
+ "typer>=0.9",
37
+ "Pillow>=9.0",
38
+ ]
39
+
40
+ [project.optional-dependencies]
41
+ # EasyOCR (PyTorch) adds Urdu address reading and an independent second
42
+ # engine that cross-checks CNIC numbers, dates and IDs.
43
+ urdu = ["easyocr>=1.7"]
44
+ pdf = ["pymupdf>=1.23"]
45
+ all = ["docling-pk[urdu,pdf]"]
46
+ dev = [
47
+ "pytest>=7",
48
+ "pytest-cov>=4",
49
+ "ruff>=0.5",
50
+ "mypy>=1.8",
51
+ "build>=1.0",
52
+ "twine>=5",
53
+ ]
54
+
55
+ [project.scripts]
56
+ docling-pk = "docling_pk.cli:app"
57
+
58
+ [project.urls]
59
+ Homepage = "https://github.com/Inference-LAB/docling-pk"
60
+ Repository = "https://github.com/Inference-LAB/docling-pk"
61
+ Issues = "https://github.com/Inference-LAB/docling-pk/issues"
62
+ Changelog = "https://github.com/Inference-LAB/docling-pk/blob/main/CHANGELOG.md"
63
+
64
+ [tool.setuptools.dynamic]
65
+ version = { attr = "docling_pk.__version__" }
66
+
67
+ [tool.setuptools.packages.find]
68
+ where = ["src"]
69
+
70
+ [tool.setuptools.package-data]
71
+ docling_pk = ["py.typed"]
72
+
73
+ [tool.pytest.ini_options]
74
+ testpaths = ["tests"]
75
+ addopts = "-ra --strict-markers"
76
+ markers = [
77
+ "ocr: needs the EasyOCR models (downloaded on first run); deselect with -m 'not ocr'",
78
+ ]
79
+
80
+ [tool.coverage.run]
81
+ source = ["docling_pk"]
82
+ branch = true
83
+
84
+ [tool.coverage.report]
85
+ show_missing = true
86
+ skip_covered = false
87
+ exclude_lines = ["pragma: no cover", "if __name__ == .__main__.:", "raise NotImplementedError"]
88
+
89
+ [tool.ruff]
90
+ line-length = 120
91
+ target-version = "py39"
92
+ src = ["src", "tests"]
93
+
94
+ [tool.ruff.lint]
95
+ select = ["E", "F", "W", "I", "B", "UP", "SIM"]
96
+ # UP045/UP007: keep Optional[X]; Typer evaluates CLI annotations at runtime and
97
+ # "X | None" fails there on Python 3.9.
98
+ ignore = ["E501", "UP006", "UP007", "UP035", "UP045", "B008", "SIM108"]
99
+
100
+ [tool.mypy]
101
+ python_version = "3.10"
102
+ ignore_missing_imports = true
103
+ warn_unused_ignores = true
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,28 @@
1
+ """
2
+ docling-pk: structured data extraction from Pakistani identity and education
3
+ documents.
4
+
5
+ Example:
6
+ >>> from docling_pk import extract
7
+ >>> result = extract("cnic.jpg", document_type="cnic") # doctest: +SKIP
8
+ >>> result.fields["cnic_number"].value # doctest: +SKIP
9
+ '35202-1234567-9'
10
+ """
11
+
12
+ __version__ = "1.0.0"
13
+
14
+ from docling_pk.errors import DoclingPKError, ImageLoadError, UnsupportedDocumentTypeError # noqa: E402
15
+ from docling_pk.extractor import SUPPORTED_TYPES, extract # noqa: E402
16
+ from docling_pk.schema import DocumentResult, FieldResult, FieldStatus # noqa: E402
17
+
18
+ __all__ = [
19
+ "extract",
20
+ "SUPPORTED_TYPES",
21
+ "DocumentResult",
22
+ "FieldResult",
23
+ "FieldStatus",
24
+ "DoclingPKError",
25
+ "ImageLoadError",
26
+ "UnsupportedDocumentTypeError",
27
+ "__version__",
28
+ ]