parisaocr 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. parisaocr-0.2.0/LICENSE +73 -0
  2. parisaocr-0.2.0/NOTICE +23 -0
  3. parisaocr-0.2.0/PKG-INFO +223 -0
  4. parisaocr-0.2.0/README.md +190 -0
  5. parisaocr-0.2.0/parisaocr/__init__.py +27 -0
  6. parisaocr-0.2.0/parisaocr/__main__.py +3 -0
  7. parisaocr-0.2.0/parisaocr/api.py +54 -0
  8. parisaocr-0.2.0/parisaocr/bidi.py +37 -0
  9. parisaocr-0.2.0/parisaocr/cli.py +256 -0
  10. parisaocr-0.2.0/parisaocr/cut.py +113 -0
  11. parisaocr-0.2.0/parisaocr/detect.py +87 -0
  12. parisaocr-0.2.0/parisaocr/detectors.py +117 -0
  13. parisaocr-0.2.0/parisaocr/fa_text.py +72 -0
  14. parisaocr-0.2.0/parisaocr/fonts/OFL-Vazirmatn.txt +93 -0
  15. parisaocr-0.2.0/parisaocr/fonts/Vazirmatn-Regular.ttf +0 -0
  16. parisaocr-0.2.0/parisaocr/kraken_reader.py +195 -0
  17. parisaocr-0.2.0/parisaocr/models/parisaocr-fa-0.1.safetensors +0 -0
  18. parisaocr-0.2.0/parisaocr/models/ppocrv6-det-small.onnx +0 -0
  19. parisaocr-0.2.0/parisaocr/order.py +29 -0
  20. parisaocr-0.2.0/parisaocr/output.py +59 -0
  21. parisaocr-0.2.0/parisaocr/pages.py +211 -0
  22. parisaocr-0.2.0/parisaocr/pdfout.py +163 -0
  23. parisaocr-0.2.0/parisaocr/pipeline.py +206 -0
  24. parisaocr-0.2.0/parisaocr/reader.py +62 -0
  25. parisaocr-0.2.0/parisaocr.egg-info/PKG-INFO +223 -0
  26. parisaocr-0.2.0/parisaocr.egg-info/SOURCES.txt +33 -0
  27. parisaocr-0.2.0/parisaocr.egg-info/dependency_links.txt +1 -0
  28. parisaocr-0.2.0/parisaocr.egg-info/entry_points.txt +2 -0
  29. parisaocr-0.2.0/parisaocr.egg-info/requires.txt +10 -0
  30. parisaocr-0.2.0/parisaocr.egg-info/top_level.txt +1 -0
  31. parisaocr-0.2.0/pyproject.toml +50 -0
  32. parisaocr-0.2.0/setup.cfg +4 -0
  33. parisaocr-0.2.0/tests/test_ocr.py +33 -0
  34. parisaocr-0.2.0/tests/test_pdf.py +61 -0
  35. parisaocr-0.2.0/tests/test_text.py +32 -0
@@ -0,0 +1,73 @@
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction, and distribution as defined by Sections 1 through 9 of this document.
10
+
11
+ "Licensor" shall mean the copyright owner or entity authorized by the copyright owner that is granting the License.
12
+
13
+ "Legal Entity" shall mean the union of the acting entity and all other entities that control, are controlled by, or are under common control with that entity. For the purposes of this definition, "control" means (i) the power, direct or indirect, to cause the direction or management of such entity, whether by contract or otherwise, or (ii) ownership of fifty percent (50%) or more of the outstanding shares, or (iii) beneficial ownership of such entity.
14
+
15
+ "You" (or "Your") shall mean an individual or Legal Entity exercising permissions granted by this License.
16
+
17
+ "Source" form shall mean the preferred form for making modifications, including but not limited to software source code, documentation source, and configuration files.
18
+
19
+ "Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types.
20
+
21
+ "Work" shall mean the work of authorship, whether in Source or Object form, made available under the License, as indicated by a copyright notice that is included in or attached to the work (an example is provided in the Appendix below).
22
+
23
+ "Derivative Works" shall mean any work, whether in Source or Object form, that is based on (or derived from) the Work and for which the editorial revisions, annotations, elaborations, or other modifications represent, as a whole, an original work of authorship. For the purposes of this License, Derivative Works shall not include works that remain separable from, or merely link (or bind by name) to the interfaces of, the Work and Derivative Works thereof.
24
+
25
+ "Contribution" shall mean any work of authorship, including the original version of the Work and any modifications or additions to that Work or Derivative Works thereof, that is intentionally submitted to Licensor for inclusion in the Work by the copyright owner or by an individual or Legal Entity authorized to submit on behalf of the copyright owner. For the purposes of this definition, "submitted" means any form of electronic, verbal, or written communication sent to the Licensor or its representatives, including but not limited to communication on electronic mailing lists, source code control systems, and issue tracking systems that are managed by, or on behalf of, the Licensor for the purpose of discussing and improving the Work, but excluding communication that is conspicuously marked or otherwise designated in writing by the copyright owner as "Not a Contribution."
26
+
27
+ "Contributor" shall mean Licensor and any individual or Legal Entity on behalf of whom a Contribution has been received by Licensor and subsequently incorporated within the Work.
28
+
29
+ 2. Grant of Copyright License. Subject to the terms and conditions of this License, each Contributor hereby grants to You a perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable copyright license to reproduce, prepare Derivative Works of, publicly display, publicly perform, sublicense, and distribute the Work and such Derivative Works in Source or Object form.
30
+
31
+ 3. Grant of Patent License. Subject to the terms and conditions of this License, each Contributor hereby grants to You a perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable (except as stated in this section) patent license to make, have made, use, offer to sell, sell, import, and otherwise transfer the Work, where such license applies only to those patent claims licensable by such Contributor that are necessarily infringed by their Contribution(s) alone or by combination of their Contribution(s) with the Work to which such Contribution(s) was submitted. If You institute patent litigation against any entity (including a cross-claim or counterclaim in a lawsuit) alleging that the Work or a Contribution incorporated within the Work constitutes direct or contributory patent infringement, then any patent licenses granted to You under this License for that Work shall terminate as of the date such litigation is filed.
32
+
33
+ 4. Redistribution. You may reproduce and distribute copies of the Work or Derivative Works thereof in any medium, with or without modifications, and in Source or Object form, provided that You meet the following conditions:
34
+
35
+ (a) You must give any other recipients of the Work or Derivative Works a copy of this License; and
36
+
37
+ (b) You must cause any modified files to carry prominent notices stating that You changed the files; and
38
+
39
+ (c) You must retain, in the Source form of any Derivative Works that You distribute, all copyright, patent, trademark, and attribution notices from the Source form of the Work, excluding those notices that do not pertain to any part of the Derivative Works; and
40
+
41
+ (d) If the Work includes a "NOTICE" text file as part of its distribution, then any Derivative Works that You distribute must include a readable copy of the attribution notices contained within such NOTICE file, excluding those notices that do not pertain to any part of the Derivative Works, in at least one of the following places: within a NOTICE text file distributed as part of the Derivative Works; within the Source form or documentation, if provided along with the Derivative Works; or, within a display generated by the Derivative Works, if and wherever such third-party notices normally appear. The contents of the NOTICE file are for informational purposes only and do not modify the License. You may add Your own attribution notices within Derivative Works that You distribute, alongside or as an addendum to the NOTICE text from the Work, provided that such additional attribution notices cannot be construed as modifying the License.
42
+
43
+ You may add Your own copyright statement to Your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of Your modifications, or for any such Derivative Works as a whole, provided Your use, reproduction, and distribution of the Work otherwise complies with the conditions stated in this License.
44
+
45
+ 5. Submission of Contributions. Unless You explicitly state otherwise, any Contribution intentionally submitted for inclusion in the Work by You to the Licensor shall be under the terms and conditions of this License, without any additional terms or conditions. Notwithstanding the above, nothing herein shall supersede or modify the terms of any separate license agreement you may have executed with Licensor regarding such Contributions.
46
+
47
+ 6. Trademarks. This License does not grant permission to use the trade names, trademarks, service marks, or product names of the Licensor, except as required for reasonable and customary use in describing the origin of the Work and reproducing the content of the NOTICE file.
48
+
49
+ 7. Disclaimer of Warranty. Unless required by applicable law or agreed to in writing, Licensor provides the Work (and each Contributor provides its Contributions) on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied, including, without limitation, any warranties or conditions of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A PARTICULAR PURPOSE. You are solely responsible for determining the appropriateness of using or redistributing the Work and assume any risks associated with Your exercise of permissions under this License.
50
+
51
+ 8. Limitation of Liability. In no event and under no legal theory, whether in tort (including negligence), contract, or otherwise, unless required by applicable law (such as deliberate and grossly negligent acts) or agreed to in writing, shall any Contributor be liable to You for damages, including any direct, indirect, special, incidental, or consequential damages of any character arising as a result of this License or out of the use or inability to use the Work (including but not limited to damages for loss of goodwill, work stoppage, computer failure or malfunction, or any and all other commercial damages or losses), even if such Contributor has been advised of the possibility of such damages.
52
+
53
+ 9. Accepting Warranty or Additional Liability. While redistributing the Work or Derivative Works thereof, You may choose to offer, and charge a fee for, acceptance of support, warranty, indemnity, or other liability obligations and/or rights consistent with this License. However, in accepting such obligations, You may act only on Your own behalf and on Your sole responsibility, not on behalf of any other Contributor, and only if You agree to indemnify, defend, and hold each Contributor harmless for any liability incurred by, or claims asserted against, such Contributor by reason of your accepting any such warranty or additional liability.
54
+
55
+ END OF TERMS AND CONDITIONS
56
+
57
+ APPENDIX: How to apply the Apache License to your work.
58
+
59
+ To apply the Apache License to your work, attach the following boilerplate notice, with the fields enclosed by brackets "[]" replaced with your own identifying information. (Don't include the brackets!) The text should be enclosed in the appropriate comment syntax for the file format. We also recommend that a file or class name and description of purpose be included on the same "printed page" as the copyright notice for easier identification within third-party archives.
60
+
61
+ Copyright [yyyy] [name of copyright owner]
62
+
63
+ Licensed under the Apache License, Version 2.0 (the "License");
64
+ you may not use this file except in compliance with the License.
65
+ You may obtain a copy of the License at
66
+
67
+ http://www.apache.org/licenses/LICENSE-2.0
68
+
69
+ Unless required by applicable law or agreed to in writing, software
70
+ distributed under the License is distributed on an "AS IS" BASIS,
71
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
72
+ See the License for the specific language governing permissions and
73
+ limitations under the License.
parisaocr-0.2.0/NOTICE ADDED
@@ -0,0 +1,23 @@
1
+ ParisaOCR
2
+ Copyright 2026 Ali Javadi
3
+
4
+ This product is licensed under the Apache License, Version 2.0 (see LICENSE).
5
+
6
+ It includes:
7
+
8
+ - parisaocr/models/ppocrv6-det-small.onnx: the PP-OCRv6 small text detection
9
+ model of PaddleOCR (https://github.com/PaddlePaddle/PaddleOCR), Copyright
10
+ PaddlePaddle Authors, licensed under the Apache License, Version 2.0, in the
11
+ ONNX conversion distributed by RapidOCR (https://github.com/RapidAI/RapidOCR),
12
+ also under the Apache License, Version 2.0.
13
+
14
+ - parisaocr/models/parisaocr-fa-0.1.safetensors: a text recognition model
15
+ trained with Kraken (https://github.com/mittagessen/kraken, Apache License,
16
+ Version 2.0). See MODEL_CARD.md for its training data, including synthetic
17
+ lines rendered from Persian and English Wikipedia text (CC BY-SA 4.0).
18
+
19
+ - parisaocr/fonts/Vazirmatn-Regular.ttf: the Vazirmatn font
20
+ (https://github.com/rastikerdar/vazirmatn), Copyright 2015 The Vazirmatn
21
+ Project Authors, licensed under the SIL Open Font License 1.1
22
+ (parisaocr/fonts/OFL-Vazirmatn.txt). It is embedded, subset, in the
23
+ invisible text layer of the searchable PDFs ParisaOCR writes.
@@ -0,0 +1,223 @@
1
+ Metadata-Version: 2.4
2
+ Name: parisaocr
3
+ Version: 0.2.0
4
+ Summary: Open-source OCR for printed Persian: PP-OCR line detection and a Kraken recognizer trained on Persian books
5
+ Author: Ali Javadi
6
+ License-Expression: Apache-2.0
7
+ Project-URL: Homepage, https://github.com/givia/ParisaOCR
8
+ Project-URL: Issues, https://github.com/givia/ParisaOCR/issues
9
+ Project-URL: Model, https://huggingface.co/givia/ParisaOCR
10
+ Keywords: ocr,persian,farsi,text recognition,kraken,rtl
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Natural Language :: Persian
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Topic :: Scientific/Engineering :: Image Recognition
18
+ Classifier: Topic :: Text Processing :: Linguistic
19
+ Requires-Python: <3.14,>=3.10
20
+ Description-Content-Type: text/markdown
21
+ License-File: LICENSE
22
+ License-File: NOTICE
23
+ Requires-Dist: pillow
24
+ Requires-Dist: numpy
25
+ Requires-Dist: rapidocr>=3.9
26
+ Requires-Dist: onnxruntime
27
+ Requires-Dist: kraken==7.1.1
28
+ Requires-Dist: reportlab
29
+ Requires-Dist: pikepdf
30
+ Provides-Extra: surya
31
+ Requires-Dist: surya-ocr; extra == "surya"
32
+ Dynamic: license-file
33
+
34
+ # ParisaOCR (پریسا اوسی‌آر)
35
+
36
+ Open-source OCR for printed Persian. It reads scanned book pages about as well
37
+ as Google's Gemini Flash models, runs on your own machine (CPU or GPU, no
38
+ internet needed), and costs a few cents per thousand pages to run.
39
+
40
+ **Status: preview.** It works well on printed Persian books and
41
+ documents; see [Known limitations](#known-limitations). Feedback with example
42
+ pages is very welcome.
43
+
44
+ [فارسی](#فارسی) · [Model on Hugging Face](https://huggingface.co/givia/ParisaOCR) · [PyPI](https://pypi.org/project/parisaocr/)
45
+
46
+ ## Install
47
+
48
+ ```
49
+ pip install parisaocr
50
+ ```
51
+
52
+ Python 3.10 to 3.13. The two models (text-line detector and recognizer, 22 MB)
53
+ are included. PyTorch is installed as a dependency of Kraken; a GPU is used
54
+ when one is available, otherwise the CPU (about 5 to 10 seconds per page).
55
+ For PDF input, Poppler's command-line tools must be installed
56
+ (`poppler-utils` on Debian and Ubuntu, `poppler` on Arch, Homebrew and conda-forge).
57
+
58
+ ## Use
59
+
60
+ ```
61
+ parisaocr ocr page.png # print the text
62
+ parisaocr ocr scans/ --out out # out/txt/*.txt and out/hocr/*.hocr
63
+ parisaocr ocr book.pdf --out out --format txt,hocr,jsonl
64
+ parisaocr ocr book.pdf --out out --first 10 --last 20
65
+ parisaocr ocr book.pdf --out out --format pdf # out/pdf/book.pdf: searchable
66
+ ```
67
+
68
+ Inputs are images, directories of images, or a PDF (scanned pages are
69
+ extracted losslessly, born-digital pages rendered). Outputs: plain text (one
70
+ line per printed line, columns right to left), hOCR with line and word boxes on
71
+ the original page, and JSONL with text, boxes and confidences.
72
+
73
+ ### Searchable PDFs
74
+
75
+ `--format pdf` writes PDFs you can search, select and copy text in, with the
76
+ page images exactly as they were:
77
+
78
+ - **PDF input:** `out/pdf/NAME.pdf` is the original file with an invisible text
79
+ layer over each page that was read. The scans are not re-encoded; page
80
+ rotation is respected. Pages that already have a text layer (born-digital
81
+ pages, or scans someone OCRed before) are left as they are; `--pdf-text add`
82
+ adds ours to them too, e.g. over a poor earlier OCR layer.
83
+ - **Image input:** one PDF per image in `out/pdf/`, or all of them in one file
84
+ with `--merge-pdf NAME`. The page size follows the image's dpi.
85
+
86
+ The text layer is stored the way word processors store Persian text (one run
87
+ per line in visual order), so PDF viewers find and copy it in reading order.
88
+ Tested with Poppler (`pdftotext`, used by Okular and Evince) and pdf.js
89
+ (Firefox): words and their order come out right; pdftotext sometimes moves
90
+ punctuation at the edge of a word, and pdf.js drops half-spaces when copying.
91
+ Tools that do not reorder right-to-left text (such as pdfminer) return each
92
+ line reversed, as they do for any Persian PDF.
93
+
94
+ From Python:
95
+
96
+ ```python
97
+ from parisaocr import OCR
98
+
99
+ ocr = OCR() # loads the models once
100
+ print(ocr.text("page.png"))
101
+ for line in ocr.lines("page.png"): # reading order
102
+ print(line.bbox, line.conf, line.text)
103
+ ```
104
+
105
+ ## Accuracy
106
+
107
+ Measured on material the model never saw in training.
108
+
109
+ | Test set | ParisaOCR 0.1 | Gemini 3.8 Flash | Tesseract fas_print¹ |
110
+ |---|---|---|---|
111
+ | 32 book pages from 11 books, verified text: words found | **96.7%** | 94.7% | 94.7% |
112
+ | 750 lines from the same books: character error, ignoring spaces | 0.39% | 0.36% | 1.19% |
113
+ | same lines: word error | **6.1%** | 7.9% | 7.3% |
114
+ | 74 printed pages of [PersianML/persian-ocr-benchmark](https://huggingface.co/datasets/PersianML/persian-ocr-benchmark): words found | 77.9% | 81.7% | 71.1% |
115
+ | cost per 1,000 pages | about $0.14 (consumer GPU at $0.60/h) | $2.86 (list price) | about $0.10 |
116
+
117
+ ¹ Our earlier Persian Tesseract model ([tessdata_contrib](https://github.com/tesseract-ocr/tessdata_contrib/pull/19)),
118
+ reading the lines found by the Surya detector.
119
+
120
+ On the benchmark the gap to Gemini comes mostly from very small,
121
+ low-resolution scans (lines under about 16 pixels high), where Gemini reads
122
+ about 78% of the words and ParisaOCR 74%. Many benchmark pages are hard even
123
+ for a human reader, and its page-level character error (about 40% for every
124
+ reader) mostly measures reading order on multi-column pages.
125
+
126
+ The verified test set is small (750 lines from 11 books), so treat the numbers
127
+ as a guide rather than a precise ranking. Details in [MODEL_CARD.md](MODEL_CARD.md).
128
+
129
+ ## Known limitations
130
+
131
+ - **Printed Persian only.** Handwriting, Nastaliq calligraphy, manuscripts and
132
+ Urdu are not supported.
133
+ - **Very small text** (lines under about 14 pixels high, i.e. low-resolution
134
+ scans) loses accuracy; ParisaOCR prints a warning for such pages. Scan at
135
+ 300 dpi when you can.
136
+ - **Layout is simple.** Columns are read right to left and top to bottom;
137
+ tables, captions and complex magazine layouts may come out in the wrong
138
+ order. Text rotated sideways is not read.
139
+ - **Small ornaments** (a box, a dingbat) are sometimes read as a letter.
140
+ - **Spacing conventions.** The model follows standard Persian spelling for
141
+ half-spaces (ZWNJ) and sometimes differs from the printed spacing.
142
+ - **Diacritics** (vowel marks) are read less reliably than letters.
143
+
144
+ ## How it works
145
+
146
+ 1. **Lines.** The PP-OCRv6 text detector (PaddleOCR, Apache-2.0) finds the
147
+ text lines, run with ONNX Runtime.
148
+ 2. **Crops.** Each line is cut out with a margin that keeps the dots above and
149
+ below the letters; neighbouring lines inside the margin are painted over.
150
+ 3. **Reading.** A Kraken recognition model (PP-OCR-style architecture, CTC)
151
+ trained for Persian print reads each line. ParisaOCR turns the model's
152
+ visual-order output into reading order itself, keeping the half-space
153
+ (ZWNJ), and with each line's own direction, so English lines and footnotes
154
+ come out right.
155
+ 4. **Order.** Lines are grouped into columns, read right to left.
156
+
157
+ The model was trained on about 83,000 lines from about 200 scanned Persian
158
+ books, labelled automatically with Gemini, plus about 90,000 synthetic lines
159
+ (Persian and English text, special characters, low-resolution copies). See
160
+ [MODEL_CARD.md](MODEL_CARD.md), and the modules in [docs/architecture.md](docs/architecture.md).
161
+
162
+ ## Other models and detectors
163
+
164
+ `--model` accepts another Kraken model (`kraken:model.safetensors`) or a
165
+ Tesseract model (`TESSDATA_DIR:LANG`); `--detector` accepts other PP-OCR sizes
166
+ (`ppocr:v6-medium`), Kraken's segmenter (`kraken`), or Surya (`surya`, after
167
+ `pip install parisaocr[surya]`; note that Surya's weights are free only for
168
+ research, personal use and small companies). `parisaocr cut` cuts the lines of
169
+ scanned books for building training data.
170
+
171
+ ## Feedback
172
+
173
+ Please [open an issue](https://github.com/givia/ParisaOCR/issues/new/choose)
174
+ with the page (only if you are allowed to share it), the command you ran, and
175
+ what came out wrong.
176
+
177
+ ## License
178
+
179
+ Apache License 2.0, for the code and the models. The bundled PP-OCRv6 detector
180
+ is PaddleOCR's (Apache-2.0); see [NOTICE](NOTICE).
181
+
182
+ ---
183
+
184
+ ## فارسی
185
+
186
+ پریسا اوسی‌آر یک اوسی‌آر متن‌باز برای متن چاپی فارسی است. صفحه‌های اسکن‌شدهٔ کتاب را
187
+ تقریباً هم‌اندازهٔ مدل‌های Gemini Flash گوگل می‌خواند، روی رایانهٔ خودتان اجرا می‌شود
188
+ (بدون نیاز به اینترنت، با CPU یا GPU) و هزینهٔ اجرایش برای هر هزار صفحه چند سنت است.
189
+
190
+ **وضعیت: پیش‌نمایش.** روی کتاب‌ها و اسناد چاپی فارسی خوب کار می‌کند. محدودیت‌ها
191
+ را پایین‌تر ببینید. بازخورد همراه با صفحهٔ نمونه بسیار ارزشمند است.
192
+
193
+ نصب:
194
+
195
+ ```
196
+ pip install parisaocr
197
+ ```
198
+
199
+ استفاده:
200
+
201
+ ```
202
+ parisaocr ocr page.png # چاپ متن
203
+ parisaocr ocr book.pdf --out out # out/txt و out/hocr
204
+ parisaocr ocr book.pdf --out out --format pdf # out/pdf/book.pdf: پی‌دی‌اف جست‌وجوپذیر
205
+ ```
206
+
207
+ با `--format pdf` خروجی یک پی‌دی‌اف جست‌وجوپذیر است: تصویر صفحه‌ها همان است که بود و
208
+ یک لایهٔ متن نامرئی روی آن قرار می‌گیرد، تا بتوانید در آن جست‌وجو کنید و متن را انتخاب
209
+ و کپی کنید.
210
+
211
+ محدودیت‌ها:
212
+
213
+ - فقط متن چاپی فارسی. دست‌نوشته، خط نستعلیق، نسخهٔ خطی و اردو پشتیبانی نمی‌شوند.
214
+ - متن خیلی ریز (اسکن با وضوح پایین) دقت کمتری دارد. در صورت امکان با ۳۰۰ dpi اسکن کنید.
215
+ - چیدمان صفحه ساده خوانده می‌شود. جدول‌ها و صفحه‌های مجله‌ای پیچیده ممکن است ترتیب
216
+ درستی نداشته باشند.
217
+ - نیم‌فاصله‌ها طبق املای استاندارد نوشته می‌شوند و گاهی با چاپ کتاب فرق دارند.
218
+ - اِعراب‌ها کمتر از حروف دقیق خوانده می‌شوند.
219
+
220
+ بازخورد: لطفاً در [بخش Issues](https://github.com/givia/ParisaOCR/issues/new/choose) صفحه
221
+ (فقط اگر اجازهٔ انتشارش را دارید)، فرمانی که اجرا کردید و خطای خروجی را بفرستید.
222
+
223
+ مجوز: Apache 2.0، برای کد و مدل‌ها.
@@ -0,0 +1,190 @@
1
+ # ParisaOCR (پریسا اوسی‌آر)
2
+
3
+ Open-source OCR for printed Persian. It reads scanned book pages about as well
4
+ as Google's Gemini Flash models, runs on your own machine (CPU or GPU, no
5
+ internet needed), and costs a few cents per thousand pages to run.
6
+
7
+ **Status: preview.** It works well on printed Persian books and
8
+ documents; see [Known limitations](#known-limitations). Feedback with example
9
+ pages is very welcome.
10
+
11
+ [فارسی](#فارسی) · [Model on Hugging Face](https://huggingface.co/givia/ParisaOCR) · [PyPI](https://pypi.org/project/parisaocr/)
12
+
13
+ ## Install
14
+
15
+ ```
16
+ pip install parisaocr
17
+ ```
18
+
19
+ Python 3.10 to 3.13. The two models (text-line detector and recognizer, 22 MB)
20
+ are included. PyTorch is installed as a dependency of Kraken; a GPU is used
21
+ when one is available, otherwise the CPU (about 5 to 10 seconds per page).
22
+ For PDF input, Poppler's command-line tools must be installed
23
+ (`poppler-utils` on Debian and Ubuntu, `poppler` on Arch, Homebrew and conda-forge).
24
+
25
+ ## Use
26
+
27
+ ```
28
+ parisaocr ocr page.png # print the text
29
+ parisaocr ocr scans/ --out out # out/txt/*.txt and out/hocr/*.hocr
30
+ parisaocr ocr book.pdf --out out --format txt,hocr,jsonl
31
+ parisaocr ocr book.pdf --out out --first 10 --last 20
32
+ parisaocr ocr book.pdf --out out --format pdf # out/pdf/book.pdf: searchable
33
+ ```
34
+
35
+ Inputs are images, directories of images, or a PDF (scanned pages are
36
+ extracted losslessly, born-digital pages rendered). Outputs: plain text (one
37
+ line per printed line, columns right to left), hOCR with line and word boxes on
38
+ the original page, and JSONL with text, boxes and confidences.
39
+
40
+ ### Searchable PDFs
41
+
42
+ `--format pdf` writes PDFs you can search, select and copy text in, with the
43
+ page images exactly as they were:
44
+
45
+ - **PDF input:** `out/pdf/NAME.pdf` is the original file with an invisible text
46
+ layer over each page that was read. The scans are not re-encoded; page
47
+ rotation is respected. Pages that already have a text layer (born-digital
48
+ pages, or scans someone OCRed before) are left as they are; `--pdf-text add`
49
+ adds ours to them too, e.g. over a poor earlier OCR layer.
50
+ - **Image input:** one PDF per image in `out/pdf/`, or all of them in one file
51
+ with `--merge-pdf NAME`. The page size follows the image's dpi.
52
+
53
+ The text layer is stored the way word processors store Persian text (one run
54
+ per line in visual order), so PDF viewers find and copy it in reading order.
55
+ Tested with Poppler (`pdftotext`, used by Okular and Evince) and pdf.js
56
+ (Firefox): words and their order come out right; pdftotext sometimes moves
57
+ punctuation at the edge of a word, and pdf.js drops half-spaces when copying.
58
+ Tools that do not reorder right-to-left text (such as pdfminer) return each
59
+ line reversed, as they do for any Persian PDF.
60
+
61
+ From Python:
62
+
63
+ ```python
64
+ from parisaocr import OCR
65
+
66
+ ocr = OCR() # loads the models once
67
+ print(ocr.text("page.png"))
68
+ for line in ocr.lines("page.png"): # reading order
69
+ print(line.bbox, line.conf, line.text)
70
+ ```
71
+
72
+ ## Accuracy
73
+
74
+ Measured on material the model never saw in training.
75
+
76
+ | Test set | ParisaOCR 0.1 | Gemini 3.8 Flash | Tesseract fas_print¹ |
77
+ |---|---|---|---|
78
+ | 32 book pages from 11 books, verified text: words found | **96.7%** | 94.7% | 94.7% |
79
+ | 750 lines from the same books: character error, ignoring spaces | 0.39% | 0.36% | 1.19% |
80
+ | same lines: word error | **6.1%** | 7.9% | 7.3% |
81
+ | 74 printed pages of [PersianML/persian-ocr-benchmark](https://huggingface.co/datasets/PersianML/persian-ocr-benchmark): words found | 77.9% | 81.7% | 71.1% |
82
+ | cost per 1,000 pages | about $0.14 (consumer GPU at $0.60/h) | $2.86 (list price) | about $0.10 |
83
+
84
+ ¹ Our earlier Persian Tesseract model ([tessdata_contrib](https://github.com/tesseract-ocr/tessdata_contrib/pull/19)),
85
+ reading the lines found by the Surya detector.
86
+
87
+ On the benchmark the gap to Gemini comes mostly from very small,
88
+ low-resolution scans (lines under about 16 pixels high), where Gemini reads
89
+ about 78% of the words and ParisaOCR 74%. Many benchmark pages are hard even
90
+ for a human reader, and its page-level character error (about 40% for every
91
+ reader) mostly measures reading order on multi-column pages.
92
+
93
+ The verified test set is small (750 lines from 11 books), so treat the numbers
94
+ as a guide rather than a precise ranking. Details in [MODEL_CARD.md](MODEL_CARD.md).
95
+
96
+ ## Known limitations
97
+
98
+ - **Printed Persian only.** Handwriting, Nastaliq calligraphy, manuscripts and
99
+ Urdu are not supported.
100
+ - **Very small text** (lines under about 14 pixels high, i.e. low-resolution
101
+ scans) loses accuracy; ParisaOCR prints a warning for such pages. Scan at
102
+ 300 dpi when you can.
103
+ - **Layout is simple.** Columns are read right to left and top to bottom;
104
+ tables, captions and complex magazine layouts may come out in the wrong
105
+ order. Text rotated sideways is not read.
106
+ - **Small ornaments** (a box, a dingbat) are sometimes read as a letter.
107
+ - **Spacing conventions.** The model follows standard Persian spelling for
108
+ half-spaces (ZWNJ) and sometimes differs from the printed spacing.
109
+ - **Diacritics** (vowel marks) are read less reliably than letters.
110
+
111
+ ## How it works
112
+
113
+ 1. **Lines.** The PP-OCRv6 text detector (PaddleOCR, Apache-2.0) finds the
114
+ text lines, run with ONNX Runtime.
115
+ 2. **Crops.** Each line is cut out with a margin that keeps the dots above and
116
+ below the letters; neighbouring lines inside the margin are painted over.
117
+ 3. **Reading.** A Kraken recognition model (PP-OCR-style architecture, CTC)
118
+ trained for Persian print reads each line. ParisaOCR turns the model's
119
+ visual-order output into reading order itself, keeping the half-space
120
+ (ZWNJ), and with each line's own direction, so English lines and footnotes
121
+ come out right.
122
+ 4. **Order.** Lines are grouped into columns, read right to left.
123
+
124
+ The model was trained on about 83,000 lines from about 200 scanned Persian
125
+ books, labelled automatically with Gemini, plus about 90,000 synthetic lines
126
+ (Persian and English text, special characters, low-resolution copies). See
127
+ [MODEL_CARD.md](MODEL_CARD.md), and the modules in [docs/architecture.md](docs/architecture.md).
128
+
129
+ ## Other models and detectors
130
+
131
+ `--model` accepts another Kraken model (`kraken:model.safetensors`) or a
132
+ Tesseract model (`TESSDATA_DIR:LANG`); `--detector` accepts other PP-OCR sizes
133
+ (`ppocr:v6-medium`), Kraken's segmenter (`kraken`), or Surya (`surya`, after
134
+ `pip install parisaocr[surya]`; note that Surya's weights are free only for
135
+ research, personal use and small companies). `parisaocr cut` cuts the lines of
136
+ scanned books for building training data.
137
+
138
+ ## Feedback
139
+
140
+ Please [open an issue](https://github.com/givia/ParisaOCR/issues/new/choose)
141
+ with the page (only if you are allowed to share it), the command you ran, and
142
+ what came out wrong.
143
+
144
+ ## License
145
+
146
+ Apache License 2.0, for the code and the models. The bundled PP-OCRv6 detector
147
+ is PaddleOCR's (Apache-2.0); see [NOTICE](NOTICE).
148
+
149
+ ---
150
+
151
+ ## فارسی
152
+
153
+ پریسا اوسی‌آر یک اوسی‌آر متن‌باز برای متن چاپی فارسی است. صفحه‌های اسکن‌شدهٔ کتاب را
154
+ تقریباً هم‌اندازهٔ مدل‌های Gemini Flash گوگل می‌خواند، روی رایانهٔ خودتان اجرا می‌شود
155
+ (بدون نیاز به اینترنت، با CPU یا GPU) و هزینهٔ اجرایش برای هر هزار صفحه چند سنت است.
156
+
157
+ **وضعیت: پیش‌نمایش.** روی کتاب‌ها و اسناد چاپی فارسی خوب کار می‌کند. محدودیت‌ها
158
+ را پایین‌تر ببینید. بازخورد همراه با صفحهٔ نمونه بسیار ارزشمند است.
159
+
160
+ نصب:
161
+
162
+ ```
163
+ pip install parisaocr
164
+ ```
165
+
166
+ استفاده:
167
+
168
+ ```
169
+ parisaocr ocr page.png # چاپ متن
170
+ parisaocr ocr book.pdf --out out # out/txt و out/hocr
171
+ parisaocr ocr book.pdf --out out --format pdf # out/pdf/book.pdf: پی‌دی‌اف جست‌وجوپذیر
172
+ ```
173
+
174
+ با `--format pdf` خروجی یک پی‌دی‌اف جست‌وجوپذیر است: تصویر صفحه‌ها همان است که بود و
175
+ یک لایهٔ متن نامرئی روی آن قرار می‌گیرد، تا بتوانید در آن جست‌وجو کنید و متن را انتخاب
176
+ و کپی کنید.
177
+
178
+ محدودیت‌ها:
179
+
180
+ - فقط متن چاپی فارسی. دست‌نوشته، خط نستعلیق، نسخهٔ خطی و اردو پشتیبانی نمی‌شوند.
181
+ - متن خیلی ریز (اسکن با وضوح پایین) دقت کمتری دارد. در صورت امکان با ۳۰۰ dpi اسکن کنید.
182
+ - چیدمان صفحه ساده خوانده می‌شود. جدول‌ها و صفحه‌های مجله‌ای پیچیده ممکن است ترتیب
183
+ درستی نداشته باشند.
184
+ - نیم‌فاصله‌ها طبق املای استاندارد نوشته می‌شوند و گاهی با چاپ کتاب فرق دارند.
185
+ - اِعراب‌ها کمتر از حروف دقیق خوانده می‌شوند.
186
+
187
+ بازخورد: لطفاً در [بخش Issues](https://github.com/givia/ParisaOCR/issues/new/choose) صفحه
188
+ (فقط اگر اجازهٔ انتشارش را دارید)، فرمانی که اجرا کردید و خطای خروجی را بفرستید.
189
+
190
+ مجوز: Apache 2.0، برای کد و مدل‌ها.
@@ -0,0 +1,27 @@
1
+ """ParisaOCR: open-source OCR for printed Persian.
2
+
3
+ A text-line detector (PP-OCRv6, run with ONNX Runtime) finds the lines of a
4
+ page; a Kraken recognition model trained for Persian print reads each line; the
5
+ lines are ordered into right-to-left columns and written as text, hOCR (with
6
+ line and word boxes on the original page) or JSONL. Both models ship with the
7
+ package, so it runs offline, on a CPU or a GPU.
8
+
9
+ parisaocr ocr page.png # print the text
10
+ parisaocr ocr book.pdf --out out # out/txt, out/hocr
11
+ parisaocr ocr book.pdf --out out --format pdf # out/pdf/book.pdf, searchable
12
+
13
+ from parisaocr import OCR
14
+ print(OCR().text("page.png"))
15
+
16
+ The recognition model and the detector are replaceable (`--model`, `--detector`):
17
+ Tesseract models and the Surya detector are supported for comparison, and
18
+ `parisaocr cut` cuts the lines of scanned books for building training data.
19
+ """
20
+ __version__ = "0.2.0"
21
+
22
+
23
+ def __getattr__(name): # the interface imports torch and Kraken; only load them when asked for
24
+ if name == "OCR":
25
+ from .api import OCR
26
+ return OCR
27
+ raise AttributeError(name)
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ main()
@@ -0,0 +1,54 @@
1
+ """Python interface: read a page image and get its text or its lines.
2
+
3
+ from parisaocr import OCR
4
+ ocr = OCR() # loads the detector and the recognition model once
5
+ print(ocr.text("page.png")) # the page's text, lines in reading order
6
+ for line in ocr.lines("page.png"):
7
+ print(line.bbox, line.conf, line.text)
8
+
9
+ `OCR(model=..., detector=..., device=...)` takes the same model and detector
10
+ specifications as the command line (`parisaocr ocr --help`).
11
+ """
12
+ import pathlib
13
+ import tempfile
14
+ from types import SimpleNamespace
15
+
16
+ from PIL import Image
17
+
18
+ from . import output
19
+ from .cli import build_engine
20
+ from .order import order
21
+ from .pipeline import recognize_batch
22
+
23
+
24
+ class OCR:
25
+ def __init__(self, model="default", detector="ppocr:v6-small:1.3", cpu=False, line_order="rtl"):
26
+ opts = SimpleNamespace(model=model, detector=detector, cpu=cpu, min_width=1600, max_width=2000, batch=1,
27
+ pad=6, pad_frac=0.4, min_height=40, block_tall=True, fallback=True, jobs=8)
28
+ self.detector, self.reader, self.options = build_engine(opts)
29
+ self.line_order = line_order
30
+
31
+ def columns(self, image):
32
+ """Lines of one page (a path or a PIL image) as reading-order columns of `pipeline.Line`."""
33
+ with tempfile.TemporaryDirectory(prefix="parisaocr-") as tmp:
34
+ tmp = pathlib.Path(tmp)
35
+ if not isinstance(image, (str, pathlib.Path)):
36
+ path = tmp / "page.png"
37
+ image.save(path)
38
+ else:
39
+ path = pathlib.Path(image)
40
+ with Image.open(path) as page:
41
+ im, f = self.detector.prepare(page)
42
+ rec = {"page": "page", "image": str(path), "width": im.width, "height": im.height, "scale": f,
43
+ "orig_width": page.width, "orig_height": page.height}
44
+ rec["lines"] = self.detector.detect([im])[0]
45
+ lines = recognize_batch([rec], self.reader, self.options, tmp)["page"]
46
+ return order(lines, self.line_order)
47
+
48
+ def lines(self, image):
49
+ """The lines of one page in reading order (text, bbox on the page, mean confidence, words)."""
50
+ return [line for col in self.columns(image) for line in col]
51
+
52
+ def text(self, image):
53
+ """The text of one page: one line per printed line, columns right to left."""
54
+ return output.text(self.columns(image))
@@ -0,0 +1,37 @@
1
+ """Reading order <-> display order for one text line, with each line's own base direction.
2
+
3
+ Recognition models see a line image from left to right, so they are trained on
4
+ the display order of the text and their output is turned back into reading
5
+ order. Kraken can do both itself, but (mittagessen/kraken#809) its bidi code
6
+ deletes ZWNJ, the Persian half-space, and it gives every line the same base
7
+ direction, so a Latin-only line such as "1. Vercingétorix" is trained as
8
+ "Vercingétorix .1". Here the Unicode bidi algorithm of kraken.lib.bidi runs
9
+ with ZWNJ protected (swapped for U+00A6, a neutral that stays in place between
10
+ two letters) and with the base direction taken from the line itself.
11
+ """
12
+ import unicodedata
13
+
14
+ ZWNJ, PLACEHOLDER = "‌", "¦"
15
+
16
+
17
+ def base_direction(text):
18
+ """"L" for a line with more Latin than Arabic-script letters, else "R"."""
19
+ latin = sum(1 for c in text if c.isalpha() and unicodedata.bidirectional(c) == "L")
20
+ arabic = sum(1 for c in text if unicodedata.bidirectional(c) in ("R", "AL"))
21
+ return "L" if latin > arabic else "R"
22
+
23
+
24
+ def to_display(text, base=None):
25
+ """Display (left-to-right visual) order of a line in reading order; ZWNJ kept."""
26
+ from kraken.lib.bidi import get_display
27
+ base = base or base_direction(text)
28
+ return get_display(text.replace(ZWNJ, PLACEHOLDER), base_dir=base).replace(PLACEHOLDER, ZWNJ)
29
+
30
+
31
+ def to_logical(display, base=None):
32
+ """Reading order of a line in display order, and for each character of the result the index
33
+ of the display character it came from (to carry character positions along)."""
34
+ from kraken.lib.bidi import get_display_map
35
+ base = base or base_direction(display)
36
+ text, order = get_display_map(display.replace(ZWNJ, PLACEHOLDER), base_dir=base)
37
+ return text.replace(PLACEHOLDER, ZWNJ), order