parisaocr 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parisaocr-0.2.0/LICENSE +73 -0
- parisaocr-0.2.0/NOTICE +23 -0
- parisaocr-0.2.0/PKG-INFO +223 -0
- parisaocr-0.2.0/README.md +190 -0
- parisaocr-0.2.0/parisaocr/__init__.py +27 -0
- parisaocr-0.2.0/parisaocr/__main__.py +3 -0
- parisaocr-0.2.0/parisaocr/api.py +54 -0
- parisaocr-0.2.0/parisaocr/bidi.py +37 -0
- parisaocr-0.2.0/parisaocr/cli.py +256 -0
- parisaocr-0.2.0/parisaocr/cut.py +113 -0
- parisaocr-0.2.0/parisaocr/detect.py +87 -0
- parisaocr-0.2.0/parisaocr/detectors.py +117 -0
- parisaocr-0.2.0/parisaocr/fa_text.py +72 -0
- parisaocr-0.2.0/parisaocr/fonts/OFL-Vazirmatn.txt +93 -0
- parisaocr-0.2.0/parisaocr/fonts/Vazirmatn-Regular.ttf +0 -0
- parisaocr-0.2.0/parisaocr/kraken_reader.py +195 -0
- parisaocr-0.2.0/parisaocr/models/parisaocr-fa-0.1.safetensors +0 -0
- parisaocr-0.2.0/parisaocr/models/ppocrv6-det-small.onnx +0 -0
- parisaocr-0.2.0/parisaocr/order.py +29 -0
- parisaocr-0.2.0/parisaocr/output.py +59 -0
- parisaocr-0.2.0/parisaocr/pages.py +211 -0
- parisaocr-0.2.0/parisaocr/pdfout.py +163 -0
- parisaocr-0.2.0/parisaocr/pipeline.py +206 -0
- parisaocr-0.2.0/parisaocr/reader.py +62 -0
- parisaocr-0.2.0/parisaocr.egg-info/PKG-INFO +223 -0
- parisaocr-0.2.0/parisaocr.egg-info/SOURCES.txt +33 -0
- parisaocr-0.2.0/parisaocr.egg-info/dependency_links.txt +1 -0
- parisaocr-0.2.0/parisaocr.egg-info/entry_points.txt +2 -0
- parisaocr-0.2.0/parisaocr.egg-info/requires.txt +10 -0
- parisaocr-0.2.0/parisaocr.egg-info/top_level.txt +1 -0
- parisaocr-0.2.0/pyproject.toml +50 -0
- parisaocr-0.2.0/setup.cfg +4 -0
- parisaocr-0.2.0/tests/test_ocr.py +33 -0
- parisaocr-0.2.0/tests/test_pdf.py +61 -0
- parisaocr-0.2.0/tests/test_text.py +32 -0
parisaocr-0.2.0/LICENSE
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
Apache License
|
|
2
|
+
Version 2.0, January 2004
|
|
3
|
+
http://www.apache.org/licenses/
|
|
4
|
+
|
|
5
|
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
|
6
|
+
|
|
7
|
+
1. Definitions.
|
|
8
|
+
|
|
9
|
+
"License" shall mean the terms and conditions for use, reproduction, and distribution as defined by Sections 1 through 9 of this document.
|
|
10
|
+
|
|
11
|
+
"Licensor" shall mean the copyright owner or entity authorized by the copyright owner that is granting the License.
|
|
12
|
+
|
|
13
|
+
"Legal Entity" shall mean the union of the acting entity and all other entities that control, are controlled by, or are under common control with that entity. For the purposes of this definition, "control" means (i) the power, direct or indirect, to cause the direction or management of such entity, whether by contract or otherwise, or (ii) ownership of fifty percent (50%) or more of the outstanding shares, or (iii) beneficial ownership of such entity.
|
|
14
|
+
|
|
15
|
+
"You" (or "Your") shall mean an individual or Legal Entity exercising permissions granted by this License.
|
|
16
|
+
|
|
17
|
+
"Source" form shall mean the preferred form for making modifications, including but not limited to software source code, documentation source, and configuration files.
|
|
18
|
+
|
|
19
|
+
"Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types.
|
|
20
|
+
|
|
21
|
+
"Work" shall mean the work of authorship, whether in Source or Object form, made available under the License, as indicated by a copyright notice that is included in or attached to the work (an example is provided in the Appendix below).
|
|
22
|
+
|
|
23
|
+
"Derivative Works" shall mean any work, whether in Source or Object form, that is based on (or derived from) the Work and for which the editorial revisions, annotations, elaborations, or other modifications represent, as a whole, an original work of authorship. For the purposes of this License, Derivative Works shall not include works that remain separable from, or merely link (or bind by name) to the interfaces of, the Work and Derivative Works thereof.
|
|
24
|
+
|
|
25
|
+
"Contribution" shall mean any work of authorship, including the original version of the Work and any modifications or additions to that Work or Derivative Works thereof, that is intentionally submitted to Licensor for inclusion in the Work by the copyright owner or by an individual or Legal Entity authorized to submit on behalf of the copyright owner. For the purposes of this definition, "submitted" means any form of electronic, verbal, or written communication sent to the Licensor or its representatives, including but not limited to communication on electronic mailing lists, source code control systems, and issue tracking systems that are managed by, or on behalf of, the Licensor for the purpose of discussing and improving the Work, but excluding communication that is conspicuously marked or otherwise designated in writing by the copyright owner as "Not a Contribution."
|
|
26
|
+
|
|
27
|
+
"Contributor" shall mean Licensor and any individual or Legal Entity on behalf of whom a Contribution has been received by Licensor and subsequently incorporated within the Work.
|
|
28
|
+
|
|
29
|
+
2. Grant of Copyright License. Subject to the terms and conditions of this License, each Contributor hereby grants to You a perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable copyright license to reproduce, prepare Derivative Works of, publicly display, publicly perform, sublicense, and distribute the Work and such Derivative Works in Source or Object form.
|
|
30
|
+
|
|
31
|
+
3. Grant of Patent License. Subject to the terms and conditions of this License, each Contributor hereby grants to You a perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable (except as stated in this section) patent license to make, have made, use, offer to sell, sell, import, and otherwise transfer the Work, where such license applies only to those patent claims licensable by such Contributor that are necessarily infringed by their Contribution(s) alone or by combination of their Contribution(s) with the Work to which such Contribution(s) was submitted. If You institute patent litigation against any entity (including a cross-claim or counterclaim in a lawsuit) alleging that the Work or a Contribution incorporated within the Work constitutes direct or contributory patent infringement, then any patent licenses granted to You under this License for that Work shall terminate as of the date such litigation is filed.
|
|
32
|
+
|
|
33
|
+
4. Redistribution. You may reproduce and distribute copies of the Work or Derivative Works thereof in any medium, with or without modifications, and in Source or Object form, provided that You meet the following conditions:
|
|
34
|
+
|
|
35
|
+
(a) You must give any other recipients of the Work or Derivative Works a copy of this License; and
|
|
36
|
+
|
|
37
|
+
(b) You must cause any modified files to carry prominent notices stating that You changed the files; and
|
|
38
|
+
|
|
39
|
+
(c) You must retain, in the Source form of any Derivative Works that You distribute, all copyright, patent, trademark, and attribution notices from the Source form of the Work, excluding those notices that do not pertain to any part of the Derivative Works; and
|
|
40
|
+
|
|
41
|
+
(d) If the Work includes a "NOTICE" text file as part of its distribution, then any Derivative Works that You distribute must include a readable copy of the attribution notices contained within such NOTICE file, excluding those notices that do not pertain to any part of the Derivative Works, in at least one of the following places: within a NOTICE text file distributed as part of the Derivative Works; within the Source form or documentation, if provided along with the Derivative Works; or, within a display generated by the Derivative Works, if and wherever such third-party notices normally appear. The contents of the NOTICE file are for informational purposes only and do not modify the License. You may add Your own attribution notices within Derivative Works that You distribute, alongside or as an addendum to the NOTICE text from the Work, provided that such additional attribution notices cannot be construed as modifying the License.
|
|
42
|
+
|
|
43
|
+
You may add Your own copyright statement to Your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of Your modifications, or for any such Derivative Works as a whole, provided Your use, reproduction, and distribution of the Work otherwise complies with the conditions stated in this License.
|
|
44
|
+
|
|
45
|
+
5. Submission of Contributions. Unless You explicitly state otherwise, any Contribution intentionally submitted for inclusion in the Work by You to the Licensor shall be under the terms and conditions of this License, without any additional terms or conditions. Notwithstanding the above, nothing herein shall supersede or modify the terms of any separate license agreement you may have executed with Licensor regarding such Contributions.
|
|
46
|
+
|
|
47
|
+
6. Trademarks. This License does not grant permission to use the trade names, trademarks, service marks, or product names of the Licensor, except as required for reasonable and customary use in describing the origin of the Work and reproducing the content of the NOTICE file.
|
|
48
|
+
|
|
49
|
+
7. Disclaimer of Warranty. Unless required by applicable law or agreed to in writing, Licensor provides the Work (and each Contributor provides its Contributions) on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied, including, without limitation, any warranties or conditions of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A PARTICULAR PURPOSE. You are solely responsible for determining the appropriateness of using or redistributing the Work and assume any risks associated with Your exercise of permissions under this License.
|
|
50
|
+
|
|
51
|
+
8. Limitation of Liability. In no event and under no legal theory, whether in tort (including negligence), contract, or otherwise, unless required by applicable law (such as deliberate and grossly negligent acts) or agreed to in writing, shall any Contributor be liable to You for damages, including any direct, indirect, special, incidental, or consequential damages of any character arising as a result of this License or out of the use or inability to use the Work (including but not limited to damages for loss of goodwill, work stoppage, computer failure or malfunction, or any and all other commercial damages or losses), even if such Contributor has been advised of the possibility of such damages.
|
|
52
|
+
|
|
53
|
+
9. Accepting Warranty or Additional Liability. While redistributing the Work or Derivative Works thereof, You may choose to offer, and charge a fee for, acceptance of support, warranty, indemnity, or other liability obligations and/or rights consistent with this License. However, in accepting such obligations, You may act only on Your own behalf and on Your sole responsibility, not on behalf of any other Contributor, and only if You agree to indemnify, defend, and hold each Contributor harmless for any liability incurred by, or claims asserted against, such Contributor by reason of your accepting any such warranty or additional liability.
|
|
54
|
+
|
|
55
|
+
END OF TERMS AND CONDITIONS
|
|
56
|
+
|
|
57
|
+
APPENDIX: How to apply the Apache License to your work.
|
|
58
|
+
|
|
59
|
+
To apply the Apache License to your work, attach the following boilerplate notice, with the fields enclosed by brackets "[]" replaced with your own identifying information. (Don't include the brackets!) The text should be enclosed in the appropriate comment syntax for the file format. We also recommend that a file or class name and description of purpose be included on the same "printed page" as the copyright notice for easier identification within third-party archives.
|
|
60
|
+
|
|
61
|
+
Copyright [yyyy] [name of copyright owner]
|
|
62
|
+
|
|
63
|
+
Licensed under the Apache License, Version 2.0 (the "License");
|
|
64
|
+
you may not use this file except in compliance with the License.
|
|
65
|
+
You may obtain a copy of the License at
|
|
66
|
+
|
|
67
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
68
|
+
|
|
69
|
+
Unless required by applicable law or agreed to in writing, software
|
|
70
|
+
distributed under the License is distributed on an "AS IS" BASIS,
|
|
71
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
72
|
+
See the License for the specific language governing permissions and
|
|
73
|
+
limitations under the License.
|
parisaocr-0.2.0/NOTICE
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
ParisaOCR
|
|
2
|
+
Copyright 2026 Ali Javadi
|
|
3
|
+
|
|
4
|
+
This product is licensed under the Apache License, Version 2.0 (see LICENSE).
|
|
5
|
+
|
|
6
|
+
It includes:
|
|
7
|
+
|
|
8
|
+
- parisaocr/models/ppocrv6-det-small.onnx: the PP-OCRv6 small text detection
|
|
9
|
+
model of PaddleOCR (https://github.com/PaddlePaddle/PaddleOCR), Copyright
|
|
10
|
+
PaddlePaddle Authors, licensed under the Apache License, Version 2.0, in the
|
|
11
|
+
ONNX conversion distributed by RapidOCR (https://github.com/RapidAI/RapidOCR),
|
|
12
|
+
also under the Apache License, Version 2.0.
|
|
13
|
+
|
|
14
|
+
- parisaocr/models/parisaocr-fa-0.1.safetensors: a text recognition model
|
|
15
|
+
trained with Kraken (https://github.com/mittagessen/kraken, Apache License,
|
|
16
|
+
Version 2.0). See MODEL_CARD.md for its training data, including synthetic
|
|
17
|
+
lines rendered from Persian and English Wikipedia text (CC BY-SA 4.0).
|
|
18
|
+
|
|
19
|
+
- parisaocr/fonts/Vazirmatn-Regular.ttf: the Vazirmatn font
|
|
20
|
+
(https://github.com/rastikerdar/vazirmatn), Copyright 2015 The Vazirmatn
|
|
21
|
+
Project Authors, licensed under the SIL Open Font License 1.1
|
|
22
|
+
(parisaocr/fonts/OFL-Vazirmatn.txt). It is embedded, subset, in the
|
|
23
|
+
invisible text layer of the searchable PDFs ParisaOCR writes.
|
parisaocr-0.2.0/PKG-INFO
ADDED
|
@@ -0,0 +1,223 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: parisaocr
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Open-source OCR for printed Persian: PP-OCR line detection and a Kraken recognizer trained on Persian books
|
|
5
|
+
Author: Ali Javadi
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Project-URL: Homepage, https://github.com/givia/ParisaOCR
|
|
8
|
+
Project-URL: Issues, https://github.com/givia/ParisaOCR/issues
|
|
9
|
+
Project-URL: Model, https://huggingface.co/givia/ParisaOCR
|
|
10
|
+
Keywords: ocr,persian,farsi,text recognition,kraken,rtl
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Natural Language :: Persian
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Image Recognition
|
|
18
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
19
|
+
Requires-Python: <3.14,>=3.10
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
License-File: NOTICE
|
|
23
|
+
Requires-Dist: pillow
|
|
24
|
+
Requires-Dist: numpy
|
|
25
|
+
Requires-Dist: rapidocr>=3.9
|
|
26
|
+
Requires-Dist: onnxruntime
|
|
27
|
+
Requires-Dist: kraken==7.1.1
|
|
28
|
+
Requires-Dist: reportlab
|
|
29
|
+
Requires-Dist: pikepdf
|
|
30
|
+
Provides-Extra: surya
|
|
31
|
+
Requires-Dist: surya-ocr; extra == "surya"
|
|
32
|
+
Dynamic: license-file
|
|
33
|
+
|
|
34
|
+
# ParisaOCR (پریسا اوسیآر)
|
|
35
|
+
|
|
36
|
+
Open-source OCR for printed Persian. It reads scanned book pages about as well
|
|
37
|
+
as Google's Gemini Flash models, runs on your own machine (CPU or GPU, no
|
|
38
|
+
internet needed), and costs a few cents per thousand pages to run.
|
|
39
|
+
|
|
40
|
+
**Status: preview.** It works well on printed Persian books and
|
|
41
|
+
documents; see [Known limitations](#known-limitations). Feedback with example
|
|
42
|
+
pages is very welcome.
|
|
43
|
+
|
|
44
|
+
[فارسی](#فارسی) · [Model on Hugging Face](https://huggingface.co/givia/ParisaOCR) · [PyPI](https://pypi.org/project/parisaocr/)
|
|
45
|
+
|
|
46
|
+
## Install
|
|
47
|
+
|
|
48
|
+
```
|
|
49
|
+
pip install parisaocr
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Python 3.10 to 3.13. The two models (text-line detector and recognizer, 22 MB)
|
|
53
|
+
are included. PyTorch is installed as a dependency of Kraken; a GPU is used
|
|
54
|
+
when one is available, otherwise the CPU (about 5 to 10 seconds per page).
|
|
55
|
+
For PDF input, Poppler's command-line tools must be installed
|
|
56
|
+
(`poppler-utils` on Debian and Ubuntu, `poppler` on Arch, Homebrew and conda-forge).
|
|
57
|
+
|
|
58
|
+
## Use
|
|
59
|
+
|
|
60
|
+
```
|
|
61
|
+
parisaocr ocr page.png # print the text
|
|
62
|
+
parisaocr ocr scans/ --out out # out/txt/*.txt and out/hocr/*.hocr
|
|
63
|
+
parisaocr ocr book.pdf --out out --format txt,hocr,jsonl
|
|
64
|
+
parisaocr ocr book.pdf --out out --first 10 --last 20
|
|
65
|
+
parisaocr ocr book.pdf --out out --format pdf # out/pdf/book.pdf: searchable
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Inputs are images, directories of images, or a PDF (scanned pages are
|
|
69
|
+
extracted losslessly, born-digital pages rendered). Outputs: plain text (one
|
|
70
|
+
line per printed line, columns right to left), hOCR with line and word boxes on
|
|
71
|
+
the original page, and JSONL with text, boxes and confidences.
|
|
72
|
+
|
|
73
|
+
### Searchable PDFs
|
|
74
|
+
|
|
75
|
+
`--format pdf` writes PDFs you can search, select and copy text in, with the
|
|
76
|
+
page images exactly as they were:
|
|
77
|
+
|
|
78
|
+
- **PDF input:** `out/pdf/NAME.pdf` is the original file with an invisible text
|
|
79
|
+
layer over each page that was read. The scans are not re-encoded; page
|
|
80
|
+
rotation is respected. Pages that already have a text layer (born-digital
|
|
81
|
+
pages, or scans someone OCRed before) are left as they are; `--pdf-text add`
|
|
82
|
+
adds ours to them too, e.g. over a poor earlier OCR layer.
|
|
83
|
+
- **Image input:** one PDF per image in `out/pdf/`, or all of them in one file
|
|
84
|
+
with `--merge-pdf NAME`. The page size follows the image's dpi.
|
|
85
|
+
|
|
86
|
+
The text layer is stored the way word processors store Persian text (one run
|
|
87
|
+
per line in visual order), so PDF viewers find and copy it in reading order.
|
|
88
|
+
Tested with Poppler (`pdftotext`, used by Okular and Evince) and pdf.js
|
|
89
|
+
(Firefox): words and their order come out right; pdftotext sometimes moves
|
|
90
|
+
punctuation at the edge of a word, and pdf.js drops half-spaces when copying.
|
|
91
|
+
Tools that do not reorder right-to-left text (such as pdfminer) return each
|
|
92
|
+
line reversed, as they do for any Persian PDF.
|
|
93
|
+
|
|
94
|
+
From Python:
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
from parisaocr import OCR
|
|
98
|
+
|
|
99
|
+
ocr = OCR() # loads the models once
|
|
100
|
+
print(ocr.text("page.png"))
|
|
101
|
+
for line in ocr.lines("page.png"): # reading order
|
|
102
|
+
print(line.bbox, line.conf, line.text)
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
## Accuracy
|
|
106
|
+
|
|
107
|
+
Measured on material the model never saw in training.
|
|
108
|
+
|
|
109
|
+
| Test set | ParisaOCR 0.1 | Gemini 3.8 Flash | Tesseract fas_print¹ |
|
|
110
|
+
|---|---|---|---|
|
|
111
|
+
| 32 book pages from 11 books, verified text: words found | **96.7%** | 94.7% | 94.7% |
|
|
112
|
+
| 750 lines from the same books: character error, ignoring spaces | 0.39% | 0.36% | 1.19% |
|
|
113
|
+
| same lines: word error | **6.1%** | 7.9% | 7.3% |
|
|
114
|
+
| 74 printed pages of [PersianML/persian-ocr-benchmark](https://huggingface.co/datasets/PersianML/persian-ocr-benchmark): words found | 77.9% | 81.7% | 71.1% |
|
|
115
|
+
| cost per 1,000 pages | about $0.14 (consumer GPU at $0.60/h) | $2.86 (list price) | about $0.10 |
|
|
116
|
+
|
|
117
|
+
¹ Our earlier Persian Tesseract model ([tessdata_contrib](https://github.com/tesseract-ocr/tessdata_contrib/pull/19)),
|
|
118
|
+
reading the lines found by the Surya detector.
|
|
119
|
+
|
|
120
|
+
On the benchmark the gap to Gemini comes mostly from very small,
|
|
121
|
+
low-resolution scans (lines under about 16 pixels high), where Gemini reads
|
|
122
|
+
about 78% of the words and ParisaOCR 74%. Many benchmark pages are hard even
|
|
123
|
+
for a human reader, and its page-level character error (about 40% for every
|
|
124
|
+
reader) mostly measures reading order on multi-column pages.
|
|
125
|
+
|
|
126
|
+
The verified test set is small (750 lines from 11 books), so treat the numbers
|
|
127
|
+
as a guide rather than a precise ranking. Details in [MODEL_CARD.md](MODEL_CARD.md).
|
|
128
|
+
|
|
129
|
+
## Known limitations
|
|
130
|
+
|
|
131
|
+
- **Printed Persian only.** Handwriting, Nastaliq calligraphy, manuscripts and
|
|
132
|
+
Urdu are not supported.
|
|
133
|
+
- **Very small text** (lines under about 14 pixels high, i.e. low-resolution
|
|
134
|
+
scans) loses accuracy; ParisaOCR prints a warning for such pages. Scan at
|
|
135
|
+
300 dpi when you can.
|
|
136
|
+
- **Layout is simple.** Columns are read right to left and top to bottom;
|
|
137
|
+
tables, captions and complex magazine layouts may come out in the wrong
|
|
138
|
+
order. Text rotated sideways is not read.
|
|
139
|
+
- **Small ornaments** (a box, a dingbat) are sometimes read as a letter.
|
|
140
|
+
- **Spacing conventions.** The model follows standard Persian spelling for
|
|
141
|
+
half-spaces (ZWNJ) and sometimes differs from the printed spacing.
|
|
142
|
+
- **Diacritics** (vowel marks) are read less reliably than letters.
|
|
143
|
+
|
|
144
|
+
## How it works
|
|
145
|
+
|
|
146
|
+
1. **Lines.** The PP-OCRv6 text detector (PaddleOCR, Apache-2.0) finds the
|
|
147
|
+
text lines, run with ONNX Runtime.
|
|
148
|
+
2. **Crops.** Each line is cut out with a margin that keeps the dots above and
|
|
149
|
+
below the letters; neighbouring lines inside the margin are painted over.
|
|
150
|
+
3. **Reading.** A Kraken recognition model (PP-OCR-style architecture, CTC)
|
|
151
|
+
trained for Persian print reads each line. ParisaOCR turns the model's
|
|
152
|
+
visual-order output into reading order itself, keeping the half-space
|
|
153
|
+
(ZWNJ), and with each line's own direction, so English lines and footnotes
|
|
154
|
+
come out right.
|
|
155
|
+
4. **Order.** Lines are grouped into columns, read right to left.
|
|
156
|
+
|
|
157
|
+
The model was trained on about 83,000 lines from about 200 scanned Persian
|
|
158
|
+
books, labelled automatically with Gemini, plus about 90,000 synthetic lines
|
|
159
|
+
(Persian and English text, special characters, low-resolution copies). See
|
|
160
|
+
[MODEL_CARD.md](MODEL_CARD.md), and the modules in [docs/architecture.md](docs/architecture.md).
|
|
161
|
+
|
|
162
|
+
## Other models and detectors
|
|
163
|
+
|
|
164
|
+
`--model` accepts another Kraken model (`kraken:model.safetensors`) or a
|
|
165
|
+
Tesseract model (`TESSDATA_DIR:LANG`); `--detector` accepts other PP-OCR sizes
|
|
166
|
+
(`ppocr:v6-medium`), Kraken's segmenter (`kraken`), or Surya (`surya`, after
|
|
167
|
+
`pip install parisaocr[surya]`; note that Surya's weights are free only for
|
|
168
|
+
research, personal use and small companies). `parisaocr cut` cuts the lines of
|
|
169
|
+
scanned books for building training data.
|
|
170
|
+
|
|
171
|
+
## Feedback
|
|
172
|
+
|
|
173
|
+
Please [open an issue](https://github.com/givia/ParisaOCR/issues/new/choose)
|
|
174
|
+
with the page (only if you are allowed to share it), the command you ran, and
|
|
175
|
+
what came out wrong.
|
|
176
|
+
|
|
177
|
+
## License
|
|
178
|
+
|
|
179
|
+
Apache License 2.0, for the code and the models. The bundled PP-OCRv6 detector
|
|
180
|
+
is PaddleOCR's (Apache-2.0); see [NOTICE](NOTICE).
|
|
181
|
+
|
|
182
|
+
---
|
|
183
|
+
|
|
184
|
+
## فارسی
|
|
185
|
+
|
|
186
|
+
پریسا اوسیآر یک اوسیآر متنباز برای متن چاپی فارسی است. صفحههای اسکنشدهٔ کتاب را
|
|
187
|
+
تقریباً هماندازهٔ مدلهای Gemini Flash گوگل میخواند، روی رایانهٔ خودتان اجرا میشود
|
|
188
|
+
(بدون نیاز به اینترنت، با CPU یا GPU) و هزینهٔ اجرایش برای هر هزار صفحه چند سنت است.
|
|
189
|
+
|
|
190
|
+
**وضعیت: پیشنمایش.** روی کتابها و اسناد چاپی فارسی خوب کار میکند. محدودیتها
|
|
191
|
+
را پایینتر ببینید. بازخورد همراه با صفحهٔ نمونه بسیار ارزشمند است.
|
|
192
|
+
|
|
193
|
+
نصب:
|
|
194
|
+
|
|
195
|
+
```
|
|
196
|
+
pip install parisaocr
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
استفاده:
|
|
200
|
+
|
|
201
|
+
```
|
|
202
|
+
parisaocr ocr page.png # چاپ متن
|
|
203
|
+
parisaocr ocr book.pdf --out out # out/txt و out/hocr
|
|
204
|
+
parisaocr ocr book.pdf --out out --format pdf # out/pdf/book.pdf: پیدیاف جستوجوپذیر
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
با `--format pdf` خروجی یک پیدیاف جستوجوپذیر است: تصویر صفحهها همان است که بود و
|
|
208
|
+
یک لایهٔ متن نامرئی روی آن قرار میگیرد، تا بتوانید در آن جستوجو کنید و متن را انتخاب
|
|
209
|
+
و کپی کنید.
|
|
210
|
+
|
|
211
|
+
محدودیتها:
|
|
212
|
+
|
|
213
|
+
- فقط متن چاپی فارسی. دستنوشته، خط نستعلیق، نسخهٔ خطی و اردو پشتیبانی نمیشوند.
|
|
214
|
+
- متن خیلی ریز (اسکن با وضوح پایین) دقت کمتری دارد. در صورت امکان با ۳۰۰ dpi اسکن کنید.
|
|
215
|
+
- چیدمان صفحه ساده خوانده میشود. جدولها و صفحههای مجلهای پیچیده ممکن است ترتیب
|
|
216
|
+
درستی نداشته باشند.
|
|
217
|
+
- نیمفاصلهها طبق املای استاندارد نوشته میشوند و گاهی با چاپ کتاب فرق دارند.
|
|
218
|
+
- اِعرابها کمتر از حروف دقیق خوانده میشوند.
|
|
219
|
+
|
|
220
|
+
بازخورد: لطفاً در [بخش Issues](https://github.com/givia/ParisaOCR/issues/new/choose) صفحه
|
|
221
|
+
(فقط اگر اجازهٔ انتشارش را دارید)، فرمانی که اجرا کردید و خطای خروجی را بفرستید.
|
|
222
|
+
|
|
223
|
+
مجوز: Apache 2.0، برای کد و مدلها.
|
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
# ParisaOCR (پریسا اوسیآر)
|
|
2
|
+
|
|
3
|
+
Open-source OCR for printed Persian. It reads scanned book pages about as well
|
|
4
|
+
as Google's Gemini Flash models, runs on your own machine (CPU or GPU, no
|
|
5
|
+
internet needed), and costs a few cents per thousand pages to run.
|
|
6
|
+
|
|
7
|
+
**Status: preview.** It works well on printed Persian books and
|
|
8
|
+
documents; see [Known limitations](#known-limitations). Feedback with example
|
|
9
|
+
pages is very welcome.
|
|
10
|
+
|
|
11
|
+
[فارسی](#فارسی) · [Model on Hugging Face](https://huggingface.co/givia/ParisaOCR) · [PyPI](https://pypi.org/project/parisaocr/)
|
|
12
|
+
|
|
13
|
+
## Install
|
|
14
|
+
|
|
15
|
+
```
|
|
16
|
+
pip install parisaocr
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
Python 3.10 to 3.13. The two models (text-line detector and recognizer, 22 MB)
|
|
20
|
+
are included. PyTorch is installed as a dependency of Kraken; a GPU is used
|
|
21
|
+
when one is available, otherwise the CPU (about 5 to 10 seconds per page).
|
|
22
|
+
For PDF input, Poppler's command-line tools must be installed
|
|
23
|
+
(`poppler-utils` on Debian and Ubuntu, `poppler` on Arch, Homebrew and conda-forge).
|
|
24
|
+
|
|
25
|
+
## Use
|
|
26
|
+
|
|
27
|
+
```
|
|
28
|
+
parisaocr ocr page.png # print the text
|
|
29
|
+
parisaocr ocr scans/ --out out # out/txt/*.txt and out/hocr/*.hocr
|
|
30
|
+
parisaocr ocr book.pdf --out out --format txt,hocr,jsonl
|
|
31
|
+
parisaocr ocr book.pdf --out out --first 10 --last 20
|
|
32
|
+
parisaocr ocr book.pdf --out out --format pdf # out/pdf/book.pdf: searchable
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
Inputs are images, directories of images, or a PDF (scanned pages are
|
|
36
|
+
extracted losslessly, born-digital pages rendered). Outputs: plain text (one
|
|
37
|
+
line per printed line, columns right to left), hOCR with line and word boxes on
|
|
38
|
+
the original page, and JSONL with text, boxes and confidences.
|
|
39
|
+
|
|
40
|
+
### Searchable PDFs
|
|
41
|
+
|
|
42
|
+
`--format pdf` writes PDFs you can search, select and copy text in, with the
|
|
43
|
+
page images exactly as they were:
|
|
44
|
+
|
|
45
|
+
- **PDF input:** `out/pdf/NAME.pdf` is the original file with an invisible text
|
|
46
|
+
layer over each page that was read. The scans are not re-encoded; page
|
|
47
|
+
rotation is respected. Pages that already have a text layer (born-digital
|
|
48
|
+
pages, or scans someone OCRed before) are left as they are; `--pdf-text add`
|
|
49
|
+
adds ours to them too, e.g. over a poor earlier OCR layer.
|
|
50
|
+
- **Image input:** one PDF per image in `out/pdf/`, or all of them in one file
|
|
51
|
+
with `--merge-pdf NAME`. The page size follows the image's dpi.
|
|
52
|
+
|
|
53
|
+
The text layer is stored the way word processors store Persian text (one run
|
|
54
|
+
per line in visual order), so PDF viewers find and copy it in reading order.
|
|
55
|
+
Tested with Poppler (`pdftotext`, used by Okular and Evince) and pdf.js
|
|
56
|
+
(Firefox): words and their order come out right; pdftotext sometimes moves
|
|
57
|
+
punctuation at the edge of a word, and pdf.js drops half-spaces when copying.
|
|
58
|
+
Tools that do not reorder right-to-left text (such as pdfminer) return each
|
|
59
|
+
line reversed, as they do for any Persian PDF.
|
|
60
|
+
|
|
61
|
+
From Python:
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
from parisaocr import OCR
|
|
65
|
+
|
|
66
|
+
ocr = OCR() # loads the models once
|
|
67
|
+
print(ocr.text("page.png"))
|
|
68
|
+
for line in ocr.lines("page.png"): # reading order
|
|
69
|
+
print(line.bbox, line.conf, line.text)
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## Accuracy
|
|
73
|
+
|
|
74
|
+
Measured on material the model never saw in training.
|
|
75
|
+
|
|
76
|
+
| Test set | ParisaOCR 0.1 | Gemini 3.8 Flash | Tesseract fas_print¹ |
|
|
77
|
+
|---|---|---|---|
|
|
78
|
+
| 32 book pages from 11 books, verified text: words found | **96.7%** | 94.7% | 94.7% |
|
|
79
|
+
| 750 lines from the same books: character error, ignoring spaces | 0.39% | 0.36% | 1.19% |
|
|
80
|
+
| same lines: word error | **6.1%** | 7.9% | 7.3% |
|
|
81
|
+
| 74 printed pages of [PersianML/persian-ocr-benchmark](https://huggingface.co/datasets/PersianML/persian-ocr-benchmark): words found | 77.9% | 81.7% | 71.1% |
|
|
82
|
+
| cost per 1,000 pages | about $0.14 (consumer GPU at $0.60/h) | $2.86 (list price) | about $0.10 |
|
|
83
|
+
|
|
84
|
+
¹ Our earlier Persian Tesseract model ([tessdata_contrib](https://github.com/tesseract-ocr/tessdata_contrib/pull/19)),
|
|
85
|
+
reading the lines found by the Surya detector.
|
|
86
|
+
|
|
87
|
+
On the benchmark the gap to Gemini comes mostly from very small,
|
|
88
|
+
low-resolution scans (lines under about 16 pixels high), where Gemini reads
|
|
89
|
+
about 78% of the words and ParisaOCR 74%. Many benchmark pages are hard even
|
|
90
|
+
for a human reader, and its page-level character error (about 40% for every
|
|
91
|
+
reader) mostly measures reading order on multi-column pages.
|
|
92
|
+
|
|
93
|
+
The verified test set is small (750 lines from 11 books), so treat the numbers
|
|
94
|
+
as a guide rather than a precise ranking. Details in [MODEL_CARD.md](MODEL_CARD.md).
|
|
95
|
+
|
|
96
|
+
## Known limitations
|
|
97
|
+
|
|
98
|
+
- **Printed Persian only.** Handwriting, Nastaliq calligraphy, manuscripts and
|
|
99
|
+
Urdu are not supported.
|
|
100
|
+
- **Very small text** (lines under about 14 pixels high, i.e. low-resolution
|
|
101
|
+
scans) loses accuracy; ParisaOCR prints a warning for such pages. Scan at
|
|
102
|
+
300 dpi when you can.
|
|
103
|
+
- **Layout is simple.** Columns are read right to left and top to bottom;
|
|
104
|
+
tables, captions and complex magazine layouts may come out in the wrong
|
|
105
|
+
order. Text rotated sideways is not read.
|
|
106
|
+
- **Small ornaments** (a box, a dingbat) are sometimes read as a letter.
|
|
107
|
+
- **Spacing conventions.** The model follows standard Persian spelling for
|
|
108
|
+
half-spaces (ZWNJ) and sometimes differs from the printed spacing.
|
|
109
|
+
- **Diacritics** (vowel marks) are read less reliably than letters.
|
|
110
|
+
|
|
111
|
+
## How it works
|
|
112
|
+
|
|
113
|
+
1. **Lines.** The PP-OCRv6 text detector (PaddleOCR, Apache-2.0) finds the
|
|
114
|
+
text lines, run with ONNX Runtime.
|
|
115
|
+
2. **Crops.** Each line is cut out with a margin that keeps the dots above and
|
|
116
|
+
below the letters; neighbouring lines inside the margin are painted over.
|
|
117
|
+
3. **Reading.** A Kraken recognition model (PP-OCR-style architecture, CTC)
|
|
118
|
+
trained for Persian print reads each line. ParisaOCR turns the model's
|
|
119
|
+
visual-order output into reading order itself, keeping the half-space
|
|
120
|
+
(ZWNJ), and with each line's own direction, so English lines and footnotes
|
|
121
|
+
come out right.
|
|
122
|
+
4. **Order.** Lines are grouped into columns, read right to left.
|
|
123
|
+
|
|
124
|
+
The model was trained on about 83,000 lines from about 200 scanned Persian
|
|
125
|
+
books, labelled automatically with Gemini, plus about 90,000 synthetic lines
|
|
126
|
+
(Persian and English text, special characters, low-resolution copies). See
|
|
127
|
+
[MODEL_CARD.md](MODEL_CARD.md), and the modules in [docs/architecture.md](docs/architecture.md).
|
|
128
|
+
|
|
129
|
+
## Other models and detectors
|
|
130
|
+
|
|
131
|
+
`--model` accepts another Kraken model (`kraken:model.safetensors`) or a
|
|
132
|
+
Tesseract model (`TESSDATA_DIR:LANG`); `--detector` accepts other PP-OCR sizes
|
|
133
|
+
(`ppocr:v6-medium`), Kraken's segmenter (`kraken`), or Surya (`surya`, after
|
|
134
|
+
`pip install parisaocr[surya]`; note that Surya's weights are free only for
|
|
135
|
+
research, personal use and small companies). `parisaocr cut` cuts the lines of
|
|
136
|
+
scanned books for building training data.
|
|
137
|
+
|
|
138
|
+
## Feedback
|
|
139
|
+
|
|
140
|
+
Please [open an issue](https://github.com/givia/ParisaOCR/issues/new/choose)
|
|
141
|
+
with the page (only if you are allowed to share it), the command you ran, and
|
|
142
|
+
what came out wrong.
|
|
143
|
+
|
|
144
|
+
## License
|
|
145
|
+
|
|
146
|
+
Apache License 2.0, for the code and the models. The bundled PP-OCRv6 detector
|
|
147
|
+
is PaddleOCR's (Apache-2.0); see [NOTICE](NOTICE).
|
|
148
|
+
|
|
149
|
+
---
|
|
150
|
+
|
|
151
|
+
## فارسی
|
|
152
|
+
|
|
153
|
+
پریسا اوسیآر یک اوسیآر متنباز برای متن چاپی فارسی است. صفحههای اسکنشدهٔ کتاب را
|
|
154
|
+
تقریباً هماندازهٔ مدلهای Gemini Flash گوگل میخواند، روی رایانهٔ خودتان اجرا میشود
|
|
155
|
+
(بدون نیاز به اینترنت، با CPU یا GPU) و هزینهٔ اجرایش برای هر هزار صفحه چند سنت است.
|
|
156
|
+
|
|
157
|
+
**وضعیت: پیشنمایش.** روی کتابها و اسناد چاپی فارسی خوب کار میکند. محدودیتها
|
|
158
|
+
را پایینتر ببینید. بازخورد همراه با صفحهٔ نمونه بسیار ارزشمند است.
|
|
159
|
+
|
|
160
|
+
نصب:
|
|
161
|
+
|
|
162
|
+
```
|
|
163
|
+
pip install parisaocr
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
استفاده:
|
|
167
|
+
|
|
168
|
+
```
|
|
169
|
+
parisaocr ocr page.png # چاپ متن
|
|
170
|
+
parisaocr ocr book.pdf --out out # out/txt و out/hocr
|
|
171
|
+
parisaocr ocr book.pdf --out out --format pdf # out/pdf/book.pdf: پیدیاف جستوجوپذیر
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
با `--format pdf` خروجی یک پیدیاف جستوجوپذیر است: تصویر صفحهها همان است که بود و
|
|
175
|
+
یک لایهٔ متن نامرئی روی آن قرار میگیرد، تا بتوانید در آن جستوجو کنید و متن را انتخاب
|
|
176
|
+
و کپی کنید.
|
|
177
|
+
|
|
178
|
+
محدودیتها:
|
|
179
|
+
|
|
180
|
+
- فقط متن چاپی فارسی. دستنوشته، خط نستعلیق، نسخهٔ خطی و اردو پشتیبانی نمیشوند.
|
|
181
|
+
- متن خیلی ریز (اسکن با وضوح پایین) دقت کمتری دارد. در صورت امکان با ۳۰۰ dpi اسکن کنید.
|
|
182
|
+
- چیدمان صفحه ساده خوانده میشود. جدولها و صفحههای مجلهای پیچیده ممکن است ترتیب
|
|
183
|
+
درستی نداشته باشند.
|
|
184
|
+
- نیمفاصلهها طبق املای استاندارد نوشته میشوند و گاهی با چاپ کتاب فرق دارند.
|
|
185
|
+
- اِعرابها کمتر از حروف دقیق خوانده میشوند.
|
|
186
|
+
|
|
187
|
+
بازخورد: لطفاً در [بخش Issues](https://github.com/givia/ParisaOCR/issues/new/choose) صفحه
|
|
188
|
+
(فقط اگر اجازهٔ انتشارش را دارید)، فرمانی که اجرا کردید و خطای خروجی را بفرستید.
|
|
189
|
+
|
|
190
|
+
مجوز: Apache 2.0، برای کد و مدلها.
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""ParisaOCR: open-source OCR for printed Persian.
|
|
2
|
+
|
|
3
|
+
A text-line detector (PP-OCRv6, run with ONNX Runtime) finds the lines of a
|
|
4
|
+
page; a Kraken recognition model trained for Persian print reads each line; the
|
|
5
|
+
lines are ordered into right-to-left columns and written as text, hOCR (with
|
|
6
|
+
line and word boxes on the original page) or JSONL. Both models ship with the
|
|
7
|
+
package, so it runs offline, on a CPU or a GPU.
|
|
8
|
+
|
|
9
|
+
parisaocr ocr page.png # print the text
|
|
10
|
+
parisaocr ocr book.pdf --out out # out/txt, out/hocr
|
|
11
|
+
parisaocr ocr book.pdf --out out --format pdf # out/pdf/book.pdf, searchable
|
|
12
|
+
|
|
13
|
+
from parisaocr import OCR
|
|
14
|
+
print(OCR().text("page.png"))
|
|
15
|
+
|
|
16
|
+
The recognition model and the detector are replaceable (`--model`, `--detector`):
|
|
17
|
+
Tesseract models and the Surya detector are supported for comparison, and
|
|
18
|
+
`parisaocr cut` cuts the lines of scanned books for building training data.
|
|
19
|
+
"""
|
|
20
|
+
__version__ = "0.2.0"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def __getattr__(name): # the interface imports torch and Kraken; only load them when asked for
|
|
24
|
+
if name == "OCR":
|
|
25
|
+
from .api import OCR
|
|
26
|
+
return OCR
|
|
27
|
+
raise AttributeError(name)
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Python interface: read a page image and get its text or its lines.
|
|
2
|
+
|
|
3
|
+
from parisaocr import OCR
|
|
4
|
+
ocr = OCR() # loads the detector and the recognition model once
|
|
5
|
+
print(ocr.text("page.png")) # the page's text, lines in reading order
|
|
6
|
+
for line in ocr.lines("page.png"):
|
|
7
|
+
print(line.bbox, line.conf, line.text)
|
|
8
|
+
|
|
9
|
+
`OCR(model=..., detector=..., device=...)` takes the same model and detector
|
|
10
|
+
specifications as the command line (`parisaocr ocr --help`).
|
|
11
|
+
"""
|
|
12
|
+
import pathlib
|
|
13
|
+
import tempfile
|
|
14
|
+
from types import SimpleNamespace
|
|
15
|
+
|
|
16
|
+
from PIL import Image
|
|
17
|
+
|
|
18
|
+
from . import output
|
|
19
|
+
from .cli import build_engine
|
|
20
|
+
from .order import order
|
|
21
|
+
from .pipeline import recognize_batch
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class OCR:
|
|
25
|
+
def __init__(self, model="default", detector="ppocr:v6-small:1.3", cpu=False, line_order="rtl"):
|
|
26
|
+
opts = SimpleNamespace(model=model, detector=detector, cpu=cpu, min_width=1600, max_width=2000, batch=1,
|
|
27
|
+
pad=6, pad_frac=0.4, min_height=40, block_tall=True, fallback=True, jobs=8)
|
|
28
|
+
self.detector, self.reader, self.options = build_engine(opts)
|
|
29
|
+
self.line_order = line_order
|
|
30
|
+
|
|
31
|
+
def columns(self, image):
|
|
32
|
+
"""Lines of one page (a path or a PIL image) as reading-order columns of `pipeline.Line`."""
|
|
33
|
+
with tempfile.TemporaryDirectory(prefix="parisaocr-") as tmp:
|
|
34
|
+
tmp = pathlib.Path(tmp)
|
|
35
|
+
if not isinstance(image, (str, pathlib.Path)):
|
|
36
|
+
path = tmp / "page.png"
|
|
37
|
+
image.save(path)
|
|
38
|
+
else:
|
|
39
|
+
path = pathlib.Path(image)
|
|
40
|
+
with Image.open(path) as page:
|
|
41
|
+
im, f = self.detector.prepare(page)
|
|
42
|
+
rec = {"page": "page", "image": str(path), "width": im.width, "height": im.height, "scale": f,
|
|
43
|
+
"orig_width": page.width, "orig_height": page.height}
|
|
44
|
+
rec["lines"] = self.detector.detect([im])[0]
|
|
45
|
+
lines = recognize_batch([rec], self.reader, self.options, tmp)["page"]
|
|
46
|
+
return order(lines, self.line_order)
|
|
47
|
+
|
|
48
|
+
def lines(self, image):
|
|
49
|
+
"""The lines of one page in reading order (text, bbox on the page, mean confidence, words)."""
|
|
50
|
+
return [line for col in self.columns(image) for line in col]
|
|
51
|
+
|
|
52
|
+
def text(self, image):
|
|
53
|
+
"""The text of one page: one line per printed line, columns right to left."""
|
|
54
|
+
return output.text(self.columns(image))
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Reading order <-> display order for one text line, with each line's own base direction.
|
|
2
|
+
|
|
3
|
+
Recognition models see a line image from left to right, so they are trained on
|
|
4
|
+
the display order of the text and their output is turned back into reading
|
|
5
|
+
order. Kraken can do both itself, but (mittagessen/kraken#809) its bidi code
|
|
6
|
+
deletes ZWNJ, the Persian half-space, and it gives every line the same base
|
|
7
|
+
direction, so a Latin-only line such as "1. Vercingétorix" is trained as
|
|
8
|
+
"Vercingétorix .1". Here the Unicode bidi algorithm of kraken.lib.bidi runs
|
|
9
|
+
with ZWNJ protected (swapped for U+00A6, a neutral that stays in place between
|
|
10
|
+
two letters) and with the base direction taken from the line itself.
|
|
11
|
+
"""
|
|
12
|
+
import unicodedata
|
|
13
|
+
|
|
14
|
+
ZWNJ, PLACEHOLDER = "", "¦"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def base_direction(text):
|
|
18
|
+
""""L" for a line with more Latin than Arabic-script letters, else "R"."""
|
|
19
|
+
latin = sum(1 for c in text if c.isalpha() and unicodedata.bidirectional(c) == "L")
|
|
20
|
+
arabic = sum(1 for c in text if unicodedata.bidirectional(c) in ("R", "AL"))
|
|
21
|
+
return "L" if latin > arabic else "R"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def to_display(text, base=None):
|
|
25
|
+
"""Display (left-to-right visual) order of a line in reading order; ZWNJ kept."""
|
|
26
|
+
from kraken.lib.bidi import get_display
|
|
27
|
+
base = base or base_direction(text)
|
|
28
|
+
return get_display(text.replace(ZWNJ, PLACEHOLDER), base_dir=base).replace(PLACEHOLDER, ZWNJ)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def to_logical(display, base=None):
|
|
32
|
+
"""Reading order of a line in display order, and for each character of the result the index
|
|
33
|
+
of the display character it came from (to carry character positions along)."""
|
|
34
|
+
from kraken.lib.bidi import get_display_map
|
|
35
|
+
base = base or base_direction(display)
|
|
36
|
+
text, order = get_display_map(display.replace(ZWNJ, PLACEHOLDER), base_dir=base)
|
|
37
|
+
return text.replace(PLACEHOLDER, ZWNJ), order
|