ocr-cleaner 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ocr_cleaner-0.1.0/.gitignore +30 -0
- ocr_cleaner-0.1.0/LICENSE +21 -0
- ocr_cleaner-0.1.0/PKG-INFO +187 -0
- ocr_cleaner-0.1.0/README.md +160 -0
- ocr_cleaner-0.1.0/pyproject.toml +54 -0
- ocr_cleaner-0.1.0/src/ocr_cleaner/__init__.py +55 -0
- ocr_cleaner-0.1.0/src/ocr_cleaner/_analysis.py +633 -0
- ocr_cleaner-0.1.0/src/ocr_cleaner/_core.py +461 -0
- ocr_cleaner-0.1.0/src/ocr_cleaner/_images.py +227 -0
- ocr_cleaner-0.1.0/src/ocr_cleaner/_result.py +266 -0
- ocr_cleaner-0.1.0/src/ocr_cleaner/_steps.py +456 -0
- ocr_cleaner-0.1.0/src/ocr_cleaner/cli.py +331 -0
- ocr_cleaner-0.1.0/tests/conftest.py +293 -0
- ocr_cleaner-0.1.0/tests/test_clean.py +314 -0
- ocr_cleaner-0.1.0/tests/test_cli.py +231 -0
- ocr_cleaner-0.1.0/tests/test_docs.py +109 -0
- ocr_cleaner-0.1.0/tests/test_edge_cases.py +205 -0
- ocr_cleaner-0.1.0/tests/test_result.py +160 -0
- ocr_cleaner-0.1.0/tests/test_skew.py +119 -0
- ocr_cleaner-0.1.0/tests/test_steps.py +303 -0
- ocr_cleaner-0.1.0/tests/test_text_height.py +112 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Virtual environments and scratch space used while building. Never committed:
|
|
2
|
+
# they are large, machine-specific, and rebuilt from pyproject.toml anyway.
|
|
3
|
+
_venvs/
|
|
4
|
+
_proof/
|
|
5
|
+
.venv*/
|
|
6
|
+
|
|
7
|
+
# Build output. Wheels are built by the release workflow from this source, so a
|
|
8
|
+
# wheel in git could silently differ from the code beside it.
|
|
9
|
+
dist/
|
|
10
|
+
build/
|
|
11
|
+
*.egg-info/
|
|
12
|
+
src/*.egg-info/
|
|
13
|
+
|
|
14
|
+
# Python noise
|
|
15
|
+
__pycache__/
|
|
16
|
+
*.py[cod]
|
|
17
|
+
.pytest_cache/
|
|
18
|
+
.mypy_cache/
|
|
19
|
+
.ruff_cache/
|
|
20
|
+
|
|
21
|
+
# Local verification state, not source
|
|
22
|
+
verify_results.json
|
|
23
|
+
publish-log.txt
|
|
24
|
+
published.json
|
|
25
|
+
|
|
26
|
+
# Credentials. None of these belong here, and this line is the backstop.
|
|
27
|
+
.pypirc
|
|
28
|
+
.env
|
|
29
|
+
*.pem
|
|
30
|
+
*.key
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Pranay Mahendrakar
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: ocr-cleaner
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Prepare a scanned page so OCR reads it better: deskew, denoise, threshold
|
|
5
|
+
Project-URL: Homepage, https://pypi.org/project/ocr-cleaner/
|
|
6
|
+
Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
|
|
7
|
+
Author: Pranay Mahendrakar
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: binarization,denoise,deskew,document,image-processing,ocr,pillow,preprocessing,scan,threshold
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Topic :: Multimedia :: Graphics
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Image Processing
|
|
20
|
+
Classifier: Topic :: Text Processing
|
|
21
|
+
Requires-Python: >=3.9
|
|
22
|
+
Requires-Dist: numpy>=1.23
|
|
23
|
+
Requires-Dist: pillow>=9
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# ocr-cleaner
|
|
29
|
+
|
|
30
|
+
An OCR engine reads what you hand it. Hand it a page that is two degrees off
|
|
31
|
+
horizontal, speckled, and darker at one edge than the other, and it will read
|
|
32
|
+
that. This straightens the page, takes the grain off it and binarises it
|
|
33
|
+
against the lighting it actually has - and tells you which of those it did,
|
|
34
|
+
which it skipped, and why.
|
|
35
|
+
|
|
36
|
+
## Install
|
|
37
|
+
|
|
38
|
+
```
|
|
39
|
+
pip install ocr-cleaner
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Quickstart
|
|
43
|
+
|
|
44
|
+
```python
|
|
45
|
+
import numpy as np, ocr_cleaner
|
|
46
|
+
|
|
47
|
+
page = np.full((1100, 850), 246, dtype=np.uint8) # a sheet of paper
|
|
48
|
+
for top in range(120, 1000, 34): # rows of text on it
|
|
49
|
+
page[top:top + 11, 90:760] = 50
|
|
50
|
+
|
|
51
|
+
result = ocr_cleaner.clean(page)
|
|
52
|
+
print(result.summary())
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
```
|
|
56
|
+
ocr-cleaner: <array 850x1100>
|
|
57
|
+
page document, 26 line-shaped bands of text, 79% of the page bare paper
|
|
58
|
+
size 850 x 1100 in, 850 x 1100 out
|
|
59
|
+
text lines about 11 px tall
|
|
60
|
+
skew +0.00 degrees measured, not corrected
|
|
61
|
+
result black and white
|
|
62
|
+
steps
|
|
63
|
+
grayscale skipped already 8-bit greyscale
|
|
64
|
+
deskew skipped the page is already straight at +0.00 degrees, under the 0.05 degree
|
|
65
|
+
floor worth an interpolation pass
|
|
66
|
+
border skipped no scanner edge or black margin found; every side of the page is
|
|
67
|
+
already paper
|
|
68
|
+
denoise skipped the page is already clean: the paper measures 0.0 grey levels of grain
|
|
69
|
+
and 0.00% specks, under the 1.5 level floor - filtering it would only
|
|
70
|
+
soften the text
|
|
71
|
+
threshold applied local mean over a 33 x 33 window (3x the 11 px text height), ink is 24
|
|
72
|
+
levels below its surroundings
|
|
73
|
+
upscale skipped no upscale_to_dpi was asked for, so the page keeps its own resolution
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
`result.image` is the cleaned page. `result.steps` is the list above, and it is
|
|
77
|
+
the point: a page that comes back looking much as it went in tells you which
|
|
78
|
+
steps stood down rather than leaving you to guess.
|
|
79
|
+
|
|
80
|
+
For a file in and a file out, one line:
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
result = ocr_cleaner.clean_file("scan.tif", "clean.png", dpi=200, upscale_to_dpi=300)
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
## What it does
|
|
87
|
+
|
|
88
|
+
Six steps, in this order, each one reported in `result.steps` whether it ran
|
|
89
|
+
or not:
|
|
90
|
+
|
|
91
|
+
- **grayscale** - luminance, from any Pillow mode: colour, palette, 16-bit,
|
|
92
|
+
1-bit, or transparency composited onto white. An EXIF orientation tag is
|
|
93
|
+
honoured first, so a page photographed sideways is not reported as ninety
|
|
94
|
+
degrees of skew.
|
|
95
|
+
- **deskew** - the angle of the text, to a fraction of a degree, then a rotation
|
|
96
|
+
back to horizontal with the new corners filled in the page's own paper colour
|
|
97
|
+
rather than white. A page already straight is left alone and says so.
|
|
98
|
+
- **border** - the black band a scanner leaves down the side of a page smaller
|
|
99
|
+
than its glass. A band is solid black across its whole width, where even
|
|
100
|
+
heavy display type leaves paper between the words, so the two are told apart
|
|
101
|
+
by counting rather than guessing.
|
|
102
|
+
- **denoise** - a median filter, sized so its window stays well inside the width
|
|
103
|
+
of a stroke: 3 x 3 for ordinary body text, 5 x 5 only for text large enough to
|
|
104
|
+
survive it. Grain is measured on the paper, where paper is supposed to be
|
|
105
|
+
flat, so a crisp page is not softened for nothing.
|
|
106
|
+
- **threshold** - `"adaptive"` compares every pixel to the mean of its own
|
|
107
|
+
neighbourhood, with a window taken from the height of the text. That is the
|
|
108
|
+
default because a page lit unevenly is the normal case, and one global cut has
|
|
109
|
+
to choose between losing the text at the dark end and flooding the bright one.
|
|
110
|
+
`"otsu"` is that single cut, for a page lit evenly end to end. `"none"` leaves
|
|
111
|
+
the page in greyscale.
|
|
112
|
+
- **upscale** - to `upscale_to_dpi`, and only when `dpi` says what the page is
|
|
113
|
+
now. OCR engines do better at 300 dpi, and enlarging blind makes a page worse
|
|
114
|
+
rather than better, so with only one of the two numbers this step does nothing
|
|
115
|
+
and says so.
|
|
116
|
+
|
|
117
|
+
And three pages that are not put through any of it:
|
|
118
|
+
|
|
119
|
+
- **A blank sheet** is reported blank and handed back untouched. Thresholding
|
|
120
|
+
blank paper turns its grain into a field of speckles that an OCR engine reads
|
|
121
|
+
as text, which is worse than doing nothing.
|
|
122
|
+
- **A photograph** is reported as a photograph. It has no paper level, no lines
|
|
123
|
+
and no skew, and binarising one destroys it.
|
|
124
|
+
- **A page that is already clean** passes through with only the threshold
|
|
125
|
+
applied, and the other four steps each say what they measured and why they
|
|
126
|
+
stood down.
|
|
127
|
+
|
|
128
|
+
Pure numpy and Pillow. No OpenCV, no OCR engine, no model download, no network,
|
|
129
|
+
and the same page always gives the same result. A 4000 x 3000 scan cleans in a
|
|
130
|
+
couple of seconds, and your image is never modified.
|
|
131
|
+
|
|
132
|
+
This is the companion to `document-quality`, which decides whether a page is
|
|
133
|
+
worth OCRing. This one improves it.
|
|
134
|
+
|
|
135
|
+
## API
|
|
136
|
+
|
|
137
|
+
| Call | What you get |
|
|
138
|
+
| --- | --- |
|
|
139
|
+
| `clean(image, *, deskew=True, denoise=True, threshold="adaptive", border=True, upscale_to_dpi=None, dpi=None)` | a `CleanResult` |
|
|
140
|
+
| `clean_file(src, dst, **kw)` | the same, written to `dst` |
|
|
141
|
+
| `estimate_skew(image)` | degrees counter-clockwise off horizontal, as a float |
|
|
142
|
+
|
|
143
|
+
`image` is a path, a `PIL.Image.Image`, or a numpy array shaped `(h, w)`,
|
|
144
|
+
`(h, w, 1)`, `(h, w, 3)` or `(h, w, 4)`. Greyscale and colour are both fine.
|
|
145
|
+
|
|
146
|
+
**`CleanResult`**
|
|
147
|
+
|
|
148
|
+
| Attribute | Meaning |
|
|
149
|
+
| --- | --- |
|
|
150
|
+
| `.image` | the cleaned page, a `PIL.Image.Image` in mode `L` |
|
|
151
|
+
| `.steps` | `list[Step]`, one per stage, in order |
|
|
152
|
+
| `.skew_corrected_degrees` | how far the page was actually turned |
|
|
153
|
+
| `.estimated_text_height_px` | line height in source pixels, ascender top to descender foot, or `None` |
|
|
154
|
+
| `.page_kind` | `"document"`, `"blank"` or `"photograph"` |
|
|
155
|
+
| `.estimated_skew_degrees` | what was measured, corrected or not |
|
|
156
|
+
| `.applied` / `.skipped` | step names, as two lists |
|
|
157
|
+
| `.is_blank` / `.is_photograph` / `.binary` | the one-word answers |
|
|
158
|
+
| `.summary()` | the report above, as plain ASCII text |
|
|
159
|
+
| `.to_dict()` / `.to_json()` | the same, JSON-safe, minus the pixels |
|
|
160
|
+
| `.save(path)` | write `.image`, creating parent directories |
|
|
161
|
+
|
|
162
|
+
**`Step`** has `.name`, `.applied` and `.detail` - and `.detail` is never empty,
|
|
163
|
+
including when `.applied` is `False`.
|
|
164
|
+
|
|
165
|
+
`estimate_skew` is positive counter-clockwise, matching `PIL.Image.rotate`, so
|
|
166
|
+
`image.rotate(-ocr_cleaner.estimate_skew(image))` straightens a page by hand.
|
|
167
|
+
|
|
168
|
+
## CLI
|
|
169
|
+
|
|
170
|
+
```
|
|
171
|
+
ocr-cleaner scan.png # report, writes nothing
|
|
172
|
+
ocr-cleaner scan.png --output clean.png
|
|
173
|
+
ocr-cleaner scans/ --out-dir cleaned/ --suffix -clean
|
|
174
|
+
ocr-cleaner scan.tif --dpi 200 --upscale-to-dpi 300 --output big.tif
|
|
175
|
+
ocr-cleaner scan.png --threshold otsu --no-denoise
|
|
176
|
+
ocr-cleaner scans/ --recursive --quiet
|
|
177
|
+
ocr-cleaner scan.png --json
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
Nothing is written unless you ask with `--output` or `--out-dir`.
|
|
181
|
+
`--require-document` exits 2 when a page turns out to be blank or not a
|
|
182
|
+
document, which is the flag a batch job wants. `ocr-cleaner --help` lists the
|
|
183
|
+
rest.
|
|
184
|
+
|
|
185
|
+
## License
|
|
186
|
+
|
|
187
|
+
MIT
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
# ocr-cleaner
|
|
2
|
+
|
|
3
|
+
An OCR engine reads what you hand it. Hand it a page that is two degrees off
|
|
4
|
+
horizontal, speckled, and darker at one edge than the other, and it will read
|
|
5
|
+
that. This straightens the page, takes the grain off it and binarises it
|
|
6
|
+
against the lighting it actually has - and tells you which of those it did,
|
|
7
|
+
which it skipped, and why.
|
|
8
|
+
|
|
9
|
+
## Install
|
|
10
|
+
|
|
11
|
+
```
|
|
12
|
+
pip install ocr-cleaner
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
## Quickstart
|
|
16
|
+
|
|
17
|
+
```python
|
|
18
|
+
import numpy as np, ocr_cleaner
|
|
19
|
+
|
|
20
|
+
page = np.full((1100, 850), 246, dtype=np.uint8) # a sheet of paper
|
|
21
|
+
for top in range(120, 1000, 34): # rows of text on it
|
|
22
|
+
page[top:top + 11, 90:760] = 50
|
|
23
|
+
|
|
24
|
+
result = ocr_cleaner.clean(page)
|
|
25
|
+
print(result.summary())
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
```
|
|
29
|
+
ocr-cleaner: <array 850x1100>
|
|
30
|
+
page document, 26 line-shaped bands of text, 79% of the page bare paper
|
|
31
|
+
size 850 x 1100 in, 850 x 1100 out
|
|
32
|
+
text lines about 11 px tall
|
|
33
|
+
skew +0.00 degrees measured, not corrected
|
|
34
|
+
result black and white
|
|
35
|
+
steps
|
|
36
|
+
grayscale skipped already 8-bit greyscale
|
|
37
|
+
deskew skipped the page is already straight at +0.00 degrees, under the 0.05 degree
|
|
38
|
+
floor worth an interpolation pass
|
|
39
|
+
border skipped no scanner edge or black margin found; every side of the page is
|
|
40
|
+
already paper
|
|
41
|
+
denoise skipped the page is already clean: the paper measures 0.0 grey levels of grain
|
|
42
|
+
and 0.00% specks, under the 1.5 level floor - filtering it would only
|
|
43
|
+
soften the text
|
|
44
|
+
threshold applied local mean over a 33 x 33 window (3x the 11 px text height), ink is 24
|
|
45
|
+
levels below its surroundings
|
|
46
|
+
upscale skipped no upscale_to_dpi was asked for, so the page keeps its own resolution
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
`result.image` is the cleaned page. `result.steps` is the list above, and it is
|
|
50
|
+
the point: a page that comes back looking much as it went in tells you which
|
|
51
|
+
steps stood down rather than leaving you to guess.
|
|
52
|
+
|
|
53
|
+
For a file in and a file out, one line:
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
result = ocr_cleaner.clean_file("scan.tif", "clean.png", dpi=200, upscale_to_dpi=300)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## What it does
|
|
60
|
+
|
|
61
|
+
Six steps, in this order, each one reported in `result.steps` whether it ran
|
|
62
|
+
or not:
|
|
63
|
+
|
|
64
|
+
- **grayscale** - luminance, from any Pillow mode: colour, palette, 16-bit,
|
|
65
|
+
1-bit, or transparency composited onto white. An EXIF orientation tag is
|
|
66
|
+
honoured first, so a page photographed sideways is not reported as ninety
|
|
67
|
+
degrees of skew.
|
|
68
|
+
- **deskew** - the angle of the text, to a fraction of a degree, then a rotation
|
|
69
|
+
back to horizontal with the new corners filled in the page's own paper colour
|
|
70
|
+
rather than white. A page already straight is left alone and says so.
|
|
71
|
+
- **border** - the black band a scanner leaves down the side of a page smaller
|
|
72
|
+
than its glass. A band is solid black across its whole width, where even
|
|
73
|
+
heavy display type leaves paper between the words, so the two are told apart
|
|
74
|
+
by counting rather than guessing.
|
|
75
|
+
- **denoise** - a median filter, sized so its window stays well inside the width
|
|
76
|
+
of a stroke: 3 x 3 for ordinary body text, 5 x 5 only for text large enough to
|
|
77
|
+
survive it. Grain is measured on the paper, where paper is supposed to be
|
|
78
|
+
flat, so a crisp page is not softened for nothing.
|
|
79
|
+
- **threshold** - `"adaptive"` compares every pixel to the mean of its own
|
|
80
|
+
neighbourhood, with a window taken from the height of the text. That is the
|
|
81
|
+
default because a page lit unevenly is the normal case, and one global cut has
|
|
82
|
+
to choose between losing the text at the dark end and flooding the bright one.
|
|
83
|
+
`"otsu"` is that single cut, for a page lit evenly end to end. `"none"` leaves
|
|
84
|
+
the page in greyscale.
|
|
85
|
+
- **upscale** - to `upscale_to_dpi`, and only when `dpi` says what the page is
|
|
86
|
+
now. OCR engines do better at 300 dpi, and enlarging blind makes a page worse
|
|
87
|
+
rather than better, so with only one of the two numbers this step does nothing
|
|
88
|
+
and says so.
|
|
89
|
+
|
|
90
|
+
And three pages that are not put through any of it:
|
|
91
|
+
|
|
92
|
+
- **A blank sheet** is reported blank and handed back untouched. Thresholding
|
|
93
|
+
blank paper turns its grain into a field of speckles that an OCR engine reads
|
|
94
|
+
as text, which is worse than doing nothing.
|
|
95
|
+
- **A photograph** is reported as a photograph. It has no paper level, no lines
|
|
96
|
+
and no skew, and binarising one destroys it.
|
|
97
|
+
- **A page that is already clean** passes through with only the threshold
|
|
98
|
+
applied, and the other four steps each say what they measured and why they
|
|
99
|
+
stood down.
|
|
100
|
+
|
|
101
|
+
Pure numpy and Pillow. No OpenCV, no OCR engine, no model download, no network,
|
|
102
|
+
and the same page always gives the same result. A 4000 x 3000 scan cleans in a
|
|
103
|
+
couple of seconds, and your image is never modified.
|
|
104
|
+
|
|
105
|
+
This is the companion to `document-quality`, which decides whether a page is
|
|
106
|
+
worth OCRing. This one improves it.
|
|
107
|
+
|
|
108
|
+
## API
|
|
109
|
+
|
|
110
|
+
| Call | What you get |
|
|
111
|
+
| --- | --- |
|
|
112
|
+
| `clean(image, *, deskew=True, denoise=True, threshold="adaptive", border=True, upscale_to_dpi=None, dpi=None)` | a `CleanResult` |
|
|
113
|
+
| `clean_file(src, dst, **kw)` | the same, written to `dst` |
|
|
114
|
+
| `estimate_skew(image)` | degrees counter-clockwise off horizontal, as a float |
|
|
115
|
+
|
|
116
|
+
`image` is a path, a `PIL.Image.Image`, or a numpy array shaped `(h, w)`,
|
|
117
|
+
`(h, w, 1)`, `(h, w, 3)` or `(h, w, 4)`. Greyscale and colour are both fine.
|
|
118
|
+
|
|
119
|
+
**`CleanResult`**
|
|
120
|
+
|
|
121
|
+
| Attribute | Meaning |
|
|
122
|
+
| --- | --- |
|
|
123
|
+
| `.image` | the cleaned page, a `PIL.Image.Image` in mode `L` |
|
|
124
|
+
| `.steps` | `list[Step]`, one per stage, in order |
|
|
125
|
+
| `.skew_corrected_degrees` | how far the page was actually turned |
|
|
126
|
+
| `.estimated_text_height_px` | line height in source pixels, ascender top to descender foot, or `None` |
|
|
127
|
+
| `.page_kind` | `"document"`, `"blank"` or `"photograph"` |
|
|
128
|
+
| `.estimated_skew_degrees` | what was measured, corrected or not |
|
|
129
|
+
| `.applied` / `.skipped` | step names, as two lists |
|
|
130
|
+
| `.is_blank` / `.is_photograph` / `.binary` | the one-word answers |
|
|
131
|
+
| `.summary()` | the report above, as plain ASCII text |
|
|
132
|
+
| `.to_dict()` / `.to_json()` | the same, JSON-safe, minus the pixels |
|
|
133
|
+
| `.save(path)` | write `.image`, creating parent directories |
|
|
134
|
+
|
|
135
|
+
**`Step`** has `.name`, `.applied` and `.detail` - and `.detail` is never empty,
|
|
136
|
+
including when `.applied` is `False`.
|
|
137
|
+
|
|
138
|
+
`estimate_skew` is positive counter-clockwise, matching `PIL.Image.rotate`, so
|
|
139
|
+
`image.rotate(-ocr_cleaner.estimate_skew(image))` straightens a page by hand.
|
|
140
|
+
|
|
141
|
+
## CLI
|
|
142
|
+
|
|
143
|
+
```
|
|
144
|
+
ocr-cleaner scan.png # report, writes nothing
|
|
145
|
+
ocr-cleaner scan.png --output clean.png
|
|
146
|
+
ocr-cleaner scans/ --out-dir cleaned/ --suffix -clean
|
|
147
|
+
ocr-cleaner scan.tif --dpi 200 --upscale-to-dpi 300 --output big.tif
|
|
148
|
+
ocr-cleaner scan.png --threshold otsu --no-denoise
|
|
149
|
+
ocr-cleaner scans/ --recursive --quiet
|
|
150
|
+
ocr-cleaner scan.png --json
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
Nothing is written unless you ask with `--output` or `--out-dir`.
|
|
154
|
+
`--require-document` exits 2 when a page turns out to be blank or not a
|
|
155
|
+
document, which is the flag a batch job wants. `ocr-cleaner --help` lists the
|
|
156
|
+
rest.
|
|
157
|
+
|
|
158
|
+
## License
|
|
159
|
+
|
|
160
|
+
MIT
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.27"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "ocr-cleaner"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Prepare a scanned page so OCR reads it better: deskew, denoise, threshold"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Pranay Mahendrakar" }]
|
|
14
|
+
keywords = [
|
|
15
|
+
"ocr",
|
|
16
|
+
"preprocessing",
|
|
17
|
+
"deskew",
|
|
18
|
+
"binarization",
|
|
19
|
+
"threshold",
|
|
20
|
+
"denoise",
|
|
21
|
+
"scan",
|
|
22
|
+
"document",
|
|
23
|
+
"image-processing",
|
|
24
|
+
"pillow",
|
|
25
|
+
]
|
|
26
|
+
classifiers = [
|
|
27
|
+
"Development Status :: 4 - Beta",
|
|
28
|
+
"Intended Audience :: Developers",
|
|
29
|
+
"Intended Audience :: Science/Research",
|
|
30
|
+
"Programming Language :: Python :: 3",
|
|
31
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
32
|
+
"Operating System :: OS Independent",
|
|
33
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
34
|
+
"Topic :: Scientific/Engineering :: Image Processing",
|
|
35
|
+
"Topic :: Multimedia :: Graphics",
|
|
36
|
+
"Topic :: Text Processing",
|
|
37
|
+
]
|
|
38
|
+
dependencies = [
|
|
39
|
+
"numpy>=1.23",
|
|
40
|
+
"Pillow>=9",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
[project.optional-dependencies]
|
|
44
|
+
dev = ["pytest>=7"]
|
|
45
|
+
|
|
46
|
+
[project.scripts]
|
|
47
|
+
ocr-cleaner = "ocr_cleaner.cli:main"
|
|
48
|
+
|
|
49
|
+
[project.urls]
|
|
50
|
+
Homepage = "https://pypi.org/project/ocr-cleaner/"
|
|
51
|
+
Author = "https://pypi.org/user/pranaymahendrakar/"
|
|
52
|
+
|
|
53
|
+
[tool.hatch.build.targets.wheel]
|
|
54
|
+
packages = ["src/ocr_cleaner"]
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""ocr-cleaner: prepare a scanned page so OCR reads it better.
|
|
2
|
+
|
|
3
|
+
Deskew, denoise and threshold, in one call, with every step reporting what it
|
|
4
|
+
did to the page and why - including the steps that looked and decided to do
|
|
5
|
+
nothing.
|
|
6
|
+
|
|
7
|
+
>>> import numpy as np, ocr_cleaner
|
|
8
|
+
>>> page = np.full((1100, 850), 246, dtype=np.uint8) # a sheet of paper
|
|
9
|
+
>>> for top in range(120, 1000, 34): # rows of text on it
|
|
10
|
+
... page[top:top + 11, 90:760] = 50
|
|
11
|
+
>>> result = ocr_cleaner.clean(page)
|
|
12
|
+
>>> result.page_kind
|
|
13
|
+
'document'
|
|
14
|
+
>>> "threshold" in result.applied
|
|
15
|
+
True
|
|
16
|
+
|
|
17
|
+
A blank sheet is reported blank and handed back untouched rather than
|
|
18
|
+
thresholded into a field of speckles, and a photograph is reported as a
|
|
19
|
+
photograph rather than cleaned as a bad scan.
|
|
20
|
+
|
|
21
|
+
Pure numpy and Pillow. No OpenCV, no OCR engine, no model download, nothing
|
|
22
|
+
touches the network, and the same page always gives the same result.
|
|
23
|
+
"""
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from ._analysis import MAX_SKEW_DEGREES, PAGE_KINDS
|
|
27
|
+
from ._core import (
|
|
28
|
+
ANALYSIS_MAX_SIDE,
|
|
29
|
+
DEFAULT_THRESHOLD,
|
|
30
|
+
THRESHOLD_MODES,
|
|
31
|
+
clean,
|
|
32
|
+
clean_file,
|
|
33
|
+
estimate_skew,
|
|
34
|
+
)
|
|
35
|
+
from ._images import IMAGE_SUFFIXES, open_image
|
|
36
|
+
from ._result import STEP_NAMES, CleanResult, Step
|
|
37
|
+
|
|
38
|
+
__version__ = "0.1.0"
|
|
39
|
+
|
|
40
|
+
__all__ = [
|
|
41
|
+
"clean",
|
|
42
|
+
"clean_file",
|
|
43
|
+
"estimate_skew",
|
|
44
|
+
"CleanResult",
|
|
45
|
+
"Step",
|
|
46
|
+
"STEP_NAMES",
|
|
47
|
+
"THRESHOLD_MODES",
|
|
48
|
+
"DEFAULT_THRESHOLD",
|
|
49
|
+
"PAGE_KINDS",
|
|
50
|
+
"MAX_SKEW_DEGREES",
|
|
51
|
+
"ANALYSIS_MAX_SIDE",
|
|
52
|
+
"IMAGE_SUFFIXES",
|
|
53
|
+
"open_image",
|
|
54
|
+
"__version__",
|
|
55
|
+
]
|