ocr-cleaner 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,30 @@
1
+ # Virtual environments and scratch space used while building. Never committed:
2
+ # they are large, machine-specific, and rebuilt from pyproject.toml anyway.
3
+ _venvs/
4
+ _proof/
5
+ .venv*/
6
+
7
+ # Build output. Wheels are built by the release workflow from this source, so a
8
+ # wheel in git could silently differ from the code beside it.
9
+ dist/
10
+ build/
11
+ *.egg-info/
12
+ src/*.egg-info/
13
+
14
+ # Python noise
15
+ __pycache__/
16
+ *.py[cod]
17
+ .pytest_cache/
18
+ .mypy_cache/
19
+ .ruff_cache/
20
+
21
+ # Local verification state, not source
22
+ verify_results.json
23
+ publish-log.txt
24
+ published.json
25
+
26
+ # Credentials. None of these belong here, and this line is the backstop.
27
+ .pypirc
28
+ .env
29
+ *.pem
30
+ *.key
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Pranay Mahendrakar
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,187 @@
1
+ Metadata-Version: 2.5
2
+ Name: ocr-cleaner
3
+ Version: 0.1.0
4
+ Summary: Prepare a scanned page so OCR reads it better: deskew, denoise, threshold
5
+ Project-URL: Homepage, https://pypi.org/project/ocr-cleaner/
6
+ Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
7
+ Author: Pranay Mahendrakar
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: binarization,denoise,deskew,document,image-processing,ocr,pillow,preprocessing,scan,threshold
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Topic :: Multimedia :: Graphics
18
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
19
+ Classifier: Topic :: Scientific/Engineering :: Image Processing
20
+ Classifier: Topic :: Text Processing
21
+ Requires-Python: >=3.9
22
+ Requires-Dist: numpy>=1.23
23
+ Requires-Dist: pillow>=9
24
+ Provides-Extra: dev
25
+ Requires-Dist: pytest>=7; extra == 'dev'
26
+ Description-Content-Type: text/markdown
27
+
28
+ # ocr-cleaner
29
+
30
+ An OCR engine reads what you hand it. Hand it a page that is two degrees off
31
+ horizontal, speckled, and darker at one edge than the other, and it will read
32
+ that. This straightens the page, takes the grain off it and binarises it
33
+ against the lighting it actually has - and tells you which of those it did,
34
+ which it skipped, and why.
35
+
36
+ ## Install
37
+
38
+ ```
39
+ pip install ocr-cleaner
40
+ ```
41
+
42
+ ## Quickstart
43
+
44
+ ```python
45
+ import numpy as np, ocr_cleaner
46
+
47
+ page = np.full((1100, 850), 246, dtype=np.uint8) # a sheet of paper
48
+ for top in range(120, 1000, 34): # rows of text on it
49
+ page[top:top + 11, 90:760] = 50
50
+
51
+ result = ocr_cleaner.clean(page)
52
+ print(result.summary())
53
+ ```
54
+
55
+ ```
56
+ ocr-cleaner: <array 850x1100>
57
+ page document, 26 line-shaped bands of text, 79% of the page bare paper
58
+ size 850 x 1100 in, 850 x 1100 out
59
+ text lines about 11 px tall
60
+ skew +0.00 degrees measured, not corrected
61
+ result black and white
62
+ steps
63
+ grayscale skipped already 8-bit greyscale
64
+ deskew skipped the page is already straight at +0.00 degrees, under the 0.05 degree
65
+ floor worth an interpolation pass
66
+ border skipped no scanner edge or black margin found; every side of the page is
67
+ already paper
68
+ denoise skipped the page is already clean: the paper measures 0.0 grey levels of grain
69
+ and 0.00% specks, under the 1.5 level floor - filtering it would only
70
+ soften the text
71
+ threshold applied local mean over a 33 x 33 window (3x the 11 px text height), ink is 24
72
+ levels below its surroundings
73
+ upscale skipped no upscale_to_dpi was asked for, so the page keeps its own resolution
74
+ ```
75
+
76
+ `result.image` is the cleaned page. `result.steps` is the list above, and it is
77
+ the point: a page that comes back looking much as it went in tells you which
78
+ steps stood down rather than leaving you to guess.
79
+
80
+ For a file in and a file out, one line:
81
+
82
+ ```python
83
+ result = ocr_cleaner.clean_file("scan.tif", "clean.png", dpi=200, upscale_to_dpi=300)
84
+ ```
85
+
86
+ ## What it does
87
+
88
+ Six steps, in this order, each one reported in `result.steps` whether it ran
89
+ or not:
90
+
91
+ - **grayscale** - luminance, from any Pillow mode: colour, palette, 16-bit,
92
+ 1-bit, or transparency composited onto white. An EXIF orientation tag is
93
+ honoured first, so a page photographed sideways is not reported as ninety
94
+ degrees of skew.
95
+ - **deskew** - the angle of the text, to a fraction of a degree, then a rotation
96
+ back to horizontal with the new corners filled in the page's own paper colour
97
+ rather than white. A page already straight is left alone and says so.
98
+ - **border** - the black band a scanner leaves down the side of a page smaller
99
+ than its glass. A band is solid black across its whole width, where even
100
+ heavy display type leaves paper between the words, so the two are told apart
101
+ by counting rather than guessing.
102
+ - **denoise** - a median filter, sized so its window stays well inside the width
103
+ of a stroke: 3 x 3 for ordinary body text, 5 x 5 only for text large enough to
104
+ survive it. Grain is measured on the paper, where paper is supposed to be
105
+ flat, so a crisp page is not softened for nothing.
106
+ - **threshold** - `"adaptive"` compares every pixel to the mean of its own
107
+ neighbourhood, with a window taken from the height of the text. That is the
108
+ default because a page lit unevenly is the normal case, and one global cut has
109
+ to choose between losing the text at the dark end and flooding the bright one.
110
+ `"otsu"` is that single cut, for a page lit evenly end to end. `"none"` leaves
111
+ the page in greyscale.
112
+ - **upscale** - to `upscale_to_dpi`, and only when `dpi` says what the page is
113
+ now. OCR engines do better at 300 dpi, and enlarging blind makes a page worse
114
+ rather than better, so with only one of the two numbers this step does nothing
115
+ and says so.
116
+
117
+ And three pages that are not put through any of it:
118
+
119
+ - **A blank sheet** is reported blank and handed back untouched. Thresholding
120
+ blank paper turns its grain into a field of speckles that an OCR engine reads
121
+ as text, which is worse than doing nothing.
122
+ - **A photograph** is reported as a photograph. It has no paper level, no lines
123
+ and no skew, and binarising one destroys it.
124
+ - **A page that is already clean** passes through with only the threshold
125
+ applied, and the other four steps each say what they measured and why they
126
+ stood down.
127
+
128
+ Pure numpy and Pillow. No OpenCV, no OCR engine, no model download, no network,
129
+ and the same page always gives the same result. A 4000 x 3000 scan cleans in a
130
+ couple of seconds, and your image is never modified.
131
+
132
+ This is the companion to `document-quality`, which decides whether a page is
133
+ worth OCRing. This one improves it.
134
+
135
+ ## API
136
+
137
+ | Call | What you get |
138
+ | --- | --- |
139
+ | `clean(image, *, deskew=True, denoise=True, threshold="adaptive", border=True, upscale_to_dpi=None, dpi=None)` | a `CleanResult` |
140
+ | `clean_file(src, dst, **kw)` | the same, written to `dst` |
141
+ | `estimate_skew(image)` | degrees counter-clockwise off horizontal, as a float |
142
+
143
+ `image` is a path, a `PIL.Image.Image`, or a numpy array shaped `(h, w)`,
144
+ `(h, w, 1)`, `(h, w, 3)` or `(h, w, 4)`. Greyscale and colour are both fine.
145
+
146
+ **`CleanResult`**
147
+
148
+ | Attribute | Meaning |
149
+ | --- | --- |
150
+ | `.image` | the cleaned page, a `PIL.Image.Image` in mode `L` |
151
+ | `.steps` | `list[Step]`, one per stage, in order |
152
+ | `.skew_corrected_degrees` | how far the page was actually turned |
153
+ | `.estimated_text_height_px` | line height in source pixels, ascender top to descender foot, or `None` |
154
+ | `.page_kind` | `"document"`, `"blank"` or `"photograph"` |
155
+ | `.estimated_skew_degrees` | what was measured, corrected or not |
156
+ | `.applied` / `.skipped` | step names, as two lists |
157
+ | `.is_blank` / `.is_photograph` / `.binary` | the one-word answers |
158
+ | `.summary()` | the report above, as plain ASCII text |
159
+ | `.to_dict()` / `.to_json()` | the same, JSON-safe, minus the pixels |
160
+ | `.save(path)` | write `.image`, creating parent directories |
161
+
162
+ **`Step`** has `.name`, `.applied` and `.detail` - and `.detail` is never empty,
163
+ including when `.applied` is `False`.
164
+
165
+ `estimate_skew` is positive counter-clockwise, matching `PIL.Image.rotate`, so
166
+ `image.rotate(-ocr_cleaner.estimate_skew(image))` straightens a page by hand.
167
+
168
+ ## CLI
169
+
170
+ ```
171
+ ocr-cleaner scan.png # report, writes nothing
172
+ ocr-cleaner scan.png --output clean.png
173
+ ocr-cleaner scans/ --out-dir cleaned/ --suffix -clean
174
+ ocr-cleaner scan.tif --dpi 200 --upscale-to-dpi 300 --output big.tif
175
+ ocr-cleaner scan.png --threshold otsu --no-denoise
176
+ ocr-cleaner scans/ --recursive --quiet
177
+ ocr-cleaner scan.png --json
178
+ ```
179
+
180
+ Nothing is written unless you ask with `--output` or `--out-dir`.
181
+ `--require-document` exits 2 when a page turns out to be blank or not a
182
+ document, which is the flag a batch job wants. `ocr-cleaner --help` lists the
183
+ rest.
184
+
185
+ ## License
186
+
187
+ MIT
@@ -0,0 +1,160 @@
1
+ # ocr-cleaner
2
+
3
+ An OCR engine reads what you hand it. Hand it a page that is two degrees off
4
+ horizontal, speckled, and darker at one edge than the other, and it will read
5
+ that. This straightens the page, takes the grain off it and binarises it
6
+ against the lighting it actually has - and tells you which of those it did,
7
+ which it skipped, and why.
8
+
9
+ ## Install
10
+
11
+ ```
12
+ pip install ocr-cleaner
13
+ ```
14
+
15
+ ## Quickstart
16
+
17
+ ```python
18
+ import numpy as np, ocr_cleaner
19
+
20
+ page = np.full((1100, 850), 246, dtype=np.uint8) # a sheet of paper
21
+ for top in range(120, 1000, 34): # rows of text on it
22
+ page[top:top + 11, 90:760] = 50
23
+
24
+ result = ocr_cleaner.clean(page)
25
+ print(result.summary())
26
+ ```
27
+
28
+ ```
29
+ ocr-cleaner: <array 850x1100>
30
+ page document, 26 line-shaped bands of text, 79% of the page bare paper
31
+ size 850 x 1100 in, 850 x 1100 out
32
+ text lines about 11 px tall
33
+ skew +0.00 degrees measured, not corrected
34
+ result black and white
35
+ steps
36
+ grayscale skipped already 8-bit greyscale
37
+ deskew skipped the page is already straight at +0.00 degrees, under the 0.05 degree
38
+ floor worth an interpolation pass
39
+ border skipped no scanner edge or black margin found; every side of the page is
40
+ already paper
41
+ denoise skipped the page is already clean: the paper measures 0.0 grey levels of grain
42
+ and 0.00% specks, under the 1.5 level floor - filtering it would only
43
+ soften the text
44
+ threshold applied local mean over a 33 x 33 window (3x the 11 px text height), ink is 24
45
+ levels below its surroundings
46
+ upscale skipped no upscale_to_dpi was asked for, so the page keeps its own resolution
47
+ ```
48
+
49
+ `result.image` is the cleaned page. `result.steps` is the list above, and it is
50
+ the point: a page that comes back looking much as it went in tells you which
51
+ steps stood down rather than leaving you to guess.
52
+
53
+ For a file in and a file out, one line:
54
+
55
+ ```python
56
+ result = ocr_cleaner.clean_file("scan.tif", "clean.png", dpi=200, upscale_to_dpi=300)
57
+ ```
58
+
59
+ ## What it does
60
+
61
+ Six steps, in this order, each one reported in `result.steps` whether it ran
62
+ or not:
63
+
64
+ - **grayscale** - luminance, from any Pillow mode: colour, palette, 16-bit,
65
+ 1-bit, or transparency composited onto white. An EXIF orientation tag is
66
+ honoured first, so a page photographed sideways is not reported as ninety
67
+ degrees of skew.
68
+ - **deskew** - the angle of the text, to a fraction of a degree, then a rotation
69
+ back to horizontal with the new corners filled in the page's own paper colour
70
+ rather than white. A page already straight is left alone and says so.
71
+ - **border** - the black band a scanner leaves down the side of a page smaller
72
+ than its glass. A band is solid black across its whole width, where even
73
+ heavy display type leaves paper between the words, so the two are told apart
74
+ by counting rather than guessing.
75
+ - **denoise** - a median filter, sized so its window stays well inside the width
76
+ of a stroke: 3 x 3 for ordinary body text, 5 x 5 only for text large enough to
77
+ survive it. Grain is measured on the paper, where paper is supposed to be
78
+ flat, so a crisp page is not softened for nothing.
79
+ - **threshold** - `"adaptive"` compares every pixel to the mean of its own
80
+ neighbourhood, with a window taken from the height of the text. That is the
81
+ default because a page lit unevenly is the normal case, and one global cut has
82
+ to choose between losing the text at the dark end and flooding the bright one.
83
+ `"otsu"` is that single cut, for a page lit evenly end to end. `"none"` leaves
84
+ the page in greyscale.
85
+ - **upscale** - to `upscale_to_dpi`, and only when `dpi` says what the page is
86
+ now. OCR engines do better at 300 dpi, and enlarging blind makes a page worse
87
+ rather than better, so with only one of the two numbers this step does nothing
88
+ and says so.
89
+
90
+ And three pages that are not put through any of it:
91
+
92
+ - **A blank sheet** is reported blank and handed back untouched. Thresholding
93
+ blank paper turns its grain into a field of speckles that an OCR engine reads
94
+ as text, which is worse than doing nothing.
95
+ - **A photograph** is reported as a photograph. It has no paper level, no lines
96
+ and no skew, and binarising one destroys it.
97
+ - **A page that is already clean** passes through with only the threshold
98
+ applied, and the other four steps each say what they measured and why they
99
+ stood down.
100
+
101
+ Pure numpy and Pillow. No OpenCV, no OCR engine, no model download, no network,
102
+ and the same page always gives the same result. A 4000 x 3000 scan cleans in a
103
+ couple of seconds, and your image is never modified.
104
+
105
+ This is the companion to `document-quality`, which decides whether a page is
106
+ worth OCRing. This one improves it.
107
+
108
+ ## API
109
+
110
+ | Call | What you get |
111
+ | --- | --- |
112
+ | `clean(image, *, deskew=True, denoise=True, threshold="adaptive", border=True, upscale_to_dpi=None, dpi=None)` | a `CleanResult` |
113
+ | `clean_file(src, dst, **kw)` | the same, written to `dst` |
114
+ | `estimate_skew(image)` | degrees counter-clockwise off horizontal, as a float |
115
+
116
+ `image` is a path, a `PIL.Image.Image`, or a numpy array shaped `(h, w)`,
117
+ `(h, w, 1)`, `(h, w, 3)` or `(h, w, 4)`. Greyscale and colour are both fine.
118
+
119
+ **`CleanResult`**
120
+
121
+ | Attribute | Meaning |
122
+ | --- | --- |
123
+ | `.image` | the cleaned page, a `PIL.Image.Image` in mode `L` |
124
+ | `.steps` | `list[Step]`, one per stage, in order |
125
+ | `.skew_corrected_degrees` | how far the page was actually turned |
126
+ | `.estimated_text_height_px` | line height in source pixels, ascender top to descender foot, or `None` |
127
+ | `.page_kind` | `"document"`, `"blank"` or `"photograph"` |
128
+ | `.estimated_skew_degrees` | what was measured, corrected or not |
129
+ | `.applied` / `.skipped` | step names, as two lists |
130
+ | `.is_blank` / `.is_photograph` / `.binary` | the one-word answers |
131
+ | `.summary()` | the report above, as plain ASCII text |
132
+ | `.to_dict()` / `.to_json()` | the same, JSON-safe, minus the pixels |
133
+ | `.save(path)` | write `.image`, creating parent directories |
134
+
135
+ **`Step`** has `.name`, `.applied` and `.detail` - and `.detail` is never empty,
136
+ including when `.applied` is `False`.
137
+
138
+ `estimate_skew` is positive counter-clockwise, matching `PIL.Image.rotate`, so
139
+ `image.rotate(-ocr_cleaner.estimate_skew(image))` straightens a page by hand.
140
+
141
+ ## CLI
142
+
143
+ ```
144
+ ocr-cleaner scan.png # report, writes nothing
145
+ ocr-cleaner scan.png --output clean.png
146
+ ocr-cleaner scans/ --out-dir cleaned/ --suffix -clean
147
+ ocr-cleaner scan.tif --dpi 200 --upscale-to-dpi 300 --output big.tif
148
+ ocr-cleaner scan.png --threshold otsu --no-denoise
149
+ ocr-cleaner scans/ --recursive --quiet
150
+ ocr-cleaner scan.png --json
151
+ ```
152
+
153
+ Nothing is written unless you ask with `--output` or `--out-dir`.
154
+ `--require-document` exits 2 when a page turns out to be blank or not a
155
+ document, which is the flag a batch job wants. `ocr-cleaner --help` lists the
156
+ rest.
157
+
158
+ ## License
159
+
160
+ MIT
@@ -0,0 +1,54 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.27"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "ocr-cleaner"
7
+ version = "0.1.0"
8
+ description = "Prepare a scanned page so OCR reads it better: deskew, denoise, threshold"
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Pranay Mahendrakar" }]
14
+ keywords = [
15
+ "ocr",
16
+ "preprocessing",
17
+ "deskew",
18
+ "binarization",
19
+ "threshold",
20
+ "denoise",
21
+ "scan",
22
+ "document",
23
+ "image-processing",
24
+ "pillow",
25
+ ]
26
+ classifiers = [
27
+ "Development Status :: 4 - Beta",
28
+ "Intended Audience :: Developers",
29
+ "Intended Audience :: Science/Research",
30
+ "Programming Language :: Python :: 3",
31
+ "Programming Language :: Python :: 3 :: Only",
32
+ "Operating System :: OS Independent",
33
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
34
+ "Topic :: Scientific/Engineering :: Image Processing",
35
+ "Topic :: Multimedia :: Graphics",
36
+ "Topic :: Text Processing",
37
+ ]
38
+ dependencies = [
39
+ "numpy>=1.23",
40
+ "Pillow>=9",
41
+ ]
42
+
43
+ [project.optional-dependencies]
44
+ dev = ["pytest>=7"]
45
+
46
+ [project.scripts]
47
+ ocr-cleaner = "ocr_cleaner.cli:main"
48
+
49
+ [project.urls]
50
+ Homepage = "https://pypi.org/project/ocr-cleaner/"
51
+ Author = "https://pypi.org/user/pranaymahendrakar/"
52
+
53
+ [tool.hatch.build.targets.wheel]
54
+ packages = ["src/ocr_cleaner"]
@@ -0,0 +1,55 @@
1
+ """ocr-cleaner: prepare a scanned page so OCR reads it better.
2
+
3
+ Deskew, denoise and threshold, in one call, with every step reporting what it
4
+ did to the page and why - including the steps that looked and decided to do
5
+ nothing.
6
+
7
+ >>> import numpy as np, ocr_cleaner
8
+ >>> page = np.full((1100, 850), 246, dtype=np.uint8) # a sheet of paper
9
+ >>> for top in range(120, 1000, 34): # rows of text on it
10
+ ... page[top:top + 11, 90:760] = 50
11
+ >>> result = ocr_cleaner.clean(page)
12
+ >>> result.page_kind
13
+ 'document'
14
+ >>> "threshold" in result.applied
15
+ True
16
+
17
+ A blank sheet is reported blank and handed back untouched rather than
18
+ thresholded into a field of speckles, and a photograph is reported as a
19
+ photograph rather than cleaned as a bad scan.
20
+
21
+ Pure numpy and Pillow. No OpenCV, no OCR engine, no model download, nothing
22
+ touches the network, and the same page always gives the same result.
23
+ """
24
+ from __future__ import annotations
25
+
26
+ from ._analysis import MAX_SKEW_DEGREES, PAGE_KINDS
27
+ from ._core import (
28
+ ANALYSIS_MAX_SIDE,
29
+ DEFAULT_THRESHOLD,
30
+ THRESHOLD_MODES,
31
+ clean,
32
+ clean_file,
33
+ estimate_skew,
34
+ )
35
+ from ._images import IMAGE_SUFFIXES, open_image
36
+ from ._result import STEP_NAMES, CleanResult, Step
37
+
38
+ __version__ = "0.1.0"
39
+
40
+ __all__ = [
41
+ "clean",
42
+ "clean_file",
43
+ "estimate_skew",
44
+ "CleanResult",
45
+ "Step",
46
+ "STEP_NAMES",
47
+ "THRESHOLD_MODES",
48
+ "DEFAULT_THRESHOLD",
49
+ "PAGE_KINDS",
50
+ "MAX_SKEW_DEGREES",
51
+ "ANALYSIS_MAX_SIDE",
52
+ "IMAGE_SUFFIXES",
53
+ "open_image",
54
+ "__version__",
55
+ ]