document-quality 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- document_quality-0.1.0/.gitignore +30 -0
- document_quality-0.1.0/LICENSE +21 -0
- document_quality-0.1.0/PKG-INFO +192 -0
- document_quality-0.1.0/README.md +166 -0
- document_quality-0.1.0/pyproject.toml +53 -0
- document_quality-0.1.0/src/document_quality/__init__.py +67 -0
- document_quality-0.1.0/src/document_quality/__main__.py +7 -0
- document_quality-0.1.0/src/document_quality/_assess.py +477 -0
- document_quality-0.1.0/src/document_quality/_images.py +508 -0
- document_quality-0.1.0/src/document_quality/_lighting.py +636 -0
- document_quality-0.1.0/src/document_quality/_measures.py +1070 -0
- document_quality-0.1.0/src/document_quality/_report.py +522 -0
- document_quality-0.1.0/src/document_quality/_skew.py +1009 -0
- document_quality-0.1.0/src/document_quality/_thresholds.py +295 -0
- document_quality-0.1.0/src/document_quality/cli.py +253 -0
- document_quality-0.1.0/tests/_synthetic.py +252 -0
- document_quality-0.1.0/tests/conftest.py +47 -0
- document_quality-0.1.0/tests/test_batch_cli.py +168 -0
- document_quality-0.1.0/tests/test_inputs.py +202 -0
- document_quality-0.1.0/tests/test_lighting.py +168 -0
- document_quality-0.1.0/tests/test_measures.py +176 -0
- document_quality-0.1.0/tests/test_page_kinds.py +173 -0
- document_quality-0.1.0/tests/test_quickstart.py +110 -0
- document_quality-0.1.0/tests/test_real_type.py +128 -0
- document_quality-0.1.0/tests/test_skew.py +97 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Virtual environments and scratch space used while building. Never committed:
|
|
2
|
+
# they are large, machine-specific, and rebuilt from pyproject.toml anyway.
|
|
3
|
+
_venvs/
|
|
4
|
+
_proof/
|
|
5
|
+
.venv*/
|
|
6
|
+
|
|
7
|
+
# Build output. Wheels are built by the release workflow from this source, so a
|
|
8
|
+
# wheel in git could silently differ from the code beside it.
|
|
9
|
+
dist/
|
|
10
|
+
build/
|
|
11
|
+
*.egg-info/
|
|
12
|
+
src/*.egg-info/
|
|
13
|
+
|
|
14
|
+
# Python noise
|
|
15
|
+
__pycache__/
|
|
16
|
+
*.py[cod]
|
|
17
|
+
.pytest_cache/
|
|
18
|
+
.mypy_cache/
|
|
19
|
+
.ruff_cache/
|
|
20
|
+
|
|
21
|
+
# Local verification state, not source
|
|
22
|
+
verify_results.json
|
|
23
|
+
publish-log.txt
|
|
24
|
+
published.json
|
|
25
|
+
|
|
26
|
+
# Credentials. None of these belong here, and this line is the backstop.
|
|
27
|
+
.pypirc
|
|
28
|
+
.env
|
|
29
|
+
*.pem
|
|
30
|
+
*.key
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Pranay Mahendrakar
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: document-quality
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Decide whether a scanned page is good enough to OCR before you spend money OCRing it
|
|
5
|
+
Project-URL: Homepage, https://pypi.org/project/document-quality/
|
|
6
|
+
Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
|
|
7
|
+
Author: Pranay Mahendrakar
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: deskew,document,image-quality,ocr,pillow,preflight,scan,sharpness,show-through,skew
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Topic :: Multimedia :: Graphics
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
19
|
+
Classifier: Topic :: Text Processing
|
|
20
|
+
Requires-Python: >=3.9
|
|
21
|
+
Requires-Dist: numpy>=1.23
|
|
22
|
+
Requires-Dist: pillow>=9
|
|
23
|
+
Provides-Extra: dev
|
|
24
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
# document-quality
|
|
28
|
+
|
|
29
|
+
OCR is billed per page, and a bad scan costs exactly as much as a good one before
|
|
30
|
+
anybody notices it was unreadable. This reads the page first and tells you whether
|
|
31
|
+
it is worth sending, and what to fix if it is not.
|
|
32
|
+
|
|
33
|
+
## Install
|
|
34
|
+
|
|
35
|
+
```
|
|
36
|
+
pip install document-quality
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Quickstart
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
import numpy as np, document_quality
|
|
43
|
+
from PIL import Image
|
|
44
|
+
|
|
45
|
+
page = np.full((1100, 850), 245, dtype=np.uint8) # a sheet of paper
|
|
46
|
+
for top in range(100, 1000, 40): # rows of word-shaped ink
|
|
47
|
+
for left in range(80, 740, 66):
|
|
48
|
+
page[top:top + 22, left:left + 50] = 30
|
|
49
|
+
scan = Image.fromarray(page).rotate(2.3, fillcolor=245) # fed in crooked
|
|
50
|
+
|
|
51
|
+
report = document_quality.assess(scan, dpi=300)
|
|
52
|
+
print(report.summary())
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
```
|
|
56
|
+
<image 850x1100>: not ready to OCR - score 91 of 100, held back by skew
|
|
57
|
+
Document page, 850 x 1100 px, 300 dpi (from argument), 2.8 x 3.7 in.
|
|
58
|
+
Text lines about 24 px tall, skew +2.29 degrees.
|
|
59
|
+
Note: The letters give no clear sign of which way up the page reads, so it was taken to be upright; an upside-down page cannot be ruled out.
|
|
60
|
+
|
|
61
|
+
What to do, worst first:
|
|
62
|
+
[failure] The page is turned 2.29 degrees counter-clockwise of horizontal.
|
|
63
|
+
fix: deskew by 2.3 degrees clockwise
|
|
64
|
+
|
|
65
|
+
Measures:
|
|
66
|
+
resolution 300.000 100 Scanned at 300 dpi (argument), ...
|
|
67
|
+
text_size 23.641 91 Text lines stand about 24 px tall across 23 row(s); ...
|
|
68
|
+
skew 2.288 40 Text runs 2.29 degrees counter-clockwise of horizontal; ...
|
|
69
|
+
...
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
The first three lines are the answer. (The note is there because this page's
|
|
73
|
+
"words" are solid bars: with no ascenders or descenders to read, the report
|
|
74
|
+
says it could not tell upright from upside down rather than guess.) `report.ocr_ready` is the yes or no,
|
|
75
|
+
`report.score` is 0-100, and every entry in `report.issues` carries the concrete
|
|
76
|
+
remedy in `.fix` - "rescan at 300 dpi", "deskew by 2.3 degrees clockwise",
|
|
77
|
+
"increase lighting on the left edge", "crop the black border before OCR".
|
|
78
|
+
|
|
79
|
+
For a file on disk it is the same call: `document_quality.assess("scan.png")`.
|
|
80
|
+
|
|
81
|
+
## What it checks
|
|
82
|
+
|
|
83
|
+
- **Effective resolution** - dots per inch and the sheet size that implies. Left out
|
|
84
|
+
of the report entirely when no dpi is known, rather than guessed from pixel count.
|
|
85
|
+
- **Skew** - how far the text runs off horizontal, typically to within a tenth of
|
|
86
|
+
a degree across plus or minus 15 degrees, in either direction.
|
|
87
|
+
- **Ink-to-paper contrast** - how far the darkest ink sits from clean paper.
|
|
88
|
+
- **Sharpness** - how many pixels ink takes to become paper, scaled by the size of
|
|
89
|
+
the text so a 600 dpi scan is not marked down for spreading the same edge wider.
|
|
90
|
+
- **Uneven lighting** - measured as a gradient: the paper level is estimated
|
|
91
|
+
everywhere on the page and the report says how far it falls and towards where,
|
|
92
|
+
in words: "the left edge", "the top-right corner". Everything else - including
|
|
93
|
+
the text lines, their height and which way up they read - is measured against
|
|
94
|
+
that local paper level, so a shadow is reported as a shadow and not as low
|
|
95
|
+
contrast, show-through, a scanner border or an upside-down page.
|
|
96
|
+
- **Scanner borders** - only genuinely black, empty bands running in from the
|
|
97
|
+
edge of the image and ending in the sharp edge of the sheet count; a deep
|
|
98
|
+
shadow, which fades back to paper and still has text on it, never does. Borders
|
|
99
|
+
are cropped off before anything else is measured and reported with the pixels
|
|
100
|
+
to crop from each side. The black corners around a sheet scanned crooked on a
|
|
101
|
+
dark lid are left out of every measure too.
|
|
102
|
+
- **Show-through** - pale, soft marks bleeding through from the reverse side.
|
|
103
|
+
- **Black and white clipping** - strokes crushed to solid black, faint content
|
|
104
|
+
erased by the white point.
|
|
105
|
+
- **Text line height in pixels** - the number that actually decides whether an OCR
|
|
106
|
+
engine has enough pixels per character.
|
|
107
|
+
- **How much of the page looks like text** - the quantity behind the blank and
|
|
108
|
+
photograph verdicts.
|
|
109
|
+
- **Orientation** - a page lying on its side or upside down is caught, and a
|
|
110
|
+
sideways page is measured as the upright page it will be once rotated.
|
|
111
|
+
Upright is told from upside down by where each line's ink sits: Latin type
|
|
112
|
+
carries far more of it in the zone above its x-height than in the zone below
|
|
113
|
+
its baseline. When the letters give no clear sign - a soft scan, text in
|
|
114
|
+
capitals - the report says so in a note instead of guessing.
|
|
115
|
+
|
|
116
|
+
It also answers the two questions that come before all of those: is this sheet
|
|
117
|
+
blank, and is this a document page at all. A blank page is reported as blank, not
|
|
118
|
+
as eight failures about text it does not have - dust specks and a crooked scan on
|
|
119
|
+
a black lid included - and a faint page that still has rows of text is not
|
|
120
|
+
mistaken for a blank one. A photograph, or a sheet holding only a solid shape, is
|
|
121
|
+
reported as not a document page, not as a badly scanned one. A document page with
|
|
122
|
+
no rows of text on it at all is never called ready.
|
|
123
|
+
|
|
124
|
+
Greyscale and colour scans are both read, 8-bit, 16-bit or float, each on its own
|
|
125
|
+
fixed scale, so a faded 16-bit scan is as faded as the 8-bit one. A 4000 x 3000
|
|
126
|
+
scan takes well under a second. Your image is never modified. Pure numpy and
|
|
127
|
+
Pillow: no OpenCV, no OCR engine, no model download, no network.
|
|
128
|
+
|
|
129
|
+
## API
|
|
130
|
+
|
|
131
|
+
| Call | What you get |
|
|
132
|
+
| --- | --- |
|
|
133
|
+
| `assess(image, *, dpi=None)` | a `PageReport` for one page |
|
|
134
|
+
| `assess_batch(images)` | a `BatchReport` for many, unreadable files recorded not raised |
|
|
135
|
+
| `estimate_skew(image)` | degrees counter-clockwise off horizontal, as a float; `0.0` with no text lines |
|
|
136
|
+
| `detect_orientation(image)` | `0`, `90`, `180` or `270` - the counter-clockwise turn that sets it upright |
|
|
137
|
+
| `describe_thresholds()` | every boundary, its value and what it means |
|
|
138
|
+
|
|
139
|
+
`image` is a path, a `PIL.Image.Image`, or a numpy array shaped `(h, w)`,
|
|
140
|
+
`(h, w, 1)`, `(h, w, 3)` or `(h, w, 4)`. `assess` also takes `thresholds=`,
|
|
141
|
+
`source=` (a name for the page) and `check_orientation=False` for pages known to
|
|
142
|
+
be upright.
|
|
143
|
+
|
|
144
|
+
**`PageReport`**
|
|
145
|
+
|
|
146
|
+
| Attribute | Meaning |
|
|
147
|
+
| --- | --- |
|
|
148
|
+
| `.ocr_ready` | `True` when this page is worth sending to an OCR engine |
|
|
149
|
+
| `.score` | overall quality for OCR, 0-100 |
|
|
150
|
+
| `.kind` | `"document"`, `"blank"` or `"photograph"` |
|
|
151
|
+
| `.skew_degrees` | counter-clockwise off horizontal; `image.rotate(-skew_degrees)` straightens it |
|
|
152
|
+
| `.estimated_text_height_px` | inked line height, or `None` when no rows of text were found |
|
|
153
|
+
| `.issues` | `list[Issue(kind, severity, message, fix)]`, worst first |
|
|
154
|
+
| `.fixes` | just the remedies, worst first, de-duplicated |
|
|
155
|
+
| `.measures` | every measure by name, with its value, score and message |
|
|
156
|
+
| `.dpi`, `.dpi_source` | the resolution used and where it came from, or `None` |
|
|
157
|
+
| `.summary()`, `.to_dict()`, `.to_json()` | the whole report as text, as a dict, as JSON |
|
|
158
|
+
|
|
159
|
+
`Issue.kind` is one of `resolution`, `dpi_tag`, `text_size`, `skew`,
|
|
160
|
+
`orientation`, `contrast`, `sharpness`, `lighting`, `show_through`, `clipping`,
|
|
161
|
+
`border`, `blank` or `not_a_document`, and `Issue.severity` is `failure`,
|
|
162
|
+
`warning` or `info`. Any failure keeps a page from being OCR-ready.
|
|
163
|
+
|
|
164
|
+
**`BatchReport`**: `.not_ready` (worst first), `.ready`, `.blank`, `.photographs`,
|
|
165
|
+
`.failures`, `.rows()`, `.issue_counts()`, `.summary()`, `.to_dict()`, `.to_json()`.
|
|
166
|
+
|
|
167
|
+
**Thresholds.** Every boundary is a field you can override, for pages that are not
|
|
168
|
+
300 dpi office scans:
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
report = document_quality.assess(page, thresholds={"target_dpi": 600})
|
|
172
|
+
print(document_quality.describe_thresholds())
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
## CLI
|
|
176
|
+
|
|
177
|
+
```
|
|
178
|
+
document-quality scan.png # the summary for one page
|
|
179
|
+
document-quality scans/ --dpi 300 # every image in a folder
|
|
180
|
+
document-quality scans/ --json # to_dict() as JSON
|
|
181
|
+
document-quality scan.png --output report.json
|
|
182
|
+
document-quality scans/ --only-problems # just the pages needing work
|
|
183
|
+
document-quality --list-thresholds # every boundary and what it means
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
The exit code is `0` when every page assessed is ready to OCR, `1` when any page is
|
|
187
|
+
not, and `2` when no page could be read at all - so it drops straight into a shell
|
|
188
|
+
script guarding an OCR run. `python -m document_quality` works the same way.
|
|
189
|
+
|
|
190
|
+
## License
|
|
191
|
+
|
|
192
|
+
MIT
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
# document-quality
|
|
2
|
+
|
|
3
|
+
OCR is billed per page, and a bad scan costs exactly as much as a good one before
|
|
4
|
+
anybody notices it was unreadable. This reads the page first and tells you whether
|
|
5
|
+
it is worth sending, and what to fix if it is not.
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
```
|
|
10
|
+
pip install document-quality
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Quickstart
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
import numpy as np, document_quality
|
|
17
|
+
from PIL import Image
|
|
18
|
+
|
|
19
|
+
page = np.full((1100, 850), 245, dtype=np.uint8) # a sheet of paper
|
|
20
|
+
for top in range(100, 1000, 40): # rows of word-shaped ink
|
|
21
|
+
for left in range(80, 740, 66):
|
|
22
|
+
page[top:top + 22, left:left + 50] = 30
|
|
23
|
+
scan = Image.fromarray(page).rotate(2.3, fillcolor=245) # fed in crooked
|
|
24
|
+
|
|
25
|
+
report = document_quality.assess(scan, dpi=300)
|
|
26
|
+
print(report.summary())
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
```
|
|
30
|
+
<image 850x1100>: not ready to OCR - score 91 of 100, held back by skew
|
|
31
|
+
Document page, 850 x 1100 px, 300 dpi (from argument), 2.8 x 3.7 in.
|
|
32
|
+
Text lines about 24 px tall, skew +2.29 degrees.
|
|
33
|
+
Note: The letters give no clear sign of which way up the page reads, so it was taken to be upright; an upside-down page cannot be ruled out.
|
|
34
|
+
|
|
35
|
+
What to do, worst first:
|
|
36
|
+
[failure] The page is turned 2.29 degrees counter-clockwise of horizontal.
|
|
37
|
+
fix: deskew by 2.3 degrees clockwise
|
|
38
|
+
|
|
39
|
+
Measures:
|
|
40
|
+
resolution 300.000 100 Scanned at 300 dpi (argument), ...
|
|
41
|
+
text_size 23.641 91 Text lines stand about 24 px tall across 23 row(s); ...
|
|
42
|
+
skew 2.288 40 Text runs 2.29 degrees counter-clockwise of horizontal; ...
|
|
43
|
+
...
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
The first three lines are the answer. (The note is there because this page's
|
|
47
|
+
"words" are solid bars: with no ascenders or descenders to read, the report
|
|
48
|
+
says it could not tell upright from upside down rather than guess.) `report.ocr_ready` is the yes or no,
|
|
49
|
+
`report.score` is 0-100, and every entry in `report.issues` carries the concrete
|
|
50
|
+
remedy in `.fix` - "rescan at 300 dpi", "deskew by 2.3 degrees clockwise",
|
|
51
|
+
"increase lighting on the left edge", "crop the black border before OCR".
|
|
52
|
+
|
|
53
|
+
For a file on disk it is the same call: `document_quality.assess("scan.png")`.
|
|
54
|
+
|
|
55
|
+
## What it checks
|
|
56
|
+
|
|
57
|
+
- **Effective resolution** - dots per inch and the sheet size that implies. Left out
|
|
58
|
+
of the report entirely when no dpi is known, rather than guessed from pixel count.
|
|
59
|
+
- **Skew** - how far the text runs off horizontal, typically to within a tenth of
|
|
60
|
+
a degree across plus or minus 15 degrees, in either direction.
|
|
61
|
+
- **Ink-to-paper contrast** - how far the darkest ink sits from clean paper.
|
|
62
|
+
- **Sharpness** - how many pixels ink takes to become paper, scaled by the size of
|
|
63
|
+
the text so a 600 dpi scan is not marked down for spreading the same edge wider.
|
|
64
|
+
- **Uneven lighting** - measured as a gradient: the paper level is estimated
|
|
65
|
+
everywhere on the page and the report says how far it falls and towards where,
|
|
66
|
+
in words: "the left edge", "the top-right corner". Everything else - including
|
|
67
|
+
the text lines, their height and which way up they read - is measured against
|
|
68
|
+
that local paper level, so a shadow is reported as a shadow and not as low
|
|
69
|
+
contrast, show-through, a scanner border or an upside-down page.
|
|
70
|
+
- **Scanner borders** - only genuinely black, empty bands running in from the
|
|
71
|
+
edge of the image and ending in the sharp edge of the sheet count; a deep
|
|
72
|
+
shadow, which fades back to paper and still has text on it, never does. Borders
|
|
73
|
+
are cropped off before anything else is measured and reported with the pixels
|
|
74
|
+
to crop from each side. The black corners around a sheet scanned crooked on a
|
|
75
|
+
dark lid are left out of every measure too.
|
|
76
|
+
- **Show-through** - pale, soft marks bleeding through from the reverse side.
|
|
77
|
+
- **Black and white clipping** - strokes crushed to solid black, faint content
|
|
78
|
+
erased by the white point.
|
|
79
|
+
- **Text line height in pixels** - the number that actually decides whether an OCR
|
|
80
|
+
engine has enough pixels per character.
|
|
81
|
+
- **How much of the page looks like text** - the quantity behind the blank and
|
|
82
|
+
photograph verdicts.
|
|
83
|
+
- **Orientation** - a page lying on its side or upside down is caught, and a
|
|
84
|
+
sideways page is measured as the upright page it will be once rotated.
|
|
85
|
+
Upright is told from upside down by where each line's ink sits: Latin type
|
|
86
|
+
carries far more of it in the zone above its x-height than in the zone below
|
|
87
|
+
its baseline. When the letters give no clear sign - a soft scan, text in
|
|
88
|
+
capitals - the report says so in a note instead of guessing.
|
|
89
|
+
|
|
90
|
+
It also answers the two questions that come before all of those: is this sheet
|
|
91
|
+
blank, and is this a document page at all. A blank page is reported as blank, not
|
|
92
|
+
as eight failures about text it does not have - dust specks and a crooked scan on
|
|
93
|
+
a black lid included - and a faint page that still has rows of text is not
|
|
94
|
+
mistaken for a blank one. A photograph, or a sheet holding only a solid shape, is
|
|
95
|
+
reported as not a document page, not as a badly scanned one. A document page with
|
|
96
|
+
no rows of text on it at all is never called ready.
|
|
97
|
+
|
|
98
|
+
Greyscale and colour scans are both read, 8-bit, 16-bit or float, each on its own
|
|
99
|
+
fixed scale, so a faded 16-bit scan is as faded as the 8-bit one. A 4000 x 3000
|
|
100
|
+
scan takes well under a second. Your image is never modified. Pure numpy and
|
|
101
|
+
Pillow: no OpenCV, no OCR engine, no model download, no network.
|
|
102
|
+
|
|
103
|
+
## API
|
|
104
|
+
|
|
105
|
+
| Call | What you get |
|
|
106
|
+
| --- | --- |
|
|
107
|
+
| `assess(image, *, dpi=None)` | a `PageReport` for one page |
|
|
108
|
+
| `assess_batch(images)` | a `BatchReport` for many, unreadable files recorded not raised |
|
|
109
|
+
| `estimate_skew(image)` | degrees counter-clockwise off horizontal, as a float; `0.0` with no text lines |
|
|
110
|
+
| `detect_orientation(image)` | `0`, `90`, `180` or `270` - the counter-clockwise turn that sets it upright |
|
|
111
|
+
| `describe_thresholds()` | every boundary, its value and what it means |
|
|
112
|
+
|
|
113
|
+
`image` is a path, a `PIL.Image.Image`, or a numpy array shaped `(h, w)`,
|
|
114
|
+
`(h, w, 1)`, `(h, w, 3)` or `(h, w, 4)`. `assess` also takes `thresholds=`,
|
|
115
|
+
`source=` (a name for the page) and `check_orientation=False` for pages known to
|
|
116
|
+
be upright.
|
|
117
|
+
|
|
118
|
+
**`PageReport`**
|
|
119
|
+
|
|
120
|
+
| Attribute | Meaning |
|
|
121
|
+
| --- | --- |
|
|
122
|
+
| `.ocr_ready` | `True` when this page is worth sending to an OCR engine |
|
|
123
|
+
| `.score` | overall quality for OCR, 0-100 |
|
|
124
|
+
| `.kind` | `"document"`, `"blank"` or `"photograph"` |
|
|
125
|
+
| `.skew_degrees` | counter-clockwise off horizontal; `image.rotate(-skew_degrees)` straightens it |
|
|
126
|
+
| `.estimated_text_height_px` | inked line height, or `None` when no rows of text were found |
|
|
127
|
+
| `.issues` | `list[Issue(kind, severity, message, fix)]`, worst first |
|
|
128
|
+
| `.fixes` | just the remedies, worst first, de-duplicated |
|
|
129
|
+
| `.measures` | every measure by name, with its value, score and message |
|
|
130
|
+
| `.dpi`, `.dpi_source` | the resolution used and where it came from, or `None` |
|
|
131
|
+
| `.summary()`, `.to_dict()`, `.to_json()` | the whole report as text, as a dict, as JSON |
|
|
132
|
+
|
|
133
|
+
`Issue.kind` is one of `resolution`, `dpi_tag`, `text_size`, `skew`,
|
|
134
|
+
`orientation`, `contrast`, `sharpness`, `lighting`, `show_through`, `clipping`,
|
|
135
|
+
`border`, `blank` or `not_a_document`, and `Issue.severity` is `failure`,
|
|
136
|
+
`warning` or `info`. Any failure keeps a page from being OCR-ready.
|
|
137
|
+
|
|
138
|
+
**`BatchReport`**: `.not_ready` (worst first), `.ready`, `.blank`, `.photographs`,
|
|
139
|
+
`.failures`, `.rows()`, `.issue_counts()`, `.summary()`, `.to_dict()`, `.to_json()`.
|
|
140
|
+
|
|
141
|
+
**Thresholds.** Every boundary is a field you can override, for pages that are not
|
|
142
|
+
300 dpi office scans:
|
|
143
|
+
|
|
144
|
+
```python
|
|
145
|
+
report = document_quality.assess(page, thresholds={"target_dpi": 600})
|
|
146
|
+
print(document_quality.describe_thresholds())
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
## CLI
|
|
150
|
+
|
|
151
|
+
```
|
|
152
|
+
document-quality scan.png # the summary for one page
|
|
153
|
+
document-quality scans/ --dpi 300 # every image in a folder
|
|
154
|
+
document-quality scans/ --json # to_dict() as JSON
|
|
155
|
+
document-quality scan.png --output report.json
|
|
156
|
+
document-quality scans/ --only-problems # just the pages needing work
|
|
157
|
+
document-quality --list-thresholds # every boundary and what it means
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
The exit code is `0` when every page assessed is ready to OCR, `1` when any page is
|
|
161
|
+
not, and `2` when no page could be read at all - so it drops straight into a shell
|
|
162
|
+
script guarding an OCR run. `python -m document_quality` works the same way.
|
|
163
|
+
|
|
164
|
+
## License
|
|
165
|
+
|
|
166
|
+
MIT
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.27"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "document-quality"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Decide whether a scanned page is good enough to OCR before you spend money OCRing it"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Pranay Mahendrakar" }]
|
|
14
|
+
keywords = [
|
|
15
|
+
"ocr",
|
|
16
|
+
"scan",
|
|
17
|
+
"document",
|
|
18
|
+
"image-quality",
|
|
19
|
+
"deskew",
|
|
20
|
+
"skew",
|
|
21
|
+
"preflight",
|
|
22
|
+
"sharpness",
|
|
23
|
+
"show-through",
|
|
24
|
+
"pillow",
|
|
25
|
+
]
|
|
26
|
+
classifiers = [
|
|
27
|
+
"Development Status :: 4 - Beta",
|
|
28
|
+
"Intended Audience :: Developers",
|
|
29
|
+
"Intended Audience :: Science/Research",
|
|
30
|
+
"Programming Language :: Python :: 3",
|
|
31
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
32
|
+
"Operating System :: OS Independent",
|
|
33
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
34
|
+
"Topic :: Multimedia :: Graphics",
|
|
35
|
+
"Topic :: Text Processing",
|
|
36
|
+
]
|
|
37
|
+
dependencies = [
|
|
38
|
+
"numpy>=1.23",
|
|
39
|
+
"Pillow>=9",
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
[project.optional-dependencies]
|
|
43
|
+
dev = ["pytest>=7"]
|
|
44
|
+
|
|
45
|
+
[project.scripts]
|
|
46
|
+
document-quality = "document_quality.cli:main"
|
|
47
|
+
|
|
48
|
+
[project.urls]
|
|
49
|
+
Homepage = "https://pypi.org/project/document-quality/"
|
|
50
|
+
Author = "https://pypi.org/user/pranaymahendrakar/"
|
|
51
|
+
|
|
52
|
+
[tool.hatch.build.targets.wheel]
|
|
53
|
+
packages = ["src/document_quality"]
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Decide whether a scanned page is good enough to OCR before you pay to OCR it.
|
|
2
|
+
|
|
3
|
+
OCR is billed per page and a bad scan costs the same as a good one, then has to
|
|
4
|
+
be caught, re-scanned and re-run. This package reads the page first and answers
|
|
5
|
+
in one line::
|
|
6
|
+
|
|
7
|
+
import document_quality
|
|
8
|
+
|
|
9
|
+
report = document_quality.assess("invoice.png")
|
|
10
|
+
print(report.summary())
|
|
11
|
+
|
|
12
|
+
if not report.ocr_ready:
|
|
13
|
+
for issue in report.issues:
|
|
14
|
+
print(issue.kind, "->", issue.fix)
|
|
15
|
+
|
|
16
|
+
Every verdict carries the number behind it and the boundary it was compared
|
|
17
|
+
against, and every problem carries the concrete remedy: "rescan at 300 dpi",
|
|
18
|
+
"deskew by 2.3 degrees clockwise", "increase lighting on the left edge",
|
|
19
|
+
"crop the black border before OCR". Nothing here
|
|
20
|
+
downloads a model, calls out to a network, or needs OpenCV - it is numpy and
|
|
21
|
+
Pillow measuring a page.
|
|
22
|
+
|
|
23
|
+
What gets measured: effective resolution, skew angle, ink-to-paper contrast,
|
|
24
|
+
sharpness, uneven lighting (as a gradient of the paper level, so a shadow is
|
|
25
|
+
never mistaken for a scanner border), show-through from the reverse side, black
|
|
26
|
+
and white clipping, genuinely black scanner borders, text line height in
|
|
27
|
+
pixels, and how much of the page looks like text.
|
|
28
|
+
A blank sheet is reported as blank rather than as eight failures, and a
|
|
29
|
+
photograph is reported as not a document page rather than as a bad one.
|
|
30
|
+
"""
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
from ._assess import (
|
|
34
|
+
SCORE_WEIGHTS,
|
|
35
|
+
assess,
|
|
36
|
+
assess_batch,
|
|
37
|
+
detect_orientation,
|
|
38
|
+
estimate_skew,
|
|
39
|
+
)
|
|
40
|
+
from ._report import PAGE_KINDS, SEVERITIES, BatchReport, Issue, Measure, PageReport
|
|
41
|
+
from ._thresholds import (
|
|
42
|
+
DEFAULT_THRESHOLDS,
|
|
43
|
+
PASS_SCORE,
|
|
44
|
+
Thresholds,
|
|
45
|
+
describe_thresholds,
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
__version__ = "0.1.0"
|
|
49
|
+
|
|
50
|
+
__all__ = [
|
|
51
|
+
"assess",
|
|
52
|
+
"assess_batch",
|
|
53
|
+
"estimate_skew",
|
|
54
|
+
"detect_orientation",
|
|
55
|
+
"PageReport",
|
|
56
|
+
"BatchReport",
|
|
57
|
+
"Issue",
|
|
58
|
+
"Measure",
|
|
59
|
+
"Thresholds",
|
|
60
|
+
"DEFAULT_THRESHOLDS",
|
|
61
|
+
"describe_thresholds",
|
|
62
|
+
"PAGE_KINDS",
|
|
63
|
+
"SEVERITIES",
|
|
64
|
+
"SCORE_WEIGHTS",
|
|
65
|
+
"PASS_SCORE",
|
|
66
|
+
"__version__",
|
|
67
|
+
]
|