document-quality 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,30 @@
1
+ # Virtual environments and scratch space used while building. Never committed:
2
+ # they are large, machine-specific, and rebuilt from pyproject.toml anyway.
3
+ _venvs/
4
+ _proof/
5
+ .venv*/
6
+
7
+ # Build output. Wheels are built by the release workflow from this source, so a
8
+ # wheel in git could silently differ from the code beside it.
9
+ dist/
10
+ build/
11
+ *.egg-info/
12
+ src/*.egg-info/
13
+
14
+ # Python noise
15
+ __pycache__/
16
+ *.py[cod]
17
+ .pytest_cache/
18
+ .mypy_cache/
19
+ .ruff_cache/
20
+
21
+ # Local verification state, not source
22
+ verify_results.json
23
+ publish-log.txt
24
+ published.json
25
+
26
+ # Credentials. None of these belong here, and this line is the backstop.
27
+ .pypirc
28
+ .env
29
+ *.pem
30
+ *.key
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Pranay Mahendrakar
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,192 @@
1
+ Metadata-Version: 2.5
2
+ Name: document-quality
3
+ Version: 0.1.0
4
+ Summary: Decide whether a scanned page is good enough to OCR before you spend money OCRing it
5
+ Project-URL: Homepage, https://pypi.org/project/document-quality/
6
+ Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
7
+ Author: Pranay Mahendrakar
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: deskew,document,image-quality,ocr,pillow,preflight,scan,sharpness,show-through,skew
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Topic :: Multimedia :: Graphics
18
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
19
+ Classifier: Topic :: Text Processing
20
+ Requires-Python: >=3.9
21
+ Requires-Dist: numpy>=1.23
22
+ Requires-Dist: pillow>=9
23
+ Provides-Extra: dev
24
+ Requires-Dist: pytest>=7; extra == 'dev'
25
+ Description-Content-Type: text/markdown
26
+
27
+ # document-quality
28
+
29
+ OCR is billed per page, and a bad scan costs exactly as much as a good one before
30
+ anybody notices it was unreadable. This reads the page first and tells you whether
31
+ it is worth sending, and what to fix if it is not.
32
+
33
+ ## Install
34
+
35
+ ```
36
+ pip install document-quality
37
+ ```
38
+
39
+ ## Quickstart
40
+
41
+ ```python
42
+ import numpy as np, document_quality
43
+ from PIL import Image
44
+
45
+ page = np.full((1100, 850), 245, dtype=np.uint8) # a sheet of paper
46
+ for top in range(100, 1000, 40): # rows of word-shaped ink
47
+ for left in range(80, 740, 66):
48
+ page[top:top + 22, left:left + 50] = 30
49
+ scan = Image.fromarray(page).rotate(2.3, fillcolor=245) # fed in crooked
50
+
51
+ report = document_quality.assess(scan, dpi=300)
52
+ print(report.summary())
53
+ ```
54
+
55
+ ```
56
+ <image 850x1100>: not ready to OCR - score 91 of 100, held back by skew
57
+ Document page, 850 x 1100 px, 300 dpi (from argument), 2.8 x 3.7 in.
58
+ Text lines about 24 px tall, skew +2.29 degrees.
59
+ Note: The letters give no clear sign of which way up the page reads, so it was taken to be upright; an upside-down page cannot be ruled out.
60
+
61
+ What to do, worst first:
62
+ [failure] The page is turned 2.29 degrees counter-clockwise of horizontal.
63
+ fix: deskew by 2.3 degrees clockwise
64
+
65
+ Measures:
66
+ resolution 300.000 100 Scanned at 300 dpi (argument), ...
67
+ text_size 23.641 91 Text lines stand about 24 px tall across 23 row(s); ...
68
+ skew 2.288 40 Text runs 2.29 degrees counter-clockwise of horizontal; ...
69
+ ...
70
+ ```
71
+
72
+ The first three lines are the answer. (The note is there because this page's
73
+ "words" are solid bars: with no ascenders or descenders to read, the report
74
+ says it could not tell upright from upside down rather than guess.) `report.ocr_ready` is the yes or no,
75
+ `report.score` is 0-100, and every entry in `report.issues` carries the concrete
76
+ remedy in `.fix` - "rescan at 300 dpi", "deskew by 2.3 degrees clockwise",
77
+ "increase lighting on the left edge", "crop the black border before OCR".
78
+
79
+ For a file on disk it is the same call: `document_quality.assess("scan.png")`.
80
+
81
+ ## What it checks
82
+
83
+ - **Effective resolution** - dots per inch and the sheet size that implies. Left out
84
+ of the report entirely when no dpi is known, rather than guessed from pixel count.
85
+ - **Skew** - how far the text runs off horizontal, typically to within a tenth of
86
+ a degree across plus or minus 15 degrees, in either direction.
87
+ - **Ink-to-paper contrast** - how far the darkest ink sits from clean paper.
88
+ - **Sharpness** - how many pixels ink takes to become paper, scaled by the size of
89
+ the text so a 600 dpi scan is not marked down for spreading the same edge wider.
90
+ - **Uneven lighting** - measured as a gradient: the paper level is estimated
91
+ everywhere on the page and the report says how far it falls and towards where,
92
+ in words: "the left edge", "the top-right corner". Everything else - including
93
+ the text lines, their height and which way up they read - is measured against
94
+ that local paper level, so a shadow is reported as a shadow and not as low
95
+ contrast, show-through, a scanner border or an upside-down page.
96
+ - **Scanner borders** - only genuinely black, empty bands running in from the
97
+ edge of the image and ending in the sharp edge of the sheet count; a deep
98
+ shadow, which fades back to paper and still has text on it, never does. Borders
99
+ are cropped off before anything else is measured and reported with the pixels
100
+ to crop from each side. The black corners around a sheet scanned crooked on a
101
+ dark lid are left out of every measure too.
102
+ - **Show-through** - pale, soft marks bleeding through from the reverse side.
103
+ - **Black and white clipping** - strokes crushed to solid black, faint content
104
+ erased by the white point.
105
+ - **Text line height in pixels** - the number that actually decides whether an OCR
106
+ engine has enough pixels per character.
107
+ - **How much of the page looks like text** - the quantity behind the blank and
108
+ photograph verdicts.
109
+ - **Orientation** - a page lying on its side or upside down is caught, and a
110
+ sideways page is measured as the upright page it will be once rotated.
111
+ Upright is told from upside down by where each line's ink sits: Latin type
112
+ carries far more of it in the zone above its x-height than in the zone below
113
+ its baseline. When the letters give no clear sign - a soft scan, text in
114
+ capitals - the report says so in a note instead of guessing.
115
+
116
+ It also answers the two questions that come before all of those: is this sheet
117
+ blank, and is this a document page at all. A blank page is reported as blank, not
118
+ as eight failures about text it does not have - dust specks and a crooked scan on
119
+ a black lid included - and a faint page that still has rows of text is not
120
+ mistaken for a blank one. A photograph, or a sheet holding only a solid shape, is
121
+ reported as not a document page, not as a badly scanned one. A document page with
122
+ no rows of text on it at all is never called ready.
123
+
124
+ Greyscale and colour scans are both read, 8-bit, 16-bit or float, each on its own
125
+ fixed scale, so a faded 16-bit scan is as faded as the 8-bit one. A 4000 x 3000
126
+ scan takes well under a second. Your image is never modified. Pure numpy and
127
+ Pillow: no OpenCV, no OCR engine, no model download, no network.
128
+
129
+ ## API
130
+
131
+ | Call | What you get |
132
+ | --- | --- |
133
+ | `assess(image, *, dpi=None)` | a `PageReport` for one page |
134
+ | `assess_batch(images)` | a `BatchReport` for many, unreadable files recorded not raised |
135
+ | `estimate_skew(image)` | degrees counter-clockwise off horizontal, as a float; `0.0` with no text lines |
136
+ | `detect_orientation(image)` | `0`, `90`, `180` or `270` - the counter-clockwise turn that sets it upright |
137
+ | `describe_thresholds()` | every boundary, its value and what it means |
138
+
139
+ `image` is a path, a `PIL.Image.Image`, or a numpy array shaped `(h, w)`,
140
+ `(h, w, 1)`, `(h, w, 3)` or `(h, w, 4)`. `assess` also takes `thresholds=`,
141
+ `source=` (a name for the page) and `check_orientation=False` for pages known to
142
+ be upright.
143
+
144
+ **`PageReport`**
145
+
146
+ | Attribute | Meaning |
147
+ | --- | --- |
148
+ | `.ocr_ready` | `True` when this page is worth sending to an OCR engine |
149
+ | `.score` | overall quality for OCR, 0-100 |
150
+ | `.kind` | `"document"`, `"blank"` or `"photograph"` |
151
+ | `.skew_degrees` | counter-clockwise off horizontal; `image.rotate(-skew_degrees)` straightens it |
152
+ | `.estimated_text_height_px` | inked line height, or `None` when no rows of text were found |
153
+ | `.issues` | `list[Issue(kind, severity, message, fix)]`, worst first |
154
+ | `.fixes` | just the remedies, worst first, de-duplicated |
155
+ | `.measures` | every measure by name, with its value, score and message |
156
+ | `.dpi`, `.dpi_source` | the resolution used and where it came from, or `None` |
157
+ | `.summary()`, `.to_dict()`, `.to_json()` | the whole report as text, as a dict, as JSON |
158
+
159
+ `Issue.kind` is one of `resolution`, `dpi_tag`, `text_size`, `skew`,
160
+ `orientation`, `contrast`, `sharpness`, `lighting`, `show_through`, `clipping`,
161
+ `border`, `blank` or `not_a_document`, and `Issue.severity` is `failure`,
162
+ `warning` or `info`. Any failure keeps a page from being OCR-ready.
163
+
164
+ **`BatchReport`**: `.not_ready` (worst first), `.ready`, `.blank`, `.photographs`,
165
+ `.failures`, `.rows()`, `.issue_counts()`, `.summary()`, `.to_dict()`, `.to_json()`.
166
+
167
+ **Thresholds.** Every boundary is a field you can override, for pages that are not
168
+ 300 dpi office scans:
169
+
170
+ ```python
171
+ report = document_quality.assess(page, thresholds={"target_dpi": 600})
172
+ print(document_quality.describe_thresholds())
173
+ ```
174
+
175
+ ## CLI
176
+
177
+ ```
178
+ document-quality scan.png # the summary for one page
179
+ document-quality scans/ --dpi 300 # every image in a folder
180
+ document-quality scans/ --json # to_dict() as JSON
181
+ document-quality scan.png --output report.json
182
+ document-quality scans/ --only-problems # just the pages needing work
183
+ document-quality --list-thresholds # every boundary and what it means
184
+ ```
185
+
186
+ The exit code is `0` when every page assessed is ready to OCR, `1` when any page is
187
+ not, and `2` when no page could be read at all - so it drops straight into a shell
188
+ script guarding an OCR run. `python -m document_quality` works the same way.
189
+
190
+ ## License
191
+
192
+ MIT
@@ -0,0 +1,166 @@
1
+ # document-quality
2
+
3
+ OCR is billed per page, and a bad scan costs exactly as much as a good one before
4
+ anybody notices it was unreadable. This reads the page first and tells you whether
5
+ it is worth sending, and what to fix if it is not.
6
+
7
+ ## Install
8
+
9
+ ```
10
+ pip install document-quality
11
+ ```
12
+
13
+ ## Quickstart
14
+
15
+ ```python
16
+ import numpy as np, document_quality
17
+ from PIL import Image
18
+
19
+ page = np.full((1100, 850), 245, dtype=np.uint8) # a sheet of paper
20
+ for top in range(100, 1000, 40): # rows of word-shaped ink
21
+ for left in range(80, 740, 66):
22
+ page[top:top + 22, left:left + 50] = 30
23
+ scan = Image.fromarray(page).rotate(2.3, fillcolor=245) # fed in crooked
24
+
25
+ report = document_quality.assess(scan, dpi=300)
26
+ print(report.summary())
27
+ ```
28
+
29
+ ```
30
+ <image 850x1100>: not ready to OCR - score 91 of 100, held back by skew
31
+ Document page, 850 x 1100 px, 300 dpi (from argument), 2.8 x 3.7 in.
32
+ Text lines about 24 px tall, skew +2.29 degrees.
33
+ Note: The letters give no clear sign of which way up the page reads, so it was taken to be upright; an upside-down page cannot be ruled out.
34
+
35
+ What to do, worst first:
36
+ [failure] The page is turned 2.29 degrees counter-clockwise of horizontal.
37
+ fix: deskew by 2.3 degrees clockwise
38
+
39
+ Measures:
40
+ resolution 300.000 100 Scanned at 300 dpi (argument), ...
41
+ text_size 23.641 91 Text lines stand about 24 px tall across 23 row(s); ...
42
+ skew 2.288 40 Text runs 2.29 degrees counter-clockwise of horizontal; ...
43
+ ...
44
+ ```
45
+
46
+ The first three lines are the answer. (The note is there because this page's
47
+ "words" are solid bars: with no ascenders or descenders to read, the report
48
+ says it could not tell upright from upside down rather than guess.) `report.ocr_ready` is the yes or no,
49
+ `report.score` is 0-100, and every entry in `report.issues` carries the concrete
50
+ remedy in `.fix` - "rescan at 300 dpi", "deskew by 2.3 degrees clockwise",
51
+ "increase lighting on the left edge", "crop the black border before OCR".
52
+
53
+ For a file on disk it is the same call: `document_quality.assess("scan.png")`.
54
+
55
+ ## What it checks
56
+
57
+ - **Effective resolution** - dots per inch and the sheet size that implies. Left out
58
+ of the report entirely when no dpi is known, rather than guessed from pixel count.
59
+ - **Skew** - how far the text runs off horizontal, typically to within a tenth of
60
+ a degree across plus or minus 15 degrees, in either direction.
61
+ - **Ink-to-paper contrast** - how far the darkest ink sits from clean paper.
62
+ - **Sharpness** - how many pixels ink takes to become paper, scaled by the size of
63
+ the text so a 600 dpi scan is not marked down for spreading the same edge wider.
64
+ - **Uneven lighting** - measured as a gradient: the paper level is estimated
65
+ everywhere on the page and the report says how far it falls and towards where,
66
+ in words: "the left edge", "the top-right corner". Everything else - including
67
+ the text lines, their height and which way up they read - is measured against
68
+ that local paper level, so a shadow is reported as a shadow and not as low
69
+ contrast, show-through, a scanner border or an upside-down page.
70
+ - **Scanner borders** - only genuinely black, empty bands running in from the
71
+ edge of the image and ending in the sharp edge of the sheet count; a deep
72
+ shadow, which fades back to paper and still has text on it, never does. Borders
73
+ are cropped off before anything else is measured and reported with the pixels
74
+ to crop from each side. The black corners around a sheet scanned crooked on a
75
+ dark lid are left out of every measure too.
76
+ - **Show-through** - pale, soft marks bleeding through from the reverse side.
77
+ - **Black and white clipping** - strokes crushed to solid black, faint content
78
+ erased by the white point.
79
+ - **Text line height in pixels** - the number that actually decides whether an OCR
80
+ engine has enough pixels per character.
81
+ - **How much of the page looks like text** - the quantity behind the blank and
82
+ photograph verdicts.
83
+ - **Orientation** - a page lying on its side or upside down is caught, and a
84
+ sideways page is measured as the upright page it will be once rotated.
85
+ Upright is told from upside down by where each line's ink sits: Latin type
86
+ carries far more of it in the zone above its x-height than in the zone below
87
+ its baseline. When the letters give no clear sign - a soft scan, text in
88
+ capitals - the report says so in a note instead of guessing.
89
+
90
+ It also answers the two questions that come before all of those: is this sheet
91
+ blank, and is this a document page at all. A blank page is reported as blank, not
92
+ as eight failures about text it does not have - dust specks and a crooked scan on
93
+ a black lid included - and a faint page that still has rows of text is not
94
+ mistaken for a blank one. A photograph, or a sheet holding only a solid shape, is
95
+ reported as not a document page, not as a badly scanned one. A document page with
96
+ no rows of text on it at all is never called ready.
97
+
98
+ Greyscale and colour scans are both read, 8-bit, 16-bit or float, each on its own
99
+ fixed scale, so a faded 16-bit scan is as faded as the 8-bit one. A 4000 x 3000
100
+ scan takes well under a second. Your image is never modified. Pure numpy and
101
+ Pillow: no OpenCV, no OCR engine, no model download, no network.
102
+
103
+ ## API
104
+
105
+ | Call | What you get |
106
+ | --- | --- |
107
+ | `assess(image, *, dpi=None)` | a `PageReport` for one page |
108
+ | `assess_batch(images)` | a `BatchReport` for many, unreadable files recorded not raised |
109
+ | `estimate_skew(image)` | degrees counter-clockwise off horizontal, as a float; `0.0` with no text lines |
110
+ | `detect_orientation(image)` | `0`, `90`, `180` or `270` - the counter-clockwise turn that sets it upright |
111
+ | `describe_thresholds()` | every boundary, its value and what it means |
112
+
113
+ `image` is a path, a `PIL.Image.Image`, or a numpy array shaped `(h, w)`,
114
+ `(h, w, 1)`, `(h, w, 3)` or `(h, w, 4)`. `assess` also takes `thresholds=`,
115
+ `source=` (a name for the page) and `check_orientation=False` for pages known to
116
+ be upright.
117
+
118
+ **`PageReport`**
119
+
120
+ | Attribute | Meaning |
121
+ | --- | --- |
122
+ | `.ocr_ready` | `True` when this page is worth sending to an OCR engine |
123
+ | `.score` | overall quality for OCR, 0-100 |
124
+ | `.kind` | `"document"`, `"blank"` or `"photograph"` |
125
+ | `.skew_degrees` | counter-clockwise off horizontal; `image.rotate(-skew_degrees)` straightens it |
126
+ | `.estimated_text_height_px` | inked line height, or `None` when no rows of text were found |
127
+ | `.issues` | `list[Issue(kind, severity, message, fix)]`, worst first |
128
+ | `.fixes` | just the remedies, worst first, de-duplicated |
129
+ | `.measures` | every measure by name, with its value, score and message |
130
+ | `.dpi`, `.dpi_source` | the resolution used and where it came from, or `None` |
131
+ | `.summary()`, `.to_dict()`, `.to_json()` | the whole report as text, as a dict, as JSON |
132
+
133
+ `Issue.kind` is one of `resolution`, `dpi_tag`, `text_size`, `skew`,
134
+ `orientation`, `contrast`, `sharpness`, `lighting`, `show_through`, `clipping`,
135
+ `border`, `blank` or `not_a_document`, and `Issue.severity` is `failure`,
136
+ `warning` or `info`. Any failure keeps a page from being OCR-ready.
137
+
138
+ **`BatchReport`**: `.not_ready` (worst first), `.ready`, `.blank`, `.photographs`,
139
+ `.failures`, `.rows()`, `.issue_counts()`, `.summary()`, `.to_dict()`, `.to_json()`.
140
+
141
+ **Thresholds.** Every boundary is a field you can override, for pages that are not
142
+ 300 dpi office scans:
143
+
144
+ ```python
145
+ report = document_quality.assess(page, thresholds={"target_dpi": 600})
146
+ print(document_quality.describe_thresholds())
147
+ ```
148
+
149
+ ## CLI
150
+
151
+ ```
152
+ document-quality scan.png # the summary for one page
153
+ document-quality scans/ --dpi 300 # every image in a folder
154
+ document-quality scans/ --json # to_dict() as JSON
155
+ document-quality scan.png --output report.json
156
+ document-quality scans/ --only-problems # just the pages needing work
157
+ document-quality --list-thresholds # every boundary and what it means
158
+ ```
159
+
160
+ The exit code is `0` when every page assessed is ready to OCR, `1` when any page is
161
+ not, and `2` when no page could be read at all - so it drops straight into a shell
162
+ script guarding an OCR run. `python -m document_quality` works the same way.
163
+
164
+ ## License
165
+
166
+ MIT
@@ -0,0 +1,53 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.27"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "document-quality"
7
+ version = "0.1.0"
8
+ description = "Decide whether a scanned page is good enough to OCR before you spend money OCRing it"
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Pranay Mahendrakar" }]
14
+ keywords = [
15
+ "ocr",
16
+ "scan",
17
+ "document",
18
+ "image-quality",
19
+ "deskew",
20
+ "skew",
21
+ "preflight",
22
+ "sharpness",
23
+ "show-through",
24
+ "pillow",
25
+ ]
26
+ classifiers = [
27
+ "Development Status :: 4 - Beta",
28
+ "Intended Audience :: Developers",
29
+ "Intended Audience :: Science/Research",
30
+ "Programming Language :: Python :: 3",
31
+ "Programming Language :: Python :: 3 :: Only",
32
+ "Operating System :: OS Independent",
33
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
34
+ "Topic :: Multimedia :: Graphics",
35
+ "Topic :: Text Processing",
36
+ ]
37
+ dependencies = [
38
+ "numpy>=1.23",
39
+ "Pillow>=9",
40
+ ]
41
+
42
+ [project.optional-dependencies]
43
+ dev = ["pytest>=7"]
44
+
45
+ [project.scripts]
46
+ document-quality = "document_quality.cli:main"
47
+
48
+ [project.urls]
49
+ Homepage = "https://pypi.org/project/document-quality/"
50
+ Author = "https://pypi.org/user/pranaymahendrakar/"
51
+
52
+ [tool.hatch.build.targets.wheel]
53
+ packages = ["src/document_quality"]
@@ -0,0 +1,67 @@
1
+ """Decide whether a scanned page is good enough to OCR before you pay to OCR it.
2
+
3
+ OCR is billed per page and a bad scan costs the same as a good one, then has to
4
+ be caught, re-scanned and re-run. This package reads the page first and answers
5
+ in one line::
6
+
7
+ import document_quality
8
+
9
+ report = document_quality.assess("invoice.png")
10
+ print(report.summary())
11
+
12
+ if not report.ocr_ready:
13
+ for issue in report.issues:
14
+ print(issue.kind, "->", issue.fix)
15
+
16
+ Every verdict carries the number behind it and the boundary it was compared
17
+ against, and every problem carries the concrete remedy: "rescan at 300 dpi",
18
+ "deskew by 2.3 degrees clockwise", "increase lighting on the left edge",
19
+ "crop the black border before OCR". Nothing here
20
+ downloads a model, calls out to a network, or needs OpenCV - it is numpy and
21
+ Pillow measuring a page.
22
+
23
+ What gets measured: effective resolution, skew angle, ink-to-paper contrast,
24
+ sharpness, uneven lighting (as a gradient of the paper level, so a shadow is
25
+ never mistaken for a scanner border), show-through from the reverse side, black
26
+ and white clipping, genuinely black scanner borders, text line height in
27
+ pixels, and how much of the page looks like text.
28
+ A blank sheet is reported as blank rather than as eight failures, and a
29
+ photograph is reported as not a document page rather than as a bad one.
30
+ """
31
+ from __future__ import annotations
32
+
33
+ from ._assess import (
34
+ SCORE_WEIGHTS,
35
+ assess,
36
+ assess_batch,
37
+ detect_orientation,
38
+ estimate_skew,
39
+ )
40
+ from ._report import PAGE_KINDS, SEVERITIES, BatchReport, Issue, Measure, PageReport
41
+ from ._thresholds import (
42
+ DEFAULT_THRESHOLDS,
43
+ PASS_SCORE,
44
+ Thresholds,
45
+ describe_thresholds,
46
+ )
47
+
48
+ __version__ = "0.1.0"
49
+
50
+ __all__ = [
51
+ "assess",
52
+ "assess_batch",
53
+ "estimate_skew",
54
+ "detect_orientation",
55
+ "PageReport",
56
+ "BatchReport",
57
+ "Issue",
58
+ "Measure",
59
+ "Thresholds",
60
+ "DEFAULT_THRESHOLDS",
61
+ "describe_thresholds",
62
+ "PAGE_KINDS",
63
+ "SEVERITIES",
64
+ "SCORE_WEIGHTS",
65
+ "PASS_SCORE",
66
+ "__version__",
67
+ ]
@@ -0,0 +1,7 @@
1
+ """``python -m document_quality`` runs the same command line as ``document-quality``."""
2
+ from __future__ import annotations
3
+
4
+ from .cli import main
5
+
6
+ if __name__ == "__main__": # pragma: no cover - module entry point
7
+ raise SystemExit(main())