ocr-cleaner 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ocr_cleaner/__init__.py +55 -0
- ocr_cleaner/_analysis.py +633 -0
- ocr_cleaner/_core.py +461 -0
- ocr_cleaner/_images.py +227 -0
- ocr_cleaner/_result.py +266 -0
- ocr_cleaner/_steps.py +456 -0
- ocr_cleaner/cli.py +331 -0
- ocr_cleaner-0.1.0.dist-info/METADATA +187 -0
- ocr_cleaner-0.1.0.dist-info/RECORD +12 -0
- ocr_cleaner-0.1.0.dist-info/WHEEL +4 -0
- ocr_cleaner-0.1.0.dist-info/entry_points.txt +2 -0
- ocr_cleaner-0.1.0.dist-info/licenses/LICENSE +21 -0
ocr_cleaner/__init__.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""ocr-cleaner: prepare a scanned page so OCR reads it better.
|
|
2
|
+
|
|
3
|
+
Deskew, denoise and threshold, in one call, with every step reporting what it
|
|
4
|
+
did to the page and why - including the steps that looked and decided to do
|
|
5
|
+
nothing.
|
|
6
|
+
|
|
7
|
+
>>> import numpy as np, ocr_cleaner
|
|
8
|
+
>>> page = np.full((1100, 850), 246, dtype=np.uint8) # a sheet of paper
|
|
9
|
+
>>> for top in range(120, 1000, 34): # rows of text on it
|
|
10
|
+
... page[top:top + 11, 90:760] = 50
|
|
11
|
+
>>> result = ocr_cleaner.clean(page)
|
|
12
|
+
>>> result.page_kind
|
|
13
|
+
'document'
|
|
14
|
+
>>> "threshold" in result.applied
|
|
15
|
+
True
|
|
16
|
+
|
|
17
|
+
A blank sheet is reported blank and handed back untouched rather than
|
|
18
|
+
thresholded into a field of speckles, and a photograph is reported as a
|
|
19
|
+
photograph rather than cleaned as a bad scan.
|
|
20
|
+
|
|
21
|
+
Pure numpy and Pillow. No OpenCV, no OCR engine, no model download, nothing
|
|
22
|
+
touches the network, and the same page always gives the same result.
|
|
23
|
+
"""
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from ._analysis import MAX_SKEW_DEGREES, PAGE_KINDS
|
|
27
|
+
from ._core import (
|
|
28
|
+
ANALYSIS_MAX_SIDE,
|
|
29
|
+
DEFAULT_THRESHOLD,
|
|
30
|
+
THRESHOLD_MODES,
|
|
31
|
+
clean,
|
|
32
|
+
clean_file,
|
|
33
|
+
estimate_skew,
|
|
34
|
+
)
|
|
35
|
+
from ._images import IMAGE_SUFFIXES, open_image
|
|
36
|
+
from ._result import STEP_NAMES, CleanResult, Step
|
|
37
|
+
|
|
38
|
+
__version__ = "0.1.0"
|
|
39
|
+
|
|
40
|
+
__all__ = [
|
|
41
|
+
"clean",
|
|
42
|
+
"clean_file",
|
|
43
|
+
"estimate_skew",
|
|
44
|
+
"CleanResult",
|
|
45
|
+
"Step",
|
|
46
|
+
"STEP_NAMES",
|
|
47
|
+
"THRESHOLD_MODES",
|
|
48
|
+
"DEFAULT_THRESHOLD",
|
|
49
|
+
"PAGE_KINDS",
|
|
50
|
+
"MAX_SKEW_DEGREES",
|
|
51
|
+
"ANALYSIS_MAX_SIDE",
|
|
52
|
+
"IMAGE_SUFFIXES",
|
|
53
|
+
"open_image",
|
|
54
|
+
"__version__",
|
|
55
|
+
]
|
ocr_cleaner/_analysis.py
ADDED
|
@@ -0,0 +1,633 @@
|
|
|
1
|
+
"""Everything the pipeline needs to know before it changes a pixel.
|
|
2
|
+
|
|
3
|
+
Three questions get answered here, and all three come out of one projection
|
|
4
|
+
profile:
|
|
5
|
+
|
|
6
|
+
1. **How far is the text turned?** Lines of text are the only strong periodic
|
|
7
|
+
structure on a page. Shear the ink by a candidate angle, sum each row, and
|
|
8
|
+
the profile becomes a comb: tall teeth where lines are, near-zero between
|
|
9
|
+
them. The comb is sharpest at exactly one angle, and that angle is the skew.
|
|
10
|
+
2. **How tall is a line of text?** The teeth of that comb, measured at the angle
|
|
11
|
+
that sharpened them. This is the number every later step is sized from - the
|
|
12
|
+
median window, the adaptive threshold window, the upscale target.
|
|
13
|
+
3. **What kind of page is this?** Blank paper, a photograph, or a document. A
|
|
14
|
+
blank sheet must not be thresholded into a field of noise, and a photograph
|
|
15
|
+
must not be binarised at all, so both have to be recognised before anything
|
|
16
|
+
is applied.
|
|
17
|
+
|
|
18
|
+
Nothing is rotated or resampled to measure an angle. Shearing by row offsets is
|
|
19
|
+
both faster than bicubic rotation and more honest here, because rotation smears
|
|
20
|
+
the very edges being counted.
|
|
21
|
+
|
|
22
|
+
Angles follow ``PIL.Image.rotate``: positive is counter-clockwise, so a page
|
|
23
|
+
whose text runs downhill to the right has a negative skew, and
|
|
24
|
+
``image.rotate(-estimate_skew(image))`` puts it straight.
|
|
25
|
+
"""
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import logging
|
|
29
|
+
from dataclasses import dataclass
|
|
30
|
+
from typing import List, NamedTuple, Optional, Sequence, Tuple
|
|
31
|
+
|
|
32
|
+
import numpy as np
|
|
33
|
+
|
|
34
|
+
from . import _images
|
|
35
|
+
|
|
36
|
+
logger = logging.getLogger(__name__)
|
|
37
|
+
|
|
38
|
+
#: Widest skew the search looks at, in degrees either side of upright. Beyond
|
|
39
|
+
#: this a page is not skewed, it is turned, which is a different repair.
|
|
40
|
+
MAX_SKEW_DEGREES = 15.0
|
|
41
|
+
#: The search, stage by stage: ``(strip width in columns, half-width of the
|
|
42
|
+
#: window in degrees, step in degrees, use the smooth shear)``. ``0`` columns
|
|
43
|
+
#: means the whole page in one piece. Each stage widens the aperture - which
|
|
44
|
+
#: sharpens the peak - and narrows the window around what the last stage found.
|
|
45
|
+
#: :func:`_search` explains why the aperture is the thing that matters.
|
|
46
|
+
SEARCH_STAGES = (
|
|
47
|
+
(96, MAX_SKEW_DEGREES, 1.0, False),
|
|
48
|
+
(288, 1.1, 0.2, True),
|
|
49
|
+
(0, 0.28, 0.035, True),
|
|
50
|
+
)
|
|
51
|
+
#: Narrowest strip worth shearing on its own.
|
|
52
|
+
MIN_STRIP_COLUMNS = 24
|
|
53
|
+
#: Longest side the angle search itself runs on. Not a free choice: the comb
|
|
54
|
+
#: the search looks for only exists while the plane still has a few rows per
|
|
55
|
+
#: line of text, and a 300 dpi page of six-point type reduced to 800 pixels is
|
|
56
|
+
#: down to four. Reduced to 500 it is down to two and a half, and the measured
|
|
57
|
+
#: angle starts coming back degrees wrong rather than hundredths. Going the
|
|
58
|
+
#: other way buys nothing, so this sits where the margin is comfortable and the
|
|
59
|
+
#: cost is not. Line height is measured separately on the full analysis plane,
|
|
60
|
+
#: at the angle the search found.
|
|
61
|
+
SEARCH_MAX_SIDE = 800
|
|
62
|
+
#: Smallest plane side worth measuring lines on.
|
|
63
|
+
MIN_PLANE_SIDE = 24
|
|
64
|
+
#: Share of the profile swing that separates a line of text from the gap above.
|
|
65
|
+
LINE_THRESHOLD = 0.35
|
|
66
|
+
#: Where the cut for the full inked extent sits, as a share of a line's own
|
|
67
|
+
#: height above the paper between lines. It is deliberately close to the paper,
|
|
68
|
+
#: because only a handful of glyphs in a line carry an ascender or a descender:
|
|
69
|
+
#: measured on rendered Times and Arial from 10 to 60 pixels, those rows sum to
|
|
70
|
+
#: between 6 and 20 percent of an x-height row. A cut set anywhere inside that
|
|
71
|
+
#: band chops the tails off, and one set above it breaks the line into slivers.
|
|
72
|
+
#: Sweeping this against the true ascender-to-descender extent of rendered text
|
|
73
|
+
#: - eight sizes, both faces, tight and loose leading, grain to sigma 18 - every
|
|
74
|
+
#: value from 0.02 to 0.04 lands within a tenth of the truth; 0.02 starts
|
|
75
|
+
#: reading grain as a descender on the noisiest page, so the cut sits at the top
|
|
76
|
+
#: of the range that still measures the whole line.
|
|
77
|
+
EXTENT_SHARE = 0.04
|
|
78
|
+
#: Shortest run of rows that can be part of a line of text.
|
|
79
|
+
MIN_LINE_ROWS = 2
|
|
80
|
+
#: Profile swing relative to its own mean, below which there are no text rows.
|
|
81
|
+
#: Pages of text measure 0.75 and up, and a sparse one measures several;
|
|
82
|
+
#: photographs and grainy blank paper sit between 0.05 and 0.2.
|
|
83
|
+
TEXT_LINE_CONTRAST = 0.5
|
|
84
|
+
#: How many line-shaped bands must repeat down the page before it is text. A
|
|
85
|
+
#: photograph's broad tonal bands can out-swing a text comb, but they repeat
|
|
86
|
+
#: three or four times in a frame where text repeats dozens.
|
|
87
|
+
MIN_TEXT_LINES = 6
|
|
88
|
+
#: Share of the page that must come out as bare paper for it to be a document.
|
|
89
|
+
#: Scans of text run 0.9 and up; photographs rarely clear a half.
|
|
90
|
+
MIN_PAPER_SHARE = 0.72
|
|
91
|
+
#: Ink strength, on the 0 to 1 scale of :func:`ink_plane`, at which a pixel is a
|
|
92
|
+
#: stroke rather than paper.
|
|
93
|
+
INK_CUT = 0.25
|
|
94
|
+
#: Ink strength below which a pixel is bare paper.
|
|
95
|
+
PAPER_CUT = 0.08
|
|
96
|
+
#: Ink coverage at or below which a page holds nothing worth cleaning.
|
|
97
|
+
BLANK_INK_SHARE = 0.0015
|
|
98
|
+
#: Denominator for the background radius, as a share of the plane's short side.
|
|
99
|
+
#: Wide enough to ignore the text, narrow enough to follow a lighting gradient.
|
|
100
|
+
BACKGROUND_DIVISOR = 24.0
|
|
101
|
+
#: Floor for that radius, so a small page still gets a usable background.
|
|
102
|
+
BACKGROUND_MIN_RADIUS = 4
|
|
103
|
+
#: Percentile of the ink plane taken as "the darkest stroke".
|
|
104
|
+
INK_PEAK_PERCENTILE = 99.7
|
|
105
|
+
#: How far the darkest stroke must sit below the paper beside it, in grey
|
|
106
|
+
#: levels, before a page has ink on it at all. The ink plane is scaled to its
|
|
107
|
+
#: own darkest mark, which is what lets it read a faint pencil page - and which
|
|
108
|
+
#: would just as happily scale the grain on an empty sheet into what looks like
|
|
109
|
+
#: dense text. This is the floor that stops it.
|
|
110
|
+
MIN_INK_DEPTH = 14.0
|
|
111
|
+
#: How much the ink has to be gathered into marks, rather than spread evenly as
|
|
112
|
+
#: grain, before a page has anything on it. Measured by :func:`ink_structure`.
|
|
113
|
+
#: Pages of text measure 17 and up; empty paper, however grainy, stays under 5.
|
|
114
|
+
MIN_INK_STRUCTURE = 8.0
|
|
115
|
+
#: Denominator for the radius that blur is done at, as a share of the short side.
|
|
116
|
+
STRUCTURE_DIVISOR = 64.0
|
|
117
|
+
#: Floor for that radius.
|
|
118
|
+
STRUCTURE_MIN_RADIUS = 3
|
|
119
|
+
|
|
120
|
+
#: The three answers :attr:`PageStats.kind` can give.
|
|
121
|
+
PAGE_KINDS = ("document", "blank", "photograph")
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def background_radius(plane: np.ndarray) -> int:
|
|
125
|
+
"""Box radius that follows the lighting on a page without following its text."""
|
|
126
|
+
return int(max(BACKGROUND_MIN_RADIUS, round(min(plane.shape) / BACKGROUND_DIVISOR)))
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
class InkPlane(NamedTuple):
|
|
130
|
+
"""The ink on a page, and the two numbers that say whether it is ink at all."""
|
|
131
|
+
|
|
132
|
+
#: 0.0 where the paper is, 1.0 at the darkest stroke.
|
|
133
|
+
values: np.ndarray
|
|
134
|
+
#: How far that darkest stroke ran below the paper beside it, in grey levels.
|
|
135
|
+
depth: float
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def ink_plane(plane: np.ndarray) -> InkPlane:
|
|
139
|
+
"""Turn luminance into ink. See :class:`InkPlane` for what comes back.
|
|
140
|
+
|
|
141
|
+
``values`` runs 0.0 where the paper is to 1.0 at the darkest stroke. That
|
|
142
|
+
scaling is what lets the same code read a faint pencil page and a crisp
|
|
143
|
+
laser one - and it would just as happily scale the grain on an empty sheet
|
|
144
|
+
into what looks like dense text, which is why ``depth`` and ``light_depth``
|
|
145
|
+
come back with it.
|
|
146
|
+
|
|
147
|
+
Ink is measured against the paper *beside it*, not against one number for
|
|
148
|
+
the whole sheet. A scan lit from one side can be seventy grey levels darker
|
|
149
|
+
at the far edge than at the near one, and a single paper level then reads
|
|
150
|
+
that whole edge as ink - which buries the comb the skew search is looking
|
|
151
|
+
for, and makes a perfectly ordinary page look like a photograph. Subtracting
|
|
152
|
+
a wide local mean instead removes any lighting that varies more slowly than
|
|
153
|
+
the text does, and it flattens a black scanner margin to nothing at the same
|
|
154
|
+
time, so the border stops pulling on the angle.
|
|
155
|
+
|
|
156
|
+
The scale comes from a high percentile rather than the maximum, so a single
|
|
157
|
+
hot pixel of impulse noise cannot decide what "the darkest stroke" means.
|
|
158
|
+
"""
|
|
159
|
+
floats = plane.astype(np.float64)
|
|
160
|
+
background = _images.local_mean(plane, background_radius(plane))
|
|
161
|
+
ink = np.clip(background - floats, 0.0, None)
|
|
162
|
+
peak = float(np.percentile(ink, INK_PEAK_PERCENTILE))
|
|
163
|
+
if peak <= 0.0:
|
|
164
|
+
return InkPlane(np.zeros_like(ink), 0.0)
|
|
165
|
+
return InkPlane(np.clip(ink / peak, 0.0, 1.0), peak)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def ink_structure(ink: np.ndarray) -> float:
|
|
169
|
+
"""How far from evenly spread the ink on a page is, in grey levels.
|
|
170
|
+
|
|
171
|
+
Ink gathers. Grain does not. Blur the ink plane over a small neighbourhood
|
|
172
|
+
and grain averages away to a flat field, because it is as likely to land
|
|
173
|
+
anywhere as anywhere else, while strokes and margins keep their difference,
|
|
174
|
+
because ink is somewhere and not elsewhere. What comes back is how much of
|
|
175
|
+
that difference is left, and it is the measurement that tells a faint page
|
|
176
|
+
of pencil from a dusty empty one - the two look identical to anything that
|
|
177
|
+
only counts how dark the darkest mark is.
|
|
178
|
+
"""
|
|
179
|
+
if ink.size == 0: # pragma: no cover - guarded by callers
|
|
180
|
+
return 0.0
|
|
181
|
+
radius = int(max(STRUCTURE_MIN_RADIUS, round(min(ink.shape) / STRUCTURE_DIVISOR)))
|
|
182
|
+
levels = np.clip(ink * 255.0, 0.0, 255.0).astype(np.uint8)
|
|
183
|
+
return float(_images.local_mean(levels, radius).std())
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _shear_margin(width: int, degrees: float) -> int:
|
|
187
|
+
"""Rows of headroom a shear of ``degrees`` needs at each end of the profile."""
|
|
188
|
+
return int(np.ceil(width * 0.5 * abs(np.tan(np.radians(degrees))))) + 1
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
class ProfileBank:
|
|
192
|
+
"""One ink plane, ready to be sheared to any angle inside its margin.
|
|
193
|
+
|
|
194
|
+
A shear moves every pixel in a column by the same amount, because the
|
|
195
|
+
offset depends on the column and not on the row. That one fact is what
|
|
196
|
+
makes this cheap. Rather than computing a destination row for each of the
|
|
197
|
+
plane's pixels and scattering them - which is a random write per pixel, and
|
|
198
|
+
the slowest thing numpy does - the columns are grouped by the whole number
|
|
199
|
+
of rows they move, each group is summed with a single ``add.reduceat``, and
|
|
200
|
+
the handful of group totals are added into the profile at their offsets.
|
|
201
|
+
The work stops being one pass per pixel and becomes two passes plus a short
|
|
202
|
+
loop, and the profile that comes out is identical to the last float.
|
|
203
|
+
|
|
204
|
+
Two shears are offered, and they cost a few times apart. :meth:`profile`
|
|
205
|
+
rounds each column to a whole row, which is cheap enough to sweep the whole
|
|
206
|
+
range. Rounding quantises the answer though: on a nearly upright page every
|
|
207
|
+
angle under about a tenth of a degree rounds to the same offsets, so a
|
|
208
|
+
sweep using it can only bracket the truth. :meth:`profile_fine` splits each
|
|
209
|
+
column between the two rows it falls between, which makes the score a
|
|
210
|
+
smooth function of the angle. The search uses the cheap one to find the
|
|
211
|
+
neighbourhood and the smooth one to land inside it.
|
|
212
|
+
"""
|
|
213
|
+
|
|
214
|
+
def __init__(self, ink: np.ndarray, max_degrees: float) -> None:
|
|
215
|
+
self.ink = np.ascontiguousarray(ink, dtype=np.float64)
|
|
216
|
+
self.height, self.width = self.ink.shape
|
|
217
|
+
self.total = float(self.ink.sum())
|
|
218
|
+
self.margin = _shear_margin(self.width, max_degrees)
|
|
219
|
+
self._length = self.height + 2 * self.margin
|
|
220
|
+
self._centred = np.arange(self.width, dtype=np.float64) - (self.width - 1) / 2.0
|
|
221
|
+
|
|
222
|
+
@property
|
|
223
|
+
def usable(self) -> bool:
|
|
224
|
+
"""True when enough whole rows survive the shear margin to mean anything."""
|
|
225
|
+
return self.total > 0.0 and self.height - 2 * self.margin >= 8
|
|
226
|
+
|
|
227
|
+
def _offsets(self, degrees: float) -> np.ndarray:
|
|
228
|
+
"""Row offset each column takes when the plane is sheared by ``degrees``."""
|
|
229
|
+
return self._centred * np.tan(np.radians(degrees))
|
|
230
|
+
|
|
231
|
+
@staticmethod
|
|
232
|
+
def _runs(shifts: np.ndarray) -> Tuple[np.ndarray, np.ndarray]:
|
|
233
|
+
"""Where each run of equal shifts starts, and what it shifts by.
|
|
234
|
+
|
|
235
|
+
The offsets grow steadily across the page, so columns that move by the
|
|
236
|
+
same whole number of rows are always next to each other and a run is
|
|
237
|
+
enough to describe them.
|
|
238
|
+
"""
|
|
239
|
+
change = np.flatnonzero(np.diff(shifts)) + 1
|
|
240
|
+
starts = np.empty(change.size + 1, dtype=np.intp)
|
|
241
|
+
starts[0] = 0
|
|
242
|
+
starts[1:] = change
|
|
243
|
+
return starts, shifts[starts]
|
|
244
|
+
|
|
245
|
+
def profile(self, degrees: float) -> np.ndarray:
|
|
246
|
+
"""Row sums after shearing by ``degrees``, each column rounded to a row."""
|
|
247
|
+
starts, shifts = self._runs(np.rint(self._offsets(degrees)).astype(np.int64))
|
|
248
|
+
grouped = np.add.reduceat(self.ink, starts, axis=1)
|
|
249
|
+
summed = np.zeros(self._length + 2)
|
|
250
|
+
for group, shift in enumerate(shifts):
|
|
251
|
+
start = self.margin + int(shift)
|
|
252
|
+
summed[start: start + self.height] += grouped[:, group]
|
|
253
|
+
return summed[2 * self.margin: self.height]
|
|
254
|
+
|
|
255
|
+
def profile_fine(self, degrees: float) -> np.ndarray:
|
|
256
|
+
"""Row sums after shearing by ``degrees``, split between adjacent rows.
|
|
257
|
+
|
|
258
|
+
Each column lands between two rows and gives each its share, so nudging
|
|
259
|
+
the angle by a hundredth of a degree moves the profile a little rather
|
|
260
|
+
than not at all. This is what makes agreement with a known rotation to
|
|
261
|
+
a fraction of a degree possible.
|
|
262
|
+
"""
|
|
263
|
+
offsets = self._offsets(degrees)
|
|
264
|
+
lower = np.floor(offsets)
|
|
265
|
+
starts, shifts = self._runs(lower.astype(np.int64))
|
|
266
|
+
upper_share = self.ink * (offsets - lower)[None, :]
|
|
267
|
+
whole = np.add.reduceat(self.ink, starts, axis=1)
|
|
268
|
+
upper = np.add.reduceat(upper_share, starts, axis=1)
|
|
269
|
+
lower_part = whole - upper
|
|
270
|
+
summed = np.zeros(self._length + 2)
|
|
271
|
+
for group, shift in enumerate(shifts):
|
|
272
|
+
start = self.margin + int(shift)
|
|
273
|
+
summed[start: start + self.height] += lower_part[:, group]
|
|
274
|
+
summed[start + 1: start + 1 + self.height] += upper[:, group]
|
|
275
|
+
return summed[2 * self.margin: self.height]
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def comb_score(profile: np.ndarray) -> float:
|
|
279
|
+
"""How comb-like a profile is: the energy of its first difference.
|
|
280
|
+
|
|
281
|
+
Row variance peaks at the right angle too, but the first difference is
|
|
282
|
+
blind to a page-wide lighting ramp, which variance happily rewards.
|
|
283
|
+
"""
|
|
284
|
+
if profile.size < 2:
|
|
285
|
+
return 0.0
|
|
286
|
+
step = np.diff(profile)
|
|
287
|
+
return float(np.dot(step, step))
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def strip_banks(ink: np.ndarray, columns: int) -> List[ProfileBank]:
|
|
291
|
+
"""Cut the ink plane into vertical strips, each ready to be sheared.
|
|
292
|
+
|
|
293
|
+
``columns`` is the width each strip should be near; ``0`` means one strip
|
|
294
|
+
covering the whole page. Strips are what set the search's aperture, and the
|
|
295
|
+
aperture is what sets how wide the peak is - see :func:`_search`.
|
|
296
|
+
"""
|
|
297
|
+
height, width = ink.shape
|
|
298
|
+
if columns <= 0 or width <= columns * 1.5:
|
|
299
|
+
return [ProfileBank(ink, MAX_SKEW_DEGREES)]
|
|
300
|
+
count = max(1, int(round(width / float(columns))))
|
|
301
|
+
edges = np.linspace(0, width, count + 1).astype(int)
|
|
302
|
+
banks = [
|
|
303
|
+
ProfileBank(np.ascontiguousarray(ink[:, start:stop]), MAX_SKEW_DEGREES)
|
|
304
|
+
for start, stop in zip(edges[:-1], edges[1:])
|
|
305
|
+
if stop - start >= MIN_STRIP_COLUMNS
|
|
306
|
+
]
|
|
307
|
+
usable = [bank for bank in banks if bank.usable]
|
|
308
|
+
return usable or [ProfileBank(ink, MAX_SKEW_DEGREES)]
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def _stage_score(banks: Sequence[ProfileBank], degrees: float, smooth: bool) -> float:
|
|
312
|
+
"""Total comb score across every strip at one angle."""
|
|
313
|
+
return sum(
|
|
314
|
+
comb_score(bank.profile_fine(degrees) if smooth else bank.profile(degrees))
|
|
315
|
+
for bank in banks
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def _refine(angles: np.ndarray, scores: np.ndarray, pick: int) -> float:
|
|
320
|
+
"""Where the peak really sits, from the three samples around the best one.
|
|
321
|
+
|
|
322
|
+
The comb score near its maximum is close enough to a parabola that fitting
|
|
323
|
+
one through three samples lands nearer the truth than the sample itself -
|
|
324
|
+
and it is free, where another sweep costs a pass over the page.
|
|
325
|
+
"""
|
|
326
|
+
best = float(angles[pick])
|
|
327
|
+
if pick <= 0 or pick >= len(scores) - 1 or len(angles) < 2:
|
|
328
|
+
return best
|
|
329
|
+
left, middle, right = float(scores[pick - 1]), float(scores[pick]), float(scores[pick + 1])
|
|
330
|
+
bend = left - 2.0 * middle + right
|
|
331
|
+
if bend == 0.0:
|
|
332
|
+
return best
|
|
333
|
+
shift = 0.5 * (left - right) / bend
|
|
334
|
+
if abs(shift) > 1.0: # pragma: no cover - a peak that is not a peak
|
|
335
|
+
return best
|
|
336
|
+
return best + shift * float(angles[1] - angles[0])
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def _search(ink: np.ndarray) -> float:
|
|
340
|
+
"""Find the angle whose comb is sharpest, in degrees counter-clockwise.
|
|
341
|
+
|
|
342
|
+
How sharply the comb answers to angle is set by a single ratio: the height
|
|
343
|
+
of a line of text over the width of the page. A line stays lined up under a
|
|
344
|
+
shear until the shear has moved one end of the page by about its own
|
|
345
|
+
height, so the peak is roughly ``line_height / page_width`` radians wide. On
|
|
346
|
+
a page of six-pixel text twenty-five hundred pixels across, that is a
|
|
347
|
+
seventh of a degree, and a sweep in one-degree steps walks straight over it
|
|
348
|
+
and reports whatever noise it lands on instead. This is not hypothetical: it
|
|
349
|
+
put a page eight degrees out.
|
|
350
|
+
|
|
351
|
+
Narrowing the *aperture* is the fix. The same lines measured across a strip
|
|
352
|
+
a quarter of the page wide give a peak four times broader, centred on the
|
|
353
|
+
same angle - and squeezing the page's columns together instead would not,
|
|
354
|
+
because that scales the angle by exactly as much as it broadens the peak.
|
|
355
|
+
So the search begins on narrow strips, where the peak is wide enough for a
|
|
356
|
+
coarse sweep to find, and widens the aperture as it narrows the window:
|
|
357
|
+
each stage sharper than the last, each looking only where the last one
|
|
358
|
+
pointed. The strips' scores are summed rather than one strip chosen, so
|
|
359
|
+
every line of ink on the page stays in the measurement.
|
|
360
|
+
"""
|
|
361
|
+
best = 0.0
|
|
362
|
+
for columns, span, step, smooth in SEARCH_STAGES:
|
|
363
|
+
banks = strip_banks(ink, columns)
|
|
364
|
+
if not banks or not any(bank.usable for bank in banks):
|
|
365
|
+
return 0.0
|
|
366
|
+
low = max(-MAX_SKEW_DEGREES, best - span)
|
|
367
|
+
high = min(MAX_SKEW_DEGREES, best + span)
|
|
368
|
+
count = max(int(round((high - low) / step)) + 1, 1)
|
|
369
|
+
angles = np.linspace(low, high, count)
|
|
370
|
+
scores = np.array([_stage_score(banks, angle, smooth) for angle in angles])
|
|
371
|
+
best = _refine(angles, scores, int(np.argmax(scores)))
|
|
372
|
+
return float(best)
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def _runs(profile: np.ndarray, threshold: float, minimum: int) -> List[Tuple[int, int]]:
|
|
376
|
+
"""Runs of rows above ``threshold``, as half-open ``(start, stop)`` pairs."""
|
|
377
|
+
above = profile > threshold
|
|
378
|
+
if not above.any():
|
|
379
|
+
return []
|
|
380
|
+
edges = np.diff(above.astype(np.int8))
|
|
381
|
+
starts = list(np.flatnonzero(edges == 1) + 1)
|
|
382
|
+
stops = list(np.flatnonzero(edges == -1) + 1)
|
|
383
|
+
if above[0]:
|
|
384
|
+
starts.insert(0, 0)
|
|
385
|
+
if above[-1]:
|
|
386
|
+
stops.append(int(above.size))
|
|
387
|
+
return [(int(a), int(b)) for a, b in zip(starts, stops) if b - a >= minimum]
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
@dataclass
|
|
391
|
+
class LineGeometry:
|
|
392
|
+
"""What the straightened profile says about the lines of text on a page."""
|
|
393
|
+
|
|
394
|
+
#: Height of an inked line, ascender top to descender foot, in plane pixels.
|
|
395
|
+
text_height: Optional[float] = None
|
|
396
|
+
#: Median baseline-to-baseline distance, in plane pixels.
|
|
397
|
+
pitch: Optional[float] = None
|
|
398
|
+
#: How many line-shaped bands were found.
|
|
399
|
+
line_count: int = 0
|
|
400
|
+
#: Profile swing relative to its own mean. A page with no lines scores ~0.
|
|
401
|
+
line_contrast: float = 0.0
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
def paper_between(profile: np.ndarray, cores: Sequence[Tuple[int, int]]) -> float:
|
|
405
|
+
"""The level the profile falls back to between two lines of text.
|
|
406
|
+
|
|
407
|
+
The lowest point in each gap, medianed over the gaps, so one wide margin or
|
|
408
|
+
one pair of lines that touch cannot move it. This is the zero that a line's
|
|
409
|
+
own height is measured from - the profile's global minimum would do on a
|
|
410
|
+
clean page, but on a lit or grainy one the darkest row of the page is not
|
|
411
|
+
the level the gaps sit at.
|
|
412
|
+
"""
|
|
413
|
+
dips = [
|
|
414
|
+
float(profile[stop:start].min())
|
|
415
|
+
for (_, stop), (start, _) in zip(cores, cores[1:])
|
|
416
|
+
if start > stop
|
|
417
|
+
]
|
|
418
|
+
return float(np.median(dips)) if dips else float(profile.min())
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def _grow_to_extent(
|
|
422
|
+
profile: np.ndarray, core: Tuple[int, int], cut: float, bounds: Tuple[int, int]
|
|
423
|
+
) -> Tuple[int, int]:
|
|
424
|
+
"""Walk out from a line's x-height band to its ascender top and descender foot.
|
|
425
|
+
|
|
426
|
+
Growing outward from the core is what ties the extent to the line it
|
|
427
|
+
belongs to. Thresholding the whole profile a second time and measuring
|
|
428
|
+
whatever comes out does not: the low cut also lifts the ascenders and
|
|
429
|
+
descenders sitting alone in the gaps into runs of their own, and they
|
|
430
|
+
outnumber the lines.
|
|
431
|
+
"""
|
|
432
|
+
start, stop = core
|
|
433
|
+
floor, ceiling = bounds
|
|
434
|
+
while start > floor and profile[start - 1] > cut:
|
|
435
|
+
start -= 1
|
|
436
|
+
while stop < ceiling and profile[stop] > cut:
|
|
437
|
+
stop += 1
|
|
438
|
+
return start, stop
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def _extent_bounds(
|
|
442
|
+
cores: Sequence[Tuple[int, int]], index: int, size: int
|
|
443
|
+
) -> Tuple[int, int]:
|
|
444
|
+
"""How far a line may grow: to the middle of the gap to each neighbour."""
|
|
445
|
+
start, stop = cores[index]
|
|
446
|
+
floor = 0 if index == 0 else (cores[index - 1][1] + start) // 2
|
|
447
|
+
ceiling = size if index + 1 == len(cores) else (stop + cores[index + 1][0] + 1) // 2
|
|
448
|
+
return min(floor, start), max(ceiling, stop)
|
|
449
|
+
|
|
450
|
+
|
|
451
|
+
def line_geometry(profile: np.ndarray) -> LineGeometry:
|
|
452
|
+
"""Measure the teeth of a straightened comb profile."""
|
|
453
|
+
if profile.size < MIN_PLANE_SIDE:
|
|
454
|
+
return LineGeometry()
|
|
455
|
+
mean = float(profile.mean())
|
|
456
|
+
if mean <= 0.0:
|
|
457
|
+
return LineGeometry()
|
|
458
|
+
low, high = float(profile.min()), float(profile.max())
|
|
459
|
+
swing = high - low
|
|
460
|
+
contrast = float(profile.std()) / mean
|
|
461
|
+
if swing <= 0.0:
|
|
462
|
+
return LineGeometry(line_contrast=contrast)
|
|
463
|
+
cores = _runs(profile, low + LINE_THRESHOLD * swing, MIN_LINE_ROWS)
|
|
464
|
+
if not cores:
|
|
465
|
+
return LineGeometry(line_contrast=contrast)
|
|
466
|
+
paper = paper_between(profile, cores)
|
|
467
|
+
body = float(np.median([profile[start:stop].max() for start, stop in cores]))
|
|
468
|
+
cut = paper + EXTENT_SHARE * max(body - paper, 0.0)
|
|
469
|
+
heights = [
|
|
470
|
+
stop - start
|
|
471
|
+
for start, stop in (
|
|
472
|
+
_grow_to_extent(
|
|
473
|
+
profile, core, cut, _extent_bounds(cores, index, int(profile.size))
|
|
474
|
+
)
|
|
475
|
+
for index, core in enumerate(cores)
|
|
476
|
+
)
|
|
477
|
+
]
|
|
478
|
+
centres = [(start + stop) / 2.0 for start, stop in cores]
|
|
479
|
+
pitch = None
|
|
480
|
+
if len(centres) >= 2:
|
|
481
|
+
gaps = np.diff(np.asarray(centres, dtype=np.float64))
|
|
482
|
+
pitch = float(np.median(gaps))
|
|
483
|
+
return LineGeometry(
|
|
484
|
+
text_height=float(np.median(np.asarray(heights, dtype=np.float64))),
|
|
485
|
+
pitch=pitch,
|
|
486
|
+
line_count=len(cores),
|
|
487
|
+
line_contrast=contrast,
|
|
488
|
+
)
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
@dataclass
|
|
492
|
+
class PageStats:
|
|
493
|
+
"""One page, measured. Every later decision is taken from these numbers.
|
|
494
|
+
|
|
495
|
+
Attributes:
|
|
496
|
+
kind: ``"document"``, ``"blank"`` or ``"photograph"``.
|
|
497
|
+
why: one sentence explaining the kind, in plain language.
|
|
498
|
+
skew_degrees: counter-clockwise degrees the text is off horizontal.
|
|
499
|
+
text_height: height of a line of text in *source* pixels, or ``None``
|
|
500
|
+
when no lines were found.
|
|
501
|
+
line_count: how many lines the profile found, on the analysis plane.
|
|
502
|
+
line_contrast: how far the profile swings relative to its own mean.
|
|
503
|
+
paper_level: luminance of clean paper, 0 to 255.
|
|
504
|
+
ink_depth: how far the darkest stroke sits below the paper beside it,
|
|
505
|
+
in grey levels. Below :data:`MIN_INK_DEPTH` there is no ink, only
|
|
506
|
+
grain.
|
|
507
|
+
ink_structure: how far from evenly spread that ink is, in grey levels.
|
|
508
|
+
Below :data:`MIN_INK_STRUCTURE` it is grain, not marks.
|
|
509
|
+
ink_share: share of pixels that are a stroke, on the local scale
|
|
510
|
+
:func:`ink_plane` measures.
|
|
511
|
+
paper_share: share of pixels that are bare paper on that same scale.
|
|
512
|
+
"""
|
|
513
|
+
|
|
514
|
+
kind: str = "document"
|
|
515
|
+
why: str = ""
|
|
516
|
+
skew_degrees: float = 0.0
|
|
517
|
+
text_height: Optional[float] = None
|
|
518
|
+
line_count: int = 0
|
|
519
|
+
line_contrast: float = 0.0
|
|
520
|
+
paper_level: float = 255.0
|
|
521
|
+
ink_depth: float = 0.0
|
|
522
|
+
ink_structure: float = 0.0
|
|
523
|
+
ink_share: float = 0.0
|
|
524
|
+
paper_share: float = 0.0
|
|
525
|
+
|
|
526
|
+
@property
|
|
527
|
+
def is_document(self) -> bool:
|
|
528
|
+
"""True when the page is worth putting through the pipeline."""
|
|
529
|
+
return self.kind == "document"
|
|
530
|
+
|
|
531
|
+
|
|
532
|
+
def measure(plane: np.ndarray, scale: float = 1.0) -> PageStats:
|
|
533
|
+
"""Measure an analysis plane: kind, skew, and the height of a line of text.
|
|
534
|
+
|
|
535
|
+
Args:
|
|
536
|
+
plane: an 8-bit luminance plane, already downscaled for analysis.
|
|
537
|
+
scale: how much that downscale shrank the source, so lengths can be
|
|
538
|
+
reported back in source pixels.
|
|
539
|
+
"""
|
|
540
|
+
stats = PageStats(paper_level=_images.paper_level(plane) if plane.size else 255.0)
|
|
541
|
+
|
|
542
|
+
if plane.size == 0 or min(plane.shape) < MIN_PLANE_SIDE:
|
|
543
|
+
stats.kind = "blank"
|
|
544
|
+
stats.why = "the page is too small to hold a line of text"
|
|
545
|
+
return stats
|
|
546
|
+
|
|
547
|
+
coarse, _ = _images.analysis_plane(plane, SEARCH_MAX_SIDE)
|
|
548
|
+
coarse_ink, ink_depth = ink_plane(coarse)
|
|
549
|
+
stats.ink_depth = ink_depth
|
|
550
|
+
ink_share = float(np.mean(coarse_ink > INK_CUT))
|
|
551
|
+
paper_share = float(np.mean(coarse_ink < PAPER_CUT))
|
|
552
|
+
stats.ink_share = ink_share
|
|
553
|
+
stats.paper_share = paper_share
|
|
554
|
+
stats.ink_structure = ink_structure(coarse_ink)
|
|
555
|
+
if not ProfileBank(coarse_ink, MAX_SKEW_DEGREES).usable:
|
|
556
|
+
stats.kind = "blank"
|
|
557
|
+
stats.why = "the page carries no ink at all"
|
|
558
|
+
return stats
|
|
559
|
+
|
|
560
|
+
degrees = _search(coarse_ink)
|
|
561
|
+
detail_ink = coarse_ink if coarse.shape == plane.shape else ink_plane(plane).values
|
|
562
|
+
detail_bank = ProfileBank(detail_ink, MAX_SKEW_DEGREES)
|
|
563
|
+
geometry = (
|
|
564
|
+
line_geometry(detail_bank.profile_fine(degrees))
|
|
565
|
+
if detail_bank.usable
|
|
566
|
+
else LineGeometry()
|
|
567
|
+
)
|
|
568
|
+
stats.skew_degrees = float(degrees)
|
|
569
|
+
stats.line_count = geometry.line_count
|
|
570
|
+
stats.line_contrast = geometry.line_contrast
|
|
571
|
+
if geometry.text_height is not None and scale > 0.0:
|
|
572
|
+
stats.text_height = float(geometry.text_height) / scale
|
|
573
|
+
|
|
574
|
+
has_lines = (
|
|
575
|
+
geometry.line_count >= MIN_TEXT_LINES
|
|
576
|
+
and geometry.line_contrast >= TEXT_LINE_CONTRAST
|
|
577
|
+
)
|
|
578
|
+
if ink_depth < MIN_INK_DEPTH:
|
|
579
|
+
stats.kind = "blank"
|
|
580
|
+
stats.why = (
|
|
581
|
+
"the darkest mark on the page is only {0:.1f} grey levels below the "
|
|
582
|
+
"paper around it, which is grain rather than ink".format(ink_depth)
|
|
583
|
+
)
|
|
584
|
+
stats.skew_degrees = 0.0
|
|
585
|
+
stats.text_height = None
|
|
586
|
+
return stats
|
|
587
|
+
if not has_lines and stats.ink_structure < MIN_INK_STRUCTURE:
|
|
588
|
+
stats.kind = "blank"
|
|
589
|
+
stats.why = (
|
|
590
|
+
"what ink there is lies evenly across the whole sheet rather than "
|
|
591
|
+
"gathered into marks (it measures {0:.1f} against a floor of "
|
|
592
|
+
"{1:g}), which is grain on empty paper".format(
|
|
593
|
+
stats.ink_structure, MIN_INK_STRUCTURE
|
|
594
|
+
)
|
|
595
|
+
)
|
|
596
|
+
stats.skew_degrees = 0.0
|
|
597
|
+
stats.text_height = None
|
|
598
|
+
return stats
|
|
599
|
+
if ink_share <= BLANK_INK_SHARE and not has_lines:
|
|
600
|
+
stats.kind = "blank"
|
|
601
|
+
stats.why = (
|
|
602
|
+
"only {0:.2f}% of the page carries ink, and no lines of text were "
|
|
603
|
+
"found".format(ink_share * 100.0)
|
|
604
|
+
)
|
|
605
|
+
stats.skew_degrees = 0.0
|
|
606
|
+
stats.text_height = None
|
|
607
|
+
return stats
|
|
608
|
+
if paper_share < MIN_PAPER_SHARE and not has_lines:
|
|
609
|
+
stats.kind = "photograph"
|
|
610
|
+
stats.why = (
|
|
611
|
+
"only {0:.0f}% of the page comes out as bare paper and no repeating "
|
|
612
|
+
"lines of text were found, so this reads as a photograph rather "
|
|
613
|
+
"than a document".format(paper_share * 100.0)
|
|
614
|
+
)
|
|
615
|
+
stats.skew_degrees = 0.0
|
|
616
|
+
return stats
|
|
617
|
+
stats.why = (
|
|
618
|
+
"{0} line-shaped bands of text, {1:.0f}% of the page bare paper".format(
|
|
619
|
+
geometry.line_count, paper_share * 100.0
|
|
620
|
+
)
|
|
621
|
+
)
|
|
622
|
+
return stats
|
|
623
|
+
|
|
624
|
+
|
|
625
|
+
def estimate_skew_on_plane(plane: np.ndarray) -> float:
|
|
626
|
+
"""Counter-clockwise degrees the text on ``plane`` is off horizontal."""
|
|
627
|
+
coarse, _ = _images.analysis_plane(plane, SEARCH_MAX_SIDE)
|
|
628
|
+
if coarse.size == 0 or min(coarse.shape) < MIN_PLANE_SIDE:
|
|
629
|
+
return 0.0
|
|
630
|
+
ink = ink_plane(coarse).values
|
|
631
|
+
if not ProfileBank(ink, MAX_SKEW_DEGREES).usable:
|
|
632
|
+
return 0.0
|
|
633
|
+
return _search(ink)
|