ocr-cleaner 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,55 @@
1
+ """ocr-cleaner: prepare a scanned page so OCR reads it better.
2
+
3
+ Deskew, denoise and threshold, in one call, with every step reporting what it
4
+ did to the page and why - including the steps that looked and decided to do
5
+ nothing.
6
+
7
+ >>> import numpy as np, ocr_cleaner
8
+ >>> page = np.full((1100, 850), 246, dtype=np.uint8) # a sheet of paper
9
+ >>> for top in range(120, 1000, 34): # rows of text on it
10
+ ... page[top:top + 11, 90:760] = 50
11
+ >>> result = ocr_cleaner.clean(page)
12
+ >>> result.page_kind
13
+ 'document'
14
+ >>> "threshold" in result.applied
15
+ True
16
+
17
+ A blank sheet is reported blank and handed back untouched rather than
18
+ thresholded into a field of speckles, and a photograph is reported as a
19
+ photograph rather than cleaned as a bad scan.
20
+
21
+ Pure numpy and Pillow. No OpenCV, no OCR engine, no model download, nothing
22
+ touches the network, and the same page always gives the same result.
23
+ """
24
+ from __future__ import annotations
25
+
26
+ from ._analysis import MAX_SKEW_DEGREES, PAGE_KINDS
27
+ from ._core import (
28
+ ANALYSIS_MAX_SIDE,
29
+ DEFAULT_THRESHOLD,
30
+ THRESHOLD_MODES,
31
+ clean,
32
+ clean_file,
33
+ estimate_skew,
34
+ )
35
+ from ._images import IMAGE_SUFFIXES, open_image
36
+ from ._result import STEP_NAMES, CleanResult, Step
37
+
38
+ __version__ = "0.1.0"
39
+
40
+ __all__ = [
41
+ "clean",
42
+ "clean_file",
43
+ "estimate_skew",
44
+ "CleanResult",
45
+ "Step",
46
+ "STEP_NAMES",
47
+ "THRESHOLD_MODES",
48
+ "DEFAULT_THRESHOLD",
49
+ "PAGE_KINDS",
50
+ "MAX_SKEW_DEGREES",
51
+ "ANALYSIS_MAX_SIDE",
52
+ "IMAGE_SUFFIXES",
53
+ "open_image",
54
+ "__version__",
55
+ ]
@@ -0,0 +1,633 @@
1
+ """Everything the pipeline needs to know before it changes a pixel.
2
+
3
+ Three questions get answered here, and all three come out of one projection
4
+ profile:
5
+
6
+ 1. **How far is the text turned?** Lines of text are the only strong periodic
7
+ structure on a page. Shear the ink by a candidate angle, sum each row, and
8
+ the profile becomes a comb: tall teeth where lines are, near-zero between
9
+ them. The comb is sharpest at exactly one angle, and that angle is the skew.
10
+ 2. **How tall is a line of text?** The teeth of that comb, measured at the angle
11
+ that sharpened them. This is the number every later step is sized from - the
12
+ median window, the adaptive threshold window, the upscale target.
13
+ 3. **What kind of page is this?** Blank paper, a photograph, or a document. A
14
+ blank sheet must not be thresholded into a field of noise, and a photograph
15
+ must not be binarised at all, so both have to be recognised before anything
16
+ is applied.
17
+
18
+ Nothing is rotated or resampled to measure an angle. Shearing by row offsets is
19
+ both faster than bicubic rotation and more honest here, because rotation smears
20
+ the very edges being counted.
21
+
22
+ Angles follow ``PIL.Image.rotate``: positive is counter-clockwise, so a page
23
+ whose text runs downhill to the right has a negative skew, and
24
+ ``image.rotate(-estimate_skew(image))`` puts it straight.
25
+ """
26
+ from __future__ import annotations
27
+
28
+ import logging
29
+ from dataclasses import dataclass
30
+ from typing import List, NamedTuple, Optional, Sequence, Tuple
31
+
32
+ import numpy as np
33
+
34
+ from . import _images
35
+
36
+ logger = logging.getLogger(__name__)
37
+
38
+ #: Widest skew the search looks at, in degrees either side of upright. Beyond
39
+ #: this a page is not skewed, it is turned, which is a different repair.
40
+ MAX_SKEW_DEGREES = 15.0
41
+ #: The search, stage by stage: ``(strip width in columns, half-width of the
42
+ #: window in degrees, step in degrees, use the smooth shear)``. ``0`` columns
43
+ #: means the whole page in one piece. Each stage widens the aperture - which
44
+ #: sharpens the peak - and narrows the window around what the last stage found.
45
+ #: :func:`_search` explains why the aperture is the thing that matters.
46
+ SEARCH_STAGES = (
47
+ (96, MAX_SKEW_DEGREES, 1.0, False),
48
+ (288, 1.1, 0.2, True),
49
+ (0, 0.28, 0.035, True),
50
+ )
51
+ #: Narrowest strip worth shearing on its own.
52
+ MIN_STRIP_COLUMNS = 24
53
+ #: Longest side the angle search itself runs on. Not a free choice: the comb
54
+ #: the search looks for only exists while the plane still has a few rows per
55
+ #: line of text, and a 300 dpi page of six-point type reduced to 800 pixels is
56
+ #: down to four. Reduced to 500 it is down to two and a half, and the measured
57
+ #: angle starts coming back degrees wrong rather than hundredths. Going the
58
+ #: other way buys nothing, so this sits where the margin is comfortable and the
59
+ #: cost is not. Line height is measured separately on the full analysis plane,
60
+ #: at the angle the search found.
61
+ SEARCH_MAX_SIDE = 800
62
+ #: Smallest plane side worth measuring lines on.
63
+ MIN_PLANE_SIDE = 24
64
+ #: Share of the profile swing that separates a line of text from the gap above.
65
+ LINE_THRESHOLD = 0.35
66
+ #: Where the cut for the full inked extent sits, as a share of a line's own
67
+ #: height above the paper between lines. It is deliberately close to the paper,
68
+ #: because only a handful of glyphs in a line carry an ascender or a descender:
69
+ #: measured on rendered Times and Arial from 10 to 60 pixels, those rows sum to
70
+ #: between 6 and 20 percent of an x-height row. A cut set anywhere inside that
71
+ #: band chops the tails off, and one set above it breaks the line into slivers.
72
+ #: Sweeping this against the true ascender-to-descender extent of rendered text
73
+ #: - eight sizes, both faces, tight and loose leading, grain to sigma 18 - every
74
+ #: value from 0.02 to 0.04 lands within a tenth of the truth; 0.02 starts
75
+ #: reading grain as a descender on the noisiest page, so the cut sits at the top
76
+ #: of the range that still measures the whole line.
77
+ EXTENT_SHARE = 0.04
78
+ #: Shortest run of rows that can be part of a line of text.
79
+ MIN_LINE_ROWS = 2
80
+ #: Profile swing relative to its own mean, below which there are no text rows.
81
+ #: Pages of text measure 0.75 and up, and a sparse one measures several;
82
+ #: photographs and grainy blank paper sit between 0.05 and 0.2.
83
+ TEXT_LINE_CONTRAST = 0.5
84
+ #: How many line-shaped bands must repeat down the page before it is text. A
85
+ #: photograph's broad tonal bands can out-swing a text comb, but they repeat
86
+ #: three or four times in a frame where text repeats dozens.
87
+ MIN_TEXT_LINES = 6
88
+ #: Share of the page that must come out as bare paper for it to be a document.
89
+ #: Scans of text run 0.9 and up; photographs rarely clear a half.
90
+ MIN_PAPER_SHARE = 0.72
91
+ #: Ink strength, on the 0 to 1 scale of :func:`ink_plane`, at which a pixel is a
92
+ #: stroke rather than paper.
93
+ INK_CUT = 0.25
94
+ #: Ink strength below which a pixel is bare paper.
95
+ PAPER_CUT = 0.08
96
+ #: Ink coverage at or below which a page holds nothing worth cleaning.
97
+ BLANK_INK_SHARE = 0.0015
98
+ #: Denominator for the background radius, as a share of the plane's short side.
99
+ #: Wide enough to ignore the text, narrow enough to follow a lighting gradient.
100
+ BACKGROUND_DIVISOR = 24.0
101
+ #: Floor for that radius, so a small page still gets a usable background.
102
+ BACKGROUND_MIN_RADIUS = 4
103
+ #: Percentile of the ink plane taken as "the darkest stroke".
104
+ INK_PEAK_PERCENTILE = 99.7
105
+ #: How far the darkest stroke must sit below the paper beside it, in grey
106
+ #: levels, before a page has ink on it at all. The ink plane is scaled to its
107
+ #: own darkest mark, which is what lets it read a faint pencil page - and which
108
+ #: would just as happily scale the grain on an empty sheet into what looks like
109
+ #: dense text. This is the floor that stops it.
110
+ MIN_INK_DEPTH = 14.0
111
+ #: How much the ink has to be gathered into marks, rather than spread evenly as
112
+ #: grain, before a page has anything on it. Measured by :func:`ink_structure`.
113
+ #: Pages of text measure 17 and up; empty paper, however grainy, stays under 5.
114
+ MIN_INK_STRUCTURE = 8.0
115
+ #: Denominator for the radius that blur is done at, as a share of the short side.
116
+ STRUCTURE_DIVISOR = 64.0
117
+ #: Floor for that radius.
118
+ STRUCTURE_MIN_RADIUS = 3
119
+
120
+ #: The three answers :attr:`PageStats.kind` can give.
121
+ PAGE_KINDS = ("document", "blank", "photograph")
122
+
123
+
124
+ def background_radius(plane: np.ndarray) -> int:
125
+ """Box radius that follows the lighting on a page without following its text."""
126
+ return int(max(BACKGROUND_MIN_RADIUS, round(min(plane.shape) / BACKGROUND_DIVISOR)))
127
+
128
+
129
+ class InkPlane(NamedTuple):
130
+ """The ink on a page, and the two numbers that say whether it is ink at all."""
131
+
132
+ #: 0.0 where the paper is, 1.0 at the darkest stroke.
133
+ values: np.ndarray
134
+ #: How far that darkest stroke ran below the paper beside it, in grey levels.
135
+ depth: float
136
+
137
+
138
+ def ink_plane(plane: np.ndarray) -> InkPlane:
139
+ """Turn luminance into ink. See :class:`InkPlane` for what comes back.
140
+
141
+ ``values`` runs 0.0 where the paper is to 1.0 at the darkest stroke. That
142
+ scaling is what lets the same code read a faint pencil page and a crisp
143
+ laser one - and it would just as happily scale the grain on an empty sheet
144
+ into what looks like dense text, which is why ``depth`` and ``light_depth``
145
+ come back with it.
146
+
147
+ Ink is measured against the paper *beside it*, not against one number for
148
+ the whole sheet. A scan lit from one side can be seventy grey levels darker
149
+ at the far edge than at the near one, and a single paper level then reads
150
+ that whole edge as ink - which buries the comb the skew search is looking
151
+ for, and makes a perfectly ordinary page look like a photograph. Subtracting
152
+ a wide local mean instead removes any lighting that varies more slowly than
153
+ the text does, and it flattens a black scanner margin to nothing at the same
154
+ time, so the border stops pulling on the angle.
155
+
156
+ The scale comes from a high percentile rather than the maximum, so a single
157
+ hot pixel of impulse noise cannot decide what "the darkest stroke" means.
158
+ """
159
+ floats = plane.astype(np.float64)
160
+ background = _images.local_mean(plane, background_radius(plane))
161
+ ink = np.clip(background - floats, 0.0, None)
162
+ peak = float(np.percentile(ink, INK_PEAK_PERCENTILE))
163
+ if peak <= 0.0:
164
+ return InkPlane(np.zeros_like(ink), 0.0)
165
+ return InkPlane(np.clip(ink / peak, 0.0, 1.0), peak)
166
+
167
+
168
+ def ink_structure(ink: np.ndarray) -> float:
169
+ """How far from evenly spread the ink on a page is, in grey levels.
170
+
171
+ Ink gathers. Grain does not. Blur the ink plane over a small neighbourhood
172
+ and grain averages away to a flat field, because it is as likely to land
173
+ anywhere as anywhere else, while strokes and margins keep their difference,
174
+ because ink is somewhere and not elsewhere. What comes back is how much of
175
+ that difference is left, and it is the measurement that tells a faint page
176
+ of pencil from a dusty empty one - the two look identical to anything that
177
+ only counts how dark the darkest mark is.
178
+ """
179
+ if ink.size == 0: # pragma: no cover - guarded by callers
180
+ return 0.0
181
+ radius = int(max(STRUCTURE_MIN_RADIUS, round(min(ink.shape) / STRUCTURE_DIVISOR)))
182
+ levels = np.clip(ink * 255.0, 0.0, 255.0).astype(np.uint8)
183
+ return float(_images.local_mean(levels, radius).std())
184
+
185
+
186
+ def _shear_margin(width: int, degrees: float) -> int:
187
+ """Rows of headroom a shear of ``degrees`` needs at each end of the profile."""
188
+ return int(np.ceil(width * 0.5 * abs(np.tan(np.radians(degrees))))) + 1
189
+
190
+
191
+ class ProfileBank:
192
+ """One ink plane, ready to be sheared to any angle inside its margin.
193
+
194
+ A shear moves every pixel in a column by the same amount, because the
195
+ offset depends on the column and not on the row. That one fact is what
196
+ makes this cheap. Rather than computing a destination row for each of the
197
+ plane's pixels and scattering them - which is a random write per pixel, and
198
+ the slowest thing numpy does - the columns are grouped by the whole number
199
+ of rows they move, each group is summed with a single ``add.reduceat``, and
200
+ the handful of group totals are added into the profile at their offsets.
201
+ The work stops being one pass per pixel and becomes two passes plus a short
202
+ loop, and the profile that comes out is identical to the last float.
203
+
204
+ Two shears are offered, and they cost a few times apart. :meth:`profile`
205
+ rounds each column to a whole row, which is cheap enough to sweep the whole
206
+ range. Rounding quantises the answer though: on a nearly upright page every
207
+ angle under about a tenth of a degree rounds to the same offsets, so a
208
+ sweep using it can only bracket the truth. :meth:`profile_fine` splits each
209
+ column between the two rows it falls between, which makes the score a
210
+ smooth function of the angle. The search uses the cheap one to find the
211
+ neighbourhood and the smooth one to land inside it.
212
+ """
213
+
214
+ def __init__(self, ink: np.ndarray, max_degrees: float) -> None:
215
+ self.ink = np.ascontiguousarray(ink, dtype=np.float64)
216
+ self.height, self.width = self.ink.shape
217
+ self.total = float(self.ink.sum())
218
+ self.margin = _shear_margin(self.width, max_degrees)
219
+ self._length = self.height + 2 * self.margin
220
+ self._centred = np.arange(self.width, dtype=np.float64) - (self.width - 1) / 2.0
221
+
222
+ @property
223
+ def usable(self) -> bool:
224
+ """True when enough whole rows survive the shear margin to mean anything."""
225
+ return self.total > 0.0 and self.height - 2 * self.margin >= 8
226
+
227
+ def _offsets(self, degrees: float) -> np.ndarray:
228
+ """Row offset each column takes when the plane is sheared by ``degrees``."""
229
+ return self._centred * np.tan(np.radians(degrees))
230
+
231
+ @staticmethod
232
+ def _runs(shifts: np.ndarray) -> Tuple[np.ndarray, np.ndarray]:
233
+ """Where each run of equal shifts starts, and what it shifts by.
234
+
235
+ The offsets grow steadily across the page, so columns that move by the
236
+ same whole number of rows are always next to each other and a run is
237
+ enough to describe them.
238
+ """
239
+ change = np.flatnonzero(np.diff(shifts)) + 1
240
+ starts = np.empty(change.size + 1, dtype=np.intp)
241
+ starts[0] = 0
242
+ starts[1:] = change
243
+ return starts, shifts[starts]
244
+
245
+ def profile(self, degrees: float) -> np.ndarray:
246
+ """Row sums after shearing by ``degrees``, each column rounded to a row."""
247
+ starts, shifts = self._runs(np.rint(self._offsets(degrees)).astype(np.int64))
248
+ grouped = np.add.reduceat(self.ink, starts, axis=1)
249
+ summed = np.zeros(self._length + 2)
250
+ for group, shift in enumerate(shifts):
251
+ start = self.margin + int(shift)
252
+ summed[start: start + self.height] += grouped[:, group]
253
+ return summed[2 * self.margin: self.height]
254
+
255
+ def profile_fine(self, degrees: float) -> np.ndarray:
256
+ """Row sums after shearing by ``degrees``, split between adjacent rows.
257
+
258
+ Each column lands between two rows and gives each its share, so nudging
259
+ the angle by a hundredth of a degree moves the profile a little rather
260
+ than not at all. This is what makes agreement with a known rotation to
261
+ a fraction of a degree possible.
262
+ """
263
+ offsets = self._offsets(degrees)
264
+ lower = np.floor(offsets)
265
+ starts, shifts = self._runs(lower.astype(np.int64))
266
+ upper_share = self.ink * (offsets - lower)[None, :]
267
+ whole = np.add.reduceat(self.ink, starts, axis=1)
268
+ upper = np.add.reduceat(upper_share, starts, axis=1)
269
+ lower_part = whole - upper
270
+ summed = np.zeros(self._length + 2)
271
+ for group, shift in enumerate(shifts):
272
+ start = self.margin + int(shift)
273
+ summed[start: start + self.height] += lower_part[:, group]
274
+ summed[start + 1: start + 1 + self.height] += upper[:, group]
275
+ return summed[2 * self.margin: self.height]
276
+
277
+
278
+ def comb_score(profile: np.ndarray) -> float:
279
+ """How comb-like a profile is: the energy of its first difference.
280
+
281
+ Row variance peaks at the right angle too, but the first difference is
282
+ blind to a page-wide lighting ramp, which variance happily rewards.
283
+ """
284
+ if profile.size < 2:
285
+ return 0.0
286
+ step = np.diff(profile)
287
+ return float(np.dot(step, step))
288
+
289
+
290
+ def strip_banks(ink: np.ndarray, columns: int) -> List[ProfileBank]:
291
+ """Cut the ink plane into vertical strips, each ready to be sheared.
292
+
293
+ ``columns`` is the width each strip should be near; ``0`` means one strip
294
+ covering the whole page. Strips are what set the search's aperture, and the
295
+ aperture is what sets how wide the peak is - see :func:`_search`.
296
+ """
297
+ height, width = ink.shape
298
+ if columns <= 0 or width <= columns * 1.5:
299
+ return [ProfileBank(ink, MAX_SKEW_DEGREES)]
300
+ count = max(1, int(round(width / float(columns))))
301
+ edges = np.linspace(0, width, count + 1).astype(int)
302
+ banks = [
303
+ ProfileBank(np.ascontiguousarray(ink[:, start:stop]), MAX_SKEW_DEGREES)
304
+ for start, stop in zip(edges[:-1], edges[1:])
305
+ if stop - start >= MIN_STRIP_COLUMNS
306
+ ]
307
+ usable = [bank for bank in banks if bank.usable]
308
+ return usable or [ProfileBank(ink, MAX_SKEW_DEGREES)]
309
+
310
+
311
+ def _stage_score(banks: Sequence[ProfileBank], degrees: float, smooth: bool) -> float:
312
+ """Total comb score across every strip at one angle."""
313
+ return sum(
314
+ comb_score(bank.profile_fine(degrees) if smooth else bank.profile(degrees))
315
+ for bank in banks
316
+ )
317
+
318
+
319
+ def _refine(angles: np.ndarray, scores: np.ndarray, pick: int) -> float:
320
+ """Where the peak really sits, from the three samples around the best one.
321
+
322
+ The comb score near its maximum is close enough to a parabola that fitting
323
+ one through three samples lands nearer the truth than the sample itself -
324
+ and it is free, where another sweep costs a pass over the page.
325
+ """
326
+ best = float(angles[pick])
327
+ if pick <= 0 or pick >= len(scores) - 1 or len(angles) < 2:
328
+ return best
329
+ left, middle, right = float(scores[pick - 1]), float(scores[pick]), float(scores[pick + 1])
330
+ bend = left - 2.0 * middle + right
331
+ if bend == 0.0:
332
+ return best
333
+ shift = 0.5 * (left - right) / bend
334
+ if abs(shift) > 1.0: # pragma: no cover - a peak that is not a peak
335
+ return best
336
+ return best + shift * float(angles[1] - angles[0])
337
+
338
+
339
+ def _search(ink: np.ndarray) -> float:
340
+ """Find the angle whose comb is sharpest, in degrees counter-clockwise.
341
+
342
+ How sharply the comb answers to angle is set by a single ratio: the height
343
+ of a line of text over the width of the page. A line stays lined up under a
344
+ shear until the shear has moved one end of the page by about its own
345
+ height, so the peak is roughly ``line_height / page_width`` radians wide. On
346
+ a page of six-pixel text twenty-five hundred pixels across, that is a
347
+ seventh of a degree, and a sweep in one-degree steps walks straight over it
348
+ and reports whatever noise it lands on instead. This is not hypothetical: it
349
+ put a page eight degrees out.
350
+
351
+ Narrowing the *aperture* is the fix. The same lines measured across a strip
352
+ a quarter of the page wide give a peak four times broader, centred on the
353
+ same angle - and squeezing the page's columns together instead would not,
354
+ because that scales the angle by exactly as much as it broadens the peak.
355
+ So the search begins on narrow strips, where the peak is wide enough for a
356
+ coarse sweep to find, and widens the aperture as it narrows the window:
357
+ each stage sharper than the last, each looking only where the last one
358
+ pointed. The strips' scores are summed rather than one strip chosen, so
359
+ every line of ink on the page stays in the measurement.
360
+ """
361
+ best = 0.0
362
+ for columns, span, step, smooth in SEARCH_STAGES:
363
+ banks = strip_banks(ink, columns)
364
+ if not banks or not any(bank.usable for bank in banks):
365
+ return 0.0
366
+ low = max(-MAX_SKEW_DEGREES, best - span)
367
+ high = min(MAX_SKEW_DEGREES, best + span)
368
+ count = max(int(round((high - low) / step)) + 1, 1)
369
+ angles = np.linspace(low, high, count)
370
+ scores = np.array([_stage_score(banks, angle, smooth) for angle in angles])
371
+ best = _refine(angles, scores, int(np.argmax(scores)))
372
+ return float(best)
373
+
374
+
375
+ def _runs(profile: np.ndarray, threshold: float, minimum: int) -> List[Tuple[int, int]]:
376
+ """Runs of rows above ``threshold``, as half-open ``(start, stop)`` pairs."""
377
+ above = profile > threshold
378
+ if not above.any():
379
+ return []
380
+ edges = np.diff(above.astype(np.int8))
381
+ starts = list(np.flatnonzero(edges == 1) + 1)
382
+ stops = list(np.flatnonzero(edges == -1) + 1)
383
+ if above[0]:
384
+ starts.insert(0, 0)
385
+ if above[-1]:
386
+ stops.append(int(above.size))
387
+ return [(int(a), int(b)) for a, b in zip(starts, stops) if b - a >= minimum]
388
+
389
+
390
+ @dataclass
391
+ class LineGeometry:
392
+ """What the straightened profile says about the lines of text on a page."""
393
+
394
+ #: Height of an inked line, ascender top to descender foot, in plane pixels.
395
+ text_height: Optional[float] = None
396
+ #: Median baseline-to-baseline distance, in plane pixels.
397
+ pitch: Optional[float] = None
398
+ #: How many line-shaped bands were found.
399
+ line_count: int = 0
400
+ #: Profile swing relative to its own mean. A page with no lines scores ~0.
401
+ line_contrast: float = 0.0
402
+
403
+
404
+ def paper_between(profile: np.ndarray, cores: Sequence[Tuple[int, int]]) -> float:
405
+ """The level the profile falls back to between two lines of text.
406
+
407
+ The lowest point in each gap, medianed over the gaps, so one wide margin or
408
+ one pair of lines that touch cannot move it. This is the zero that a line's
409
+ own height is measured from - the profile's global minimum would do on a
410
+ clean page, but on a lit or grainy one the darkest row of the page is not
411
+ the level the gaps sit at.
412
+ """
413
+ dips = [
414
+ float(profile[stop:start].min())
415
+ for (_, stop), (start, _) in zip(cores, cores[1:])
416
+ if start > stop
417
+ ]
418
+ return float(np.median(dips)) if dips else float(profile.min())
419
+
420
+
421
+ def _grow_to_extent(
422
+ profile: np.ndarray, core: Tuple[int, int], cut: float, bounds: Tuple[int, int]
423
+ ) -> Tuple[int, int]:
424
+ """Walk out from a line's x-height band to its ascender top and descender foot.
425
+
426
+ Growing outward from the core is what ties the extent to the line it
427
+ belongs to. Thresholding the whole profile a second time and measuring
428
+ whatever comes out does not: the low cut also lifts the ascenders and
429
+ descenders sitting alone in the gaps into runs of their own, and they
430
+ outnumber the lines.
431
+ """
432
+ start, stop = core
433
+ floor, ceiling = bounds
434
+ while start > floor and profile[start - 1] > cut:
435
+ start -= 1
436
+ while stop < ceiling and profile[stop] > cut:
437
+ stop += 1
438
+ return start, stop
439
+
440
+
441
+ def _extent_bounds(
442
+ cores: Sequence[Tuple[int, int]], index: int, size: int
443
+ ) -> Tuple[int, int]:
444
+ """How far a line may grow: to the middle of the gap to each neighbour."""
445
+ start, stop = cores[index]
446
+ floor = 0 if index == 0 else (cores[index - 1][1] + start) // 2
447
+ ceiling = size if index + 1 == len(cores) else (stop + cores[index + 1][0] + 1) // 2
448
+ return min(floor, start), max(ceiling, stop)
449
+
450
+
451
+ def line_geometry(profile: np.ndarray) -> LineGeometry:
452
+ """Measure the teeth of a straightened comb profile."""
453
+ if profile.size < MIN_PLANE_SIDE:
454
+ return LineGeometry()
455
+ mean = float(profile.mean())
456
+ if mean <= 0.0:
457
+ return LineGeometry()
458
+ low, high = float(profile.min()), float(profile.max())
459
+ swing = high - low
460
+ contrast = float(profile.std()) / mean
461
+ if swing <= 0.0:
462
+ return LineGeometry(line_contrast=contrast)
463
+ cores = _runs(profile, low + LINE_THRESHOLD * swing, MIN_LINE_ROWS)
464
+ if not cores:
465
+ return LineGeometry(line_contrast=contrast)
466
+ paper = paper_between(profile, cores)
467
+ body = float(np.median([profile[start:stop].max() for start, stop in cores]))
468
+ cut = paper + EXTENT_SHARE * max(body - paper, 0.0)
469
+ heights = [
470
+ stop - start
471
+ for start, stop in (
472
+ _grow_to_extent(
473
+ profile, core, cut, _extent_bounds(cores, index, int(profile.size))
474
+ )
475
+ for index, core in enumerate(cores)
476
+ )
477
+ ]
478
+ centres = [(start + stop) / 2.0 for start, stop in cores]
479
+ pitch = None
480
+ if len(centres) >= 2:
481
+ gaps = np.diff(np.asarray(centres, dtype=np.float64))
482
+ pitch = float(np.median(gaps))
483
+ return LineGeometry(
484
+ text_height=float(np.median(np.asarray(heights, dtype=np.float64))),
485
+ pitch=pitch,
486
+ line_count=len(cores),
487
+ line_contrast=contrast,
488
+ )
489
+
490
+
491
+ @dataclass
492
+ class PageStats:
493
+ """One page, measured. Every later decision is taken from these numbers.
494
+
495
+ Attributes:
496
+ kind: ``"document"``, ``"blank"`` or ``"photograph"``.
497
+ why: one sentence explaining the kind, in plain language.
498
+ skew_degrees: counter-clockwise degrees the text is off horizontal.
499
+ text_height: height of a line of text in *source* pixels, or ``None``
500
+ when no lines were found.
501
+ line_count: how many lines the profile found, on the analysis plane.
502
+ line_contrast: how far the profile swings relative to its own mean.
503
+ paper_level: luminance of clean paper, 0 to 255.
504
+ ink_depth: how far the darkest stroke sits below the paper beside it,
505
+ in grey levels. Below :data:`MIN_INK_DEPTH` there is no ink, only
506
+ grain.
507
+ ink_structure: how far from evenly spread that ink is, in grey levels.
508
+ Below :data:`MIN_INK_STRUCTURE` it is grain, not marks.
509
+ ink_share: share of pixels that are a stroke, on the local scale
510
+ :func:`ink_plane` measures.
511
+ paper_share: share of pixels that are bare paper on that same scale.
512
+ """
513
+
514
+ kind: str = "document"
515
+ why: str = ""
516
+ skew_degrees: float = 0.0
517
+ text_height: Optional[float] = None
518
+ line_count: int = 0
519
+ line_contrast: float = 0.0
520
+ paper_level: float = 255.0
521
+ ink_depth: float = 0.0
522
+ ink_structure: float = 0.0
523
+ ink_share: float = 0.0
524
+ paper_share: float = 0.0
525
+
526
+ @property
527
+ def is_document(self) -> bool:
528
+ """True when the page is worth putting through the pipeline."""
529
+ return self.kind == "document"
530
+
531
+
532
+ def measure(plane: np.ndarray, scale: float = 1.0) -> PageStats:
533
+ """Measure an analysis plane: kind, skew, and the height of a line of text.
534
+
535
+ Args:
536
+ plane: an 8-bit luminance plane, already downscaled for analysis.
537
+ scale: how much that downscale shrank the source, so lengths can be
538
+ reported back in source pixels.
539
+ """
540
+ stats = PageStats(paper_level=_images.paper_level(plane) if plane.size else 255.0)
541
+
542
+ if plane.size == 0 or min(plane.shape) < MIN_PLANE_SIDE:
543
+ stats.kind = "blank"
544
+ stats.why = "the page is too small to hold a line of text"
545
+ return stats
546
+
547
+ coarse, _ = _images.analysis_plane(plane, SEARCH_MAX_SIDE)
548
+ coarse_ink, ink_depth = ink_plane(coarse)
549
+ stats.ink_depth = ink_depth
550
+ ink_share = float(np.mean(coarse_ink > INK_CUT))
551
+ paper_share = float(np.mean(coarse_ink < PAPER_CUT))
552
+ stats.ink_share = ink_share
553
+ stats.paper_share = paper_share
554
+ stats.ink_structure = ink_structure(coarse_ink)
555
+ if not ProfileBank(coarse_ink, MAX_SKEW_DEGREES).usable:
556
+ stats.kind = "blank"
557
+ stats.why = "the page carries no ink at all"
558
+ return stats
559
+
560
+ degrees = _search(coarse_ink)
561
+ detail_ink = coarse_ink if coarse.shape == plane.shape else ink_plane(plane).values
562
+ detail_bank = ProfileBank(detail_ink, MAX_SKEW_DEGREES)
563
+ geometry = (
564
+ line_geometry(detail_bank.profile_fine(degrees))
565
+ if detail_bank.usable
566
+ else LineGeometry()
567
+ )
568
+ stats.skew_degrees = float(degrees)
569
+ stats.line_count = geometry.line_count
570
+ stats.line_contrast = geometry.line_contrast
571
+ if geometry.text_height is not None and scale > 0.0:
572
+ stats.text_height = float(geometry.text_height) / scale
573
+
574
+ has_lines = (
575
+ geometry.line_count >= MIN_TEXT_LINES
576
+ and geometry.line_contrast >= TEXT_LINE_CONTRAST
577
+ )
578
+ if ink_depth < MIN_INK_DEPTH:
579
+ stats.kind = "blank"
580
+ stats.why = (
581
+ "the darkest mark on the page is only {0:.1f} grey levels below the "
582
+ "paper around it, which is grain rather than ink".format(ink_depth)
583
+ )
584
+ stats.skew_degrees = 0.0
585
+ stats.text_height = None
586
+ return stats
587
+ if not has_lines and stats.ink_structure < MIN_INK_STRUCTURE:
588
+ stats.kind = "blank"
589
+ stats.why = (
590
+ "what ink there is lies evenly across the whole sheet rather than "
591
+ "gathered into marks (it measures {0:.1f} against a floor of "
592
+ "{1:g}), which is grain on empty paper".format(
593
+ stats.ink_structure, MIN_INK_STRUCTURE
594
+ )
595
+ )
596
+ stats.skew_degrees = 0.0
597
+ stats.text_height = None
598
+ return stats
599
+ if ink_share <= BLANK_INK_SHARE and not has_lines:
600
+ stats.kind = "blank"
601
+ stats.why = (
602
+ "only {0:.2f}% of the page carries ink, and no lines of text were "
603
+ "found".format(ink_share * 100.0)
604
+ )
605
+ stats.skew_degrees = 0.0
606
+ stats.text_height = None
607
+ return stats
608
+ if paper_share < MIN_PAPER_SHARE and not has_lines:
609
+ stats.kind = "photograph"
610
+ stats.why = (
611
+ "only {0:.0f}% of the page comes out as bare paper and no repeating "
612
+ "lines of text were found, so this reads as a photograph rather "
613
+ "than a document".format(paper_share * 100.0)
614
+ )
615
+ stats.skew_degrees = 0.0
616
+ return stats
617
+ stats.why = (
618
+ "{0} line-shaped bands of text, {1:.0f}% of the page bare paper".format(
619
+ geometry.line_count, paper_share * 100.0
620
+ )
621
+ )
622
+ return stats
623
+
624
+
625
+ def estimate_skew_on_plane(plane: np.ndarray) -> float:
626
+ """Counter-clockwise degrees the text on ``plane`` is off horizontal."""
627
+ coarse, _ = _images.analysis_plane(plane, SEARCH_MAX_SIDE)
628
+ if coarse.size == 0 or min(coarse.shape) < MIN_PLANE_SIDE:
629
+ return 0.0
630
+ ink = ink_plane(coarse).values
631
+ if not ProfileBank(ink, MAX_SKEW_DEGREES).usable:
632
+ return 0.0
633
+ return _search(ink)