scanlayer 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
scanlayer/config.py ADDED
@@ -0,0 +1,267 @@
1
+ """
2
+ Central configuration for scanlayer.
3
+
4
+ Resolves Tesseract path at import time (TESSERACT_CMD / TESSDATA_DIR),
5
+ overridable at runtime via configure(). All tunable OCR, preprocessing,
6
+ and PDF-output parameters are defined here.
7
+ """
8
+
9
+ import json
10
+ import os
11
+ import platform
12
+ import shutil
13
+ import sys
14
+
15
+ if getattr(sys, "frozen", False):
16
+ BASE_DIR = os.path.dirname(sys.executable)
17
+ else:
18
+ BASE_DIR = os.path.dirname(os.path.abspath(__file__))
19
+
20
+ DEV_TESSERACT_OVERRIDE = None
21
+
22
+
23
+ def _find_bundled_tesseract():
24
+ """Look for a Tesseract binary bundled next to this package."""
25
+ exe_name = "tesseract.exe" if platform.system() == "Windows" else "tesseract"
26
+ cmd = os.path.join(BASE_DIR, "bin", "tesseract", exe_name)
27
+ if os.path.exists(cmd):
28
+ return cmd, os.path.join(BASE_DIR, "bin", "tesseract", "tessdata")
29
+ return None, None
30
+
31
+
32
+ def _find_system_tesseract():
33
+ """Find Tesseract on PATH or common install locations."""
34
+ on_path = shutil.which("tesseract")
35
+ if on_path:
36
+ return on_path
37
+
38
+ system = platform.system()
39
+ if system == "Windows":
40
+ candidates = [
41
+ r"C:\Program Files\Tesseract-OCR\tesseract.exe",
42
+ r"C:\Program Files (x86)\Tesseract-OCR\tesseract.exe",
43
+ ]
44
+ elif system == "Darwin":
45
+ candidates = [
46
+ "/opt/homebrew/bin/tesseract",
47
+ "/usr/local/bin/tesseract",
48
+ ]
49
+ else:
50
+ candidates = [
51
+ "/usr/bin/tesseract",
52
+ "/usr/local/bin/tesseract",
53
+ ]
54
+
55
+ for candidate in candidates:
56
+ if os.path.exists(candidate):
57
+ return candidate
58
+
59
+ return "tesseract"
60
+
61
+
62
+ # Resolution order: env var > DEV_TESSERACT_OVERRIDE > bundled > PATH
63
+ _env_override = os.environ.get("TESSERACT_CMD", "").strip()
64
+ if _env_override:
65
+ TESSERACT_CMD = _env_override
66
+ TESSDATA_DIR = os.environ.get("TESSDATA_PREFIX") or None
67
+ elif DEV_TESSERACT_OVERRIDE and os.path.exists(DEV_TESSERACT_OVERRIDE):
68
+ TESSERACT_CMD = DEV_TESSERACT_OVERRIDE
69
+ TESSDATA_DIR = None
70
+ else:
71
+ _bundled_cmd, _bundled_tessdata = _find_bundled_tesseract()
72
+ if _bundled_cmd:
73
+ TESSERACT_CMD = _bundled_cmd
74
+ TESSDATA_DIR = _bundled_tessdata
75
+ else:
76
+ TESSERACT_CMD = _find_system_tesseract()
77
+ TESSDATA_DIR = None
78
+
79
+
80
+ DEFAULT_OCR_LANG = "fra+eng"
81
+ DEFAULT_DPI = 300
82
+ OCR_UPSCALE_MIN_WIDTH = 2500
83
+ OCR_DOWNSCALE_MAX_WIDTH = 3500
84
+ MIN_WORD_CONFIDENCE = 35
85
+
86
+ PSM_EARLY_EXIT_CONFIDENCE = 80.0
87
+
88
+ # Tesseract releases the GIL, so threads give real speedup.
89
+ PSM_PARALLEL = True
90
+ PSM_MAX_WORKERS = None
91
+
92
+ OCR_TIMEOUT_SECONDS = 45
93
+
94
+ # PSM candidates: engine.py runs one pass per PSM, keeps highest mean confidence.
95
+ # 3=fully automatic, 4=single column, 6=uniform block, 11=sparse text
96
+ TESSERACT_PSM_CANDIDATES = [3, 4, 6, 11]
97
+
98
+ # 0=legacy, 1=LSTM, 2=both, 3=auto. 1 is best accuracy/speed tradeoff.
99
+ TESSERACT_OEM = 1
100
+
101
+ TESSERACT_CHAR_BLACKLIST = None
102
+ TESSERACT_CHAR_WHITELIST = None
103
+
104
+ DROP_NON_PRINTABLE_WORDS = True
105
+
106
+
107
+ MULTI_COLUMN_DETECTION = True
108
+ COLUMN_MIN_GUTTER_FRACTION = 0.03
109
+ COLUMN_FULL_WIDTH_LINE_FRACTION = 0.62
110
+ COLUMN_MIN_LINES_FOR_DETECTION = 6
111
+ COLUMN_GUTTER_VOTE_FRACTION = 0.6
112
+
113
+
114
+ APPLY_EXIF_ORIENTATION = True
115
+
116
+ UNSHARP_AMOUNT = 0.5
117
+ UNSHARP_SIGMA = 1.2
118
+
119
+ PREPROCESS_TRY_ADAPTIVE_THRESHOLD = False
120
+
121
+ DENOISING_H = 10
122
+
123
+ CLAHE_CLIP_LIMIT = 2.0
124
+ CLAHE_TILE_GRID = (8, 8)
125
+
126
+ DESKEW_MIN_PIXELS = 100
127
+ DESKEW_MIN_STD = 5.0
128
+ DESKEW_MIN_ANGLE = 0.3
129
+ DESKEW_MAX_ANGLE = 30.0
130
+
131
+
132
+ PDF_JPEG_QUALITY = 82
133
+ PDF_ADAPTIVE_COMPRESSION = True
134
+ PDF_GRAYSCALE_THRESHOLD = 0.02
135
+ FONT_PATH = None
136
+
137
+ PDF_METADATA = {
138
+ "title": "Image converted to searchable PDF",
139
+ "author": "scanlayer",
140
+ "subject": "Scanned document with OCR text layer",
141
+ "creator": "scanlayer (Tesseract + ReportLab)",
142
+ }
143
+
144
+
145
+ LOG_LEVEL = os.environ.get("SCANLAYER_LOG_LEVEL", "INFO")
146
+ LOG_TIMING = True
147
+
148
+
149
+ # Runtime configuration API
150
+ #
151
+ # Precedence: 1. convert() args, 2. configure() calls,
152
+ # 3. env vars, 4. defaults here.
153
+
154
+ _CONFIGURABLE_KEYS = {
155
+ "tesseract_cmd": "TESSERACT_CMD",
156
+ "tessdata_dir": "TESSDATA_DIR",
157
+ "lang": "DEFAULT_OCR_LANG",
158
+ "default_dpi": "DEFAULT_DPI",
159
+ "min_word_confidence": "MIN_WORD_CONFIDENCE",
160
+ "psm_candidates": "TESSERACT_PSM_CANDIDATES",
161
+ "psm_early_exit_confidence": "PSM_EARLY_EXIT_CONFIDENCE",
162
+ "psm_parallel": "PSM_PARALLEL",
163
+ "psm_max_workers": "PSM_MAX_WORKERS",
164
+ "ocr_timeout_seconds": "OCR_TIMEOUT_SECONDS",
165
+ "char_whitelist": "TESSERACT_CHAR_WHITELIST",
166
+ "char_blacklist": "TESSERACT_CHAR_BLACKLIST",
167
+ "jpeg_quality": "PDF_JPEG_QUALITY",
168
+ "font_path": "FONT_PATH",
169
+ "log_level": "LOG_LEVEL",
170
+ "log_timing": "LOG_TIMING",
171
+ "multi_column_detection": "MULTI_COLUMN_DETECTION",
172
+ "column_min_gutter_fraction": "COLUMN_MIN_GUTTER_FRACTION",
173
+ "column_gutter_vote_fraction": "COLUMN_GUTTER_VOTE_FRACTION",
174
+ "column_full_width_line_fraction": "COLUMN_FULL_WIDTH_LINE_FRACTION",
175
+ "column_min_lines_for_detection": "COLUMN_MIN_LINES_FOR_DETECTION",
176
+ }
177
+
178
+
179
+ def configure(**overrides) -> None:
180
+ """Override configuration at runtime.
181
+
182
+ Tesseract is usually found automatically. Use tesseract_cmd/tessdata_dir
183
+ only to point at a specific install.
184
+
185
+ Example:
186
+ scanlayer.configure(lang="eng", min_word_confidence=50)
187
+
188
+ Raises ValueError on unknown keywords.
189
+ """
190
+ unknown = set(overrides) - set(_CONFIGURABLE_KEYS)
191
+ if unknown:
192
+ raise ValueError(
193
+ f"configure() got unknown option(s): {sorted(unknown)}. "
194
+ f"Valid options: {sorted(_CONFIGURABLE_KEYS)}"
195
+ )
196
+
197
+ module = sys.modules[__name__]
198
+ for key, value in overrides.items():
199
+ attr = _CONFIGURABLE_KEYS[key]
200
+ setattr(module, attr, value)
201
+
202
+ if "log_level" in overrides:
203
+ from scanlayer.utils.logger import set_log_level
204
+ set_log_level(overrides["log_level"])
205
+
206
+
207
+ def get_settings() -> dict:
208
+ """Return the current value of every configurable setting."""
209
+ module = sys.modules[__name__]
210
+ return {key: getattr(module, attr) for key, attr in _CONFIGURABLE_KEYS.items()}
211
+
212
+
213
+ def load_config_file(path: str) -> dict:
214
+ """Read a .yaml/.yml/.json config profile and return its contents.
215
+
216
+ Raises FileNotFoundError if path doesn't exist, ValueError for
217
+ unsupported extension or unparseable content.
218
+ """
219
+ if not os.path.exists(path):
220
+ raise FileNotFoundError(f"Config file not found: {path}")
221
+
222
+ ext = os.path.splitext(path)[1].lower()
223
+ with open(path, "r", encoding="utf-8") as f:
224
+ raw = f.read()
225
+
226
+ if ext in (".yaml", ".yml"):
227
+ try:
228
+ import yaml
229
+ except ImportError as exc:
230
+ raise ValueError(
231
+ "Reading a .yaml/.yml config file requires PyYAML "
232
+ "('pip install pyyaml'). Use a .json config file instead."
233
+ ) from exc
234
+ try:
235
+ data = yaml.safe_load(raw)
236
+ except yaml.YAMLError as exc:
237
+ raise ValueError(f"Could not parse YAML config file {path}: {exc}") from exc
238
+ elif ext == ".json":
239
+ try:
240
+ data = json.loads(raw)
241
+ except json.JSONDecodeError as exc:
242
+ raise ValueError(f"Could not parse JSON config file {path}: {exc}") from exc
243
+ else:
244
+ raise ValueError(
245
+ f"Unsupported config file extension: {ext!r} (expected .yaml, "
246
+ f".yml, or .json)"
247
+ )
248
+
249
+ if data is None:
250
+ return {}
251
+ if not isinstance(data, dict):
252
+ raise ValueError(
253
+ f"Config file {path} must contain a mapping of setting "
254
+ f"names to values at the top level, got {type(data).__name__}."
255
+ )
256
+ return data
257
+
258
+
259
+ def configure_from_file(path: str) -> dict:
260
+ """Load config file and apply it via configure(). Returns applied settings.
261
+
262
+ Example:
263
+ scanlayer.configure_from_file("profile.yaml")
264
+ """
265
+ settings = load_config_file(path)
266
+ configure(**settings)
267
+ return settings
Binary file
File without changes
@@ -0,0 +1,264 @@
1
+ """
2
+ Multi-column reading-order reconstruction.
3
+
4
+ Tesseract's block/paragraph numbering frequently interleaves columns on
5
+ two-column pages. This module re-derives reading order geometrically:
6
+
7
+ 1. Group words into rows (Tesseract's vertical grouping is reliable).
8
+ 2. Split rows into segments at wide gaps (catches column fusions).
9
+ 3. Detect column gutters by per-row gap voting.
10
+ 4. Assign segments to gutter-delimited bands, sort top-to-bottom.
11
+ 5. Place full-width lines (headers/footers) before/after columns.
12
+
13
+ Tuned for prose-style layouts, not tables. Only produces left-to-right
14
+ column order.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import math
20
+ from dataclasses import dataclass
21
+
22
+ from scanlayer import config
23
+ from scanlayer.ocr.engine import Word
24
+ from scanlayer.utils.logger import get_logger, log_warning
25
+
26
+ log = get_logger(__name__)
27
+
28
+
29
+ def _group_into_lines(words: list[Word]) -> list[list[Word]]:
30
+ """Group words, preserving Tesseract line order via Word.line_id."""
31
+ lines: list[list[Word]] = []
32
+ current: list[Word] = []
33
+ current_id = None
34
+ for w in words:
35
+ if current_id is None or w.line_id == current_id:
36
+ current.append(w)
37
+ else:
38
+ lines.append(current)
39
+ current = [w]
40
+ current_id = w.line_id
41
+ if current:
42
+ lines.append(current)
43
+ return lines
44
+
45
+
46
+ @dataclass
47
+ class _Line:
48
+ words: list[Word]
49
+ x_min: float
50
+ x_max: float
51
+ y_min: float
52
+ y_max: float
53
+
54
+ @property
55
+ def width(self) -> float:
56
+ return self.x_max - self.x_min
57
+
58
+ @property
59
+ def y_center(self) -> float:
60
+ return (self.y_min + self.y_max) / 2
61
+
62
+
63
+ def _line_bounds(words: list[Word]) -> _Line:
64
+ return _Line(
65
+ words=words,
66
+ x_min=min(w.x for w in words),
67
+ x_max=max(w.x + w.width for w in words),
68
+ y_min=min(w.y for w in words),
69
+ y_max=max(w.y + w.height for w in words),
70
+ )
71
+
72
+
73
+ def _split_row_into_segments(
74
+ row_words: list[Word], min_gap: float
75
+ ) -> list[list[Word]]:
76
+ """Split a Tesseract line into horizontal segments at wide gaps.
77
+
78
+ Under column-blind PSMs, Tesseract fuses same-row text from two
79
+ columns into one line. This re-splits on the horizontal gap signal.
80
+ """
81
+ if not row_words:
82
+ return []
83
+ ordered = sorted(row_words, key=lambda w: w.x)
84
+ segments = [[ordered[0]]]
85
+ for prev, w in zip(ordered, ordered[1:]):
86
+ gap = w.x - (prev.x + prev.width)
87
+ if gap >= min_gap:
88
+ segments.append([w])
89
+ else:
90
+ segments[-1].append(w)
91
+ return segments
92
+
93
+
94
+ def _build_segmented_lines(words: list[Word], page_width: float) -> list[_Line]:
95
+ """Group words into rows, then split each row into segments at wide gaps."""
96
+ min_gap = page_width * config.COLUMN_MIN_GUTTER_FRACTION
97
+ segments: list[_Line] = []
98
+ for row in _group_into_lines(words):
99
+ for segment_words in _split_row_into_segments(row, min_gap):
100
+ segments.append(_line_bounds(segment_words))
101
+ return segments
102
+
103
+
104
+ def _row_gap_votes(
105
+ rows: list[list[Word]], page_width: float
106
+ ) -> tuple[list[int], float]:
107
+ """Build per-x-bin profile counting rows with word gaps covering each bin."""
108
+ bin_w = max(1.0, page_width / 500)
109
+ n_bins = int(page_width / bin_w) + 2
110
+ votes: list[int] = [0] * n_bins
111
+ for row in rows:
112
+ ordered = sorted(row, key=lambda w: w.x)
113
+ voted: set[int] = set()
114
+ for prev, nxt in zip(ordered, ordered[1:]):
115
+ gap_start = prev.x + prev.width
116
+ gap_end = nxt.x
117
+ if gap_end - gap_start < bin_w:
118
+ continue
119
+ lo = max(0, int(gap_start / bin_w))
120
+ hi = min(n_bins - 1, int(gap_end / bin_w))
121
+ for i in range(lo, hi + 1):
122
+ if i not in voted:
123
+ voted.add(i)
124
+ votes[i] += 1
125
+ return votes, bin_w
126
+
127
+
128
+ def detect_column_gutters(
129
+ rows: list[list[Word]], page_width: float
130
+ ) -> list[float]:
131
+ """Return x-positions of detected column gutters, or empty list if single-column.
132
+
133
+ rows are RAW Tesseract line groupings (pre-segmentation).
134
+ """
135
+ if not rows or page_width <= 0:
136
+ return []
137
+
138
+ min_gap = page_width * config.COLUMN_MIN_GUTTER_FRACTION
139
+ full_width_threshold = page_width * config.COLUMN_FULL_WIDTH_LINE_FRACTION
140
+ segments = [
141
+ seg for row in rows for seg in _split_row_into_segments(row, min_gap)
142
+ ]
143
+ narrow_lines = [
144
+ ln
145
+ for ln in (_line_bounds(s) for s in segments)
146
+ if ln.width <= full_width_threshold
147
+ ]
148
+ if len(narrow_lines) < config.COLUMN_MIN_LINES_FOR_DETECTION:
149
+ return []
150
+
151
+ votes, bin_w = _row_gap_votes(rows, page_width)
152
+ strongest = max(votes, default=0)
153
+ if strongest < 2:
154
+ return []
155
+ min_votes = max(2, math.ceil(strongest * config.COLUMN_GUTTER_VOTE_FRACTION))
156
+
157
+ tolerance = page_width * 0.005
158
+ gutters: list[float] = []
159
+ start: int | None = None
160
+ for i, v in enumerate([*votes, 0]):
161
+ if v >= min_votes:
162
+ if start is None:
163
+ start = i
164
+ continue
165
+ if start is None:
166
+ continue
167
+ band_start, band_end = start * bin_w, i * bin_w
168
+ start = None
169
+ if band_end - band_start < min_gap:
170
+ continue
171
+ left_support = sum(
172
+ 1 for ln in narrow_lines if ln.x_max <= band_start + tolerance
173
+ )
174
+ right_support = sum(
175
+ 1 for ln in narrow_lines if ln.x_min >= band_end - tolerance
176
+ )
177
+ if left_support >= 2 and right_support >= 2:
178
+ gutters.append((band_start + band_end) / 2.0)
179
+ return gutters
180
+
181
+
182
+ def reorder_reading_order(
183
+ words: list[Word], page_width: float
184
+ ) -> tuple[list[Word], int]:
185
+ """Reorder words into left-to-right, top-to-bottom column reading order.
186
+
187
+ Returns (words, 1) unchanged if single-column, not enough text,
188
+ no gutters found, or reorder would drop/duplicate words.
189
+ Otherwise returns (reordered_words, column_count).
190
+ """
191
+ if not config.MULTI_COLUMN_DETECTION or not words or page_width <= 0:
192
+ return words, 1
193
+
194
+ lines = _build_segmented_lines(words, page_width)
195
+ if len(lines) < config.COLUMN_MIN_LINES_FOR_DETECTION:
196
+ return words, 1
197
+
198
+ gutters = detect_column_gutters(_group_into_lines(words), page_width)
199
+ if not gutters:
200
+ return words, 1
201
+
202
+ boundaries = [0.0, *gutters, page_width]
203
+ full_width_threshold = page_width * config.COLUMN_FULL_WIDTH_LINE_FRACTION
204
+ crossing_margin = page_width * 0.005
205
+
206
+ columns: list[list[_Line]] = [[] for _ in range(len(boundaries) - 1)]
207
+ full_width_lines: list[_Line] = []
208
+ for line in lines:
209
+ crosses_gutter = any(
210
+ line.x_min < g - crossing_margin and line.x_max > g + crossing_margin
211
+ for g in gutters
212
+ )
213
+ if line.width > full_width_threshold or crosses_gutter:
214
+ full_width_lines.append(line)
215
+ continue
216
+ center = (line.x_min + line.x_max) / 2.0
217
+ col_index = len(columns) - 1
218
+ for i in range(len(boundaries) - 1):
219
+ if boundaries[i] <= center < boundaries[i + 1]:
220
+ col_index = i
221
+ break
222
+ columns[col_index].append(line)
223
+
224
+ for col in columns:
225
+ col.sort(key=lambda ln: ln.y_min)
226
+
227
+ column_ys = [ln.y_min for col in columns for ln in col] or [0.0]
228
+ column_ys_max = [ln.y_max for col in columns for ln in col] or [0.0]
229
+ top_of_columns = min(column_ys)
230
+ bottom_of_columns = max(column_ys_max)
231
+ midpoint = (top_of_columns + bottom_of_columns) / 2.0
232
+
233
+ headers = sorted(
234
+ (ln for ln in full_width_lines if ln.y_center <= midpoint),
235
+ key=lambda ln: ln.y_min,
236
+ )
237
+ footers = sorted(
238
+ (ln for ln in full_width_lines if ln.y_center > midpoint),
239
+ key=lambda ln: ln.y_min,
240
+ )
241
+
242
+ ordered_lines = headers
243
+ for col in columns:
244
+ ordered_lines += col
245
+ ordered_lines += footers
246
+
247
+ reordered_words = [w for line in ordered_lines for w in line.words]
248
+
249
+ if len(reordered_words) != len(words):
250
+ log_warning(
251
+ log,
252
+ "Column reordering produced a different word count than the "
253
+ "input (bug in the layout logic), keeping original word "
254
+ "order as a safety fallback.",
255
+ )
256
+ return words, 1
257
+
258
+ non_empty_columns = sum(1 for col in columns if col)
259
+ log.info(
260
+ f"Multi-column layout detected: {non_empty_columns} column(s), "
261
+ f"gutter(s) at x={[round(g, 1) for g in gutters]}px "
262
+ f"(page width {page_width:.0f}px)"
263
+ )
264
+ return reordered_words, non_empty_columns