scanlayer 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scanlayer/__init__.py +32 -0
- scanlayer/__main__.py +10 -0
- scanlayer/cli/__init__.py +10 -0
- scanlayer/cli/dry_run.py +191 -0
- scanlayer/cli/parser.py +199 -0
- scanlayer/cli/run.py +217 -0
- scanlayer/config.py +267 -0
- scanlayer/fonts/DejaVuSans.ttf +0 -0
- scanlayer/layout/__init__.py +0 -0
- scanlayer/layout/columns.py +264 -0
- scanlayer/main.py +666 -0
- scanlayer/ocr/__init__.py +5 -0
- scanlayer/ocr/engine.py +335 -0
- scanlayer/ocr/export.py +264 -0
- scanlayer/pdf/__init__.py +5 -0
- scanlayer/pdf/builder.py +309 -0
- scanlayer/pdf/fonts.py +110 -0
- scanlayer/preprocessing/__init__.py +5 -0
- scanlayer/preprocessing/enhance.py +367 -0
- scanlayer/utils/__init__.py +19 -0
- scanlayer/utils/debug_image.py +85 -0
- scanlayer/utils/errors.py +42 -0
- scanlayer/utils/logger.py +100 -0
- scanlayer/utils/validators.py +199 -0
- scanlayer-1.0.0.dist-info/METADATA +17 -0
- scanlayer-1.0.0.dist-info/RECORD +30 -0
- scanlayer-1.0.0.dist-info/WHEEL +5 -0
- scanlayer-1.0.0.dist-info/entry_points.txt +2 -0
- scanlayer-1.0.0.dist-info/licenses/LICENSE.md +21 -0
- scanlayer-1.0.0.dist-info/top_level.txt +1 -0
scanlayer/config.py
ADDED
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Central configuration for scanlayer.
|
|
3
|
+
|
|
4
|
+
Resolves Tesseract path at import time (TESSERACT_CMD / TESSDATA_DIR),
|
|
5
|
+
overridable at runtime via configure(). All tunable OCR, preprocessing,
|
|
6
|
+
and PDF-output parameters are defined here.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import os
|
|
11
|
+
import platform
|
|
12
|
+
import shutil
|
|
13
|
+
import sys
|
|
14
|
+
|
|
15
|
+
if getattr(sys, "frozen", False):
|
|
16
|
+
BASE_DIR = os.path.dirname(sys.executable)
|
|
17
|
+
else:
|
|
18
|
+
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
|
|
19
|
+
|
|
20
|
+
DEV_TESSERACT_OVERRIDE = None
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _find_bundled_tesseract():
|
|
24
|
+
"""Look for a Tesseract binary bundled next to this package."""
|
|
25
|
+
exe_name = "tesseract.exe" if platform.system() == "Windows" else "tesseract"
|
|
26
|
+
cmd = os.path.join(BASE_DIR, "bin", "tesseract", exe_name)
|
|
27
|
+
if os.path.exists(cmd):
|
|
28
|
+
return cmd, os.path.join(BASE_DIR, "bin", "tesseract", "tessdata")
|
|
29
|
+
return None, None
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _find_system_tesseract():
|
|
33
|
+
"""Find Tesseract on PATH or common install locations."""
|
|
34
|
+
on_path = shutil.which("tesseract")
|
|
35
|
+
if on_path:
|
|
36
|
+
return on_path
|
|
37
|
+
|
|
38
|
+
system = platform.system()
|
|
39
|
+
if system == "Windows":
|
|
40
|
+
candidates = [
|
|
41
|
+
r"C:\Program Files\Tesseract-OCR\tesseract.exe",
|
|
42
|
+
r"C:\Program Files (x86)\Tesseract-OCR\tesseract.exe",
|
|
43
|
+
]
|
|
44
|
+
elif system == "Darwin":
|
|
45
|
+
candidates = [
|
|
46
|
+
"/opt/homebrew/bin/tesseract",
|
|
47
|
+
"/usr/local/bin/tesseract",
|
|
48
|
+
]
|
|
49
|
+
else:
|
|
50
|
+
candidates = [
|
|
51
|
+
"/usr/bin/tesseract",
|
|
52
|
+
"/usr/local/bin/tesseract",
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
for candidate in candidates:
|
|
56
|
+
if os.path.exists(candidate):
|
|
57
|
+
return candidate
|
|
58
|
+
|
|
59
|
+
return "tesseract"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
# Resolution order: env var > DEV_TESSERACT_OVERRIDE > bundled > PATH
|
|
63
|
+
_env_override = os.environ.get("TESSERACT_CMD", "").strip()
|
|
64
|
+
if _env_override:
|
|
65
|
+
TESSERACT_CMD = _env_override
|
|
66
|
+
TESSDATA_DIR = os.environ.get("TESSDATA_PREFIX") or None
|
|
67
|
+
elif DEV_TESSERACT_OVERRIDE and os.path.exists(DEV_TESSERACT_OVERRIDE):
|
|
68
|
+
TESSERACT_CMD = DEV_TESSERACT_OVERRIDE
|
|
69
|
+
TESSDATA_DIR = None
|
|
70
|
+
else:
|
|
71
|
+
_bundled_cmd, _bundled_tessdata = _find_bundled_tesseract()
|
|
72
|
+
if _bundled_cmd:
|
|
73
|
+
TESSERACT_CMD = _bundled_cmd
|
|
74
|
+
TESSDATA_DIR = _bundled_tessdata
|
|
75
|
+
else:
|
|
76
|
+
TESSERACT_CMD = _find_system_tesseract()
|
|
77
|
+
TESSDATA_DIR = None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
DEFAULT_OCR_LANG = "fra+eng"
|
|
81
|
+
DEFAULT_DPI = 300
|
|
82
|
+
OCR_UPSCALE_MIN_WIDTH = 2500
|
|
83
|
+
OCR_DOWNSCALE_MAX_WIDTH = 3500
|
|
84
|
+
MIN_WORD_CONFIDENCE = 35
|
|
85
|
+
|
|
86
|
+
PSM_EARLY_EXIT_CONFIDENCE = 80.0
|
|
87
|
+
|
|
88
|
+
# Tesseract releases the GIL, so threads give real speedup.
|
|
89
|
+
PSM_PARALLEL = True
|
|
90
|
+
PSM_MAX_WORKERS = None
|
|
91
|
+
|
|
92
|
+
OCR_TIMEOUT_SECONDS = 45
|
|
93
|
+
|
|
94
|
+
# PSM candidates: engine.py runs one pass per PSM, keeps highest mean confidence.
|
|
95
|
+
# 3=fully automatic, 4=single column, 6=uniform block, 11=sparse text
|
|
96
|
+
TESSERACT_PSM_CANDIDATES = [3, 4, 6, 11]
|
|
97
|
+
|
|
98
|
+
# 0=legacy, 1=LSTM, 2=both, 3=auto. 1 is best accuracy/speed tradeoff.
|
|
99
|
+
TESSERACT_OEM = 1
|
|
100
|
+
|
|
101
|
+
TESSERACT_CHAR_BLACKLIST = None
|
|
102
|
+
TESSERACT_CHAR_WHITELIST = None
|
|
103
|
+
|
|
104
|
+
DROP_NON_PRINTABLE_WORDS = True
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
MULTI_COLUMN_DETECTION = True
|
|
108
|
+
COLUMN_MIN_GUTTER_FRACTION = 0.03
|
|
109
|
+
COLUMN_FULL_WIDTH_LINE_FRACTION = 0.62
|
|
110
|
+
COLUMN_MIN_LINES_FOR_DETECTION = 6
|
|
111
|
+
COLUMN_GUTTER_VOTE_FRACTION = 0.6
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
APPLY_EXIF_ORIENTATION = True
|
|
115
|
+
|
|
116
|
+
UNSHARP_AMOUNT = 0.5
|
|
117
|
+
UNSHARP_SIGMA = 1.2
|
|
118
|
+
|
|
119
|
+
PREPROCESS_TRY_ADAPTIVE_THRESHOLD = False
|
|
120
|
+
|
|
121
|
+
DENOISING_H = 10
|
|
122
|
+
|
|
123
|
+
CLAHE_CLIP_LIMIT = 2.0
|
|
124
|
+
CLAHE_TILE_GRID = (8, 8)
|
|
125
|
+
|
|
126
|
+
DESKEW_MIN_PIXELS = 100
|
|
127
|
+
DESKEW_MIN_STD = 5.0
|
|
128
|
+
DESKEW_MIN_ANGLE = 0.3
|
|
129
|
+
DESKEW_MAX_ANGLE = 30.0
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
PDF_JPEG_QUALITY = 82
|
|
133
|
+
PDF_ADAPTIVE_COMPRESSION = True
|
|
134
|
+
PDF_GRAYSCALE_THRESHOLD = 0.02
|
|
135
|
+
FONT_PATH = None
|
|
136
|
+
|
|
137
|
+
PDF_METADATA = {
|
|
138
|
+
"title": "Image converted to searchable PDF",
|
|
139
|
+
"author": "scanlayer",
|
|
140
|
+
"subject": "Scanned document with OCR text layer",
|
|
141
|
+
"creator": "scanlayer (Tesseract + ReportLab)",
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
LOG_LEVEL = os.environ.get("SCANLAYER_LOG_LEVEL", "INFO")
|
|
146
|
+
LOG_TIMING = True
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
# Runtime configuration API
|
|
150
|
+
#
|
|
151
|
+
# Precedence: 1. convert() args, 2. configure() calls,
|
|
152
|
+
# 3. env vars, 4. defaults here.
|
|
153
|
+
|
|
154
|
+
_CONFIGURABLE_KEYS = {
|
|
155
|
+
"tesseract_cmd": "TESSERACT_CMD",
|
|
156
|
+
"tessdata_dir": "TESSDATA_DIR",
|
|
157
|
+
"lang": "DEFAULT_OCR_LANG",
|
|
158
|
+
"default_dpi": "DEFAULT_DPI",
|
|
159
|
+
"min_word_confidence": "MIN_WORD_CONFIDENCE",
|
|
160
|
+
"psm_candidates": "TESSERACT_PSM_CANDIDATES",
|
|
161
|
+
"psm_early_exit_confidence": "PSM_EARLY_EXIT_CONFIDENCE",
|
|
162
|
+
"psm_parallel": "PSM_PARALLEL",
|
|
163
|
+
"psm_max_workers": "PSM_MAX_WORKERS",
|
|
164
|
+
"ocr_timeout_seconds": "OCR_TIMEOUT_SECONDS",
|
|
165
|
+
"char_whitelist": "TESSERACT_CHAR_WHITELIST",
|
|
166
|
+
"char_blacklist": "TESSERACT_CHAR_BLACKLIST",
|
|
167
|
+
"jpeg_quality": "PDF_JPEG_QUALITY",
|
|
168
|
+
"font_path": "FONT_PATH",
|
|
169
|
+
"log_level": "LOG_LEVEL",
|
|
170
|
+
"log_timing": "LOG_TIMING",
|
|
171
|
+
"multi_column_detection": "MULTI_COLUMN_DETECTION",
|
|
172
|
+
"column_min_gutter_fraction": "COLUMN_MIN_GUTTER_FRACTION",
|
|
173
|
+
"column_gutter_vote_fraction": "COLUMN_GUTTER_VOTE_FRACTION",
|
|
174
|
+
"column_full_width_line_fraction": "COLUMN_FULL_WIDTH_LINE_FRACTION",
|
|
175
|
+
"column_min_lines_for_detection": "COLUMN_MIN_LINES_FOR_DETECTION",
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def configure(**overrides) -> None:
|
|
180
|
+
"""Override configuration at runtime.
|
|
181
|
+
|
|
182
|
+
Tesseract is usually found automatically. Use tesseract_cmd/tessdata_dir
|
|
183
|
+
only to point at a specific install.
|
|
184
|
+
|
|
185
|
+
Example:
|
|
186
|
+
scanlayer.configure(lang="eng", min_word_confidence=50)
|
|
187
|
+
|
|
188
|
+
Raises ValueError on unknown keywords.
|
|
189
|
+
"""
|
|
190
|
+
unknown = set(overrides) - set(_CONFIGURABLE_KEYS)
|
|
191
|
+
if unknown:
|
|
192
|
+
raise ValueError(
|
|
193
|
+
f"configure() got unknown option(s): {sorted(unknown)}. "
|
|
194
|
+
f"Valid options: {sorted(_CONFIGURABLE_KEYS)}"
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
module = sys.modules[__name__]
|
|
198
|
+
for key, value in overrides.items():
|
|
199
|
+
attr = _CONFIGURABLE_KEYS[key]
|
|
200
|
+
setattr(module, attr, value)
|
|
201
|
+
|
|
202
|
+
if "log_level" in overrides:
|
|
203
|
+
from scanlayer.utils.logger import set_log_level
|
|
204
|
+
set_log_level(overrides["log_level"])
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def get_settings() -> dict:
|
|
208
|
+
"""Return the current value of every configurable setting."""
|
|
209
|
+
module = sys.modules[__name__]
|
|
210
|
+
return {key: getattr(module, attr) for key, attr in _CONFIGURABLE_KEYS.items()}
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def load_config_file(path: str) -> dict:
|
|
214
|
+
"""Read a .yaml/.yml/.json config profile and return its contents.
|
|
215
|
+
|
|
216
|
+
Raises FileNotFoundError if path doesn't exist, ValueError for
|
|
217
|
+
unsupported extension or unparseable content.
|
|
218
|
+
"""
|
|
219
|
+
if not os.path.exists(path):
|
|
220
|
+
raise FileNotFoundError(f"Config file not found: {path}")
|
|
221
|
+
|
|
222
|
+
ext = os.path.splitext(path)[1].lower()
|
|
223
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
224
|
+
raw = f.read()
|
|
225
|
+
|
|
226
|
+
if ext in (".yaml", ".yml"):
|
|
227
|
+
try:
|
|
228
|
+
import yaml
|
|
229
|
+
except ImportError as exc:
|
|
230
|
+
raise ValueError(
|
|
231
|
+
"Reading a .yaml/.yml config file requires PyYAML "
|
|
232
|
+
"('pip install pyyaml'). Use a .json config file instead."
|
|
233
|
+
) from exc
|
|
234
|
+
try:
|
|
235
|
+
data = yaml.safe_load(raw)
|
|
236
|
+
except yaml.YAMLError as exc:
|
|
237
|
+
raise ValueError(f"Could not parse YAML config file {path}: {exc}") from exc
|
|
238
|
+
elif ext == ".json":
|
|
239
|
+
try:
|
|
240
|
+
data = json.loads(raw)
|
|
241
|
+
except json.JSONDecodeError as exc:
|
|
242
|
+
raise ValueError(f"Could not parse JSON config file {path}: {exc}") from exc
|
|
243
|
+
else:
|
|
244
|
+
raise ValueError(
|
|
245
|
+
f"Unsupported config file extension: {ext!r} (expected .yaml, "
|
|
246
|
+
f".yml, or .json)"
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
if data is None:
|
|
250
|
+
return {}
|
|
251
|
+
if not isinstance(data, dict):
|
|
252
|
+
raise ValueError(
|
|
253
|
+
f"Config file {path} must contain a mapping of setting "
|
|
254
|
+
f"names to values at the top level, got {type(data).__name__}."
|
|
255
|
+
)
|
|
256
|
+
return data
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def configure_from_file(path: str) -> dict:
|
|
260
|
+
"""Load config file and apply it via configure(). Returns applied settings.
|
|
261
|
+
|
|
262
|
+
Example:
|
|
263
|
+
scanlayer.configure_from_file("profile.yaml")
|
|
264
|
+
"""
|
|
265
|
+
settings = load_config_file(path)
|
|
266
|
+
configure(**settings)
|
|
267
|
+
return settings
|
|
Binary file
|
|
File without changes
|
|
@@ -0,0 +1,264 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Multi-column reading-order reconstruction.
|
|
3
|
+
|
|
4
|
+
Tesseract's block/paragraph numbering frequently interleaves columns on
|
|
5
|
+
two-column pages. This module re-derives reading order geometrically:
|
|
6
|
+
|
|
7
|
+
1. Group words into rows (Tesseract's vertical grouping is reliable).
|
|
8
|
+
2. Split rows into segments at wide gaps (catches column fusions).
|
|
9
|
+
3. Detect column gutters by per-row gap voting.
|
|
10
|
+
4. Assign segments to gutter-delimited bands, sort top-to-bottom.
|
|
11
|
+
5. Place full-width lines (headers/footers) before/after columns.
|
|
12
|
+
|
|
13
|
+
Tuned for prose-style layouts, not tables. Only produces left-to-right
|
|
14
|
+
column order.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import math
|
|
20
|
+
from dataclasses import dataclass
|
|
21
|
+
|
|
22
|
+
from scanlayer import config
|
|
23
|
+
from scanlayer.ocr.engine import Word
|
|
24
|
+
from scanlayer.utils.logger import get_logger, log_warning
|
|
25
|
+
|
|
26
|
+
log = get_logger(__name__)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _group_into_lines(words: list[Word]) -> list[list[Word]]:
|
|
30
|
+
"""Group words, preserving Tesseract line order via Word.line_id."""
|
|
31
|
+
lines: list[list[Word]] = []
|
|
32
|
+
current: list[Word] = []
|
|
33
|
+
current_id = None
|
|
34
|
+
for w in words:
|
|
35
|
+
if current_id is None or w.line_id == current_id:
|
|
36
|
+
current.append(w)
|
|
37
|
+
else:
|
|
38
|
+
lines.append(current)
|
|
39
|
+
current = [w]
|
|
40
|
+
current_id = w.line_id
|
|
41
|
+
if current:
|
|
42
|
+
lines.append(current)
|
|
43
|
+
return lines
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass
|
|
47
|
+
class _Line:
|
|
48
|
+
words: list[Word]
|
|
49
|
+
x_min: float
|
|
50
|
+
x_max: float
|
|
51
|
+
y_min: float
|
|
52
|
+
y_max: float
|
|
53
|
+
|
|
54
|
+
@property
|
|
55
|
+
def width(self) -> float:
|
|
56
|
+
return self.x_max - self.x_min
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def y_center(self) -> float:
|
|
60
|
+
return (self.y_min + self.y_max) / 2
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _line_bounds(words: list[Word]) -> _Line:
|
|
64
|
+
return _Line(
|
|
65
|
+
words=words,
|
|
66
|
+
x_min=min(w.x for w in words),
|
|
67
|
+
x_max=max(w.x + w.width for w in words),
|
|
68
|
+
y_min=min(w.y for w in words),
|
|
69
|
+
y_max=max(w.y + w.height for w in words),
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _split_row_into_segments(
|
|
74
|
+
row_words: list[Word], min_gap: float
|
|
75
|
+
) -> list[list[Word]]:
|
|
76
|
+
"""Split a Tesseract line into horizontal segments at wide gaps.
|
|
77
|
+
|
|
78
|
+
Under column-blind PSMs, Tesseract fuses same-row text from two
|
|
79
|
+
columns into one line. This re-splits on the horizontal gap signal.
|
|
80
|
+
"""
|
|
81
|
+
if not row_words:
|
|
82
|
+
return []
|
|
83
|
+
ordered = sorted(row_words, key=lambda w: w.x)
|
|
84
|
+
segments = [[ordered[0]]]
|
|
85
|
+
for prev, w in zip(ordered, ordered[1:]):
|
|
86
|
+
gap = w.x - (prev.x + prev.width)
|
|
87
|
+
if gap >= min_gap:
|
|
88
|
+
segments.append([w])
|
|
89
|
+
else:
|
|
90
|
+
segments[-1].append(w)
|
|
91
|
+
return segments
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _build_segmented_lines(words: list[Word], page_width: float) -> list[_Line]:
|
|
95
|
+
"""Group words into rows, then split each row into segments at wide gaps."""
|
|
96
|
+
min_gap = page_width * config.COLUMN_MIN_GUTTER_FRACTION
|
|
97
|
+
segments: list[_Line] = []
|
|
98
|
+
for row in _group_into_lines(words):
|
|
99
|
+
for segment_words in _split_row_into_segments(row, min_gap):
|
|
100
|
+
segments.append(_line_bounds(segment_words))
|
|
101
|
+
return segments
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _row_gap_votes(
|
|
105
|
+
rows: list[list[Word]], page_width: float
|
|
106
|
+
) -> tuple[list[int], float]:
|
|
107
|
+
"""Build per-x-bin profile counting rows with word gaps covering each bin."""
|
|
108
|
+
bin_w = max(1.0, page_width / 500)
|
|
109
|
+
n_bins = int(page_width / bin_w) + 2
|
|
110
|
+
votes: list[int] = [0] * n_bins
|
|
111
|
+
for row in rows:
|
|
112
|
+
ordered = sorted(row, key=lambda w: w.x)
|
|
113
|
+
voted: set[int] = set()
|
|
114
|
+
for prev, nxt in zip(ordered, ordered[1:]):
|
|
115
|
+
gap_start = prev.x + prev.width
|
|
116
|
+
gap_end = nxt.x
|
|
117
|
+
if gap_end - gap_start < bin_w:
|
|
118
|
+
continue
|
|
119
|
+
lo = max(0, int(gap_start / bin_w))
|
|
120
|
+
hi = min(n_bins - 1, int(gap_end / bin_w))
|
|
121
|
+
for i in range(lo, hi + 1):
|
|
122
|
+
if i not in voted:
|
|
123
|
+
voted.add(i)
|
|
124
|
+
votes[i] += 1
|
|
125
|
+
return votes, bin_w
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def detect_column_gutters(
|
|
129
|
+
rows: list[list[Word]], page_width: float
|
|
130
|
+
) -> list[float]:
|
|
131
|
+
"""Return x-positions of detected column gutters, or empty list if single-column.
|
|
132
|
+
|
|
133
|
+
rows are RAW Tesseract line groupings (pre-segmentation).
|
|
134
|
+
"""
|
|
135
|
+
if not rows or page_width <= 0:
|
|
136
|
+
return []
|
|
137
|
+
|
|
138
|
+
min_gap = page_width * config.COLUMN_MIN_GUTTER_FRACTION
|
|
139
|
+
full_width_threshold = page_width * config.COLUMN_FULL_WIDTH_LINE_FRACTION
|
|
140
|
+
segments = [
|
|
141
|
+
seg for row in rows for seg in _split_row_into_segments(row, min_gap)
|
|
142
|
+
]
|
|
143
|
+
narrow_lines = [
|
|
144
|
+
ln
|
|
145
|
+
for ln in (_line_bounds(s) for s in segments)
|
|
146
|
+
if ln.width <= full_width_threshold
|
|
147
|
+
]
|
|
148
|
+
if len(narrow_lines) < config.COLUMN_MIN_LINES_FOR_DETECTION:
|
|
149
|
+
return []
|
|
150
|
+
|
|
151
|
+
votes, bin_w = _row_gap_votes(rows, page_width)
|
|
152
|
+
strongest = max(votes, default=0)
|
|
153
|
+
if strongest < 2:
|
|
154
|
+
return []
|
|
155
|
+
min_votes = max(2, math.ceil(strongest * config.COLUMN_GUTTER_VOTE_FRACTION))
|
|
156
|
+
|
|
157
|
+
tolerance = page_width * 0.005
|
|
158
|
+
gutters: list[float] = []
|
|
159
|
+
start: int | None = None
|
|
160
|
+
for i, v in enumerate([*votes, 0]):
|
|
161
|
+
if v >= min_votes:
|
|
162
|
+
if start is None:
|
|
163
|
+
start = i
|
|
164
|
+
continue
|
|
165
|
+
if start is None:
|
|
166
|
+
continue
|
|
167
|
+
band_start, band_end = start * bin_w, i * bin_w
|
|
168
|
+
start = None
|
|
169
|
+
if band_end - band_start < min_gap:
|
|
170
|
+
continue
|
|
171
|
+
left_support = sum(
|
|
172
|
+
1 for ln in narrow_lines if ln.x_max <= band_start + tolerance
|
|
173
|
+
)
|
|
174
|
+
right_support = sum(
|
|
175
|
+
1 for ln in narrow_lines if ln.x_min >= band_end - tolerance
|
|
176
|
+
)
|
|
177
|
+
if left_support >= 2 and right_support >= 2:
|
|
178
|
+
gutters.append((band_start + band_end) / 2.0)
|
|
179
|
+
return gutters
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def reorder_reading_order(
|
|
183
|
+
words: list[Word], page_width: float
|
|
184
|
+
) -> tuple[list[Word], int]:
|
|
185
|
+
"""Reorder words into left-to-right, top-to-bottom column reading order.
|
|
186
|
+
|
|
187
|
+
Returns (words, 1) unchanged if single-column, not enough text,
|
|
188
|
+
no gutters found, or reorder would drop/duplicate words.
|
|
189
|
+
Otherwise returns (reordered_words, column_count).
|
|
190
|
+
"""
|
|
191
|
+
if not config.MULTI_COLUMN_DETECTION or not words or page_width <= 0:
|
|
192
|
+
return words, 1
|
|
193
|
+
|
|
194
|
+
lines = _build_segmented_lines(words, page_width)
|
|
195
|
+
if len(lines) < config.COLUMN_MIN_LINES_FOR_DETECTION:
|
|
196
|
+
return words, 1
|
|
197
|
+
|
|
198
|
+
gutters = detect_column_gutters(_group_into_lines(words), page_width)
|
|
199
|
+
if not gutters:
|
|
200
|
+
return words, 1
|
|
201
|
+
|
|
202
|
+
boundaries = [0.0, *gutters, page_width]
|
|
203
|
+
full_width_threshold = page_width * config.COLUMN_FULL_WIDTH_LINE_FRACTION
|
|
204
|
+
crossing_margin = page_width * 0.005
|
|
205
|
+
|
|
206
|
+
columns: list[list[_Line]] = [[] for _ in range(len(boundaries) - 1)]
|
|
207
|
+
full_width_lines: list[_Line] = []
|
|
208
|
+
for line in lines:
|
|
209
|
+
crosses_gutter = any(
|
|
210
|
+
line.x_min < g - crossing_margin and line.x_max > g + crossing_margin
|
|
211
|
+
for g in gutters
|
|
212
|
+
)
|
|
213
|
+
if line.width > full_width_threshold or crosses_gutter:
|
|
214
|
+
full_width_lines.append(line)
|
|
215
|
+
continue
|
|
216
|
+
center = (line.x_min + line.x_max) / 2.0
|
|
217
|
+
col_index = len(columns) - 1
|
|
218
|
+
for i in range(len(boundaries) - 1):
|
|
219
|
+
if boundaries[i] <= center < boundaries[i + 1]:
|
|
220
|
+
col_index = i
|
|
221
|
+
break
|
|
222
|
+
columns[col_index].append(line)
|
|
223
|
+
|
|
224
|
+
for col in columns:
|
|
225
|
+
col.sort(key=lambda ln: ln.y_min)
|
|
226
|
+
|
|
227
|
+
column_ys = [ln.y_min for col in columns for ln in col] or [0.0]
|
|
228
|
+
column_ys_max = [ln.y_max for col in columns for ln in col] or [0.0]
|
|
229
|
+
top_of_columns = min(column_ys)
|
|
230
|
+
bottom_of_columns = max(column_ys_max)
|
|
231
|
+
midpoint = (top_of_columns + bottom_of_columns) / 2.0
|
|
232
|
+
|
|
233
|
+
headers = sorted(
|
|
234
|
+
(ln for ln in full_width_lines if ln.y_center <= midpoint),
|
|
235
|
+
key=lambda ln: ln.y_min,
|
|
236
|
+
)
|
|
237
|
+
footers = sorted(
|
|
238
|
+
(ln for ln in full_width_lines if ln.y_center > midpoint),
|
|
239
|
+
key=lambda ln: ln.y_min,
|
|
240
|
+
)
|
|
241
|
+
|
|
242
|
+
ordered_lines = headers
|
|
243
|
+
for col in columns:
|
|
244
|
+
ordered_lines += col
|
|
245
|
+
ordered_lines += footers
|
|
246
|
+
|
|
247
|
+
reordered_words = [w for line in ordered_lines for w in line.words]
|
|
248
|
+
|
|
249
|
+
if len(reordered_words) != len(words):
|
|
250
|
+
log_warning(
|
|
251
|
+
log,
|
|
252
|
+
"Column reordering produced a different word count than the "
|
|
253
|
+
"input (bug in the layout logic), keeping original word "
|
|
254
|
+
"order as a safety fallback.",
|
|
255
|
+
)
|
|
256
|
+
return words, 1
|
|
257
|
+
|
|
258
|
+
non_empty_columns = sum(1 for col in columns if col)
|
|
259
|
+
log.info(
|
|
260
|
+
f"Multi-column layout detected: {non_empty_columns} column(s), "
|
|
261
|
+
f"gutter(s) at x={[round(g, 1) for g in gutters]}px "
|
|
262
|
+
f"(page width {page_width:.0f}px)"
|
|
263
|
+
)
|
|
264
|
+
return reordered_words, non_empty_columns
|