scanlayer 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scanlayer/__init__.py +32 -0
- scanlayer/__main__.py +10 -0
- scanlayer/cli/__init__.py +10 -0
- scanlayer/cli/dry_run.py +191 -0
- scanlayer/cli/parser.py +199 -0
- scanlayer/cli/run.py +217 -0
- scanlayer/config.py +267 -0
- scanlayer/fonts/DejaVuSans.ttf +0 -0
- scanlayer/layout/__init__.py +0 -0
- scanlayer/layout/columns.py +264 -0
- scanlayer/main.py +666 -0
- scanlayer/ocr/__init__.py +5 -0
- scanlayer/ocr/engine.py +335 -0
- scanlayer/ocr/export.py +264 -0
- scanlayer/pdf/__init__.py +5 -0
- scanlayer/pdf/builder.py +309 -0
- scanlayer/pdf/fonts.py +110 -0
- scanlayer/preprocessing/__init__.py +5 -0
- scanlayer/preprocessing/enhance.py +367 -0
- scanlayer/utils/__init__.py +19 -0
- scanlayer/utils/debug_image.py +85 -0
- scanlayer/utils/errors.py +42 -0
- scanlayer/utils/logger.py +100 -0
- scanlayer/utils/validators.py +199 -0
- scanlayer-1.0.0.dist-info/METADATA +17 -0
- scanlayer-1.0.0.dist-info/RECORD +30 -0
- scanlayer-1.0.0.dist-info/WHEEL +5 -0
- scanlayer-1.0.0.dist-info/entry_points.txt +2 -0
- scanlayer-1.0.0.dist-info/licenses/LICENSE.md +21 -0
- scanlayer-1.0.0.dist-info/top_level.txt +1 -0
scanlayer/ocr/engine.py
ADDED
|
@@ -0,0 +1,335 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Tesseract wrapper: extracts words with coordinates.
|
|
3
|
+
|
|
4
|
+
Tries multiple PSM candidates and keeps the one with highest mean
|
|
5
|
+
confidence. Supports early-exit and parallel execution via threads.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import os
|
|
11
|
+
import time
|
|
12
|
+
import unicodedata
|
|
13
|
+
from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
from typing import Optional
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
import pytesseract
|
|
19
|
+
|
|
20
|
+
from scanlayer import config
|
|
21
|
+
from scanlayer.utils.errors import OCRProcessingError
|
|
22
|
+
from scanlayer.utils.logger import get_logger, log_warning, stage_timer
|
|
23
|
+
|
|
24
|
+
log = get_logger(__name__)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class Word:
|
|
29
|
+
text: str
|
|
30
|
+
x: float
|
|
31
|
+
y: float
|
|
32
|
+
width: float
|
|
33
|
+
height: float
|
|
34
|
+
confidence: float
|
|
35
|
+
line_id: int = -1
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class PsmAttempt:
|
|
40
|
+
"""Result of a single OCR pass for a given PSM, for diagnostics."""
|
|
41
|
+
psm: int
|
|
42
|
+
words: list[Word]
|
|
43
|
+
mean_confidence: float
|
|
44
|
+
elapsed_ms: float
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass
|
|
48
|
+
class OcrResult:
|
|
49
|
+
"""Enriched OCR result, returns words + statistics."""
|
|
50
|
+
words: list[Word]
|
|
51
|
+
best_psm: int
|
|
52
|
+
best_confidence: float
|
|
53
|
+
attempts: list[PsmAttempt] = field(default_factory=list)
|
|
54
|
+
early_exited: bool = False
|
|
55
|
+
language_used: str = ""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _is_printable(text: str) -> bool:
|
|
59
|
+
"""Return True if text contains only printable characters."""
|
|
60
|
+
for ch in text:
|
|
61
|
+
cat = unicodedata.category(ch)
|
|
62
|
+
if cat[0] == "C": # Cc, Cf, Cs, Co, Cn
|
|
63
|
+
return False
|
|
64
|
+
return True
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _build_tess_config(
|
|
68
|
+
tess_base: str,
|
|
69
|
+
psm: int,
|
|
70
|
+
effective_dpi: int,
|
|
71
|
+
char_whitelist: Optional[str],
|
|
72
|
+
char_blacklist: Optional[str],
|
|
73
|
+
) -> str:
|
|
74
|
+
"""Builds the Tesseract CLI argument string from the given options."""
|
|
75
|
+
parts: list[str] = []
|
|
76
|
+
if tess_base:
|
|
77
|
+
parts.append(tess_base)
|
|
78
|
+
parts.append(f"--psm {psm}")
|
|
79
|
+
parts.append(f"--oem {config.TESSERACT_OEM}")
|
|
80
|
+
parts.append(f"-c user_defined_dpi={effective_dpi}")
|
|
81
|
+
if char_whitelist:
|
|
82
|
+
parts.append(f"-c tessedit_char_whitelist={char_whitelist}")
|
|
83
|
+
if char_blacklist:
|
|
84
|
+
parts.append(f"-c tessedit_char_blacklist={char_blacklist}")
|
|
85
|
+
return " ".join(parts)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _run_ocr(
|
|
89
|
+
ocr_image: np.ndarray,
|
|
90
|
+
lang: str,
|
|
91
|
+
psm: int,
|
|
92
|
+
tess_base: str,
|
|
93
|
+
effective_dpi: int,
|
|
94
|
+
char_whitelist: Optional[str],
|
|
95
|
+
char_blacklist: Optional[str],
|
|
96
|
+
) -> PsmAttempt:
|
|
97
|
+
"""Run a single OCR pass with a given PSM. Raises on genuine failure."""
|
|
98
|
+
t0 = time.perf_counter()
|
|
99
|
+
|
|
100
|
+
tess_config = _build_tess_config(
|
|
101
|
+
tess_base, psm, effective_dpi, char_whitelist, char_blacklist,
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
try:
|
|
105
|
+
data = pytesseract.image_to_data(
|
|
106
|
+
ocr_image, lang=lang, config=tess_config,
|
|
107
|
+
output_type=pytesseract.Output.DICT,
|
|
108
|
+
timeout=config.OCR_TIMEOUT_SECONDS,
|
|
109
|
+
)
|
|
110
|
+
except pytesseract.TesseractNotFoundError:
|
|
111
|
+
raise
|
|
112
|
+
except pytesseract.TesseractError:
|
|
113
|
+
raise
|
|
114
|
+
except RuntimeError as exc:
|
|
115
|
+
raise OCRProcessingError(
|
|
116
|
+
f"PSM {psm}: Tesseract timeout (> {config.OCR_TIMEOUT_SECONDS}s). "
|
|
117
|
+
"Image likely too large/noisy, or the Tesseract binary is stuck."
|
|
118
|
+
) from exc
|
|
119
|
+
|
|
120
|
+
words: list[Word] = []
|
|
121
|
+
n = len(data["text"])
|
|
122
|
+
for i in range(n):
|
|
123
|
+
raw_text = data["text"][i].strip()
|
|
124
|
+
if not raw_text:
|
|
125
|
+
continue
|
|
126
|
+
|
|
127
|
+
if config.DROP_NON_PRINTABLE_WORDS and not _is_printable(raw_text):
|
|
128
|
+
log.debug(f"PSM {psm}: word rejected (non-printable): {raw_text!r}")
|
|
129
|
+
continue
|
|
130
|
+
|
|
131
|
+
try:
|
|
132
|
+
conf = float(data["conf"][i])
|
|
133
|
+
except (ValueError, TypeError):
|
|
134
|
+
conf = -1.0
|
|
135
|
+
|
|
136
|
+
if conf < config.MIN_WORD_CONFIDENCE:
|
|
137
|
+
continue
|
|
138
|
+
|
|
139
|
+
width = float(data["width"][i])
|
|
140
|
+
height = float(data["height"][i])
|
|
141
|
+
if width <= 0 or height <= 0:
|
|
142
|
+
continue
|
|
143
|
+
|
|
144
|
+
try:
|
|
145
|
+
line_id = (
|
|
146
|
+
int(data["block_num"][i]) * 1_000_000
|
|
147
|
+
+ int(data["par_num"][i]) * 1_000
|
|
148
|
+
+ int(data["line_num"][i])
|
|
149
|
+
)
|
|
150
|
+
except (ValueError, TypeError, KeyError):
|
|
151
|
+
line_id = -1
|
|
152
|
+
|
|
153
|
+
words.append(Word(
|
|
154
|
+
text=raw_text,
|
|
155
|
+
x=float(data["left"][i]),
|
|
156
|
+
y=float(data["top"][i]),
|
|
157
|
+
width=width,
|
|
158
|
+
height=height,
|
|
159
|
+
confidence=conf,
|
|
160
|
+
line_id=line_id,
|
|
161
|
+
))
|
|
162
|
+
|
|
163
|
+
mean_conf = sum(w.confidence for w in words) / len(words) if words else 0.0
|
|
164
|
+
elapsed_ms = (time.perf_counter() - t0) * 1000
|
|
165
|
+
|
|
166
|
+
return PsmAttempt(
|
|
167
|
+
psm=psm, words=words, mean_confidence=mean_conf, elapsed_ms=elapsed_ms,
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _safe_run_ocr(*args, psm: int, **kwargs) -> tuple[PsmAttempt | None, Exception | None]:
|
|
172
|
+
"""Run one PSM candidate, converting exceptions into return values."""
|
|
173
|
+
try:
|
|
174
|
+
return _run_ocr(*args, **kwargs), None
|
|
175
|
+
except pytesseract.TesseractNotFoundError as exc:
|
|
176
|
+
return None, exc
|
|
177
|
+
except pytesseract.TesseractError as exc:
|
|
178
|
+
log_warning(log, f"PSM {psm} failed (TesseractError: {exc}), pass skipped.")
|
|
179
|
+
return None, exc
|
|
180
|
+
except OCRProcessingError as exc:
|
|
181
|
+
log_warning(log, f"PSM {psm} failed: {exc}")
|
|
182
|
+
return None, exc
|
|
183
|
+
except Exception as exc: # genuinely unexpected, still surfaced, not swallowed
|
|
184
|
+
log_warning(log, f"PSM {psm}, unexpected error ({type(exc).__name__}: {exc})")
|
|
185
|
+
return None, exc
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def extract_words(
|
|
189
|
+
ocr_image: np.ndarray,
|
|
190
|
+
ocr_scale: float,
|
|
191
|
+
effective_dpi: int,
|
|
192
|
+
lang: Optional[str] = None,
|
|
193
|
+
char_whitelist: Optional[str] = None,
|
|
194
|
+
char_blacklist: Optional[str] = None,
|
|
195
|
+
) -> OcrResult:
|
|
196
|
+
"""Run OCR with automatic PSM selection.
|
|
197
|
+
|
|
198
|
+
Raises OCRProcessingError if every PSM candidate fails.
|
|
199
|
+
"""
|
|
200
|
+
lang = lang or config.DEFAULT_OCR_LANG
|
|
201
|
+
wl = char_whitelist if char_whitelist is not None else config.TESSERACT_CHAR_WHITELIST
|
|
202
|
+
bl = char_blacklist if char_blacklist is not None else config.TESSERACT_CHAR_BLACKLIST
|
|
203
|
+
|
|
204
|
+
pytesseract.pytesseract.tesseract_cmd = config.TESSERACT_CMD
|
|
205
|
+
tess_base = f'--tessdata-dir "{config.TESSDATA_DIR}"' if config.TESSDATA_DIR else ""
|
|
206
|
+
|
|
207
|
+
psm_candidates = list(config.TESSERACT_PSM_CANDIDATES)
|
|
208
|
+
if not psm_candidates:
|
|
209
|
+
log_warning(log, "No PSM candidates configured, falling back to PSM 3.")
|
|
210
|
+
psm_candidates = [3]
|
|
211
|
+
|
|
212
|
+
attempts: list[PsmAttempt] = []
|
|
213
|
+
errors: list[Exception] = []
|
|
214
|
+
early_exited = False
|
|
215
|
+
best: PsmAttempt | None = None
|
|
216
|
+
|
|
217
|
+
use_parallel = config.PSM_PARALLEL and len(psm_candidates) > 1
|
|
218
|
+
log.info(
|
|
219
|
+
f"OCR started: lang={lang}, dpi={effective_dpi}, "
|
|
220
|
+
f"PSM candidates={psm_candidates}, parallel={use_parallel}"
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
with stage_timer(log, "OCR"):
|
|
224
|
+
if len(psm_candidates) == 1:
|
|
225
|
+
attempt, error = _safe_run_ocr(
|
|
226
|
+
ocr_image, lang, psm_candidates[0], tess_base,
|
|
227
|
+
effective_dpi, wl, bl, psm=psm_candidates[0],
|
|
228
|
+
)
|
|
229
|
+
if attempt is not None:
|
|
230
|
+
attempts.append(attempt)
|
|
231
|
+
best = attempt
|
|
232
|
+
else:
|
|
233
|
+
errors.append(error)
|
|
234
|
+
|
|
235
|
+
elif use_parallel:
|
|
236
|
+
max_workers = config.PSM_MAX_WORKERS or min(32, (os.cpu_count() or 1) + 4)
|
|
237
|
+
max_workers = min(max_workers, len(psm_candidates))
|
|
238
|
+
can_skip_unstarted = max_workers < len(psm_candidates)
|
|
239
|
+
|
|
240
|
+
executor = ThreadPoolExecutor(max_workers=max_workers)
|
|
241
|
+
futures = {
|
|
242
|
+
executor.submit(
|
|
243
|
+
_safe_run_ocr, ocr_image, lang, psm, tess_base,
|
|
244
|
+
effective_dpi, wl, bl, psm=psm,
|
|
245
|
+
): psm
|
|
246
|
+
for psm in psm_candidates
|
|
247
|
+
}
|
|
248
|
+
pending = set(futures)
|
|
249
|
+
try:
|
|
250
|
+
while pending:
|
|
251
|
+
done, pending = wait(pending, return_when=FIRST_COMPLETED)
|
|
252
|
+
for future in done:
|
|
253
|
+
attempt, error = future.result()
|
|
254
|
+
if attempt is not None:
|
|
255
|
+
attempts.append(attempt)
|
|
256
|
+
if best is None or attempt.mean_confidence > best.mean_confidence:
|
|
257
|
+
best = attempt
|
|
258
|
+
else:
|
|
259
|
+
errors.append(error)
|
|
260
|
+
|
|
261
|
+
if (
|
|
262
|
+
can_skip_unstarted
|
|
263
|
+
and best is not None
|
|
264
|
+
and best.mean_confidence >= config.PSM_EARLY_EXIT_CONFIDENCE
|
|
265
|
+
):
|
|
266
|
+
early_exited = True
|
|
267
|
+
log.info(
|
|
268
|
+
f"Early-exit: confidence {best.mean_confidence:.1f}% "
|
|
269
|
+
f">= threshold {config.PSM_EARLY_EXIT_CONFIDENCE}%, "
|
|
270
|
+
f"skipping {len(pending)} not-yet-started candidate(s)."
|
|
271
|
+
)
|
|
272
|
+
break
|
|
273
|
+
finally:
|
|
274
|
+
executor.shutdown(wait=False, cancel_futures=True)
|
|
275
|
+
|
|
276
|
+
else:
|
|
277
|
+
for psm in psm_candidates:
|
|
278
|
+
attempt, error = _safe_run_ocr(
|
|
279
|
+
ocr_image, lang, psm, tess_base,
|
|
280
|
+
effective_dpi, wl, bl, psm=psm,
|
|
281
|
+
)
|
|
282
|
+
if attempt is not None:
|
|
283
|
+
attempts.append(attempt)
|
|
284
|
+
if best is None or attempt.mean_confidence > best.mean_confidence:
|
|
285
|
+
best = attempt
|
|
286
|
+
else:
|
|
287
|
+
errors.append(error)
|
|
288
|
+
|
|
289
|
+
if best is not None and best.mean_confidence >= config.PSM_EARLY_EXIT_CONFIDENCE:
|
|
290
|
+
early_exited = True
|
|
291
|
+
log.info(
|
|
292
|
+
f"Early-exit: confidence {best.mean_confidence:.1f}% "
|
|
293
|
+
f">= threshold {config.PSM_EARLY_EXIT_CONFIDENCE}%, "
|
|
294
|
+
f"remaining PSMs skipped."
|
|
295
|
+
)
|
|
296
|
+
break
|
|
297
|
+
|
|
298
|
+
if best is None:
|
|
299
|
+
detail = "; ".join(f"{type(e).__name__}: {e}" for e in errors) or "no candidates ran"
|
|
300
|
+
raise OCRProcessingError(
|
|
301
|
+
f"All {len(psm_candidates)} PSM candidate(s) failed: {detail}"
|
|
302
|
+
)
|
|
303
|
+
|
|
304
|
+
for a in sorted(attempts, key=lambda x: x.psm):
|
|
305
|
+
log.debug(
|
|
306
|
+
f"PSM {a.psm}: {len(a.words)} words, "
|
|
307
|
+
f"mean confidence {a.mean_confidence:.1f}%, "
|
|
308
|
+
f"{a.elapsed_ms:.0f} ms"
|
|
309
|
+
)
|
|
310
|
+
log.info(
|
|
311
|
+
f"OCR complete: PSM {best.psm} selected "
|
|
312
|
+
f"({len(best.words)} words, confidence {best.mean_confidence:.1f}%)"
|
|
313
|
+
+ (" [early-exit]" if early_exited else "")
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
result_words: list[Word] = []
|
|
317
|
+
for w in best.words:
|
|
318
|
+
result_words.append(Word(
|
|
319
|
+
text=w.text,
|
|
320
|
+
x=w.x / ocr_scale,
|
|
321
|
+
y=w.y / ocr_scale,
|
|
322
|
+
width=w.width / ocr_scale,
|
|
323
|
+
height=w.height / ocr_scale,
|
|
324
|
+
confidence=w.confidence,
|
|
325
|
+
line_id=w.line_id,
|
|
326
|
+
))
|
|
327
|
+
|
|
328
|
+
return OcrResult(
|
|
329
|
+
words=result_words,
|
|
330
|
+
best_psm=best.psm,
|
|
331
|
+
best_confidence=best.mean_confidence,
|
|
332
|
+
attempts=attempts,
|
|
333
|
+
early_exited=early_exited,
|
|
334
|
+
language_used=lang,
|
|
335
|
+
)
|
scanlayer/ocr/export.py
ADDED
|
@@ -0,0 +1,264 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Structured OCR result export.
|
|
3
|
+
|
|
4
|
+
Turns the internal `Word` list produced by `ocr.engine.extract_words` into
|
|
5
|
+
TXT, JSON, TSV, or hOCR output, independent of PDF generation.
|
|
6
|
+
|
|
7
|
+
Callers (see `main.py`) run words through `layout.columns.reorder_reading_order`
|
|
8
|
+
before handing them to this module, so word order already reflects geometric
|
|
9
|
+
multi-column reading order. `_group_by_line` groups by Tesseract's `line_id`
|
|
10
|
+
without trusting it blindly: see its docstring for why.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json as _json
|
|
16
|
+
from typing import Optional
|
|
17
|
+
|
|
18
|
+
from scanlayer.ocr.engine import Word
|
|
19
|
+
|
|
20
|
+
FORMATS = ("pdf", "txt", "json", "tsv", "hocr")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _xml_escape(text: str) -> str:
|
|
24
|
+
return (
|
|
25
|
+
text.replace("&", "&")
|
|
26
|
+
.replace("<", "<")
|
|
27
|
+
.replace(">", ">")
|
|
28
|
+
.replace('"', """)
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
# A horizontal gap above this multiple of the reference word height is a
|
|
33
|
+
# column gutter; real inter-word gaps stay well under line height.
|
|
34
|
+
_MAX_INTRA_LINE_GAP_HEIGHT_RATIO = 4.0
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _same_visual_line(current: list[Word], w: Word) -> bool:
|
|
38
|
+
"""Whether `w` plausibly continues the text line formed by `current`.
|
|
39
|
+
|
|
40
|
+
Genuine line-mates sit close horizontally and overlap vertically; a
|
|
41
|
+
pair that straddles a column gutter fails at least one of these checks,
|
|
42
|
+
which is the signal `_group_by_line` uses to refuse the merge.
|
|
43
|
+
"""
|
|
44
|
+
prev = current[-1]
|
|
45
|
+
ref_height = max(prev.height, w.height, 1.0)
|
|
46
|
+
|
|
47
|
+
gap = w.x - (prev.x + prev.width)
|
|
48
|
+
if gap > ref_height * _MAX_INTRA_LINE_GAP_HEIGHT_RATIO:
|
|
49
|
+
return False
|
|
50
|
+
|
|
51
|
+
cur_y0 = min(x.y for x in current)
|
|
52
|
+
cur_y1 = max(x.y + x.height for x in current)
|
|
53
|
+
overlap = min(cur_y1, w.y + w.height) - max(cur_y0, w.y)
|
|
54
|
+
if overlap < -(ref_height * 0.5):
|
|
55
|
+
return False
|
|
56
|
+
|
|
57
|
+
return True
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _group_by_line(words: list[Word]) -> list[list[Word]]:
|
|
61
|
+
"""Groups words into lines by Tesseract's `line_id`, preserving order.
|
|
62
|
+
|
|
63
|
+
Words sharing a `line_id` merge only when consecutive AND
|
|
64
|
+
geometrically consistent with one visual line (see `_same_visual_line`).
|
|
65
|
+
`line_id` adjacency alone is unreliable: column-blind line detection
|
|
66
|
+
can fuse a row that crosses a gutter, and column reordering can bring
|
|
67
|
+
one line's segments from two columns together under the same `line_id`.
|
|
68
|
+
"""
|
|
69
|
+
lines: list[list[Word]] = []
|
|
70
|
+
current: list[Word] = []
|
|
71
|
+
current_id = None
|
|
72
|
+
for w in words:
|
|
73
|
+
if current and w.line_id == current_id and _same_visual_line(current, w):
|
|
74
|
+
current.append(w)
|
|
75
|
+
else:
|
|
76
|
+
if current:
|
|
77
|
+
lines.append(current)
|
|
78
|
+
current = [w]
|
|
79
|
+
current_id = w.line_id
|
|
80
|
+
if current:
|
|
81
|
+
lines.append(current)
|
|
82
|
+
return lines
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def to_text(words: list[Word]) -> str:
|
|
86
|
+
"""Plain extracted text, one line per detected Tesseract text line."""
|
|
87
|
+
lines = _group_by_line(words)
|
|
88
|
+
return "\n".join(" ".join(w.text for w in line) for line in lines)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def to_json(
|
|
92
|
+
words: list[Word],
|
|
93
|
+
mean_confidence: float,
|
|
94
|
+
best_psm: Optional[int],
|
|
95
|
+
language_used: str,
|
|
96
|
+
image_width: int,
|
|
97
|
+
image_height: int,
|
|
98
|
+
) -> str:
|
|
99
|
+
payload = {
|
|
100
|
+
"text": to_text(words),
|
|
101
|
+
"mean_confidence": round(mean_confidence, 2),
|
|
102
|
+
"best_psm": best_psm,
|
|
103
|
+
"language": language_used,
|
|
104
|
+
"image_width": image_width,
|
|
105
|
+
"image_height": image_height,
|
|
106
|
+
"words": [
|
|
107
|
+
{
|
|
108
|
+
"text": w.text,
|
|
109
|
+
"confidence": round(w.confidence, 2),
|
|
110
|
+
"bbox": [
|
|
111
|
+
round(w.x, 1), round(w.y, 1),
|
|
112
|
+
round(w.x + w.width, 1), round(w.y + w.height, 1),
|
|
113
|
+
],
|
|
114
|
+
}
|
|
115
|
+
for w in words
|
|
116
|
+
],
|
|
117
|
+
}
|
|
118
|
+
return _json.dumps(payload, ensure_ascii=False, indent=2)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def to_tsv(words: list[Word]) -> str:
|
|
122
|
+
"""Tab-separated word list. This is NOT Tesseract's own TSV column
|
|
123
|
+
layout (level/page/block/par/line/word_num/left/top/width/height/
|
|
124
|
+
conf/text), it is a simpler, flat schema kept intentionally
|
|
125
|
+
minimal for loading straight into a spreadsheet or a pandas
|
|
126
|
+
DataFrame without extra parsing.
|
|
127
|
+
"""
|
|
128
|
+
header = "text\tconfidence\tx\ty\twidth\theight\tline_id"
|
|
129
|
+
rows = [header]
|
|
130
|
+
for w in words:
|
|
131
|
+
text = w.text.replace("\t", " ")
|
|
132
|
+
rows.append(
|
|
133
|
+
f"{text}\t{w.confidence:.2f}\t{w.x:.1f}\t{w.y:.1f}\t"
|
|
134
|
+
f"{w.width:.1f}\t{w.height:.1f}\t{w.line_id}"
|
|
135
|
+
)
|
|
136
|
+
return "\n".join(rows)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def to_hocr(
|
|
140
|
+
words: list[Word],
|
|
141
|
+
image_width: int,
|
|
142
|
+
image_height: int,
|
|
143
|
+
source_name: str = "image",
|
|
144
|
+
) -> str:
|
|
145
|
+
"""Minimal hOCR (a standard OCR layout representation). Covers
|
|
146
|
+
ocr_page, ocr_line, and ocrx_word, no ocr_carea/ocr_par distinction,
|
|
147
|
+
since `Word` does not track paragraph boundaries separately from
|
|
148
|
+
line boundaries.
|
|
149
|
+
"""
|
|
150
|
+
lines = _group_by_line(words)
|
|
151
|
+
body_lines = [
|
|
152
|
+
f'<div class="ocr_page" id="page_1" '
|
|
153
|
+
f'style="width:{image_width}px;height:{image_height}px" '
|
|
154
|
+
f'title="image \'{_xml_escape(source_name)}\'; bbox 0 0 '
|
|
155
|
+
f'{image_width} {image_height}">'
|
|
156
|
+
]
|
|
157
|
+
for i, line in enumerate(lines, start=1):
|
|
158
|
+
if not line:
|
|
159
|
+
continue
|
|
160
|
+
x0 = min(w.x for w in line)
|
|
161
|
+
y0 = min(w.y for w in line)
|
|
162
|
+
x1 = max(w.x + w.width for w in line)
|
|
163
|
+
y1 = max(w.y + w.height for w in line)
|
|
164
|
+
line_h = max(y1 - y0, 1.0)
|
|
165
|
+
font_px = max(round(line_h * 0.82), 6)
|
|
166
|
+
body_lines.append(
|
|
167
|
+
f'<span class="ocr_line" id="line_{i}" '
|
|
168
|
+
f'style="left:{x0:.0f}px;top:{y0:.0f}px;'
|
|
169
|
+
f'width:{(x1 - x0):.0f}px;height:{line_h:.0f}px;'
|
|
170
|
+
f'font-size:{font_px}px" '
|
|
171
|
+
f'title="bbox {x0:.0f} {y0:.0f} {x1:.0f} {y1:.0f}">'
|
|
172
|
+
)
|
|
173
|
+
for j, w in enumerate(line, start=1):
|
|
174
|
+
wx0, wy0 = w.x, w.y
|
|
175
|
+
wx1, wy1 = w.x + w.width, w.y + w.height
|
|
176
|
+
conf_cls = (
|
|
177
|
+
" low-conf" if w.confidence < 60
|
|
178
|
+
else " mid-conf" if w.confidence < 85
|
|
179
|
+
else ""
|
|
180
|
+
)
|
|
181
|
+
body_lines.append(
|
|
182
|
+
f'<span class="ocrx_word{conf_cls}" id="line_{i}_word_{j}" '
|
|
183
|
+
f'title="bbox {wx0:.0f} {wy0:.0f} {wx1:.0f} {wy1:.0f}; '
|
|
184
|
+
f'x_wconf {w.confidence:.0f}">{_xml_escape(w.text)}</span>'
|
|
185
|
+
)
|
|
186
|
+
body_lines.append("</span>")
|
|
187
|
+
body_lines.append("</div>")
|
|
188
|
+
body = "\n".join(body_lines)
|
|
189
|
+
|
|
190
|
+
style = """
|
|
191
|
+
:root{ --ink:#1c1c1c; --low:#d64545; --mid:#c98a1f; --hl:#fff3a3; }
|
|
192
|
+
html,body{ margin:0; padding:0; background:#e9e9e9; }
|
|
193
|
+
body{
|
|
194
|
+
padding:2.5rem 1rem;
|
|
195
|
+
font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",Helvetica,Arial,sans-serif;
|
|
196
|
+
}
|
|
197
|
+
.ocr_page{
|
|
198
|
+
position:relative;
|
|
199
|
+
margin:0 auto;
|
|
200
|
+
background:#fff;
|
|
201
|
+
border:1px solid #d8d8d8;
|
|
202
|
+
box-shadow:0 2px 14px rgba(0,0,0,.10);
|
|
203
|
+
}
|
|
204
|
+
.ocr_line{
|
|
205
|
+
position:absolute;
|
|
206
|
+
white-space:nowrap;
|
|
207
|
+
line-height:1.05;
|
|
208
|
+
color:var(--ink);
|
|
209
|
+
overflow:visible;
|
|
210
|
+
}
|
|
211
|
+
.ocrx_word{
|
|
212
|
+
padding:0 1px;
|
|
213
|
+
border-radius:2px;
|
|
214
|
+
cursor:default;
|
|
215
|
+
}
|
|
216
|
+
.ocrx_word:hover{
|
|
217
|
+
background:var(--hl);
|
|
218
|
+
}
|
|
219
|
+
.ocrx_word.low-conf{ border-bottom:1px dotted var(--low); }
|
|
220
|
+
.ocrx_word.mid-conf{ border-bottom:1px dotted var(--mid); }
|
|
221
|
+
""".strip()
|
|
222
|
+
|
|
223
|
+
return (
|
|
224
|
+
'<?xml version="1.0" encoding="UTF-8"?>\n'
|
|
225
|
+
'<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" '
|
|
226
|
+
'"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">\n'
|
|
227
|
+
'<html xmlns="http://www.w3.org/1999/xhtml">\n'
|
|
228
|
+
"<head>\n"
|
|
229
|
+
"<title>OCR Output</title>\n"
|
|
230
|
+
'<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>\n'
|
|
231
|
+
'<meta name="ocr-system" content="tesseract via scanlayer" />\n'
|
|
232
|
+
'<meta name="ocr-capabilities" content="ocr_page ocr_line ocrx_word" />\n'
|
|
233
|
+
f"<style>\n{style}\n</style>\n"
|
|
234
|
+
"</head>\n"
|
|
235
|
+
f"<body>\n{body}\n</body>\n</html>\n"
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def export_words(fmt: str, words: list[Word], **kwargs) -> str:
|
|
240
|
+
"""Dispatch helper used by main.py. `fmt` must be one of "txt",
|
|
241
|
+
"json", "tsv", "hocr" ("pdf" is not handled here, that format
|
|
242
|
+
goes through pdf.builder instead).
|
|
243
|
+
"""
|
|
244
|
+
if fmt == "txt":
|
|
245
|
+
return to_text(words)
|
|
246
|
+
if fmt == "json":
|
|
247
|
+
return to_json(
|
|
248
|
+
words,
|
|
249
|
+
mean_confidence=kwargs.get("mean_confidence", 0.0),
|
|
250
|
+
best_psm=kwargs.get("best_psm"),
|
|
251
|
+
language_used=kwargs.get("language_used", ""),
|
|
252
|
+
image_width=kwargs.get("image_width", 0),
|
|
253
|
+
image_height=kwargs.get("image_height", 0),
|
|
254
|
+
)
|
|
255
|
+
if fmt == "tsv":
|
|
256
|
+
return to_tsv(words)
|
|
257
|
+
if fmt == "hocr":
|
|
258
|
+
return to_hocr(
|
|
259
|
+
words,
|
|
260
|
+
image_width=kwargs.get("image_width", 0),
|
|
261
|
+
image_height=kwargs.get("image_height", 0),
|
|
262
|
+
source_name=kwargs.get("source_name", "image"),
|
|
263
|
+
)
|
|
264
|
+
raise ValueError(f"Unsupported export format: {fmt!r}")
|