scanlayer 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,335 @@
1
+ """
2
+ Tesseract wrapper: extracts words with coordinates.
3
+
4
+ Tries multiple PSM candidates and keeps the one with highest mean
5
+ confidence. Supports early-exit and parallel execution via threads.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import os
11
+ import time
12
+ import unicodedata
13
+ from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
14
+ from dataclasses import dataclass, field
15
+ from typing import Optional
16
+
17
+ import numpy as np
18
+ import pytesseract
19
+
20
+ from scanlayer import config
21
+ from scanlayer.utils.errors import OCRProcessingError
22
+ from scanlayer.utils.logger import get_logger, log_warning, stage_timer
23
+
24
+ log = get_logger(__name__)
25
+
26
+
27
+ @dataclass
28
+ class Word:
29
+ text: str
30
+ x: float
31
+ y: float
32
+ width: float
33
+ height: float
34
+ confidence: float
35
+ line_id: int = -1
36
+
37
+
38
+ @dataclass
39
+ class PsmAttempt:
40
+ """Result of a single OCR pass for a given PSM, for diagnostics."""
41
+ psm: int
42
+ words: list[Word]
43
+ mean_confidence: float
44
+ elapsed_ms: float
45
+
46
+
47
+ @dataclass
48
+ class OcrResult:
49
+ """Enriched OCR result, returns words + statistics."""
50
+ words: list[Word]
51
+ best_psm: int
52
+ best_confidence: float
53
+ attempts: list[PsmAttempt] = field(default_factory=list)
54
+ early_exited: bool = False
55
+ language_used: str = ""
56
+
57
+
58
+ def _is_printable(text: str) -> bool:
59
+ """Return True if text contains only printable characters."""
60
+ for ch in text:
61
+ cat = unicodedata.category(ch)
62
+ if cat[0] == "C": # Cc, Cf, Cs, Co, Cn
63
+ return False
64
+ return True
65
+
66
+
67
+ def _build_tess_config(
68
+ tess_base: str,
69
+ psm: int,
70
+ effective_dpi: int,
71
+ char_whitelist: Optional[str],
72
+ char_blacklist: Optional[str],
73
+ ) -> str:
74
+ """Builds the Tesseract CLI argument string from the given options."""
75
+ parts: list[str] = []
76
+ if tess_base:
77
+ parts.append(tess_base)
78
+ parts.append(f"--psm {psm}")
79
+ parts.append(f"--oem {config.TESSERACT_OEM}")
80
+ parts.append(f"-c user_defined_dpi={effective_dpi}")
81
+ if char_whitelist:
82
+ parts.append(f"-c tessedit_char_whitelist={char_whitelist}")
83
+ if char_blacklist:
84
+ parts.append(f"-c tessedit_char_blacklist={char_blacklist}")
85
+ return " ".join(parts)
86
+
87
+
88
+ def _run_ocr(
89
+ ocr_image: np.ndarray,
90
+ lang: str,
91
+ psm: int,
92
+ tess_base: str,
93
+ effective_dpi: int,
94
+ char_whitelist: Optional[str],
95
+ char_blacklist: Optional[str],
96
+ ) -> PsmAttempt:
97
+ """Run a single OCR pass with a given PSM. Raises on genuine failure."""
98
+ t0 = time.perf_counter()
99
+
100
+ tess_config = _build_tess_config(
101
+ tess_base, psm, effective_dpi, char_whitelist, char_blacklist,
102
+ )
103
+
104
+ try:
105
+ data = pytesseract.image_to_data(
106
+ ocr_image, lang=lang, config=tess_config,
107
+ output_type=pytesseract.Output.DICT,
108
+ timeout=config.OCR_TIMEOUT_SECONDS,
109
+ )
110
+ except pytesseract.TesseractNotFoundError:
111
+ raise
112
+ except pytesseract.TesseractError:
113
+ raise
114
+ except RuntimeError as exc:
115
+ raise OCRProcessingError(
116
+ f"PSM {psm}: Tesseract timeout (> {config.OCR_TIMEOUT_SECONDS}s). "
117
+ "Image likely too large/noisy, or the Tesseract binary is stuck."
118
+ ) from exc
119
+
120
+ words: list[Word] = []
121
+ n = len(data["text"])
122
+ for i in range(n):
123
+ raw_text = data["text"][i].strip()
124
+ if not raw_text:
125
+ continue
126
+
127
+ if config.DROP_NON_PRINTABLE_WORDS and not _is_printable(raw_text):
128
+ log.debug(f"PSM {psm}: word rejected (non-printable): {raw_text!r}")
129
+ continue
130
+
131
+ try:
132
+ conf = float(data["conf"][i])
133
+ except (ValueError, TypeError):
134
+ conf = -1.0
135
+
136
+ if conf < config.MIN_WORD_CONFIDENCE:
137
+ continue
138
+
139
+ width = float(data["width"][i])
140
+ height = float(data["height"][i])
141
+ if width <= 0 or height <= 0:
142
+ continue
143
+
144
+ try:
145
+ line_id = (
146
+ int(data["block_num"][i]) * 1_000_000
147
+ + int(data["par_num"][i]) * 1_000
148
+ + int(data["line_num"][i])
149
+ )
150
+ except (ValueError, TypeError, KeyError):
151
+ line_id = -1
152
+
153
+ words.append(Word(
154
+ text=raw_text,
155
+ x=float(data["left"][i]),
156
+ y=float(data["top"][i]),
157
+ width=width,
158
+ height=height,
159
+ confidence=conf,
160
+ line_id=line_id,
161
+ ))
162
+
163
+ mean_conf = sum(w.confidence for w in words) / len(words) if words else 0.0
164
+ elapsed_ms = (time.perf_counter() - t0) * 1000
165
+
166
+ return PsmAttempt(
167
+ psm=psm, words=words, mean_confidence=mean_conf, elapsed_ms=elapsed_ms,
168
+ )
169
+
170
+
171
+ def _safe_run_ocr(*args, psm: int, **kwargs) -> tuple[PsmAttempt | None, Exception | None]:
172
+ """Run one PSM candidate, converting exceptions into return values."""
173
+ try:
174
+ return _run_ocr(*args, **kwargs), None
175
+ except pytesseract.TesseractNotFoundError as exc:
176
+ return None, exc
177
+ except pytesseract.TesseractError as exc:
178
+ log_warning(log, f"PSM {psm} failed (TesseractError: {exc}), pass skipped.")
179
+ return None, exc
180
+ except OCRProcessingError as exc:
181
+ log_warning(log, f"PSM {psm} failed: {exc}")
182
+ return None, exc
183
+ except Exception as exc: # genuinely unexpected, still surfaced, not swallowed
184
+ log_warning(log, f"PSM {psm}, unexpected error ({type(exc).__name__}: {exc})")
185
+ return None, exc
186
+
187
+
188
+ def extract_words(
189
+ ocr_image: np.ndarray,
190
+ ocr_scale: float,
191
+ effective_dpi: int,
192
+ lang: Optional[str] = None,
193
+ char_whitelist: Optional[str] = None,
194
+ char_blacklist: Optional[str] = None,
195
+ ) -> OcrResult:
196
+ """Run OCR with automatic PSM selection.
197
+
198
+ Raises OCRProcessingError if every PSM candidate fails.
199
+ """
200
+ lang = lang or config.DEFAULT_OCR_LANG
201
+ wl = char_whitelist if char_whitelist is not None else config.TESSERACT_CHAR_WHITELIST
202
+ bl = char_blacklist if char_blacklist is not None else config.TESSERACT_CHAR_BLACKLIST
203
+
204
+ pytesseract.pytesseract.tesseract_cmd = config.TESSERACT_CMD
205
+ tess_base = f'--tessdata-dir "{config.TESSDATA_DIR}"' if config.TESSDATA_DIR else ""
206
+
207
+ psm_candidates = list(config.TESSERACT_PSM_CANDIDATES)
208
+ if not psm_candidates:
209
+ log_warning(log, "No PSM candidates configured, falling back to PSM 3.")
210
+ psm_candidates = [3]
211
+
212
+ attempts: list[PsmAttempt] = []
213
+ errors: list[Exception] = []
214
+ early_exited = False
215
+ best: PsmAttempt | None = None
216
+
217
+ use_parallel = config.PSM_PARALLEL and len(psm_candidates) > 1
218
+ log.info(
219
+ f"OCR started: lang={lang}, dpi={effective_dpi}, "
220
+ f"PSM candidates={psm_candidates}, parallel={use_parallel}"
221
+ )
222
+
223
+ with stage_timer(log, "OCR"):
224
+ if len(psm_candidates) == 1:
225
+ attempt, error = _safe_run_ocr(
226
+ ocr_image, lang, psm_candidates[0], tess_base,
227
+ effective_dpi, wl, bl, psm=psm_candidates[0],
228
+ )
229
+ if attempt is not None:
230
+ attempts.append(attempt)
231
+ best = attempt
232
+ else:
233
+ errors.append(error)
234
+
235
+ elif use_parallel:
236
+ max_workers = config.PSM_MAX_WORKERS or min(32, (os.cpu_count() or 1) + 4)
237
+ max_workers = min(max_workers, len(psm_candidates))
238
+ can_skip_unstarted = max_workers < len(psm_candidates)
239
+
240
+ executor = ThreadPoolExecutor(max_workers=max_workers)
241
+ futures = {
242
+ executor.submit(
243
+ _safe_run_ocr, ocr_image, lang, psm, tess_base,
244
+ effective_dpi, wl, bl, psm=psm,
245
+ ): psm
246
+ for psm in psm_candidates
247
+ }
248
+ pending = set(futures)
249
+ try:
250
+ while pending:
251
+ done, pending = wait(pending, return_when=FIRST_COMPLETED)
252
+ for future in done:
253
+ attempt, error = future.result()
254
+ if attempt is not None:
255
+ attempts.append(attempt)
256
+ if best is None or attempt.mean_confidence > best.mean_confidence:
257
+ best = attempt
258
+ else:
259
+ errors.append(error)
260
+
261
+ if (
262
+ can_skip_unstarted
263
+ and best is not None
264
+ and best.mean_confidence >= config.PSM_EARLY_EXIT_CONFIDENCE
265
+ ):
266
+ early_exited = True
267
+ log.info(
268
+ f"Early-exit: confidence {best.mean_confidence:.1f}% "
269
+ f">= threshold {config.PSM_EARLY_EXIT_CONFIDENCE}%, "
270
+ f"skipping {len(pending)} not-yet-started candidate(s)."
271
+ )
272
+ break
273
+ finally:
274
+ executor.shutdown(wait=False, cancel_futures=True)
275
+
276
+ else:
277
+ for psm in psm_candidates:
278
+ attempt, error = _safe_run_ocr(
279
+ ocr_image, lang, psm, tess_base,
280
+ effective_dpi, wl, bl, psm=psm,
281
+ )
282
+ if attempt is not None:
283
+ attempts.append(attempt)
284
+ if best is None or attempt.mean_confidence > best.mean_confidence:
285
+ best = attempt
286
+ else:
287
+ errors.append(error)
288
+
289
+ if best is not None and best.mean_confidence >= config.PSM_EARLY_EXIT_CONFIDENCE:
290
+ early_exited = True
291
+ log.info(
292
+ f"Early-exit: confidence {best.mean_confidence:.1f}% "
293
+ f">= threshold {config.PSM_EARLY_EXIT_CONFIDENCE}%, "
294
+ f"remaining PSMs skipped."
295
+ )
296
+ break
297
+
298
+ if best is None:
299
+ detail = "; ".join(f"{type(e).__name__}: {e}" for e in errors) or "no candidates ran"
300
+ raise OCRProcessingError(
301
+ f"All {len(psm_candidates)} PSM candidate(s) failed: {detail}"
302
+ )
303
+
304
+ for a in sorted(attempts, key=lambda x: x.psm):
305
+ log.debug(
306
+ f"PSM {a.psm}: {len(a.words)} words, "
307
+ f"mean confidence {a.mean_confidence:.1f}%, "
308
+ f"{a.elapsed_ms:.0f} ms"
309
+ )
310
+ log.info(
311
+ f"OCR complete: PSM {best.psm} selected "
312
+ f"({len(best.words)} words, confidence {best.mean_confidence:.1f}%)"
313
+ + (" [early-exit]" if early_exited else "")
314
+ )
315
+
316
+ result_words: list[Word] = []
317
+ for w in best.words:
318
+ result_words.append(Word(
319
+ text=w.text,
320
+ x=w.x / ocr_scale,
321
+ y=w.y / ocr_scale,
322
+ width=w.width / ocr_scale,
323
+ height=w.height / ocr_scale,
324
+ confidence=w.confidence,
325
+ line_id=w.line_id,
326
+ ))
327
+
328
+ return OcrResult(
329
+ words=result_words,
330
+ best_psm=best.psm,
331
+ best_confidence=best.mean_confidence,
332
+ attempts=attempts,
333
+ early_exited=early_exited,
334
+ language_used=lang,
335
+ )
@@ -0,0 +1,264 @@
1
+ """
2
+ Structured OCR result export.
3
+
4
+ Turns the internal `Word` list produced by `ocr.engine.extract_words` into
5
+ TXT, JSON, TSV, or hOCR output, independent of PDF generation.
6
+
7
+ Callers (see `main.py`) run words through `layout.columns.reorder_reading_order`
8
+ before handing them to this module, so word order already reflects geometric
9
+ multi-column reading order. `_group_by_line` groups by Tesseract's `line_id`
10
+ without trusting it blindly: see its docstring for why.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json as _json
16
+ from typing import Optional
17
+
18
+ from scanlayer.ocr.engine import Word
19
+
20
+ FORMATS = ("pdf", "txt", "json", "tsv", "hocr")
21
+
22
+
23
+ def _xml_escape(text: str) -> str:
24
+ return (
25
+ text.replace("&", "&amp;")
26
+ .replace("<", "&lt;")
27
+ .replace(">", "&gt;")
28
+ .replace('"', "&quot;")
29
+ )
30
+
31
+
32
+ # A horizontal gap above this multiple of the reference word height is a
33
+ # column gutter; real inter-word gaps stay well under line height.
34
+ _MAX_INTRA_LINE_GAP_HEIGHT_RATIO = 4.0
35
+
36
+
37
+ def _same_visual_line(current: list[Word], w: Word) -> bool:
38
+ """Whether `w` plausibly continues the text line formed by `current`.
39
+
40
+ Genuine line-mates sit close horizontally and overlap vertically; a
41
+ pair that straddles a column gutter fails at least one of these checks,
42
+ which is the signal `_group_by_line` uses to refuse the merge.
43
+ """
44
+ prev = current[-1]
45
+ ref_height = max(prev.height, w.height, 1.0)
46
+
47
+ gap = w.x - (prev.x + prev.width)
48
+ if gap > ref_height * _MAX_INTRA_LINE_GAP_HEIGHT_RATIO:
49
+ return False
50
+
51
+ cur_y0 = min(x.y for x in current)
52
+ cur_y1 = max(x.y + x.height for x in current)
53
+ overlap = min(cur_y1, w.y + w.height) - max(cur_y0, w.y)
54
+ if overlap < -(ref_height * 0.5):
55
+ return False
56
+
57
+ return True
58
+
59
+
60
+ def _group_by_line(words: list[Word]) -> list[list[Word]]:
61
+ """Groups words into lines by Tesseract's `line_id`, preserving order.
62
+
63
+ Words sharing a `line_id` merge only when consecutive AND
64
+ geometrically consistent with one visual line (see `_same_visual_line`).
65
+ `line_id` adjacency alone is unreliable: column-blind line detection
66
+ can fuse a row that crosses a gutter, and column reordering can bring
67
+ one line's segments from two columns together under the same `line_id`.
68
+ """
69
+ lines: list[list[Word]] = []
70
+ current: list[Word] = []
71
+ current_id = None
72
+ for w in words:
73
+ if current and w.line_id == current_id and _same_visual_line(current, w):
74
+ current.append(w)
75
+ else:
76
+ if current:
77
+ lines.append(current)
78
+ current = [w]
79
+ current_id = w.line_id
80
+ if current:
81
+ lines.append(current)
82
+ return lines
83
+
84
+
85
+ def to_text(words: list[Word]) -> str:
86
+ """Plain extracted text, one line per detected Tesseract text line."""
87
+ lines = _group_by_line(words)
88
+ return "\n".join(" ".join(w.text for w in line) for line in lines)
89
+
90
+
91
+ def to_json(
92
+ words: list[Word],
93
+ mean_confidence: float,
94
+ best_psm: Optional[int],
95
+ language_used: str,
96
+ image_width: int,
97
+ image_height: int,
98
+ ) -> str:
99
+ payload = {
100
+ "text": to_text(words),
101
+ "mean_confidence": round(mean_confidence, 2),
102
+ "best_psm": best_psm,
103
+ "language": language_used,
104
+ "image_width": image_width,
105
+ "image_height": image_height,
106
+ "words": [
107
+ {
108
+ "text": w.text,
109
+ "confidence": round(w.confidence, 2),
110
+ "bbox": [
111
+ round(w.x, 1), round(w.y, 1),
112
+ round(w.x + w.width, 1), round(w.y + w.height, 1),
113
+ ],
114
+ }
115
+ for w in words
116
+ ],
117
+ }
118
+ return _json.dumps(payload, ensure_ascii=False, indent=2)
119
+
120
+
121
+ def to_tsv(words: list[Word]) -> str:
122
+ """Tab-separated word list. This is NOT Tesseract's own TSV column
123
+ layout (level/page/block/par/line/word_num/left/top/width/height/
124
+ conf/text), it is a simpler, flat schema kept intentionally
125
+ minimal for loading straight into a spreadsheet or a pandas
126
+ DataFrame without extra parsing.
127
+ """
128
+ header = "text\tconfidence\tx\ty\twidth\theight\tline_id"
129
+ rows = [header]
130
+ for w in words:
131
+ text = w.text.replace("\t", " ")
132
+ rows.append(
133
+ f"{text}\t{w.confidence:.2f}\t{w.x:.1f}\t{w.y:.1f}\t"
134
+ f"{w.width:.1f}\t{w.height:.1f}\t{w.line_id}"
135
+ )
136
+ return "\n".join(rows)
137
+
138
+
139
+ def to_hocr(
140
+ words: list[Word],
141
+ image_width: int,
142
+ image_height: int,
143
+ source_name: str = "image",
144
+ ) -> str:
145
+ """Minimal hOCR (a standard OCR layout representation). Covers
146
+ ocr_page, ocr_line, and ocrx_word, no ocr_carea/ocr_par distinction,
147
+ since `Word` does not track paragraph boundaries separately from
148
+ line boundaries.
149
+ """
150
+ lines = _group_by_line(words)
151
+ body_lines = [
152
+ f'<div class="ocr_page" id="page_1" '
153
+ f'style="width:{image_width}px;height:{image_height}px" '
154
+ f'title="image \'{_xml_escape(source_name)}\'; bbox 0 0 '
155
+ f'{image_width} {image_height}">'
156
+ ]
157
+ for i, line in enumerate(lines, start=1):
158
+ if not line:
159
+ continue
160
+ x0 = min(w.x for w in line)
161
+ y0 = min(w.y for w in line)
162
+ x1 = max(w.x + w.width for w in line)
163
+ y1 = max(w.y + w.height for w in line)
164
+ line_h = max(y1 - y0, 1.0)
165
+ font_px = max(round(line_h * 0.82), 6)
166
+ body_lines.append(
167
+ f'<span class="ocr_line" id="line_{i}" '
168
+ f'style="left:{x0:.0f}px;top:{y0:.0f}px;'
169
+ f'width:{(x1 - x0):.0f}px;height:{line_h:.0f}px;'
170
+ f'font-size:{font_px}px" '
171
+ f'title="bbox {x0:.0f} {y0:.0f} {x1:.0f} {y1:.0f}">'
172
+ )
173
+ for j, w in enumerate(line, start=1):
174
+ wx0, wy0 = w.x, w.y
175
+ wx1, wy1 = w.x + w.width, w.y + w.height
176
+ conf_cls = (
177
+ " low-conf" if w.confidence < 60
178
+ else " mid-conf" if w.confidence < 85
179
+ else ""
180
+ )
181
+ body_lines.append(
182
+ f'<span class="ocrx_word{conf_cls}" id="line_{i}_word_{j}" '
183
+ f'title="bbox {wx0:.0f} {wy0:.0f} {wx1:.0f} {wy1:.0f}; '
184
+ f'x_wconf {w.confidence:.0f}">{_xml_escape(w.text)}</span>'
185
+ )
186
+ body_lines.append("</span>")
187
+ body_lines.append("</div>")
188
+ body = "\n".join(body_lines)
189
+
190
+ style = """
191
+ :root{ --ink:#1c1c1c; --low:#d64545; --mid:#c98a1f; --hl:#fff3a3; }
192
+ html,body{ margin:0; padding:0; background:#e9e9e9; }
193
+ body{
194
+ padding:2.5rem 1rem;
195
+ font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",Helvetica,Arial,sans-serif;
196
+ }
197
+ .ocr_page{
198
+ position:relative;
199
+ margin:0 auto;
200
+ background:#fff;
201
+ border:1px solid #d8d8d8;
202
+ box-shadow:0 2px 14px rgba(0,0,0,.10);
203
+ }
204
+ .ocr_line{
205
+ position:absolute;
206
+ white-space:nowrap;
207
+ line-height:1.05;
208
+ color:var(--ink);
209
+ overflow:visible;
210
+ }
211
+ .ocrx_word{
212
+ padding:0 1px;
213
+ border-radius:2px;
214
+ cursor:default;
215
+ }
216
+ .ocrx_word:hover{
217
+ background:var(--hl);
218
+ }
219
+ .ocrx_word.low-conf{ border-bottom:1px dotted var(--low); }
220
+ .ocrx_word.mid-conf{ border-bottom:1px dotted var(--mid); }
221
+ """.strip()
222
+
223
+ return (
224
+ '<?xml version="1.0" encoding="UTF-8"?>\n'
225
+ '<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" '
226
+ '"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">\n'
227
+ '<html xmlns="http://www.w3.org/1999/xhtml">\n'
228
+ "<head>\n"
229
+ "<title>OCR Output</title>\n"
230
+ '<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>\n'
231
+ '<meta name="ocr-system" content="tesseract via scanlayer" />\n'
232
+ '<meta name="ocr-capabilities" content="ocr_page ocr_line ocrx_word" />\n'
233
+ f"<style>\n{style}\n</style>\n"
234
+ "</head>\n"
235
+ f"<body>\n{body}\n</body>\n</html>\n"
236
+ )
237
+
238
+
239
+ def export_words(fmt: str, words: list[Word], **kwargs) -> str:
240
+ """Dispatch helper used by main.py. `fmt` must be one of "txt",
241
+ "json", "tsv", "hocr" ("pdf" is not handled here, that format
242
+ goes through pdf.builder instead).
243
+ """
244
+ if fmt == "txt":
245
+ return to_text(words)
246
+ if fmt == "json":
247
+ return to_json(
248
+ words,
249
+ mean_confidence=kwargs.get("mean_confidence", 0.0),
250
+ best_psm=kwargs.get("best_psm"),
251
+ language_used=kwargs.get("language_used", ""),
252
+ image_width=kwargs.get("image_width", 0),
253
+ image_height=kwargs.get("image_height", 0),
254
+ )
255
+ if fmt == "tsv":
256
+ return to_tsv(words)
257
+ if fmt == "hocr":
258
+ return to_hocr(
259
+ words,
260
+ image_width=kwargs.get("image_width", 0),
261
+ image_height=kwargs.get("image_height", 0),
262
+ source_name=kwargs.get("source_name", "image"),
263
+ )
264
+ raise ValueError(f"Unsupported export format: {fmt!r}")
@@ -0,0 +1,5 @@
1
+ """pdf - searchable PDF construction."""
2
+
3
+ from scanlayer.pdf.builder import build_searchable_pdf
4
+
5
+ __all__ = ["build_searchable_pdf"]