scanlayer 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scanlayer/__init__.py +32 -0
- scanlayer/__main__.py +10 -0
- scanlayer/cli/__init__.py +10 -0
- scanlayer/cli/dry_run.py +191 -0
- scanlayer/cli/parser.py +199 -0
- scanlayer/cli/run.py +217 -0
- scanlayer/config.py +267 -0
- scanlayer/fonts/DejaVuSans.ttf +0 -0
- scanlayer/layout/__init__.py +0 -0
- scanlayer/layout/columns.py +264 -0
- scanlayer/main.py +666 -0
- scanlayer/ocr/__init__.py +5 -0
- scanlayer/ocr/engine.py +335 -0
- scanlayer/ocr/export.py +264 -0
- scanlayer/pdf/__init__.py +5 -0
- scanlayer/pdf/builder.py +309 -0
- scanlayer/pdf/fonts.py +110 -0
- scanlayer/preprocessing/__init__.py +5 -0
- scanlayer/preprocessing/enhance.py +367 -0
- scanlayer/utils/__init__.py +19 -0
- scanlayer/utils/debug_image.py +85 -0
- scanlayer/utils/errors.py +42 -0
- scanlayer/utils/logger.py +100 -0
- scanlayer/utils/validators.py +199 -0
- scanlayer-1.0.0.dist-info/METADATA +17 -0
- scanlayer-1.0.0.dist-info/RECORD +30 -0
- scanlayer-1.0.0.dist-info/WHEEL +5 -0
- scanlayer-1.0.0.dist-info/entry_points.txt +2 -0
- scanlayer-1.0.0.dist-info/licenses/LICENSE.md +21 -0
- scanlayer-1.0.0.dist-info/top_level.txt +1 -0
scanlayer/main.py
ADDED
|
@@ -0,0 +1,666 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Library entry point for scanlayer.
|
|
3
|
+
|
|
4
|
+
For the CLI, see scanlayer.cli.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import glob
|
|
10
|
+
import os
|
|
11
|
+
import tempfile
|
|
12
|
+
import time
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from typing import Optional
|
|
15
|
+
|
|
16
|
+
from PIL import Image
|
|
17
|
+
|
|
18
|
+
from scanlayer import config
|
|
19
|
+
from scanlayer.layout.columns import reorder_reading_order
|
|
20
|
+
from scanlayer.ocr.engine import extract_words
|
|
21
|
+
from scanlayer.ocr.export import FORMATS, export_words
|
|
22
|
+
from scanlayer.pdf.builder import build_searchable_pdf, build_searchable_pdf_multipage
|
|
23
|
+
from scanlayer.preprocessing.enhance import preprocess
|
|
24
|
+
from scanlayer.utils.debug_image import build_debug_image
|
|
25
|
+
from scanlayer.utils.errors import BlankPageDetectedError
|
|
26
|
+
from scanlayer.utils.logger import get_logger, log_success, log_warning
|
|
27
|
+
from scanlayer.utils.validators import (
|
|
28
|
+
DependencyError,
|
|
29
|
+
InputFileError,
|
|
30
|
+
validate_all,
|
|
31
|
+
validate_image_readable,
|
|
32
|
+
validate_input_file,
|
|
33
|
+
validate_output_path,
|
|
34
|
+
validate_tesseract_environment,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class ConversionResult:
|
|
40
|
+
"""Summary of a successful conversion."""
|
|
41
|
+
output_path: str
|
|
42
|
+
words_count: int
|
|
43
|
+
mean_confidence: float
|
|
44
|
+
best_psm: Optional[int]
|
|
45
|
+
early_exited: bool
|
|
46
|
+
pdf_size_bytes: int
|
|
47
|
+
elapsed_ms: float
|
|
48
|
+
language_used: str
|
|
49
|
+
output_format: str = "pdf"
|
|
50
|
+
debug_image_path: Optional[str] = None
|
|
51
|
+
column_count: int = 1
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass
|
|
55
|
+
class BatchResult:
|
|
56
|
+
"""Summary of a convert_batch() run.
|
|
57
|
+
|
|
58
|
+
Check .ok / .failed / .failures after the call. Never raises for
|
|
59
|
+
per-file failures.
|
|
60
|
+
"""
|
|
61
|
+
results: list # list[ConversionResult]
|
|
62
|
+
failures: list # list[tuple[str, str]]
|
|
63
|
+
|
|
64
|
+
@property
|
|
65
|
+
def succeeded(self) -> int:
|
|
66
|
+
return len(self.results)
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def failed(self) -> int:
|
|
70
|
+
return len(self.failures)
|
|
71
|
+
|
|
72
|
+
@property
|
|
73
|
+
def ok(self) -> bool:
|
|
74
|
+
return not self.failures
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _write_export(
|
|
78
|
+
output_format: str,
|
|
79
|
+
words: list,
|
|
80
|
+
output_abs: str,
|
|
81
|
+
mean_confidence: float,
|
|
82
|
+
best_psm: Optional[int],
|
|
83
|
+
language_used: str,
|
|
84
|
+
image_width: int,
|
|
85
|
+
image_height: int,
|
|
86
|
+
source_name: str,
|
|
87
|
+
) -> int:
|
|
88
|
+
"""Serialize words into the requested non-PDF format. Returns file size in bytes."""
|
|
89
|
+
content = export_words(
|
|
90
|
+
output_format, words,
|
|
91
|
+
mean_confidence=mean_confidence, best_psm=best_psm,
|
|
92
|
+
language_used=language_used, image_width=image_width,
|
|
93
|
+
image_height=image_height, source_name=source_name,
|
|
94
|
+
)
|
|
95
|
+
with open(output_abs, "w", encoding="utf-8") as f:
|
|
96
|
+
f.write(content)
|
|
97
|
+
return os.path.getsize(output_abs)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _default_output_path(input_path: str, output_format: str) -> str:
|
|
101
|
+
"""Derive an output path next to the input file when -o/output_path
|
|
102
|
+
is omitted: same directory, same stem, extension from output_format.
|
|
103
|
+
"""
|
|
104
|
+
stem = os.path.splitext(os.path.basename(input_path))[0]
|
|
105
|
+
directory = os.path.dirname(os.path.abspath(input_path))
|
|
106
|
+
return os.path.join(directory, f"{stem}.{output_format}")
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _resolve_output_path(
|
|
110
|
+
output_arg: Optional[str], input_path: str, is_batch: bool, output_format: str = "pdf"
|
|
111
|
+
) -> str:
|
|
112
|
+
"""Resolve output path for single or batch mode.
|
|
113
|
+
|
|
114
|
+
output_arg=None -> write next to the input file (see
|
|
115
|
+
_default_output_path); no directory is created in that case since
|
|
116
|
+
the input's own directory already exists.
|
|
117
|
+
"""
|
|
118
|
+
if output_arg is None:
|
|
119
|
+
return _default_output_path(input_path, output_format)
|
|
120
|
+
if is_batch or output_arg.endswith(("/", "\\")) or os.path.isdir(output_arg):
|
|
121
|
+
os.makedirs(output_arg, exist_ok=True)
|
|
122
|
+
stem = os.path.splitext(os.path.basename(input_path))[0]
|
|
123
|
+
return os.path.join(output_arg, f"{stem}.{output_format}")
|
|
124
|
+
return output_arg
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _debug_image_path(output_abs: str) -> str:
|
|
128
|
+
stem, _ext = os.path.splitext(output_abs)
|
|
129
|
+
return f"{stem}_debug.png"
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _write_debug_image(
|
|
133
|
+
output_abs: str,
|
|
134
|
+
background,
|
|
135
|
+
words: list,
|
|
136
|
+
best_psm: Optional[int],
|
|
137
|
+
mean_confidence: float,
|
|
138
|
+
language_used: str,
|
|
139
|
+
result,
|
|
140
|
+
) -> str:
|
|
141
|
+
"""Builds and saves the debug overlay image next to `output_abs`.
|
|
142
|
+
Returns the path it was written to.
|
|
143
|
+
"""
|
|
144
|
+
rotation_note = ""
|
|
145
|
+
if result.orientation_correction_skipped:
|
|
146
|
+
rotation_note = "(orientation correction disabled)"
|
|
147
|
+
elif result.manual_rotation_applied is not None:
|
|
148
|
+
rotation_note = f"manual rotation={result.manual_rotation_applied:.2f} deg"
|
|
149
|
+
elif result.deskew_angle_applied:
|
|
150
|
+
rotation_note = f"deskew={result.deskew_angle_applied:.2f} deg"
|
|
151
|
+
|
|
152
|
+
debug_img = build_debug_image(
|
|
153
|
+
background, words,
|
|
154
|
+
best_psm=best_psm, mean_confidence=mean_confidence,
|
|
155
|
+
language_used=language_used,
|
|
156
|
+
exif_orientation=result.exif_orientation_applied,
|
|
157
|
+
gross_rotation=result.gross_rotation_applied,
|
|
158
|
+
rotation_note=rotation_note,
|
|
159
|
+
)
|
|
160
|
+
path = _debug_image_path(output_abs)
|
|
161
|
+
debug_img.save(path)
|
|
162
|
+
return path
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def convert(
|
|
166
|
+
input_path: str,
|
|
167
|
+
output_path: Optional[str] = None,
|
|
168
|
+
lang: Optional[str] = None,
|
|
169
|
+
dpi: Optional[int] = None,
|
|
170
|
+
jpeg_quality: Optional[int] = None,
|
|
171
|
+
char_whitelist: Optional[str] = None,
|
|
172
|
+
char_blacklist: Optional[str] = None,
|
|
173
|
+
pdf_metadata: Optional[dict] = None,
|
|
174
|
+
force: bool = False,
|
|
175
|
+
orientation: "str | float | None" = None,
|
|
176
|
+
output_format: str = "pdf",
|
|
177
|
+
debug_image: bool = False,
|
|
178
|
+
) -> ConversionResult:
|
|
179
|
+
"""Convert an image to a searchable PDF or export raw OCR results.
|
|
180
|
+
|
|
181
|
+
Raises ValidationError if prerequisites are not met.
|
|
182
|
+
Raises BlankPageDetectedError if the image is near-blank and force=False.
|
|
183
|
+
|
|
184
|
+
orientation: None=auto, "none"=disable, or float degrees (clockwise).
|
|
185
|
+
output_format: "pdf", "txt", "json", "tsv", or "hocr".
|
|
186
|
+
debug_image: if True, saves a _debug.png with word boxes overlay.
|
|
187
|
+
output_path: if omitted (None), defaults to the input file's own
|
|
188
|
+
directory/stem with the extension for output_format, e.g.
|
|
189
|
+
"invoice.jpg" -> "invoice.pdf". Lets debug_image=True be used
|
|
190
|
+
without having to name an output file.
|
|
191
|
+
"""
|
|
192
|
+
if output_format not in FORMATS:
|
|
193
|
+
raise ValueError(
|
|
194
|
+
f"Unsupported output_format: {output_format!r} "
|
|
195
|
+
f"(expected one of {FORMATS})"
|
|
196
|
+
)
|
|
197
|
+
if output_path is None:
|
|
198
|
+
output_path = _default_output_path(input_path, output_format)
|
|
199
|
+
log = get_logger("main")
|
|
200
|
+
t0 = time.perf_counter()
|
|
201
|
+
|
|
202
|
+
input_abs, output_abs = validate_all(input_path, output_path, output_format)
|
|
203
|
+
|
|
204
|
+
dpi = dpi or config.DEFAULT_DPI
|
|
205
|
+
with Image.open(input_abs) as img:
|
|
206
|
+
image = img.copy()
|
|
207
|
+
if "dpi" in img.info and img.info["dpi"][0] > 10:
|
|
208
|
+
detected_dpi = int(img.info["dpi"][0])
|
|
209
|
+
if dpi != detected_dpi:
|
|
210
|
+
log.info(
|
|
211
|
+
f"DPI detected in image ({detected_dpi}), used "
|
|
212
|
+
f"instead of default/config value ({dpi})."
|
|
213
|
+
)
|
|
214
|
+
dpi = detected_dpi
|
|
215
|
+
|
|
216
|
+
result = preprocess(image, dpi=dpi, orientation=orientation)
|
|
217
|
+
|
|
218
|
+
for w in result.preprocessing_warnings:
|
|
219
|
+
log_warning(log, w)
|
|
220
|
+
|
|
221
|
+
if result.likely_blank and not force:
|
|
222
|
+
raise BlankPageDetectedError(
|
|
223
|
+
f"'{input_path}': source image appears blank/near-uniform "
|
|
224
|
+
f"(grayscale std={result.source_std:.2f} < "
|
|
225
|
+
f"{config.DESKEW_MIN_STD}), no {output_format} output "
|
|
226
|
+
f"generated. Pass force=True (or --force on the CLI) to "
|
|
227
|
+
f"generate the output anyway (background-only PDF, or an "
|
|
228
|
+
f"empty result for the other formats)."
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
if result.likely_blank and force:
|
|
232
|
+
log_warning(
|
|
233
|
+
log,
|
|
234
|
+
f"'{input_path}': --force used on a likely-blank page "
|
|
235
|
+
f"(std={result.source_std:.2f}), generating {output_format} "
|
|
236
|
+
f"output with no text/words (OCR skipped)."
|
|
237
|
+
)
|
|
238
|
+
words_for_output: list = []
|
|
239
|
+
language_used = lang or config.DEFAULT_OCR_LANG
|
|
240
|
+
|
|
241
|
+
if output_format == "pdf":
|
|
242
|
+
output_size_bytes = build_searchable_pdf(
|
|
243
|
+
result.background,
|
|
244
|
+
words_for_output,
|
|
245
|
+
output_abs,
|
|
246
|
+
dpi=dpi,
|
|
247
|
+
jpeg_quality=jpeg_quality,
|
|
248
|
+
metadata=pdf_metadata,
|
|
249
|
+
lang=language_used,
|
|
250
|
+
)
|
|
251
|
+
else:
|
|
252
|
+
output_size_bytes = _write_export(
|
|
253
|
+
output_format, words_for_output, output_abs,
|
|
254
|
+
mean_confidence=0.0, best_psm=None,
|
|
255
|
+
language_used=language_used,
|
|
256
|
+
image_width=result.background.width,
|
|
257
|
+
image_height=result.background.height,
|
|
258
|
+
source_name=os.path.basename(input_path),
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
debug_image_path = None
|
|
262
|
+
if debug_image:
|
|
263
|
+
debug_image_path = _write_debug_image(
|
|
264
|
+
output_abs, result.background, words_for_output,
|
|
265
|
+
best_psm=None, mean_confidence=0.0,
|
|
266
|
+
language_used=language_used, result=result,
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
elapsed_ms = (time.perf_counter() - t0) * 1000
|
|
270
|
+
log.info(
|
|
271
|
+
f"{output_format} output generated: {output_abs} "
|
|
272
|
+
f"({output_size_bytes / 1024:.1f} KB, {elapsed_ms:.0f} ms)"
|
|
273
|
+
)
|
|
274
|
+
return ConversionResult(
|
|
275
|
+
output_path=output_abs,
|
|
276
|
+
words_count=0,
|
|
277
|
+
mean_confidence=0.0,
|
|
278
|
+
best_psm=None,
|
|
279
|
+
early_exited=False,
|
|
280
|
+
pdf_size_bytes=output_size_bytes,
|
|
281
|
+
elapsed_ms=elapsed_ms,
|
|
282
|
+
language_used=language_used,
|
|
283
|
+
output_format=output_format,
|
|
284
|
+
debug_image_path=debug_image_path,
|
|
285
|
+
)
|
|
286
|
+
|
|
287
|
+
ocr_result = extract_words(
|
|
288
|
+
result.ocr_image,
|
|
289
|
+
result.ocr_scale,
|
|
290
|
+
effective_dpi=result.effective_dpi,
|
|
291
|
+
lang=lang,
|
|
292
|
+
char_whitelist=char_whitelist,
|
|
293
|
+
char_blacklist=char_blacklist,
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
ocr_result.words, column_count = reorder_reading_order(
|
|
297
|
+
ocr_result.words, page_width=result.background.width,
|
|
298
|
+
)
|
|
299
|
+
|
|
300
|
+
mean_conf = (
|
|
301
|
+
sum(w.confidence for w in ocr_result.words) / len(ocr_result.words)
|
|
302
|
+
if ocr_result.words else 0.0
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
if output_format == "pdf":
|
|
306
|
+
output_size_bytes = build_searchable_pdf(
|
|
307
|
+
result.background,
|
|
308
|
+
ocr_result.words,
|
|
309
|
+
output_abs,
|
|
310
|
+
dpi=dpi,
|
|
311
|
+
jpeg_quality=jpeg_quality,
|
|
312
|
+
metadata=pdf_metadata,
|
|
313
|
+
lang=ocr_result.language_used,
|
|
314
|
+
)
|
|
315
|
+
else:
|
|
316
|
+
output_size_bytes = _write_export(
|
|
317
|
+
output_format, ocr_result.words, output_abs,
|
|
318
|
+
mean_confidence=mean_conf, best_psm=ocr_result.best_psm,
|
|
319
|
+
language_used=ocr_result.language_used,
|
|
320
|
+
image_width=result.background.width,
|
|
321
|
+
image_height=result.background.height,
|
|
322
|
+
source_name=os.path.basename(input_path),
|
|
323
|
+
)
|
|
324
|
+
|
|
325
|
+
debug_image_path = None
|
|
326
|
+
if debug_image:
|
|
327
|
+
debug_image_path = _write_debug_image(
|
|
328
|
+
output_abs, result.background, ocr_result.words,
|
|
329
|
+
best_psm=ocr_result.best_psm, mean_confidence=mean_conf,
|
|
330
|
+
language_used=ocr_result.language_used, result=result,
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
elapsed_ms = (time.perf_counter() - t0) * 1000
|
|
334
|
+
|
|
335
|
+
if ocr_result.words:
|
|
336
|
+
log_success(
|
|
337
|
+
log,
|
|
338
|
+
f"{len(ocr_result.words)} words detected "
|
|
339
|
+
f"(mean confidence {mean_conf:.1f}%, "
|
|
340
|
+
f"PSM {ocr_result.best_psm}"
|
|
341
|
+
+ (" [early-exit]" if ocr_result.early_exited else "")
|
|
342
|
+
+ (f", {column_count} columns" if column_count > 1 else "")
|
|
343
|
+
+ ")"
|
|
344
|
+
)
|
|
345
|
+
else:
|
|
346
|
+
log_warning(log, "No words detected (blank or unreadable page)")
|
|
347
|
+
log.info(
|
|
348
|
+
f"{output_format} output generated: {output_abs} "
|
|
349
|
+
f"({output_size_bytes / 1024:.1f} KB, {elapsed_ms:.0f} ms)"
|
|
350
|
+
)
|
|
351
|
+
|
|
352
|
+
return ConversionResult(
|
|
353
|
+
output_path=output_abs,
|
|
354
|
+
words_count=len(ocr_result.words),
|
|
355
|
+
mean_confidence=mean_conf,
|
|
356
|
+
best_psm=ocr_result.best_psm,
|
|
357
|
+
early_exited=ocr_result.early_exited,
|
|
358
|
+
pdf_size_bytes=output_size_bytes,
|
|
359
|
+
elapsed_ms=elapsed_ms,
|
|
360
|
+
language_used=ocr_result.language_used,
|
|
361
|
+
output_format=output_format,
|
|
362
|
+
debug_image_path=debug_image_path,
|
|
363
|
+
column_count=column_count,
|
|
364
|
+
)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _expand_pdf_input(path: str, dpi: Optional[int], tmp_dir: str) -> list[str]:
|
|
368
|
+
"""Rasterize PDF pages to PNGs. Non-PDF inputs pass through unchanged.
|
|
369
|
+
|
|
370
|
+
Requires poppler on PATH. Raises DependencyError if missing.
|
|
371
|
+
"""
|
|
372
|
+
if os.path.splitext(path)[1].lower() != ".pdf":
|
|
373
|
+
return [path]
|
|
374
|
+
|
|
375
|
+
log = get_logger("main")
|
|
376
|
+
try:
|
|
377
|
+
from pdf2image import convert_from_path
|
|
378
|
+
from pdf2image.exceptions import PDFInfoNotInstalledError
|
|
379
|
+
except ImportError as exc:
|
|
380
|
+
raise DependencyError(
|
|
381
|
+
"PDF input requires the 'pdf2image' package: "
|
|
382
|
+
"pip install pdf2image"
|
|
383
|
+
) from exc
|
|
384
|
+
|
|
385
|
+
raster_dpi = dpi or config.DEFAULT_DPI
|
|
386
|
+
try:
|
|
387
|
+
page_images = convert_from_path(path, dpi=raster_dpi)
|
|
388
|
+
except PDFInfoNotInstalledError as exc:
|
|
389
|
+
raise DependencyError(
|
|
390
|
+
"PDF input requires poppler-utils (the 'pdftoppm'/'pdfinfo' "
|
|
391
|
+
"binaries) on PATH. 'pip install pdf2image' alone is not "
|
|
392
|
+
"enough, poppler is a separate system package "
|
|
393
|
+
"(apt install poppler-utils / brew install poppler / "
|
|
394
|
+
"download poppler for Windows and add it to PATH)."
|
|
395
|
+
) from exc
|
|
396
|
+
except Exception as exc:
|
|
397
|
+
raise InputFileError(
|
|
398
|
+
f"'{path}': could not read as a PDF ({type(exc).__name__}: {exc})"
|
|
399
|
+
) from exc
|
|
400
|
+
|
|
401
|
+
if not page_images:
|
|
402
|
+
raise InputFileError(f"'{path}': PDF has 0 pages, nothing to convert.")
|
|
403
|
+
|
|
404
|
+
stem = os.path.splitext(os.path.basename(path))[0]
|
|
405
|
+
out_paths = []
|
|
406
|
+
for i, page_img in enumerate(page_images, start=1):
|
|
407
|
+
page_path = os.path.join(tmp_dir, f"{stem}_p{i}.png")
|
|
408
|
+
page_img.save(page_path, dpi=(raster_dpi, raster_dpi))
|
|
409
|
+
out_paths.append(page_path)
|
|
410
|
+
log.info(
|
|
411
|
+
f"'{path}': rasterized {len(out_paths)} page(s) from PDF input "
|
|
412
|
+
f"at {raster_dpi} DPI."
|
|
413
|
+
)
|
|
414
|
+
return out_paths
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def _process_page_for_merge(
|
|
418
|
+
input_path: str,
|
|
419
|
+
lang: Optional[str],
|
|
420
|
+
dpi: Optional[int],
|
|
421
|
+
char_whitelist: Optional[str],
|
|
422
|
+
char_blacklist: Optional[str],
|
|
423
|
+
orientation: "str | float | None",
|
|
424
|
+
force: bool,
|
|
425
|
+
):
|
|
426
|
+
"""Run preprocessing + OCR for one page of a --merge run.
|
|
427
|
+
|
|
428
|
+
Returns (PageInput, stats_dict). Does not touch the output path.
|
|
429
|
+
"""
|
|
430
|
+
from scanlayer.pdf.builder import PageInput
|
|
431
|
+
|
|
432
|
+
log = get_logger("main")
|
|
433
|
+
input_abs = validate_input_file(input_path)
|
|
434
|
+
validate_image_readable(input_abs)
|
|
435
|
+
validate_tesseract_environment()
|
|
436
|
+
|
|
437
|
+
page_dpi = dpi or config.DEFAULT_DPI
|
|
438
|
+
with Image.open(input_abs) as img:
|
|
439
|
+
image = img.copy()
|
|
440
|
+
if "dpi" in img.info and img.info["dpi"][0] > 10:
|
|
441
|
+
page_dpi = int(img.info["dpi"][0])
|
|
442
|
+
|
|
443
|
+
result = preprocess(image, dpi=page_dpi, orientation=orientation)
|
|
444
|
+
for w in result.preprocessing_warnings:
|
|
445
|
+
log_warning(log, w)
|
|
446
|
+
|
|
447
|
+
if result.likely_blank and not force:
|
|
448
|
+
raise BlankPageDetectedError(
|
|
449
|
+
f"'{input_path}': source image appears blank/near-uniform "
|
|
450
|
+
f"(grayscale std={result.source_std:.2f} < "
|
|
451
|
+
f"{config.DESKEW_MIN_STD}), pass force=True (or --force) to "
|
|
452
|
+
f"include it as a blank page in the merged PDF."
|
|
453
|
+
)
|
|
454
|
+
|
|
455
|
+
if result.likely_blank and force:
|
|
456
|
+
log_warning(
|
|
457
|
+
log,
|
|
458
|
+
f"'{input_path}': --force used on a likely-blank page, "
|
|
459
|
+
f"included in the merged PDF with no text layer.",
|
|
460
|
+
)
|
|
461
|
+
page = PageInput(
|
|
462
|
+
background=result.background, words=[], dpi=page_dpi,
|
|
463
|
+
lang=lang or config.DEFAULT_OCR_LANG,
|
|
464
|
+
)
|
|
465
|
+
return page, {"words": 0, "mean_conf": 0.0, "best_psm": None, "columns": 1}
|
|
466
|
+
|
|
467
|
+
ocr_result = extract_words(
|
|
468
|
+
result.ocr_image, result.ocr_scale, effective_dpi=result.effective_dpi,
|
|
469
|
+
lang=lang, char_whitelist=char_whitelist, char_blacklist=char_blacklist,
|
|
470
|
+
)
|
|
471
|
+
ocr_result.words, column_count = reorder_reading_order(
|
|
472
|
+
ocr_result.words, page_width=result.background.width,
|
|
473
|
+
)
|
|
474
|
+
mean_conf = (
|
|
475
|
+
sum(w.confidence for w in ocr_result.words) / len(ocr_result.words)
|
|
476
|
+
if ocr_result.words else 0.0
|
|
477
|
+
)
|
|
478
|
+
|
|
479
|
+
page = PageInput(
|
|
480
|
+
background=result.background, words=ocr_result.words,
|
|
481
|
+
dpi=page_dpi, lang=ocr_result.language_used,
|
|
482
|
+
)
|
|
483
|
+
stats = {
|
|
484
|
+
"words": len(ocr_result.words), "mean_conf": mean_conf,
|
|
485
|
+
"best_psm": ocr_result.best_psm, "columns": column_count,
|
|
486
|
+
}
|
|
487
|
+
return page, stats
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
def convert_merge(
|
|
491
|
+
input_paths: list[str],
|
|
492
|
+
output_path: str,
|
|
493
|
+
lang: Optional[str] = None,
|
|
494
|
+
dpi: Optional[int] = None,
|
|
495
|
+
jpeg_quality: Optional[int] = None,
|
|
496
|
+
char_whitelist: Optional[str] = None,
|
|
497
|
+
char_blacklist: Optional[str] = None,
|
|
498
|
+
pdf_metadata: Optional[dict] = None,
|
|
499
|
+
force: bool = False,
|
|
500
|
+
orientation: "str | float | None" = None,
|
|
501
|
+
) -> ConversionResult:
|
|
502
|
+
"""Combine several images into one multi-page searchable PDF.
|
|
503
|
+
|
|
504
|
+
PDF-only. Raises BlankPageDetectedError if any page is blank and
|
|
505
|
+
force=False.
|
|
506
|
+
"""
|
|
507
|
+
if not input_paths:
|
|
508
|
+
raise ValueError("convert_merge() called with no input paths.")
|
|
509
|
+
|
|
510
|
+
|
|
511
|
+
log = get_logger("main")
|
|
512
|
+
t0 = time.perf_counter()
|
|
513
|
+
|
|
514
|
+
output_abs = validate_output_path(output_path, expected_ext=".pdf")
|
|
515
|
+
|
|
516
|
+
pages = []
|
|
517
|
+
total_words = 0
|
|
518
|
+
conf_weighted_sum = 0.0
|
|
519
|
+
conf_weight_n = 0
|
|
520
|
+
max_columns = 1
|
|
521
|
+
for input_path in input_paths:
|
|
522
|
+
page, stats = _process_page_for_merge(
|
|
523
|
+
input_path, lang, dpi, char_whitelist, char_blacklist, orientation, force,
|
|
524
|
+
)
|
|
525
|
+
pages.append(page)
|
|
526
|
+
total_words += stats["words"]
|
|
527
|
+
if stats["words"]:
|
|
528
|
+
conf_weighted_sum += stats["mean_conf"] * stats["words"]
|
|
529
|
+
conf_weight_n += stats["words"]
|
|
530
|
+
max_columns = max(max_columns, stats["columns"])
|
|
531
|
+
log.info(
|
|
532
|
+
f"'{input_path}': {stats['words']} words "
|
|
533
|
+
f"(mean confidence {stats['mean_conf']:.1f}%"
|
|
534
|
+
+ (f", {stats['columns']} columns" if stats["columns"] > 1 else "")
|
|
535
|
+
+ ")"
|
|
536
|
+
)
|
|
537
|
+
|
|
538
|
+
output_size_bytes = build_searchable_pdf_multipage(
|
|
539
|
+
pages, output_abs, jpeg_quality=jpeg_quality, metadata=pdf_metadata,
|
|
540
|
+
)
|
|
541
|
+
|
|
542
|
+
elapsed_ms = (time.perf_counter() - t0) * 1000
|
|
543
|
+
mean_conf = conf_weighted_sum / conf_weight_n if conf_weight_n else 0.0
|
|
544
|
+
log_success(
|
|
545
|
+
log,
|
|
546
|
+
f"Merged {len(pages)} page(s) into {output_abs} "
|
|
547
|
+
f"({total_words} words total, mean confidence {mean_conf:.1f}%)",
|
|
548
|
+
)
|
|
549
|
+
|
|
550
|
+
return ConversionResult(
|
|
551
|
+
output_path=output_abs,
|
|
552
|
+
words_count=total_words,
|
|
553
|
+
mean_confidence=mean_conf,
|
|
554
|
+
best_psm=None, # not a single meaningful value across merged pages
|
|
555
|
+
early_exited=False,
|
|
556
|
+
pdf_size_bytes=output_size_bytes,
|
|
557
|
+
elapsed_ms=elapsed_ms,
|
|
558
|
+
language_used=lang or config.DEFAULT_OCR_LANG,
|
|
559
|
+
output_format="pdf",
|
|
560
|
+
debug_image_path=None,
|
|
561
|
+
column_count=max_columns,
|
|
562
|
+
)
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
def convert_batch(
|
|
566
|
+
inputs: "str | list[str]",
|
|
567
|
+
output: Optional[str] = None,
|
|
568
|
+
*,
|
|
569
|
+
merge: bool = False,
|
|
570
|
+
lang: Optional[str] = None,
|
|
571
|
+
dpi: Optional[int] = None,
|
|
572
|
+
jpeg_quality: Optional[int] = None,
|
|
573
|
+
char_whitelist: Optional[str] = None,
|
|
574
|
+
char_blacklist: Optional[str] = None,
|
|
575
|
+
pdf_metadata: Optional[dict] = None,
|
|
576
|
+
force: bool = False,
|
|
577
|
+
orientation: "str | float | None" = None,
|
|
578
|
+
output_format: str = "pdf",
|
|
579
|
+
debug_image: bool = False,
|
|
580
|
+
verbose: bool = False,
|
|
581
|
+
quiet: bool = False,
|
|
582
|
+
) -> BatchResult:
|
|
583
|
+
"""Batch-convert multiple images (or mixed images + PDFs).
|
|
584
|
+
|
|
585
|
+
inputs: single path/pattern or list. Glob-expanded, PDFs rasterized.
|
|
586
|
+
output: folder (one file per input) or file path (with merge=True).
|
|
587
|
+
If omitted (None), each file is written next to its own input
|
|
588
|
+
(same directory/stem, extension from output_format). Not valid
|
|
589
|
+
with merge=True, which needs one explicit shared output path.
|
|
590
|
+
merge: combine all inputs into one multi-page PDF.
|
|
591
|
+
verbose/quiet: set log level for the call.
|
|
592
|
+
|
|
593
|
+
Never raises for per-file failures. Check BatchResult.failures.
|
|
594
|
+
"""
|
|
595
|
+
if verbose and quiet:
|
|
596
|
+
raise ValueError("convert_batch(): verbose and quiet are mutually exclusive.")
|
|
597
|
+
if merge and output_format != "pdf":
|
|
598
|
+
raise ValueError("convert_batch(): merge=True only supports output_format='pdf'.")
|
|
599
|
+
if merge and output is None:
|
|
600
|
+
raise ValueError(
|
|
601
|
+
"convert_batch(): merge=True requires an explicit output "
|
|
602
|
+
"file path (there's no single input to derive a shared "
|
|
603
|
+
"output name from)."
|
|
604
|
+
)
|
|
605
|
+
if verbose:
|
|
606
|
+
config.configure(log_level="DEBUG")
|
|
607
|
+
elif quiet:
|
|
608
|
+
config.configure(log_level="ERROR")
|
|
609
|
+
|
|
610
|
+
if isinstance(inputs, str):
|
|
611
|
+
inputs = [inputs]
|
|
612
|
+
|
|
613
|
+
raw_inputs: list[str] = []
|
|
614
|
+
for pat in inputs:
|
|
615
|
+
expanded = glob.glob(pat)
|
|
616
|
+
raw_inputs.extend(expanded if expanded else [pat])
|
|
617
|
+
|
|
618
|
+
results: list = []
|
|
619
|
+
failures: list = []
|
|
620
|
+
|
|
621
|
+
with tempfile.TemporaryDirectory(prefix="scanlayer_pdf_") as tmp_dir:
|
|
622
|
+
expanded_inputs: list[str] = []
|
|
623
|
+
for input_path in raw_inputs:
|
|
624
|
+
try:
|
|
625
|
+
expanded_inputs.extend(_expand_pdf_input(input_path, dpi, tmp_dir))
|
|
626
|
+
except Exception as exc:
|
|
627
|
+
failures.append((input_path, str(exc)))
|
|
628
|
+
|
|
629
|
+
if not expanded_inputs:
|
|
630
|
+
return BatchResult(results=results, failures=failures)
|
|
631
|
+
|
|
632
|
+
if merge:
|
|
633
|
+
try:
|
|
634
|
+
results.append(convert_merge(
|
|
635
|
+
input_paths=expanded_inputs, output_path=output,
|
|
636
|
+
lang=lang, dpi=dpi, jpeg_quality=jpeg_quality,
|
|
637
|
+
char_whitelist=char_whitelist, char_blacklist=char_blacklist,
|
|
638
|
+
pdf_metadata=pdf_metadata, force=force, orientation=orientation,
|
|
639
|
+
))
|
|
640
|
+
except Exception as exc:
|
|
641
|
+
failures.append(("<merge>", str(exc)))
|
|
642
|
+
return BatchResult(results=results, failures=failures)
|
|
643
|
+
|
|
644
|
+
is_batch = len(expanded_inputs) > 1
|
|
645
|
+
for input_path in expanded_inputs:
|
|
646
|
+
if output is None:
|
|
647
|
+
output_path = None # convert() derives it next to input_path
|
|
648
|
+
else:
|
|
649
|
+
try:
|
|
650
|
+
output_path = _resolve_output_path(output, input_path, is_batch, output_format)
|
|
651
|
+
except OSError as exc:
|
|
652
|
+
failures.append((input_path, str(exc)))
|
|
653
|
+
continue
|
|
654
|
+
try:
|
|
655
|
+
results.append(convert(
|
|
656
|
+
input_path=input_path, output_path=output_path,
|
|
657
|
+
lang=lang, dpi=dpi, jpeg_quality=jpeg_quality,
|
|
658
|
+
char_whitelist=char_whitelist, char_blacklist=char_blacklist,
|
|
659
|
+
pdf_metadata=pdf_metadata, force=force, orientation=orientation,
|
|
660
|
+
output_format=output_format, debug_image=debug_image,
|
|
661
|
+
))
|
|
662
|
+
except Exception as exc:
|
|
663
|
+
failures.append((input_path, str(exc)))
|
|
664
|
+
|
|
665
|
+
return BatchResult(results=results, failures=failures)
|
|
666
|
+
|