scanlayer 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,309 @@
1
+ """
2
+ Final PDF construction.
3
+
4
+ Page sized to background image, background drawn with adaptive compression
5
+ (JPEG or PNG), invisible text layer overlaid word by word.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import io
11
+ import os
12
+ from dataclasses import dataclass
13
+ from typing import Optional
14
+
15
+ import numpy as np
16
+ from PIL import Image
17
+ from reportlab.lib.utils import ImageReader
18
+ from reportlab.pdfgen import canvas
19
+
20
+ from scanlayer import config
21
+ from scanlayer.ocr.engine import Word
22
+ from scanlayer.pdf.fonts import resolve_font
23
+ from scanlayer.utils.errors import PDFBuildError
24
+ from scanlayer.utils.logger import get_logger, log_warning, stage_timer
25
+
26
+ log = get_logger(__name__)
27
+
28
+ POINTS_PER_INCH = 72.0
29
+ INVISIBLE_RENDER_MODE = 3
30
+
31
+ _MIN_FONT_SIZE_PT = 1.0
32
+ _MAX_FONT_SIZE_PT = 400.0
33
+ _MIN_HORIZ_SCALE_PCT = 1.0
34
+ _MAX_HORIZ_SCALE_PCT = 1000.0
35
+
36
+
37
+ def _is_mostly_grayscale(image: Image.Image, threshold: float = None) -> bool:
38
+ """Return True if the image is visually near-grayscale (low chrominance)."""
39
+ if image.mode == "L":
40
+ return True
41
+ if image.mode != "RGB":
42
+ image = image.convert("RGB")
43
+ arr = np.asarray(image)
44
+ # Approximate saturation: max - min across RGB channels.
45
+ sat = arr.max(axis=2).astype(np.int16) - arr.min(axis=2).astype(np.int16)
46
+ threshold = threshold if threshold is not None else config.PDF_GRAYSCALE_THRESHOLD
47
+ fraction_colored = float((sat > 25).mean()) # sat > ~10% = colored pixel
48
+ log.debug(
49
+ f"Compression detector: colored pixel fraction = {fraction_colored:.4f} "
50
+ f"(grayscale threshold = {threshold})"
51
+ )
52
+ return fraction_colored < threshold
53
+
54
+
55
+ def _is_mostly_binary(image: Image.Image) -> bool:
56
+ """Detect if an image is near-binary (black text on white background).
57
+
58
+ For near-binary content, PNG is lighter and lossless vs JPEG.
59
+ """
60
+ if image.mode != "L":
61
+ image = image.convert("L")
62
+ arr = np.asarray(image, dtype=np.float32)
63
+ hist, _ = np.histogram(arr, bins=8, range=(0, 256))
64
+ total = arr.size
65
+ if total == 0:
66
+ return False
67
+ extreme_fraction = (hist[0] + hist[-1]) / total
68
+ is_binary = extreme_fraction > 0.80
69
+ log.debug(
70
+ f"Compression detector: extreme pixel fraction = {extreme_fraction:.4f} "
71
+ f"-> binary={'yes' if is_binary else 'no'}"
72
+ )
73
+ return is_binary
74
+
75
+
76
+ def _as_jpeg_reader(image: Image.Image, quality: int) -> ImageReader:
77
+ """Encodes the image as JPEG in a memory buffer and returns an
78
+ ImageReader for ReportLab."""
79
+ buffer = io.BytesIO()
80
+ image.convert("RGB").save(
81
+ buffer, format="JPEG", quality=quality, optimize=True,
82
+ )
83
+ buffer.seek(0)
84
+ return ImageReader(buffer)
85
+
86
+
87
+ def _as_png_reader(image: Image.Image) -> ImageReader:
88
+ """Encode image as PNG (lossless) in a memory buffer.
89
+
90
+ Converts to L mode if near-grayscale to save ~60% of size.
91
+ """
92
+ if _is_mostly_grayscale(image):
93
+ image = image.convert("L")
94
+ buffer = io.BytesIO()
95
+ image.save(buffer, format="PNG", optimize=True)
96
+ buffer.seek(0)
97
+ return ImageReader(buffer)
98
+
99
+
100
+ def _select_background_reader(
101
+ image: Image.Image, jpeg_quality: int
102
+ ) -> ImageReader:
103
+ """Chooses the best compression strategy for the PDF background.
104
+
105
+ 1. If PDF_ADAPTIVE_COMPRESSION and image is near-binary -> PNG (L mode if grayscale).
106
+ 2. Otherwise -> JPEG at the configured quality.
107
+ """
108
+ if config.PDF_ADAPTIVE_COMPRESSION and _is_mostly_binary(image):
109
+ log.info("Background compression: PNG (near-binary image detected)")
110
+ return _as_png_reader(image)
111
+ log.info(f"Background compression: JPEG q={jpeg_quality}")
112
+ return _as_jpeg_reader(image, jpeg_quality)
113
+
114
+
115
+ def _font_size_for_height(height_px: float, px_to_pt: float) -> float:
116
+ """Calculate font size so a character vertically occupies height_px pixels."""
117
+ size = height_px * px_to_pt * 0.85
118
+ return min(max(size, _MIN_FONT_SIZE_PT), _MAX_FONT_SIZE_PT)
119
+
120
+
121
+ def _draw_word(
122
+ c: canvas.Canvas,
123
+ word: Word,
124
+ px_to_pt: float,
125
+ page_height_pt: float,
126
+ font_name: str,
127
+ ) -> None:
128
+ """Draw an invisible word on the PDF canvas with horizontal stretching."""
129
+ if not word.width or not word.height:
130
+ return
131
+
132
+ font_size = _font_size_for_height(word.height, px_to_pt)
133
+ natural_width = c.stringWidth(word.text, font_name, font_size)
134
+ target_width_pt = word.width * px_to_pt
135
+
136
+ text_obj = c.beginText()
137
+ text_obj.setTextRenderMode(INVISIBLE_RENDER_MODE)
138
+ text_obj.setFont(font_name, font_size)
139
+
140
+ pdf_x = word.x * px_to_pt
141
+ pdf_y = page_height_pt - (word.y + word.height) * px_to_pt
142
+ text_obj.setTextOrigin(pdf_x, pdf_y)
143
+
144
+ if natural_width > 0 and target_width_pt > 0:
145
+ h_scale = 100.0 * target_width_pt / natural_width
146
+ h_scale = min(max(h_scale, _MIN_HORIZ_SCALE_PCT), _MAX_HORIZ_SCALE_PCT)
147
+ text_obj.setHorizScale(h_scale)
148
+
149
+ text_obj.textOut(word.text)
150
+ c.drawText(text_obj)
151
+
152
+
153
+ def _validate_page_size(page_width_pt: float, page_height_pt: float, page_num: int = 1) -> None:
154
+ """Validate PDF page size (1-14400pt per spec)."""
155
+ if page_width_pt < 1 or page_height_pt < 1:
156
+ raise PDFBuildError(
157
+ f"Invalid PDF page {page_num}: {page_width_pt}x{page_height_pt}pt"
158
+ )
159
+ if page_width_pt > 14400 or page_height_pt > 14400:
160
+ log_warning(
161
+ log,
162
+ f"PDF page {page_num} > 14400pt (PDF spec limit): "
163
+ f"{page_width_pt}x{page_height_pt}pt, may be rejected by "
164
+ f"some viewers."
165
+ )
166
+
167
+
168
+ def _draw_page(
169
+ c: canvas.Canvas,
170
+ background: Image.Image,
171
+ words: list[Word],
172
+ dpi: int,
173
+ jpeg_quality: int,
174
+ lang: Optional[str],
175
+ page_num: int = 1,
176
+ ) -> None:
177
+ """Draw one full page (background + text layer) onto an open canvas."""
178
+ font_name = resolve_font(lang or config.DEFAULT_OCR_LANG)
179
+ px_to_pt = POINTS_PER_INCH / dpi
180
+ page_width_pt = background.width * px_to_pt
181
+ page_height_pt = background.height * px_to_pt
182
+ _validate_page_size(page_width_pt, page_height_pt, page_num)
183
+
184
+ c.setPageSize((page_width_pt, page_height_pt))
185
+
186
+ bg_reader = _select_background_reader(background, jpeg_quality)
187
+ c.drawImage(
188
+ bg_reader, 0, 0,
189
+ width=page_width_pt, height=page_height_pt,
190
+ preserveAspectRatio=False,
191
+ )
192
+
193
+ skipped = 0
194
+ for word in words:
195
+ try:
196
+ _draw_word(c, word, px_to_pt, page_height_pt, font_name)
197
+ except Exception as exc:
198
+ # A single failing word must not break the entire PDF.
199
+ skipped += 1
200
+ log_warning(
201
+ log,
202
+ f"Page {page_num}: word skipped in text layer: "
203
+ f"{word.text!r} ({type(exc).__name__}: {exc})"
204
+ )
205
+
206
+ c.showPage()
207
+ log.info(
208
+ f"Page {page_num}: {len(words)} words in text layer"
209
+ + (f", {skipped} skipped" if skipped else "")
210
+ )
211
+
212
+
213
+ def build_searchable_pdf(
214
+ background: Image.Image,
215
+ words: list[Word],
216
+ output_path: str,
217
+ dpi: int,
218
+ jpeg_quality: int = None,
219
+ metadata: Optional[dict] = None,
220
+ lang: Optional[str] = None,
221
+ ) -> int:
222
+ """Build a single-page searchable PDF. Returns size in bytes."""
223
+ jpeg_quality = jpeg_quality if jpeg_quality is not None else config.PDF_JPEG_QUALITY
224
+ metadata = {**config.PDF_METADATA, **(metadata or {})}
225
+
226
+ try:
227
+ with stage_timer(log, "PDF build"):
228
+ c = canvas.Canvas(output_path)
229
+ c.setTitle(metadata.get("title", ""))
230
+ c.setAuthor(metadata.get("author", ""))
231
+ c.setSubject(metadata.get("subject", ""))
232
+ c.setCreator(metadata.get("creator", ""))
233
+ if "keywords" in metadata and metadata["keywords"]:
234
+ c.setKeywords(metadata["keywords"])
235
+
236
+ _draw_page(c, background, words, dpi, jpeg_quality, lang)
237
+ c.save()
238
+ except PDFBuildError:
239
+ raise
240
+ except OSError as exc:
241
+ raise PDFBuildError(
242
+ f"Could not write PDF to {output_path}: {exc}"
243
+ ) from exc
244
+ except Exception as exc:
245
+ raise PDFBuildError(
246
+ f"Unexpected failure while building the PDF: {type(exc).__name__}: {exc}"
247
+ ) from exc
248
+
249
+ size_bytes = os.path.getsize(output_path)
250
+ log.info(f"PDF generated: {size_bytes / 1024:.1f} KB, 1 page")
251
+ return size_bytes
252
+
253
+
254
+ @dataclass
255
+ class PageInput:
256
+ """One page's worth of content for `build_searchable_pdf_multipage`."""
257
+ background: Image.Image
258
+ words: list[Word]
259
+ dpi: int
260
+ lang: Optional[str] = None
261
+
262
+
263
+ def build_searchable_pdf_multipage(
264
+ pages: list[PageInput],
265
+ output_path: str,
266
+ jpeg_quality: int = None,
267
+ metadata: Optional[dict] = None,
268
+ ) -> int:
269
+ """Build a multi-page searchable PDF from several OCR'd pages.
270
+
271
+ Returns size in bytes. Raises PDFBuildError if pages is empty.
272
+ """
273
+ if not pages:
274
+ raise PDFBuildError("build_searchable_pdf_multipage() called with 0 pages.")
275
+
276
+ jpeg_quality = jpeg_quality if jpeg_quality is not None else config.PDF_JPEG_QUALITY
277
+ metadata = {**config.PDF_METADATA, **(metadata or {})}
278
+
279
+ try:
280
+ with stage_timer(log, f"PDF build ({len(pages)} pages)"):
281
+ c = canvas.Canvas(output_path)
282
+ c.setTitle(metadata.get("title", ""))
283
+ c.setAuthor(metadata.get("author", ""))
284
+ c.setSubject(metadata.get("subject", ""))
285
+ c.setCreator(metadata.get("creator", ""))
286
+ if "keywords" in metadata and metadata["keywords"]:
287
+ c.setKeywords(metadata["keywords"])
288
+
289
+ for i, page in enumerate(pages, start=1):
290
+ _draw_page(
291
+ c, page.background, page.words, page.dpi,
292
+ jpeg_quality, page.lang, page_num=i,
293
+ )
294
+ c.save()
295
+ except PDFBuildError:
296
+ raise
297
+ except OSError as exc:
298
+ raise PDFBuildError(
299
+ f"Could not write PDF to {output_path}: {exc}"
300
+ ) from exc
301
+ except Exception as exc:
302
+ raise PDFBuildError(
303
+ f"Unexpected failure while building the multi-page PDF: "
304
+ f"{type(exc).__name__}: {exc}"
305
+ ) from exc
306
+
307
+ size_bytes = os.path.getsize(output_path)
308
+ log.info(f"PDF generated: {size_bytes / 1024:.1f} KB, {len(pages)} pages")
309
+ return size_bytes
scanlayer/pdf/fonts.py ADDED
@@ -0,0 +1,110 @@
1
+ """
2
+ Font selection for the invisible PDF text layer.
3
+
4
+ Picks a font per document based on OCR lang string:
5
+ 1. CJK: reportlab's built-in CID fonts.
6
+ 2. Latin/Cyrillic/Greek/Vietnamese: bundled DejaVu Sans TTF.
7
+ 3. Other scripts: falls back to Helvetica.
8
+
9
+ Set config.FONT_PATH to force a specific TTF and skip auto-selection.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import os
15
+
16
+ from reportlab.pdfbase import pdfmetrics
17
+ from reportlab.pdfbase.cidfonts import UnicodeCIDFont
18
+ from reportlab.pdfbase.ttfonts import TTFont
19
+
20
+ from scanlayer import config
21
+ from scanlayer.utils.logger import get_logger, log_warning
22
+
23
+ log = get_logger(__name__)
24
+
25
+ FALLBACK_FONT = "Helvetica"
26
+
27
+ _UNICODE_TTF_NAME = "ScanlayerUnicode"
28
+ _UNICODE_TTF_PATH = os.path.join(os.path.dirname(__file__), "..", "fonts", "DejaVuSans.ttf")
29
+
30
+ _CUSTOM_TTF_NAME = "ScanlayerCustom"
31
+ _custom_registered_path: str | None = None
32
+
33
+ _CJK_FONTS = {
34
+ "chi_sim": "STSong-Light",
35
+ "chi_tra": "MSung-Light",
36
+ "jpn": "HeiseiMin-W3",
37
+ "kor": "HYSMyeongJo-Medium",
38
+ }
39
+
40
+ _registered: set[str] = set()
41
+
42
+
43
+ def _register_once(register_fn, font_name: str) -> bool:
44
+ """Register a font the first time it's needed. Returns True on success."""
45
+ if font_name in _registered:
46
+ return True
47
+ try:
48
+ register_fn()
49
+ _registered.add(font_name)
50
+ return True
51
+ except Exception as exc:
52
+ log_warning(
53
+ log,
54
+ f"Could not register font {font_name!r} ({type(exc).__name__}: "
55
+ f"{exc}), falling back to {FALLBACK_FONT}.",
56
+ )
57
+ return False
58
+
59
+
60
+ def resolve_font(lang: str | None) -> str:
61
+ """Return the reportlab font name for the text layer.
62
+
63
+ If config.FONT_PATH is set, it wins. Otherwise picks based on lang.
64
+ """
65
+ global _custom_registered_path
66
+ if config.FONT_PATH:
67
+ if config.FONT_PATH != _custom_registered_path:
68
+ _registered.discard(_CUSTOM_TTF_NAME)
69
+ _custom_registered_path = config.FONT_PATH
70
+ ok = _register_once(
71
+ lambda: pdfmetrics.registerFont(TTFont(_CUSTOM_TTF_NAME, config.FONT_PATH)),
72
+ _CUSTOM_TTF_NAME,
73
+ )
74
+ if ok:
75
+ return _CUSTOM_TTF_NAME
76
+ log_warning(
77
+ log,
78
+ f"config.FONT_PATH={config.FONT_PATH!r} could not be "
79
+ f"registered, falling back to automatic font selection.",
80
+ )
81
+
82
+ lang = (lang or "").lower()
83
+ components = [c for c in lang.split("+") if c]
84
+
85
+ for component in components:
86
+ cid_font = _CJK_FONTS.get(component)
87
+ if cid_font:
88
+ ok = _register_once(
89
+ lambda cid_font=cid_font: pdfmetrics.registerFont(UnicodeCIDFont(cid_font)),
90
+ cid_font,
91
+ )
92
+ if ok:
93
+ return cid_font
94
+ break
95
+
96
+ if os.path.exists(_UNICODE_TTF_PATH):
97
+ ok = _register_once(
98
+ lambda: pdfmetrics.registerFont(TTFont(_UNICODE_TTF_NAME, _UNICODE_TTF_PATH)),
99
+ _UNICODE_TTF_NAME,
100
+ )
101
+ if ok:
102
+ return _UNICODE_TTF_NAME
103
+ else:
104
+ log_warning(
105
+ log,
106
+ f"Bundled Unicode font not found at {_UNICODE_TTF_PATH!r}, "
107
+ f"falling back to {FALLBACK_FONT} (Latin-1 only).",
108
+ )
109
+
110
+ return FALLBACK_FONT
@@ -0,0 +1,5 @@
1
+ """preprocessing - image straightening and optimization for OCR."""
2
+
3
+ from scanlayer.preprocessing.enhance import PreprocessResult, preprocess
4
+
5
+ __all__ = ["preprocess", "PreprocessResult"]