epub-pdf-wrap 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,27 @@
1
+ Copyright (c) 2026, Andrea Esuli (andrea@esuli.it)
2
+ All rights reserved.
3
+
4
+ Redistribution and use in source and binary forms, with or without
5
+ modification, are permitted provided that the following conditions are met:
6
+
7
+ 1 Redistributions of source code must retain the above copyright notice, this
8
+ list of conditions and the following disclaimer.
9
+
10
+ 2 Redistributions in binary form must reproduce the above copyright notice,
11
+ this list of conditions and the following disclaimer in the documentation
12
+ and/or other materials provided with the distribution.
13
+
14
+ 3 Neither the name of the copyright holder nor the names of its
15
+ contributors may be used to endorse or promote products derived from
16
+ this software without specific prior written permission.
17
+
18
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
19
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
20
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
21
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
22
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
23
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
24
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
25
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
26
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
27
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,95 @@
1
+ Metadata-Version: 2.4
2
+ Name: epub-pdf-wrap
3
+ Version: 0.1.1
4
+ Summary: Convert a PDF to an EPUB by wrapping each rendered page in an EPUB page
5
+ Author-email: Andrea Esuli <andrea@esuli.it>
6
+ License-Expression: BSD-3-Clause
7
+ Project-URL: Home, https://github.com/aesuli/epub_pdf_wrap
8
+ Keywords: pdf,epub,conversion,render
9
+ Classifier: Development Status :: 3 - Alpha
10
+ Classifier: Environment :: Console
11
+ Classifier: Intended Audience :: End Users/Desktop
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Topic :: Text Processing :: Markup :: HTML
15
+ Requires-Python: >=3.9
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Requires-Dist: PyMuPDF>=1.24
19
+ Provides-Extra: dev
20
+ Requires-Dist: pytest>=7; extra == "dev"
21
+ Requires-Dist: ebooklib>=0.18; extra == "dev"
22
+ Dynamic: license-file
23
+
24
+ # EPUB PDF wrap
25
+
26
+ Converts a PDF into an EPUB by rendering and wrapping each page of the PDF in
27
+ an EPUB page. This is specifically aimed at PDF files that cannot be converted
28
+ into EPUB any other way without corrupting their visual rendering (comics,
29
+ scientific papers...).
30
+
31
+ ## Installation
32
+
33
+ ```
34
+ pip install epub-pdf-wrap
35
+ ```
36
+
37
+ From source (with dev dependencies for testing):
38
+
39
+ ```
40
+ pip install -e ".[dev]"
41
+ ```
42
+
43
+ ## Usage
44
+
45
+ ```
46
+ epub-pdf-wrap <pdf-filename> [-o <epub-filename>]
47
+ ```
48
+
49
+ By default the output filename is the input filename with the `pdf` extension
50
+ replaced by the `epub` extension. Running without `pip install` also works via
51
+ `python -m epub_pdf_wrap`.
52
+
53
+ ### Options
54
+
55
+ - `-r <num>, --resolution <num>`: target render width in pixels for the pages
56
+ in the output file. By default the pages are rendered at the resolution the
57
+ PDF itself declares.
58
+ - `-c, --crop-global`: trim the white margins around the page content using
59
+ one common inset for all pages (safe: never clips content on any page, all
60
+ pages keep the same size).
61
+ - `--crop-page`: trim the white margins around each page's own content, page
62
+ by page (trims more aggressively but page sizes may vary).
63
+
64
+ `-c/--crop-global` and `--crop-page` are mutually exclusive; with neither
65
+ flag the margins are left as-is.
66
+
67
+ ## Metadata
68
+
69
+ Document metadata (title, author, subject, keywords and creation date) is
70
+ taken from the PDF and written into the EPUB. Empty fields are omitted; the
71
+ title falls back to the input filename if the PDF has none.
72
+
73
+ ## Examples
74
+
75
+ Convert a paper at a wider resolution and name the output explicitly:
76
+
77
+ ```
78
+ epub-pdf-wrap paper.pdf -o paper.epub -r 1400
79
+ ```
80
+
81
+ ## Development
82
+
83
+ ```
84
+ python -m venv .venv
85
+ .venv\Scripts\activate # Windows (or `source .venv/bin/activate` elsewhere)
86
+ pip install -e ".[dev]"
87
+ pytest
88
+ ```
89
+
90
+ `samples/` contains real input PDFs for manually verifying the output.
91
+
92
+ ## License
93
+
94
+ Distributed under the BSD 3-Clause License; see `LICENSE`.
95
+ Copyright (c) 2026, Andrea Esuli (andrea@esuli.it).
@@ -0,0 +1,72 @@
1
+ # EPUB PDF wrap
2
+
3
+ Converts a PDF into an EPUB by rendering and wrapping each page of the PDF in
4
+ an EPUB page. This is specifically aimed at PDF files that cannot be converted
5
+ into EPUB any other way without corrupting their visual rendering (comics,
6
+ scientific papers...).
7
+
8
+ ## Installation
9
+
10
+ ```
11
+ pip install epub-pdf-wrap
12
+ ```
13
+
14
+ From source (with dev dependencies for testing):
15
+
16
+ ```
17
+ pip install -e ".[dev]"
18
+ ```
19
+
20
+ ## Usage
21
+
22
+ ```
23
+ epub-pdf-wrap <pdf-filename> [-o <epub-filename>]
24
+ ```
25
+
26
+ By default the output filename is the input filename with the `pdf` extension
27
+ replaced by the `epub` extension. Running without `pip install` also works via
28
+ `python -m epub_pdf_wrap`.
29
+
30
+ ### Options
31
+
32
+ - `-r <num>, --resolution <num>`: target render width in pixels for the pages
33
+ in the output file. By default the pages are rendered at the resolution the
34
+ PDF itself declares.
35
+ - `-c, --crop-global`: trim the white margins around the page content using
36
+ one common inset for all pages (safe: never clips content on any page, all
37
+ pages keep the same size).
38
+ - `--crop-page`: trim the white margins around each page's own content, page
39
+ by page (trims more aggressively but page sizes may vary).
40
+
41
+ `-c/--crop-global` and `--crop-page` are mutually exclusive; with neither
42
+ flag the margins are left as-is.
43
+
44
+ ## Metadata
45
+
46
+ Document metadata (title, author, subject, keywords and creation date) is
47
+ taken from the PDF and written into the EPUB. Empty fields are omitted; the
48
+ title falls back to the input filename if the PDF has none.
49
+
50
+ ## Examples
51
+
52
+ Convert a paper at a wider resolution and name the output explicitly:
53
+
54
+ ```
55
+ epub-pdf-wrap paper.pdf -o paper.epub -r 1400
56
+ ```
57
+
58
+ ## Development
59
+
60
+ ```
61
+ python -m venv .venv
62
+ .venv\Scripts\activate # Windows (or `source .venv/bin/activate` elsewhere)
63
+ pip install -e ".[dev]"
64
+ pytest
65
+ ```
66
+
67
+ `samples/` contains real input PDFs for manually verifying the output.
68
+
69
+ ## License
70
+
71
+ Distributed under the BSD 3-Clause License; see `LICENSE`.
72
+ Copyright (c) 2026, Andrea Esuli (andrea@esuli.it).
@@ -0,0 +1,14 @@
1
+ """Convert PDFs to EPUBs by rendering each page as an EPUB page."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from importlib.metadata import PackageNotFoundError, version
6
+
7
+ from .core import ConversionError, convert
8
+
9
+ try:
10
+ __version__ = version("epub-pdf-wrap")
11
+ except PackageNotFoundError: # running from a source checkout
12
+ __version__ = "0.0.0"
13
+
14
+ __all__ = ["ConversionError", "convert", "__version__"]
@@ -0,0 +1,77 @@
1
+ """Command line entry point."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import sys
7
+
8
+ from . import __version__
9
+ from .core import ConversionError, convert
10
+
11
+
12
+ def build_parser() -> argparse.ArgumentParser:
13
+ parser = argparse.ArgumentParser(
14
+ prog="epub-pdf-wrap",
15
+ description=(
16
+ "Convert a PDF into an EPUB by rendering and wrapping each page "
17
+ "of the PDF in an EPUB page. Aims at PDFs whose layout would be "
18
+ "corrupted by text-based conversion (comics, scientific papers)."
19
+ ),
20
+ epilog="Copyright (c) 2026 Andrea Esuli — BSD-3-Clause license",
21
+ )
22
+ parser.add_argument(
23
+ "-V", "--version", action="version", version=f"epub-pdf-wrap {__version__}",
24
+ )
25
+ parser.add_argument("input", help="input PDF file")
26
+ parser.add_argument(
27
+ "-o", "--output",
28
+ help="output EPUB file (default: input filename with .epub extension)",
29
+ )
30
+ parser.add_argument(
31
+ "-r", "--resolution", type=int,
32
+ help="target render width in pixels for the output (default: input resolution)",
33
+ )
34
+ crop = parser.add_mutually_exclusive_group()
35
+ crop.add_argument(
36
+ "-c", "--crop-global", action="store_true",
37
+ help="trim white margins using one common inset for all pages",
38
+ )
39
+ crop.add_argument(
40
+ "--crop-page", action="store_true",
41
+ help="trim white margins per page, to each page's own content",
42
+ )
43
+ return parser
44
+
45
+
46
+ def _progress(done: int, total: int) -> None:
47
+ # Redraw a single-line progress meter over the conversion.
48
+ bar_width = 30
49
+ filled = int(bar_width * done / total) if total else bar_width
50
+ bar = "#" * filled + "-" * (bar_width - filled)
51
+ sys.stdout.write(f"\rrendering {bar} {done}/{total}")
52
+ if done != total:
53
+ sys.stdout.flush()
54
+ else:
55
+ sys.stdout.write("\n")
56
+
57
+
58
+ def main(argv: list[str] | None = None) -> int:
59
+ args = build_parser().parse_args(argv)
60
+ crop = "global" if args.crop_global else ("page" if args.crop_page else None)
61
+ try:
62
+ out = convert(
63
+ args.input,
64
+ args.output,
65
+ args.resolution,
66
+ crop=crop,
67
+ log=lambda line: print(line),
68
+ progress=_progress,
69
+ )
70
+ except ConversionError as exc:
71
+ print(f"error: {exc}", file=sys.stderr)
72
+ return 1
73
+ return 0
74
+
75
+
76
+ if __name__ == "__main__":
77
+ raise SystemExit(main())
@@ -0,0 +1,412 @@
1
+ """Convert a PDF into an EPUB by rendering each page to an image."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import time
7
+ import uuid
8
+ from io import BytesIO
9
+ from pathlib import Path
10
+ from xml.sax.saxutils import escape
11
+ from zipfile import ZIP_DEFLATED, ZIP_STORED, ZipFile
12
+
13
+ import pymupdf
14
+
15
+
16
+ class ConversionError(RuntimeError):
17
+ """Raised when the PDF cannot be read or rendered."""
18
+
19
+
20
+ def slugify(name: str) -> str:
21
+ slug = re.sub(r"[^A-Za-z0-9]+", "-", name.strip())
22
+ return slug.strip("-") or "document"
23
+
24
+
25
+ # An edge inset counts as a real margin only if it is at least this fraction
26
+ # of the page dimension; anything smaller is treated as a sliver
27
+ # (anti-aliasing, registration marks) and the page is left untrimmed.
28
+ _MARGINS_FRACTION = 0.01
29
+
30
+
31
+ def content_bbox(page: "pymupdf.Page"):
32
+ """Union of the bounding boxes of all content on the page, or None if
33
+ the page is blank.
34
+
35
+ ``get_bboxlog`` returns a list of ``(kind, bbox)`` tuples; the union of
36
+ those boxes is the extent of the page's content.
37
+ """
38
+ log = page.get_bboxlog()
39
+ if not log:
40
+ return None
41
+ bbox = pymupdf.Rect(*log[0][1])
42
+ for _, b in log[1:]:
43
+ bbox |= pymupdf.Rect(*b)
44
+ if not bbox.is_valid or bbox.is_empty or bbox.is_infinite:
45
+ return None
46
+ return bbox
47
+
48
+
49
+ def margin_insets(bbox: "pymupdf.Rect", rect: "pymupdf.Rect") -> tuple[int, int, int, int]:
50
+ """Return (left, right, top, bottom) booleans flagging real margins.
51
+
52
+ An inset counts as a real margin when it is at least
53
+ ``_MARGINS_FRACTION`` of the corresponding page dimension.
54
+ """
55
+ left = rect.x0 < bbox.x0 - _MARGINS_FRACTION * rect.width
56
+ right = rect.x1 > bbox.x1 + _MARGINS_FRACTION * rect.width
57
+ top = rect.y0 < bbox.y0 - _MARGINS_FRACTION * rect.height
58
+ bottom = rect.y1 > bbox.y1 + _MARGINS_FRACTION * rect.height
59
+ return (1 if left else 0, 1 if right else 0, 1 if top else 0, 1 if bottom else 0)
60
+
61
+
62
+ def page_clip_rect(page: "pymupdf.Page"):
63
+ """Per-page clip: the content bbox clamped to the page, or the page rect
64
+ if the page is blank or has no real margins on every side."""
65
+ rect = page.rect
66
+ bbox = content_bbox(page)
67
+ if bbox is None or not any(margin_insets(bbox, rect)):
68
+ return rect
69
+ clip = bbox & rect # intersection, clamps the bbox within the page
70
+ if clip.is_empty:
71
+ return rect
72
+ return clip
73
+
74
+
75
+ def global_clip_rects(doc: "pymupdf.Document") -> list["pymupdf.Rect"]:
76
+ """One common clip rect for every page, safe for all of them.
77
+
78
+ For a common clip to never cut into any page's content it must fully
79
+ contain each page's content region; the union of the per-page content
80
+ boxes is the smallest box with that property, hence the most aggressive
81
+ safe common trim (and it yields a uniform size for all pages). Blank
82
+ pages fall back to their own full rect.
83
+ """
84
+ per_page = [(doc.load_page(i), content_bbox(doc.load_page(i)))
85
+ for i in range(doc.page_count)]
86
+ nonblank = [bbox for _, bbox in per_page if bbox is not None]
87
+ if not nonblank:
88
+ return [page.rect for page, _ in per_page]
89
+
90
+ common = pymupdf.Rect(
91
+ min(b.x0 for b in nonblank),
92
+ min(b.y0 for b in nonblank),
93
+ max(b.x1 for b in nonblank),
94
+ max(b.y1 for b in nonblank),
95
+ )
96
+ clips = []
97
+ for page, bbox in per_page:
98
+ if bbox is None:
99
+ clips.append(page.rect)
100
+ continue
101
+ clip = common & page.rect # clamp the common box inside this page
102
+ clips.append(clip if clip.width > 0 and clip.height > 0 else page.rect)
103
+ return clips
104
+
105
+
106
+ def _clean(value: str) -> str:
107
+ return " ".join(value.split()) if value else ""
108
+
109
+
110
+ def _w3cdtf(value: str) -> str:
111
+ """Convert a PDF creation date (``D:YYYYMMDDHHmmSSz``) to W3CDTF, or
112
+ ``""`` if it does not look parseable."""
113
+ s = value.strip()
114
+ if not s:
115
+ return ""
116
+ if s.startswith("D:"):
117
+ s = s[2:]
118
+ digits = s[:14]
119
+ if len(digits) < 8 or not digits[:8].isdigit():
120
+ return ""
121
+ if len(digits) == 8:
122
+ return f"{digits[0:4]}-{digits[4:6]}-{digits[6:8]}"
123
+ if len(digits) == 14 and digits[8:14].isdigit():
124
+ return (
125
+ f"{digits[0:4]}-{digits[4:6]}-{digits[6:8]}"
126
+ f"T{digits[8:10]}:{digits[10:12]}:{digits[12:14]}"
127
+ )
128
+ return ""
129
+
130
+
131
+ def epub_metadata(doc: "pymupdf.Document") -> dict:
132
+ """Transfer the PDF document metadata into an EPUB-friendly mapping.
133
+
134
+ Only non-empty fields are included: ``title``, ``creator`` (from the PDF
135
+ ``author``), ``subject``, ``keywords`` and ``date`` (from the PDF
136
+ ``creationDate``, W3CDTF). Tooling fields (creator/producer) are
137
+ deliberately skipped.
138
+ """
139
+ meta = doc.metadata or {}
140
+ out: dict[str, str] = {}
141
+ for key, field in (
142
+ ("title", "title"),
143
+ ("creator", "author"),
144
+ ("subject", "subject"),
145
+ ("keywords", "keywords"),
146
+ ):
147
+ if value := _clean(meta.get(field) or ""):
148
+ out[key] = value
149
+ for field in ("creationDate", "modDate"):
150
+ if value := _w3cdtf(meta.get(field) or ""):
151
+ out["date"] = value
152
+ break
153
+ return out
154
+
155
+
156
+ def render_page(page: "pymupdf.Page", resolution: int | None = None,
157
+ clip: "pymupdf.Rect | None" = None) -> bytes:
158
+ """Render one PDF page to a PNG byte string.
159
+
160
+ *resolution*, when given, is the target width in pixels; the height is
161
+ scaled proportionally. *clip*, when given, is a ``Rect`` in page
162
+ coordinates to render only that region (e.g. with margins trimmed).
163
+ """
164
+ rect = clip if clip is not None else page.rect
165
+ if resolution is not None:
166
+ if resolution <= 0:
167
+ raise ConversionError(f"resolution must be positive, got {resolution}")
168
+ scale = resolution / rect.width
169
+ else:
170
+ scale = 1.0
171
+ kwargs = {"alpha": True, "colorspace": pymupdf.csRGB}
172
+ if clip is not None:
173
+ kwargs["clip"] = clip
174
+ pix = page.get_pixmap(matrix=pymupdf.Matrix(scale, scale), **kwargs)
175
+ return pix.tobytes("png")
176
+
177
+
178
+ class _EpubWriter:
179
+ """Minimal EPUB 2 writer: one XHTML section per page, one PNG per page.
180
+
181
+ The spine is built from the XHTML sections (never directly from images),
182
+ which strict readers require.
183
+ """
184
+
185
+ def __init__(self, path: Path, metadata: dict):
186
+ self.path = Path(path)
187
+ self.metadata = metadata
188
+ self.image_names: list[str] = []
189
+ self.images: list[bytes] = []
190
+ self.uuid = str(uuid.uuid4())
191
+ self._tmp = self.path.with_suffix(self.path.suffix + ".tmp")
192
+
193
+ def add_image(self, name: str, data: bytes):
194
+ self.image_names.append(name)
195
+ self.images.append(data)
196
+
197
+ @property
198
+ def _section_ids(self) -> list[str]:
199
+ return [f"sec-{i:04d}" for i in range(len(self.image_names))]
200
+
201
+ def _container_xml(self) -> str:
202
+ return (
203
+ '<?xml version="1.0" encoding="UTF-8"?>\n'
204
+ '<container version="1.0" '
205
+ 'xmlns="urn:oasis:names:tc:opendocument:xmlns:container">'
206
+ "<rootfiles>"
207
+ '<rootfile full-path="OEBPS/content.opf" '
208
+ 'media-type="application/oebps-package+xml"/>'
209
+ "</rootfiles></container>"
210
+ )
211
+
212
+ def _metadata_xml(self) -> str:
213
+ # Order follows the OPF/DC convention; only fields present are emitted.
214
+ parts = [f'<dc:identifier id="epubid">{self.uuid}</dc:identifier>']
215
+ if "title" in self.metadata:
216
+ parts.append(f"<dc:title>{escape(self.metadata['title'])}</dc:title>")
217
+ if "creator" in self.metadata:
218
+ parts.append(f"<dc:creator>{escape(self.metadata['creator'])}</dc:creator>")
219
+ if "subject" in self.metadata:
220
+ parts.append(f"<dc:subject>{escape(self.metadata['subject'])}</dc:subject>")
221
+ if "date" in self.metadata:
222
+ parts.append(f"<dc:date>{escape(self.metadata['date'])}</dc:date>")
223
+ parts.append("<dc:language>en</dc:language>")
224
+ if "keywords" in self.metadata:
225
+ parts.append(
226
+ f'<meta name="keywords" content="{escape(self.metadata["keywords"])}"/>'
227
+ )
228
+ return "".join(parts)
229
+
230
+ def _opf_xml(self) -> str:
231
+ manifest = (
232
+ '<item id="ncx" media-type="application/x-dtbncx+xml" href="toc.ncx"/>'
233
+ + "".join(
234
+ f'<item id="{sid}" media-type="application/xhtml+xml" '
235
+ f'href="page-{i:04d}.xhtml"/>'
236
+ for i, sid in enumerate(self._section_ids, start=1)
237
+ )
238
+ + "".join(
239
+ f'<item id="img-{i:04d}" media-type="image/png" '
240
+ f'href="images/{name}"/>'
241
+ for i, name in enumerate(self.image_names, start=1)
242
+ )
243
+ )
244
+ spine = "".join(
245
+ f'<itemref idref="{sid}"/>' for sid in self._section_ids
246
+ )
247
+ return (
248
+ '<?xml version="1.0" encoding="UTF-8"?>\n'
249
+ '<package xmlns="http://www.idpf.org/2007/opf" version="2.0">'
250
+ '<metadata xmlns:dc="http://purl.org/dc/elements/1.1/">'
251
+ f"{self._metadata_xml()}"
252
+ "</metadata>"
253
+ f"<manifest>{manifest}</manifest>"
254
+ f'<spine toc="ncx">{spine}</spine></package>'
255
+ )
256
+
257
+ def _ncx_xml(self) -> str:
258
+ nav = "".join(
259
+ f'<navPoint id="nav-{i:04d}" playOrder="{i}">'
260
+ f'<navLabel><text>{i}. Page {i}</text></navLabel>'
261
+ f'<content src="page-{i:04d}.xhtml"/>'
262
+ "</navPoint>"
263
+ for i in range(1, len(self.image_names) + 1)
264
+ )
265
+ return (
266
+ '<?xml version="1.0" encoding="UTF-8"?>\n'
267
+ '<ncx xmlns="http://www.daisy.org/z3986/2005/ncx/" version="2005-1">'
268
+ f'<head><meta name="dtb:uid" content="{self.uuid}"/></head>'
269
+ f"<docTitle><text>{escape(self.metadata.get('title', 'Document'))}</text></docTitle>"
270
+ f"<navMap>{nav}</navMap></ncx>"
271
+ )
272
+
273
+ def _section_xhtml(self, index: int, name: str) -> str:
274
+ css = "html,body{margin:0;padding:0;}img{display:block;max-width:100vw;margin:auto;}"
275
+ return (
276
+ '<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.1//EN" '
277
+ '"http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd">\n'
278
+ '<html xmlns="http://www.w3.org/1999/xhtml" lang="en" '
279
+ 'xmlns:epub="http://www.idpf.org/2007/ops">'
280
+ f"<head><title>Page {index}</title>"
281
+ f'<style type="text/css">{css}</style></head>'
282
+ '<body epub:type="bodymatter">'
283
+ f'<p style="margin:0;padding:0;border:none;">\u200b</p>'
284
+ f'<img src="images/{escape(name)}" alt="Page {index}"/>'
285
+ "</body></html>"
286
+ )
287
+
288
+ def build(self) -> None:
289
+ assert len(self.image_names) == len(self.images)
290
+ buf = BytesIO()
291
+ with ZipFile(buf, "w", ZIP_DEFLATED) as z:
292
+ z.writestr("mimetype", "application/epub+zip", compress_type=ZIP_STORED)
293
+ z.writestr("META-INF/container.xml", self._container_xml())
294
+ z.writestr("OEBPS/content.opf", self._opf_xml())
295
+ z.writestr("OEBPS/toc.ncx", self._ncx_xml())
296
+ for i, name in enumerate(self.image_names, start=1):
297
+ z.writestr(f"OEBPS/page-{i:04d}.xhtml", self._section_xhtml(i, name))
298
+ for name, data in zip(self.image_names, self.images):
299
+ z.writestr(f"OEBPS/images/{name}", data)
300
+
301
+ self._tmp.write_bytes(buf.getvalue())
302
+ self._tmp.replace(self.path)
303
+
304
+
305
+ def format_size(num_bytes: int) -> str:
306
+ size = float(num_bytes)
307
+ for unit in ("B", "KB", "MB", "GB", "TB"):
308
+ if size < 1024 or unit == "TB":
309
+ if unit == "B":
310
+ return f"{int(size)} {unit}"
311
+ return f"{size:.1f} {unit}"
312
+ size /= 1024
313
+
314
+
315
+ def convert(input_path: Path, output_path: Path | None = None,
316
+ resolution: int | None = None, crop: str | None = None,
317
+ log=None, progress=None) -> Path:
318
+ """Convert *input_path* (a PDF) to an EPUB and return the output path.
319
+
320
+ Every PDF page is rendered to a PNG and placed in its own EPUB section in
321
+ document order. The output file defaults to the input name with the
322
+ extension swapped to ``.epub``. *resolution*, when given, is the target
323
+ render width in pixels. *crop*, when given, trims white margins:
324
+ ``"global"`` uses one common inset for all pages, ``"page"`` trims each
325
+ page to its own content.
326
+
327
+ *log* (optional callable(str)) receives descriptive lines (input info,
328
+ output result). *progress* (optional callable(done, total)) is called for
329
+ every rendered page.
330
+ """
331
+ if crop is not None and crop not in ("global", "page"):
332
+ raise ConversionError(f"crop must be 'global' or 'page', got {crop!r}")
333
+
334
+ def _log(message: str) -> None:
335
+ if log is not None:
336
+ log(message)
337
+
338
+ def _progress(done: int, total: int) -> None:
339
+ if progress is not None:
340
+ progress(done, total)
341
+
342
+ started = time.monotonic()
343
+
344
+ input_path = Path(input_path)
345
+ if not input_path.exists():
346
+ raise ConversionError(f"input file not found: {input_path}")
347
+ if output_path is None:
348
+ output_path = input_path.with_suffix(".epub")
349
+ output_path = Path(output_path)
350
+
351
+ input_size = input_path.stat().st_size
352
+ _log(f"input: {input_path} ({format_size(input_size)})")
353
+
354
+ try:
355
+ doc = pymupdf.open(str(input_path))
356
+ except Exception as exc:
357
+ raise ConversionError(f"could not open PDF: {exc}") from exc
358
+
359
+ try:
360
+ page_count = doc.page_count
361
+ if page_count == 0:
362
+ raise ConversionError(f"PDF has no pages: {input_path}")
363
+ metadata = epub_metadata(doc)
364
+ metadata.setdefault("title", slugify(input_path.stem))
365
+ pages_list = [doc.load_page(i) for i in range(page_count)]
366
+
367
+ first = pages_list[0]
368
+ w, h = first.rect.width, first.rect.height
369
+ uniform = all(
370
+ p.rect.width == w and p.rect.height == h for p in pages_list[1:]
371
+ )
372
+ native = f"{w:.0f} x {h:.0f} px" + ("" if uniform else " (varies)")
373
+ _log(f"pages: {page_count} (native resolution: {native})")
374
+ if crop == "global":
375
+ clips = global_clip_rects(doc)
376
+ elif crop == "page":
377
+ clips = [page_clip_rect(p) for p in pages_list]
378
+ else:
379
+ clips = [None] * page_count
380
+ if resolution is not None:
381
+ _log(f"output: {output_path} ({resolution} px wide)")
382
+ else:
383
+ _log(f"output: {output_path} (native resolution)")
384
+ if crop is not None:
385
+ trimmed = sum(
386
+ 1 for p, c in zip(pages_list, clips) if c is not None and c != p.rect
387
+ )
388
+ _log(f"crop: {crop} ({trimmed} of {page_count} pages trimmed)")
389
+
390
+ rendered: list[tuple[str, bytes]] = []
391
+ for i, (page, clip_raw) in enumerate(zip(pages_list, clips)):
392
+ clip = None if (clip_raw is None or clip_raw == page.rect) else clip_raw
393
+ png = render_page(page, resolution, clip)
394
+ rendered.append((f"page-{i + 1:04d}.png", png))
395
+ _progress(i + 1, page_count)
396
+ finally:
397
+ doc.close()
398
+
399
+ pages = rendered
400
+
401
+ output_path.parent.mkdir(parents=True, exist_ok=True)
402
+ writer = _EpubWriter(output_path, metadata)
403
+ for name, data in pages:
404
+ writer.add_image(name, data)
405
+ writer.build()
406
+
407
+ _log(
408
+ f"wrote: {output_path} "
409
+ f"({format_size(output_path.stat().st_size)}, "
410
+ f"{page_count} pages, {time.monotonic() - started:.1f}s)"
411
+ )
412
+ return output_path
@@ -0,0 +1,95 @@
1
+ Metadata-Version: 2.4
2
+ Name: epub-pdf-wrap
3
+ Version: 0.1.1
4
+ Summary: Convert a PDF to an EPUB by wrapping each rendered page in an EPUB page
5
+ Author-email: Andrea Esuli <andrea@esuli.it>
6
+ License-Expression: BSD-3-Clause
7
+ Project-URL: Home, https://github.com/aesuli/epub_pdf_wrap
8
+ Keywords: pdf,epub,conversion,render
9
+ Classifier: Development Status :: 3 - Alpha
10
+ Classifier: Environment :: Console
11
+ Classifier: Intended Audience :: End Users/Desktop
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Topic :: Text Processing :: Markup :: HTML
15
+ Requires-Python: >=3.9
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Requires-Dist: PyMuPDF>=1.24
19
+ Provides-Extra: dev
20
+ Requires-Dist: pytest>=7; extra == "dev"
21
+ Requires-Dist: ebooklib>=0.18; extra == "dev"
22
+ Dynamic: license-file
23
+
24
+ # EPUB PDF wrap
25
+
26
+ Converts a PDF into an EPUB by rendering and wrapping each page of the PDF in
27
+ an EPUB page. This is specifically aimed at PDF files that cannot be converted
28
+ into EPUB any other way without corrupting their visual rendering (comics,
29
+ scientific papers...).
30
+
31
+ ## Installation
32
+
33
+ ```
34
+ pip install epub-pdf-wrap
35
+ ```
36
+
37
+ From source (with dev dependencies for testing):
38
+
39
+ ```
40
+ pip install -e ".[dev]"
41
+ ```
42
+
43
+ ## Usage
44
+
45
+ ```
46
+ epub-pdf-wrap <pdf-filename> [-o <epub-filename>]
47
+ ```
48
+
49
+ By default the output filename is the input filename with the `pdf` extension
50
+ replaced by the `epub` extension. Running without `pip install` also works via
51
+ `python -m epub_pdf_wrap`.
52
+
53
+ ### Options
54
+
55
+ - `-r <num>, --resolution <num>`: target render width in pixels for the pages
56
+ in the output file. By default the pages are rendered at the resolution the
57
+ PDF itself declares.
58
+ - `-c, --crop-global`: trim the white margins around the page content using
59
+ one common inset for all pages (safe: never clips content on any page, all
60
+ pages keep the same size).
61
+ - `--crop-page`: trim the white margins around each page's own content, page
62
+ by page (trims more aggressively but page sizes may vary).
63
+
64
+ `-c/--crop-global` and `--crop-page` are mutually exclusive; with neither
65
+ flag the margins are left as-is.
66
+
67
+ ## Metadata
68
+
69
+ Document metadata (title, author, subject, keywords and creation date) is
70
+ taken from the PDF and written into the EPUB. Empty fields are omitted; the
71
+ title falls back to the input filename if the PDF has none.
72
+
73
+ ## Examples
74
+
75
+ Convert a paper at a wider resolution and name the output explicitly:
76
+
77
+ ```
78
+ epub-pdf-wrap paper.pdf -o paper.epub -r 1400
79
+ ```
80
+
81
+ ## Development
82
+
83
+ ```
84
+ python -m venv .venv
85
+ .venv\Scripts\activate # Windows (or `source .venv/bin/activate` elsewhere)
86
+ pip install -e ".[dev]"
87
+ pytest
88
+ ```
89
+
90
+ `samples/` contains real input PDFs for manually verifying the output.
91
+
92
+ ## License
93
+
94
+ Distributed under the BSD 3-Clause License; see `LICENSE`.
95
+ Copyright (c) 2026, Andrea Esuli (andrea@esuli.it).
@@ -0,0 +1,13 @@
1
+ LICENSE
2
+ README.md
3
+ pyproject.toml
4
+ epub_pdf_wrap/__init__.py
5
+ epub_pdf_wrap/__main__.py
6
+ epub_pdf_wrap/core.py
7
+ epub_pdf_wrap.egg-info/PKG-INFO
8
+ epub_pdf_wrap.egg-info/SOURCES.txt
9
+ epub_pdf_wrap.egg-info/dependency_links.txt
10
+ epub_pdf_wrap.egg-info/entry_points.txt
11
+ epub_pdf_wrap.egg-info/requires.txt
12
+ epub_pdf_wrap.egg-info/top_level.txt
13
+ tests/test_core.py
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ epub-pdf-wrap = epub_pdf_wrap.__main__:main
@@ -0,0 +1,5 @@
1
+ PyMuPDF>=1.24
2
+
3
+ [dev]
4
+ pytest>=7
5
+ ebooklib>=0.18
@@ -0,0 +1 @@
1
+ epub_pdf_wrap
@@ -0,0 +1,40 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "epub-pdf-wrap"
7
+ version = "0.1.1"
8
+ description = "Convert a PDF to an EPUB by wrapping each rendered page in an EPUB page"
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "BSD-3-Clause"
12
+ license-files = ["LICENSE"]
13
+ authors = [
14
+ { name = "Andrea Esuli", email = "andrea@esuli.it" },
15
+ ]
16
+ urls = { "Home" = "https://github.com/aesuli/epub_pdf_wrap" }
17
+ keywords = ["pdf", "epub", "conversion", "render"]
18
+ classifiers = [
19
+ "Development Status :: 3 - Alpha",
20
+ "Environment :: Console",
21
+ "Intended Audience :: End Users/Desktop",
22
+ "Operating System :: OS Independent",
23
+ "Programming Language :: Python :: 3",
24
+ "Topic :: Text Processing :: Markup :: HTML",
25
+ ]
26
+ dependencies = [
27
+ "PyMuPDF>=1.24",
28
+ ]
29
+
30
+ [project.optional-dependencies]
31
+ dev = [
32
+ "pytest>=7",
33
+ "ebooklib>=0.18",
34
+ ]
35
+
36
+ [project.scripts]
37
+ epub-pdf-wrap = "epub_pdf_wrap.__main__:main"
38
+
39
+ [tool.setuptools]
40
+ packages = ["epub_pdf_wrap"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,345 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+
5
+ import pymupdf
6
+ import pytest
7
+
8
+ from epub_pdf_wrap.core import ConversionError, convert, format_size, page_clip_rect, render_page, slugify
9
+
10
+
11
+ @pytest.fixture
12
+ def tiny_pdf(tmp_path: Path) -> Path:
13
+ """Three-page PDF: text page, white-blank page is omitted; use text + image pages."""
14
+ doc = pymupdf.open()
15
+ page = doc.new_page(width=200, height=260)
16
+ page.insert_text((20, 40), "First page")
17
+ page = doc.new_page(width=200, height=260)
18
+ page.insert_text((20, 40), "Second page")
19
+ out = tmp_path / "tiny.pdf"
20
+ doc.save(str(out))
21
+ doc.close()
22
+ return out
23
+
24
+
25
+ def test_convert_produces_epub(tmp_path: Path, tiny_pdf: Path) -> None:
26
+ out = convert(tiny_pdf, tmp_path / "out.epub")
27
+ assert out.exists()
28
+ import zipfile
29
+
30
+ with zipfile.ZipFile(out) as z:
31
+ names = z.namelist()
32
+ assert names[0] == "mimetype"
33
+ assert z.read("mimetype") == b"application/epub+zip"
34
+ assert "OEBPS/images/page-0001.png" in names
35
+ assert "OEBPS/images/page-0002.png" in names
36
+
37
+
38
+ def test_epub_is_readable_by_ebooklib(tmp_path: Path, tiny_pdf: Path) -> None:
39
+ """Reader-grade check: ebooklib must parse metadata, spine and content."""
40
+ import ebooklib
41
+ from ebooklib import epub
42
+
43
+ out = convert(tiny_pdf, tmp_path / "reader.epub")
44
+ book = epub.read_epub(str(out), options={"ignore_ncx": True})
45
+
46
+ # Metadata
47
+ titles = book.get_metadata("DC", "title")
48
+ assert titles and titles[0][0] == "tiny"
49
+
50
+ # Spine (reading order) must resolve to real items
51
+ spine_refs = [ref for ref, _ in book.spine]
52
+ assert spine_refs == ["sec-0000", "sec-0001"]
53
+ for ref in spine_refs:
54
+ item = book.get_item_with_id(ref)
55
+ assert item is not None
56
+ assert "<img" in item.get_content().decode("utf-8")
57
+
58
+ # Every spine image must be present
59
+ images = {i.get_name() for i in book.get_items_of_type(ebooklib.ITEM_IMAGE)}
60
+ assert images == {"images/page-0001.png", "images/page-0002.png"}
61
+
62
+ # A default load must also succeed without the ignore_ncx bypass:
63
+ # the NCX toc id has to resolve in the manifest (id="ncx").
64
+ fresh = epub.read_epub(str(out))
65
+ assert len(list(fresh.get_items_of_type(ebooklib.ITEM_DOCUMENT))) == 2
66
+ # Navigation must provide one TOC entry per page, each resolving to a
67
+ # real item (this is what readers do when a user opens the TOC).
68
+ toc = list(fresh.toc)
69
+ assert len(toc) == 2
70
+ for link in toc:
71
+ item = fresh.get_item_with_href(link.href)
72
+ assert item is not None
73
+ assert "<img" in item.get_content().decode("utf-8")
74
+
75
+
76
+ def test_convert_default_output_name(tiny_pdf: Path) -> None:
77
+ out = convert(tiny_pdf)
78
+ assert out.name == "tiny.epub"
79
+ out.unlink()
80
+
81
+
82
+ def test_resolution_flag(tmp_path: Path, tiny_pdf: Path) -> None:
83
+ out = convert(tiny_pdf, tmp_path / "res.epub", resolution=600)
84
+ import zipfile
85
+
86
+ with zipfile.ZipFile(out) as z:
87
+ png = z.read("OEBPS/images/page-0001.png")
88
+ import struct
89
+
90
+ # IHDR: width at bytes 16..20
91
+ width = struct.unpack(">I", png[16:20])[0]
92
+ assert width == 600
93
+
94
+
95
+ def test_blank_only_pdf_still_produces_epub(tmp_path: Path) -> None:
96
+ doc = pymupdf.open()
97
+ doc.new_page(width=200, height=260)
98
+ blank = tmp_path / "blank.pdf"
99
+ doc.save(str(blank))
100
+ doc.close()
101
+
102
+ out = convert(blank, tmp_path / "blank.epub")
103
+ assert out.exists()
104
+ import zipfile
105
+
106
+ with zipfile.ZipFile(out) as z:
107
+ assert "OEBPS/images/page-0001.png" in z.namelist()
108
+ out.unlink()
109
+ blank.unlink()
110
+
111
+
112
+ def test_missing_input_raises(tmp_path: Path) -> None:
113
+ with pytest.raises(ConversionError):
114
+ convert(tmp_path / "nope.pdf", tmp_path / "nope.epub")
115
+
116
+
117
+ def test_bad_resolution_raises(tiny_pdf: Path) -> None:
118
+ doc = pymupdf.open(str(tiny_pdf))
119
+ page = doc.load_page(0)
120
+ try:
121
+ with pytest.raises(ConversionError):
122
+ render_page(page, resolution=0)
123
+ with pytest.raises(ConversionError):
124
+ render_page(page, resolution=-5)
125
+ finally:
126
+ doc.close()
127
+
128
+
129
+ def test_convert_reports_input_info_and_progress(tmp_path: Path, tiny_pdf: Path) -> None:
130
+ logged: list[str] = []
131
+ steps: list[tuple[int, int]] = []
132
+ out = convert(
133
+ tiny_pdf,
134
+ tmp_path / "report.epub",
135
+ log=logged.append,
136
+ progress=lambda done, total: steps.append((done, total)),
137
+ )
138
+ joined = "\n".join(logged)
139
+ assert str(tiny_pdf) in joined
140
+ assert "pages: 2" in joined
141
+ assert str(out) in joined
142
+ assert "2 pages" in joined
143
+ assert steps[-1] == (2, 2)
144
+ assert len(steps) == 2
145
+
146
+
147
+ def test_slugify() -> None:
148
+ assert slugify("Some Comic 42") == "Some-Comic-42"
149
+ assert slugify("2406.12128v2") == "2406-12128v2"
150
+ assert slugify(" ") == "document"
151
+
152
+
153
+ def test_format_size() -> None:
154
+ assert format_size(0) == "0 B"
155
+ assert format_size(999) == "999 B"
156
+ assert format_size(1024) == "1.0 KB"
157
+ assert format_size(2048) == "2.0 KB"
158
+ assert format_size(1536) == "1.5 KB"
159
+ assert format_size(1536 * 1024) == "1.5 MB"
160
+
161
+
162
+ def _png_size(png: bytes) -> tuple[int, int]:
163
+ import struct
164
+
165
+ return struct.unpack(">II", png[16:24])
166
+
167
+
168
+ @pytest.fixture
169
+ def inset_pdf(tmp_path: Path) -> Path:
170
+ """Two pages with content inset in the middle (real margins around it).
171
+
172
+ Page 1: content from (50, 50) to (150, 210).
173
+ Page 2: content from (30, 30) to (170, 230) — wider than page 1, so the
174
+ global (union) clip keeps the wider extent on that side.
175
+ """
176
+ doc = pymupdf.open()
177
+ p = doc.new_page(width=200, height=260)
178
+ p.insert_text((50, 80), "page one")
179
+ p.draw_rect(pymupdf.Rect(50, 100, 150, 200))
180
+ p = doc.new_page(width=200, height=260)
181
+ p.insert_text((30, 60), "page two")
182
+ p.draw_rect(pymupdf.Rect(30, 80, 170, 220))
183
+ out = tmp_path / "inset.pdf"
184
+ doc.save(str(out))
185
+ doc.close()
186
+ return out
187
+
188
+
189
+ def test_crop_page_trims_to_content(tmp_path: Path, inset_pdf: Path) -> None:
190
+ out = convert(inset_pdf, tmp_path / "crop.epub", crop="page")
191
+
192
+ # Compare against the no-crop render of the same PDF
193
+ nocrop = convert(inset_pdf, tmp_path / "nocrop.epub")
194
+ import zipfile
195
+
196
+ with zipfile.ZipFile(out) as z:
197
+ cropped = _png_size(z.read("OEBPS/images/page-0001.png"))
198
+ with zipfile.ZipFile(nocrop) as z:
199
+ full = _png_size(z.read("OEBPS/images/page-0001.png"))
200
+ assert full == (200, 260)
201
+ # Page 1 content spans roughly 50..150 wide, 70..200 tall: the crop must
202
+ # clearly remove margins on all four sides.
203
+ assert cropped[0] < full[0] * 0.7
204
+ assert cropped[1] < full[1] * 0.7
205
+
206
+
207
+ def test_crop_global_uniform_but_safe(tmp_path: Path, inset_pdf: Path) -> None:
208
+ out = convert(inset_pdf, tmp_path / "g.epub", crop="global")
209
+ import zipfile
210
+
211
+ with zipfile.ZipFile(out) as z:
212
+ s1 = _png_size(z.read("OEBPS/images/page-0001.png"))
213
+ s2 = _png_size(z.read("OEBPS/images/page-0002.png"))
214
+ # Union of content boxes is the same size for both pages (each clipped
215
+ # to the same global box), so both page images have equal dimensions.
216
+ assert s1 == s2
217
+ # The global box must never clip content: page 2, which extends
218
+ # furthest (30..170 wide), must not have been cut.
219
+ assert s1[0] >= 170 - 30
220
+
221
+
222
+ def test_full_bleed_crop_unchanged(tmp_path: Path) -> None:
223
+ doc = pymupdf.open()
224
+ p = doc.new_page(width=200, height=260)
225
+ p.draw_rect(pymupdf.Rect(0, 0, 200, 260)) # content touches all edges
226
+ full = tmp_path / "full.pdf"
227
+ doc.save(str(full))
228
+ doc.close()
229
+
230
+ a = convert(full, tmp_path / "a.epub")
231
+ b = convert(full, tmp_path / "b.epub", crop="global")
232
+ c = convert(full, tmp_path / "c.epub", crop="page")
233
+ import zipfile
234
+
235
+ with zipfile.ZipFile(a) as z:
236
+ base = z.read("OEBPS/images/page-0001.png")
237
+ with zipfile.ZipFile(b) as z:
238
+ assert z.read("OEBPS/images/page-0001.png") == base
239
+ with zipfile.ZipFile(c) as z:
240
+ assert z.read("OEBPS/images/page-0001.png") == base
241
+ full.unlink()
242
+
243
+
244
+ def test_blank_page_crop_does_not_crash(tmp_path: Path) -> None:
245
+ doc = pymupdf.open()
246
+ doc.new_page(width=200, height=260)
247
+ blank = tmp_path / "blank.pdf"
248
+ doc.save(str(blank))
249
+ doc.close()
250
+
251
+ a = convert(blank, tmp_path / "gb.epub", crop="global")
252
+ b = convert(blank, tmp_path / "pb.epub", crop="page")
253
+ import zipfile
254
+
255
+ for f in (a, b):
256
+ with zipfile.ZipFile(f) as z:
257
+ assert _png_size(z.read("OEBPS/images/page-0001.png")) == (200, 260)
258
+ blank.unlink()
259
+
260
+
261
+ def test_crop_invalid_value_raises(inset_pdf: Path) -> None:
262
+ with pytest.raises(ConversionError):
263
+ convert(inset_pdf, inset_pdf.with_suffix(".epub"), crop="bogus")
264
+
265
+
266
+ def test_cli_parser_crop_flags() -> None:
267
+ from epub_pdf_wrap.__main__ import build_parser
268
+
269
+ p = build_parser()
270
+ args = p.parse_args(["in.pdf", "-c"])
271
+ assert args.crop_global is True and args.crop_page is False
272
+ args = p.parse_args(["in.pdf", "--crop-global"])
273
+ assert args.crop_global is True
274
+ args = p.parse_args(["in.pdf", "--crop-page"])
275
+ assert args.crop_global is False and args.crop_page is True
276
+ with pytest.raises(SystemExit) as exc:
277
+ p.parse_args(["in.pdf", "-c", "--crop-page"])
278
+ assert exc.value.code == 2
279
+
280
+
281
+ @pytest.fixture
282
+ def metadata_pdf(tmp_path: Path) -> Path:
283
+ doc = pymupdf.open()
284
+ page = doc.new_page(width=200, height=260)
285
+ page.insert_text((20, 40), "content")
286
+ doc.set_metadata(
287
+ {
288
+ "title": "A <Great> Book & Co",
289
+ "author": "Jane Doe, John Roe",
290
+ "subject": "Typesetting with pypdf",
291
+ "keywords": "epub, pdf, wrap",
292
+ "creationDate": "D:20041212120000+01'00'",
293
+ }
294
+ )
295
+ out = tmp_path / "meta.pdf"
296
+ doc.save(str(out))
297
+ doc.close()
298
+ return out
299
+
300
+
301
+ def test_metadata_is_transferred_to_epub(tmp_path: Path, metadata_pdf: Path) -> None:
302
+ import ebooklib
303
+ from ebooklib import epub
304
+
305
+ out = convert(metadata_pdf, tmp_path / "meta.epub")
306
+ book = epub.read_epub(str(out), options={"ignore_ncx": True})
307
+
308
+ assert book.get_metadata("DC", "title")[0][0] == "A <Great> Book & Co"
309
+ assert book.get_metadata("DC", "creator")[0][0] == "Jane Doe, John Roe"
310
+ assert book.get_metadata("DC", "subject")[0][0] == "Typesetting with pypdf"
311
+ assert book.get_metadata("DC", "date")[0][0] == "2004-12-12T12:00:00"
312
+ # Tooling fields are not transferred
313
+ assert not book.get_metadata("DC", "source")
314
+
315
+ # The OPF XML itself: identifier still present, values escaped in title,
316
+ # keywords present as an EPUB-3 meta element.
317
+ from zipfile import ZipFile
318
+
319
+ with ZipFile(out) as z:
320
+ opf = z.read("OEBPS/content.opf").decode("utf-8")
321
+ assert "A &lt;Great&gt; Book &amp; Co" in opf
322
+ assert '<dc:identifier id="epubid"' in opf
323
+ assert '<meta name="keywords" content="epub, pdf, wrap"/>' in opf
324
+
325
+
326
+ def test_metadata_fallback_title_when_empty(tmp_path: Path) -> None:
327
+ import pymupdf
328
+
329
+ doc = pymupdf.open()
330
+ doc.new_page(width=200, height=260)
331
+ blank = tmp_path / "no-meta.pdf"
332
+ doc.save(str(blank))
333
+ doc.close()
334
+
335
+ out = convert(blank, tmp_path / "no-meta.epub")
336
+ import zipfile
337
+
338
+ with zipfile.ZipFile(out) as z:
339
+ opf = z.read("OEBPS/content.opf").decode("utf-8")
340
+ assert "<dc:title>no-meta</dc:title>" in opf
341
+ # Only empty fields are omitted, not the required title
342
+ assert "<dc:creator>" not in opf
343
+ assert "<dc:subject>" not in opf
344
+ assert "<dc:date>" not in opf
345
+ blank.unlink()