epub-pdf-wrap 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- epub_pdf_wrap-0.1.1/LICENSE +27 -0
- epub_pdf_wrap-0.1.1/PKG-INFO +95 -0
- epub_pdf_wrap-0.1.1/README.md +72 -0
- epub_pdf_wrap-0.1.1/epub_pdf_wrap/__init__.py +14 -0
- epub_pdf_wrap-0.1.1/epub_pdf_wrap/__main__.py +77 -0
- epub_pdf_wrap-0.1.1/epub_pdf_wrap/core.py +412 -0
- epub_pdf_wrap-0.1.1/epub_pdf_wrap.egg-info/PKG-INFO +95 -0
- epub_pdf_wrap-0.1.1/epub_pdf_wrap.egg-info/SOURCES.txt +13 -0
- epub_pdf_wrap-0.1.1/epub_pdf_wrap.egg-info/dependency_links.txt +1 -0
- epub_pdf_wrap-0.1.1/epub_pdf_wrap.egg-info/entry_points.txt +2 -0
- epub_pdf_wrap-0.1.1/epub_pdf_wrap.egg-info/requires.txt +5 -0
- epub_pdf_wrap-0.1.1/epub_pdf_wrap.egg-info/top_level.txt +1 -0
- epub_pdf_wrap-0.1.1/pyproject.toml +40 -0
- epub_pdf_wrap-0.1.1/setup.cfg +4 -0
- epub_pdf_wrap-0.1.1/tests/test_core.py +345 -0
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
Copyright (c) 2026, Andrea Esuli (andrea@esuli.it)
|
|
2
|
+
All rights reserved.
|
|
3
|
+
|
|
4
|
+
Redistribution and use in source and binary forms, with or without
|
|
5
|
+
modification, are permitted provided that the following conditions are met:
|
|
6
|
+
|
|
7
|
+
1 Redistributions of source code must retain the above copyright notice, this
|
|
8
|
+
list of conditions and the following disclaimer.
|
|
9
|
+
|
|
10
|
+
2 Redistributions in binary form must reproduce the above copyright notice,
|
|
11
|
+
this list of conditions and the following disclaimer in the documentation
|
|
12
|
+
and/or other materials provided with the distribution.
|
|
13
|
+
|
|
14
|
+
3 Neither the name of the copyright holder nor the names of its
|
|
15
|
+
contributors may be used to endorse or promote products derived from
|
|
16
|
+
this software without specific prior written permission.
|
|
17
|
+
|
|
18
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
19
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
20
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
21
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
22
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
23
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
24
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
25
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
26
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
27
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: epub-pdf-wrap
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: Convert a PDF to an EPUB by wrapping each rendered page in an EPUB page
|
|
5
|
+
Author-email: Andrea Esuli <andrea@esuli.it>
|
|
6
|
+
License-Expression: BSD-3-Clause
|
|
7
|
+
Project-URL: Home, https://github.com/aesuli/epub_pdf_wrap
|
|
8
|
+
Keywords: pdf,epub,conversion,render
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Environment :: Console
|
|
11
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Text Processing :: Markup :: HTML
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: PyMuPDF>=1.24
|
|
19
|
+
Provides-Extra: dev
|
|
20
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
21
|
+
Requires-Dist: ebooklib>=0.18; extra == "dev"
|
|
22
|
+
Dynamic: license-file
|
|
23
|
+
|
|
24
|
+
# EPUB PDF wrap
|
|
25
|
+
|
|
26
|
+
Converts a PDF into an EPUB by rendering and wrapping each page of the PDF in
|
|
27
|
+
an EPUB page. This is specifically aimed at PDF files that cannot be converted
|
|
28
|
+
into EPUB any other way without corrupting their visual rendering (comics,
|
|
29
|
+
scientific papers...).
|
|
30
|
+
|
|
31
|
+
## Installation
|
|
32
|
+
|
|
33
|
+
```
|
|
34
|
+
pip install epub-pdf-wrap
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
From source (with dev dependencies for testing):
|
|
38
|
+
|
|
39
|
+
```
|
|
40
|
+
pip install -e ".[dev]"
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## Usage
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
epub-pdf-wrap <pdf-filename> [-o <epub-filename>]
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
By default the output filename is the input filename with the `pdf` extension
|
|
50
|
+
replaced by the `epub` extension. Running without `pip install` also works via
|
|
51
|
+
`python -m epub_pdf_wrap`.
|
|
52
|
+
|
|
53
|
+
### Options
|
|
54
|
+
|
|
55
|
+
- `-r <num>, --resolution <num>`: target render width in pixels for the pages
|
|
56
|
+
in the output file. By default the pages are rendered at the resolution the
|
|
57
|
+
PDF itself declares.
|
|
58
|
+
- `-c, --crop-global`: trim the white margins around the page content using
|
|
59
|
+
one common inset for all pages (safe: never clips content on any page, all
|
|
60
|
+
pages keep the same size).
|
|
61
|
+
- `--crop-page`: trim the white margins around each page's own content, page
|
|
62
|
+
by page (trims more aggressively but page sizes may vary).
|
|
63
|
+
|
|
64
|
+
`-c/--crop-global` and `--crop-page` are mutually exclusive; with neither
|
|
65
|
+
flag the margins are left as-is.
|
|
66
|
+
|
|
67
|
+
## Metadata
|
|
68
|
+
|
|
69
|
+
Document metadata (title, author, subject, keywords and creation date) is
|
|
70
|
+
taken from the PDF and written into the EPUB. Empty fields are omitted; the
|
|
71
|
+
title falls back to the input filename if the PDF has none.
|
|
72
|
+
|
|
73
|
+
## Examples
|
|
74
|
+
|
|
75
|
+
Convert a paper at a wider resolution and name the output explicitly:
|
|
76
|
+
|
|
77
|
+
```
|
|
78
|
+
epub-pdf-wrap paper.pdf -o paper.epub -r 1400
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Development
|
|
82
|
+
|
|
83
|
+
```
|
|
84
|
+
python -m venv .venv
|
|
85
|
+
.venv\Scripts\activate # Windows (or `source .venv/bin/activate` elsewhere)
|
|
86
|
+
pip install -e ".[dev]"
|
|
87
|
+
pytest
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
`samples/` contains real input PDFs for manually verifying the output.
|
|
91
|
+
|
|
92
|
+
## License
|
|
93
|
+
|
|
94
|
+
Distributed under the BSD 3-Clause License; see `LICENSE`.
|
|
95
|
+
Copyright (c) 2026, Andrea Esuli (andrea@esuli.it).
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
# EPUB PDF wrap
|
|
2
|
+
|
|
3
|
+
Converts a PDF into an EPUB by rendering and wrapping each page of the PDF in
|
|
4
|
+
an EPUB page. This is specifically aimed at PDF files that cannot be converted
|
|
5
|
+
into EPUB any other way without corrupting their visual rendering (comics,
|
|
6
|
+
scientific papers...).
|
|
7
|
+
|
|
8
|
+
## Installation
|
|
9
|
+
|
|
10
|
+
```
|
|
11
|
+
pip install epub-pdf-wrap
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
From source (with dev dependencies for testing):
|
|
15
|
+
|
|
16
|
+
```
|
|
17
|
+
pip install -e ".[dev]"
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
## Usage
|
|
21
|
+
|
|
22
|
+
```
|
|
23
|
+
epub-pdf-wrap <pdf-filename> [-o <epub-filename>]
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
By default the output filename is the input filename with the `pdf` extension
|
|
27
|
+
replaced by the `epub` extension. Running without `pip install` also works via
|
|
28
|
+
`python -m epub_pdf_wrap`.
|
|
29
|
+
|
|
30
|
+
### Options
|
|
31
|
+
|
|
32
|
+
- `-r <num>, --resolution <num>`: target render width in pixels for the pages
|
|
33
|
+
in the output file. By default the pages are rendered at the resolution the
|
|
34
|
+
PDF itself declares.
|
|
35
|
+
- `-c, --crop-global`: trim the white margins around the page content using
|
|
36
|
+
one common inset for all pages (safe: never clips content on any page, all
|
|
37
|
+
pages keep the same size).
|
|
38
|
+
- `--crop-page`: trim the white margins around each page's own content, page
|
|
39
|
+
by page (trims more aggressively but page sizes may vary).
|
|
40
|
+
|
|
41
|
+
`-c/--crop-global` and `--crop-page` are mutually exclusive; with neither
|
|
42
|
+
flag the margins are left as-is.
|
|
43
|
+
|
|
44
|
+
## Metadata
|
|
45
|
+
|
|
46
|
+
Document metadata (title, author, subject, keywords and creation date) is
|
|
47
|
+
taken from the PDF and written into the EPUB. Empty fields are omitted; the
|
|
48
|
+
title falls back to the input filename if the PDF has none.
|
|
49
|
+
|
|
50
|
+
## Examples
|
|
51
|
+
|
|
52
|
+
Convert a paper at a wider resolution and name the output explicitly:
|
|
53
|
+
|
|
54
|
+
```
|
|
55
|
+
epub-pdf-wrap paper.pdf -o paper.epub -r 1400
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
## Development
|
|
59
|
+
|
|
60
|
+
```
|
|
61
|
+
python -m venv .venv
|
|
62
|
+
.venv\Scripts\activate # Windows (or `source .venv/bin/activate` elsewhere)
|
|
63
|
+
pip install -e ".[dev]"
|
|
64
|
+
pytest
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
`samples/` contains real input PDFs for manually verifying the output.
|
|
68
|
+
|
|
69
|
+
## License
|
|
70
|
+
|
|
71
|
+
Distributed under the BSD 3-Clause License; see `LICENSE`.
|
|
72
|
+
Copyright (c) 2026, Andrea Esuli (andrea@esuli.it).
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Convert PDFs to EPUBs by rendering each page as an EPUB page."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
6
|
+
|
|
7
|
+
from .core import ConversionError, convert
|
|
8
|
+
|
|
9
|
+
try:
|
|
10
|
+
__version__ = version("epub-pdf-wrap")
|
|
11
|
+
except PackageNotFoundError: # running from a source checkout
|
|
12
|
+
__version__ = "0.0.0"
|
|
13
|
+
|
|
14
|
+
__all__ = ["ConversionError", "convert", "__version__"]
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""Command line entry point."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
|
|
8
|
+
from . import __version__
|
|
9
|
+
from .core import ConversionError, convert
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
13
|
+
parser = argparse.ArgumentParser(
|
|
14
|
+
prog="epub-pdf-wrap",
|
|
15
|
+
description=(
|
|
16
|
+
"Convert a PDF into an EPUB by rendering and wrapping each page "
|
|
17
|
+
"of the PDF in an EPUB page. Aims at PDFs whose layout would be "
|
|
18
|
+
"corrupted by text-based conversion (comics, scientific papers)."
|
|
19
|
+
),
|
|
20
|
+
epilog="Copyright (c) 2026 Andrea Esuli — BSD-3-Clause license",
|
|
21
|
+
)
|
|
22
|
+
parser.add_argument(
|
|
23
|
+
"-V", "--version", action="version", version=f"epub-pdf-wrap {__version__}",
|
|
24
|
+
)
|
|
25
|
+
parser.add_argument("input", help="input PDF file")
|
|
26
|
+
parser.add_argument(
|
|
27
|
+
"-o", "--output",
|
|
28
|
+
help="output EPUB file (default: input filename with .epub extension)",
|
|
29
|
+
)
|
|
30
|
+
parser.add_argument(
|
|
31
|
+
"-r", "--resolution", type=int,
|
|
32
|
+
help="target render width in pixels for the output (default: input resolution)",
|
|
33
|
+
)
|
|
34
|
+
crop = parser.add_mutually_exclusive_group()
|
|
35
|
+
crop.add_argument(
|
|
36
|
+
"-c", "--crop-global", action="store_true",
|
|
37
|
+
help="trim white margins using one common inset for all pages",
|
|
38
|
+
)
|
|
39
|
+
crop.add_argument(
|
|
40
|
+
"--crop-page", action="store_true",
|
|
41
|
+
help="trim white margins per page, to each page's own content",
|
|
42
|
+
)
|
|
43
|
+
return parser
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _progress(done: int, total: int) -> None:
|
|
47
|
+
# Redraw a single-line progress meter over the conversion.
|
|
48
|
+
bar_width = 30
|
|
49
|
+
filled = int(bar_width * done / total) if total else bar_width
|
|
50
|
+
bar = "#" * filled + "-" * (bar_width - filled)
|
|
51
|
+
sys.stdout.write(f"\rrendering {bar} {done}/{total}")
|
|
52
|
+
if done != total:
|
|
53
|
+
sys.stdout.flush()
|
|
54
|
+
else:
|
|
55
|
+
sys.stdout.write("\n")
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def main(argv: list[str] | None = None) -> int:
|
|
59
|
+
args = build_parser().parse_args(argv)
|
|
60
|
+
crop = "global" if args.crop_global else ("page" if args.crop_page else None)
|
|
61
|
+
try:
|
|
62
|
+
out = convert(
|
|
63
|
+
args.input,
|
|
64
|
+
args.output,
|
|
65
|
+
args.resolution,
|
|
66
|
+
crop=crop,
|
|
67
|
+
log=lambda line: print(line),
|
|
68
|
+
progress=_progress,
|
|
69
|
+
)
|
|
70
|
+
except ConversionError as exc:
|
|
71
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
72
|
+
return 1
|
|
73
|
+
return 0
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
if __name__ == "__main__":
|
|
77
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,412 @@
|
|
|
1
|
+
"""Convert a PDF into an EPUB by rendering each page to an image."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
import time
|
|
7
|
+
import uuid
|
|
8
|
+
from io import BytesIO
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from xml.sax.saxutils import escape
|
|
11
|
+
from zipfile import ZIP_DEFLATED, ZIP_STORED, ZipFile
|
|
12
|
+
|
|
13
|
+
import pymupdf
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class ConversionError(RuntimeError):
|
|
17
|
+
"""Raised when the PDF cannot be read or rendered."""
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def slugify(name: str) -> str:
|
|
21
|
+
slug = re.sub(r"[^A-Za-z0-9]+", "-", name.strip())
|
|
22
|
+
return slug.strip("-") or "document"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
# An edge inset counts as a real margin only if it is at least this fraction
|
|
26
|
+
# of the page dimension; anything smaller is treated as a sliver
|
|
27
|
+
# (anti-aliasing, registration marks) and the page is left untrimmed.
|
|
28
|
+
_MARGINS_FRACTION = 0.01
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def content_bbox(page: "pymupdf.Page"):
|
|
32
|
+
"""Union of the bounding boxes of all content on the page, or None if
|
|
33
|
+
the page is blank.
|
|
34
|
+
|
|
35
|
+
``get_bboxlog`` returns a list of ``(kind, bbox)`` tuples; the union of
|
|
36
|
+
those boxes is the extent of the page's content.
|
|
37
|
+
"""
|
|
38
|
+
log = page.get_bboxlog()
|
|
39
|
+
if not log:
|
|
40
|
+
return None
|
|
41
|
+
bbox = pymupdf.Rect(*log[0][1])
|
|
42
|
+
for _, b in log[1:]:
|
|
43
|
+
bbox |= pymupdf.Rect(*b)
|
|
44
|
+
if not bbox.is_valid or bbox.is_empty or bbox.is_infinite:
|
|
45
|
+
return None
|
|
46
|
+
return bbox
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def margin_insets(bbox: "pymupdf.Rect", rect: "pymupdf.Rect") -> tuple[int, int, int, int]:
|
|
50
|
+
"""Return (left, right, top, bottom) booleans flagging real margins.
|
|
51
|
+
|
|
52
|
+
An inset counts as a real margin when it is at least
|
|
53
|
+
``_MARGINS_FRACTION`` of the corresponding page dimension.
|
|
54
|
+
"""
|
|
55
|
+
left = rect.x0 < bbox.x0 - _MARGINS_FRACTION * rect.width
|
|
56
|
+
right = rect.x1 > bbox.x1 + _MARGINS_FRACTION * rect.width
|
|
57
|
+
top = rect.y0 < bbox.y0 - _MARGINS_FRACTION * rect.height
|
|
58
|
+
bottom = rect.y1 > bbox.y1 + _MARGINS_FRACTION * rect.height
|
|
59
|
+
return (1 if left else 0, 1 if right else 0, 1 if top else 0, 1 if bottom else 0)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def page_clip_rect(page: "pymupdf.Page"):
|
|
63
|
+
"""Per-page clip: the content bbox clamped to the page, or the page rect
|
|
64
|
+
if the page is blank or has no real margins on every side."""
|
|
65
|
+
rect = page.rect
|
|
66
|
+
bbox = content_bbox(page)
|
|
67
|
+
if bbox is None or not any(margin_insets(bbox, rect)):
|
|
68
|
+
return rect
|
|
69
|
+
clip = bbox & rect # intersection, clamps the bbox within the page
|
|
70
|
+
if clip.is_empty:
|
|
71
|
+
return rect
|
|
72
|
+
return clip
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def global_clip_rects(doc: "pymupdf.Document") -> list["pymupdf.Rect"]:
|
|
76
|
+
"""One common clip rect for every page, safe for all of them.
|
|
77
|
+
|
|
78
|
+
For a common clip to never cut into any page's content it must fully
|
|
79
|
+
contain each page's content region; the union of the per-page content
|
|
80
|
+
boxes is the smallest box with that property, hence the most aggressive
|
|
81
|
+
safe common trim (and it yields a uniform size for all pages). Blank
|
|
82
|
+
pages fall back to their own full rect.
|
|
83
|
+
"""
|
|
84
|
+
per_page = [(doc.load_page(i), content_bbox(doc.load_page(i)))
|
|
85
|
+
for i in range(doc.page_count)]
|
|
86
|
+
nonblank = [bbox for _, bbox in per_page if bbox is not None]
|
|
87
|
+
if not nonblank:
|
|
88
|
+
return [page.rect for page, _ in per_page]
|
|
89
|
+
|
|
90
|
+
common = pymupdf.Rect(
|
|
91
|
+
min(b.x0 for b in nonblank),
|
|
92
|
+
min(b.y0 for b in nonblank),
|
|
93
|
+
max(b.x1 for b in nonblank),
|
|
94
|
+
max(b.y1 for b in nonblank),
|
|
95
|
+
)
|
|
96
|
+
clips = []
|
|
97
|
+
for page, bbox in per_page:
|
|
98
|
+
if bbox is None:
|
|
99
|
+
clips.append(page.rect)
|
|
100
|
+
continue
|
|
101
|
+
clip = common & page.rect # clamp the common box inside this page
|
|
102
|
+
clips.append(clip if clip.width > 0 and clip.height > 0 else page.rect)
|
|
103
|
+
return clips
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _clean(value: str) -> str:
|
|
107
|
+
return " ".join(value.split()) if value else ""
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _w3cdtf(value: str) -> str:
|
|
111
|
+
"""Convert a PDF creation date (``D:YYYYMMDDHHmmSSz``) to W3CDTF, or
|
|
112
|
+
``""`` if it does not look parseable."""
|
|
113
|
+
s = value.strip()
|
|
114
|
+
if not s:
|
|
115
|
+
return ""
|
|
116
|
+
if s.startswith("D:"):
|
|
117
|
+
s = s[2:]
|
|
118
|
+
digits = s[:14]
|
|
119
|
+
if len(digits) < 8 or not digits[:8].isdigit():
|
|
120
|
+
return ""
|
|
121
|
+
if len(digits) == 8:
|
|
122
|
+
return f"{digits[0:4]}-{digits[4:6]}-{digits[6:8]}"
|
|
123
|
+
if len(digits) == 14 and digits[8:14].isdigit():
|
|
124
|
+
return (
|
|
125
|
+
f"{digits[0:4]}-{digits[4:6]}-{digits[6:8]}"
|
|
126
|
+
f"T{digits[8:10]}:{digits[10:12]}:{digits[12:14]}"
|
|
127
|
+
)
|
|
128
|
+
return ""
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def epub_metadata(doc: "pymupdf.Document") -> dict:
|
|
132
|
+
"""Transfer the PDF document metadata into an EPUB-friendly mapping.
|
|
133
|
+
|
|
134
|
+
Only non-empty fields are included: ``title``, ``creator`` (from the PDF
|
|
135
|
+
``author``), ``subject``, ``keywords`` and ``date`` (from the PDF
|
|
136
|
+
``creationDate``, W3CDTF). Tooling fields (creator/producer) are
|
|
137
|
+
deliberately skipped.
|
|
138
|
+
"""
|
|
139
|
+
meta = doc.metadata or {}
|
|
140
|
+
out: dict[str, str] = {}
|
|
141
|
+
for key, field in (
|
|
142
|
+
("title", "title"),
|
|
143
|
+
("creator", "author"),
|
|
144
|
+
("subject", "subject"),
|
|
145
|
+
("keywords", "keywords"),
|
|
146
|
+
):
|
|
147
|
+
if value := _clean(meta.get(field) or ""):
|
|
148
|
+
out[key] = value
|
|
149
|
+
for field in ("creationDate", "modDate"):
|
|
150
|
+
if value := _w3cdtf(meta.get(field) or ""):
|
|
151
|
+
out["date"] = value
|
|
152
|
+
break
|
|
153
|
+
return out
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def render_page(page: "pymupdf.Page", resolution: int | None = None,
|
|
157
|
+
clip: "pymupdf.Rect | None" = None) -> bytes:
|
|
158
|
+
"""Render one PDF page to a PNG byte string.
|
|
159
|
+
|
|
160
|
+
*resolution*, when given, is the target width in pixels; the height is
|
|
161
|
+
scaled proportionally. *clip*, when given, is a ``Rect`` in page
|
|
162
|
+
coordinates to render only that region (e.g. with margins trimmed).
|
|
163
|
+
"""
|
|
164
|
+
rect = clip if clip is not None else page.rect
|
|
165
|
+
if resolution is not None:
|
|
166
|
+
if resolution <= 0:
|
|
167
|
+
raise ConversionError(f"resolution must be positive, got {resolution}")
|
|
168
|
+
scale = resolution / rect.width
|
|
169
|
+
else:
|
|
170
|
+
scale = 1.0
|
|
171
|
+
kwargs = {"alpha": True, "colorspace": pymupdf.csRGB}
|
|
172
|
+
if clip is not None:
|
|
173
|
+
kwargs["clip"] = clip
|
|
174
|
+
pix = page.get_pixmap(matrix=pymupdf.Matrix(scale, scale), **kwargs)
|
|
175
|
+
return pix.tobytes("png")
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
class _EpubWriter:
|
|
179
|
+
"""Minimal EPUB 2 writer: one XHTML section per page, one PNG per page.
|
|
180
|
+
|
|
181
|
+
The spine is built from the XHTML sections (never directly from images),
|
|
182
|
+
which strict readers require.
|
|
183
|
+
"""
|
|
184
|
+
|
|
185
|
+
def __init__(self, path: Path, metadata: dict):
|
|
186
|
+
self.path = Path(path)
|
|
187
|
+
self.metadata = metadata
|
|
188
|
+
self.image_names: list[str] = []
|
|
189
|
+
self.images: list[bytes] = []
|
|
190
|
+
self.uuid = str(uuid.uuid4())
|
|
191
|
+
self._tmp = self.path.with_suffix(self.path.suffix + ".tmp")
|
|
192
|
+
|
|
193
|
+
def add_image(self, name: str, data: bytes):
|
|
194
|
+
self.image_names.append(name)
|
|
195
|
+
self.images.append(data)
|
|
196
|
+
|
|
197
|
+
@property
|
|
198
|
+
def _section_ids(self) -> list[str]:
|
|
199
|
+
return [f"sec-{i:04d}" for i in range(len(self.image_names))]
|
|
200
|
+
|
|
201
|
+
def _container_xml(self) -> str:
|
|
202
|
+
return (
|
|
203
|
+
'<?xml version="1.0" encoding="UTF-8"?>\n'
|
|
204
|
+
'<container version="1.0" '
|
|
205
|
+
'xmlns="urn:oasis:names:tc:opendocument:xmlns:container">'
|
|
206
|
+
"<rootfiles>"
|
|
207
|
+
'<rootfile full-path="OEBPS/content.opf" '
|
|
208
|
+
'media-type="application/oebps-package+xml"/>'
|
|
209
|
+
"</rootfiles></container>"
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
def _metadata_xml(self) -> str:
|
|
213
|
+
# Order follows the OPF/DC convention; only fields present are emitted.
|
|
214
|
+
parts = [f'<dc:identifier id="epubid">{self.uuid}</dc:identifier>']
|
|
215
|
+
if "title" in self.metadata:
|
|
216
|
+
parts.append(f"<dc:title>{escape(self.metadata['title'])}</dc:title>")
|
|
217
|
+
if "creator" in self.metadata:
|
|
218
|
+
parts.append(f"<dc:creator>{escape(self.metadata['creator'])}</dc:creator>")
|
|
219
|
+
if "subject" in self.metadata:
|
|
220
|
+
parts.append(f"<dc:subject>{escape(self.metadata['subject'])}</dc:subject>")
|
|
221
|
+
if "date" in self.metadata:
|
|
222
|
+
parts.append(f"<dc:date>{escape(self.metadata['date'])}</dc:date>")
|
|
223
|
+
parts.append("<dc:language>en</dc:language>")
|
|
224
|
+
if "keywords" in self.metadata:
|
|
225
|
+
parts.append(
|
|
226
|
+
f'<meta name="keywords" content="{escape(self.metadata["keywords"])}"/>'
|
|
227
|
+
)
|
|
228
|
+
return "".join(parts)
|
|
229
|
+
|
|
230
|
+
def _opf_xml(self) -> str:
|
|
231
|
+
manifest = (
|
|
232
|
+
'<item id="ncx" media-type="application/x-dtbncx+xml" href="toc.ncx"/>'
|
|
233
|
+
+ "".join(
|
|
234
|
+
f'<item id="{sid}" media-type="application/xhtml+xml" '
|
|
235
|
+
f'href="page-{i:04d}.xhtml"/>'
|
|
236
|
+
for i, sid in enumerate(self._section_ids, start=1)
|
|
237
|
+
)
|
|
238
|
+
+ "".join(
|
|
239
|
+
f'<item id="img-{i:04d}" media-type="image/png" '
|
|
240
|
+
f'href="images/{name}"/>'
|
|
241
|
+
for i, name in enumerate(self.image_names, start=1)
|
|
242
|
+
)
|
|
243
|
+
)
|
|
244
|
+
spine = "".join(
|
|
245
|
+
f'<itemref idref="{sid}"/>' for sid in self._section_ids
|
|
246
|
+
)
|
|
247
|
+
return (
|
|
248
|
+
'<?xml version="1.0" encoding="UTF-8"?>\n'
|
|
249
|
+
'<package xmlns="http://www.idpf.org/2007/opf" version="2.0">'
|
|
250
|
+
'<metadata xmlns:dc="http://purl.org/dc/elements/1.1/">'
|
|
251
|
+
f"{self._metadata_xml()}"
|
|
252
|
+
"</metadata>"
|
|
253
|
+
f"<manifest>{manifest}</manifest>"
|
|
254
|
+
f'<spine toc="ncx">{spine}</spine></package>'
|
|
255
|
+
)
|
|
256
|
+
|
|
257
|
+
def _ncx_xml(self) -> str:
|
|
258
|
+
nav = "".join(
|
|
259
|
+
f'<navPoint id="nav-{i:04d}" playOrder="{i}">'
|
|
260
|
+
f'<navLabel><text>{i}. Page {i}</text></navLabel>'
|
|
261
|
+
f'<content src="page-{i:04d}.xhtml"/>'
|
|
262
|
+
"</navPoint>"
|
|
263
|
+
for i in range(1, len(self.image_names) + 1)
|
|
264
|
+
)
|
|
265
|
+
return (
|
|
266
|
+
'<?xml version="1.0" encoding="UTF-8"?>\n'
|
|
267
|
+
'<ncx xmlns="http://www.daisy.org/z3986/2005/ncx/" version="2005-1">'
|
|
268
|
+
f'<head><meta name="dtb:uid" content="{self.uuid}"/></head>'
|
|
269
|
+
f"<docTitle><text>{escape(self.metadata.get('title', 'Document'))}</text></docTitle>"
|
|
270
|
+
f"<navMap>{nav}</navMap></ncx>"
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
def _section_xhtml(self, index: int, name: str) -> str:
|
|
274
|
+
css = "html,body{margin:0;padding:0;}img{display:block;max-width:100vw;margin:auto;}"
|
|
275
|
+
return (
|
|
276
|
+
'<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.1//EN" '
|
|
277
|
+
'"http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd">\n'
|
|
278
|
+
'<html xmlns="http://www.w3.org/1999/xhtml" lang="en" '
|
|
279
|
+
'xmlns:epub="http://www.idpf.org/2007/ops">'
|
|
280
|
+
f"<head><title>Page {index}</title>"
|
|
281
|
+
f'<style type="text/css">{css}</style></head>'
|
|
282
|
+
'<body epub:type="bodymatter">'
|
|
283
|
+
f'<p style="margin:0;padding:0;border:none;">\u200b</p>'
|
|
284
|
+
f'<img src="images/{escape(name)}" alt="Page {index}"/>'
|
|
285
|
+
"</body></html>"
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
def build(self) -> None:
|
|
289
|
+
assert len(self.image_names) == len(self.images)
|
|
290
|
+
buf = BytesIO()
|
|
291
|
+
with ZipFile(buf, "w", ZIP_DEFLATED) as z:
|
|
292
|
+
z.writestr("mimetype", "application/epub+zip", compress_type=ZIP_STORED)
|
|
293
|
+
z.writestr("META-INF/container.xml", self._container_xml())
|
|
294
|
+
z.writestr("OEBPS/content.opf", self._opf_xml())
|
|
295
|
+
z.writestr("OEBPS/toc.ncx", self._ncx_xml())
|
|
296
|
+
for i, name in enumerate(self.image_names, start=1):
|
|
297
|
+
z.writestr(f"OEBPS/page-{i:04d}.xhtml", self._section_xhtml(i, name))
|
|
298
|
+
for name, data in zip(self.image_names, self.images):
|
|
299
|
+
z.writestr(f"OEBPS/images/{name}", data)
|
|
300
|
+
|
|
301
|
+
self._tmp.write_bytes(buf.getvalue())
|
|
302
|
+
self._tmp.replace(self.path)
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def format_size(num_bytes: int) -> str:
|
|
306
|
+
size = float(num_bytes)
|
|
307
|
+
for unit in ("B", "KB", "MB", "GB", "TB"):
|
|
308
|
+
if size < 1024 or unit == "TB":
|
|
309
|
+
if unit == "B":
|
|
310
|
+
return f"{int(size)} {unit}"
|
|
311
|
+
return f"{size:.1f} {unit}"
|
|
312
|
+
size /= 1024
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def convert(input_path: Path, output_path: Path | None = None,
|
|
316
|
+
resolution: int | None = None, crop: str | None = None,
|
|
317
|
+
log=None, progress=None) -> Path:
|
|
318
|
+
"""Convert *input_path* (a PDF) to an EPUB and return the output path.
|
|
319
|
+
|
|
320
|
+
Every PDF page is rendered to a PNG and placed in its own EPUB section in
|
|
321
|
+
document order. The output file defaults to the input name with the
|
|
322
|
+
extension swapped to ``.epub``. *resolution*, when given, is the target
|
|
323
|
+
render width in pixels. *crop*, when given, trims white margins:
|
|
324
|
+
``"global"`` uses one common inset for all pages, ``"page"`` trims each
|
|
325
|
+
page to its own content.
|
|
326
|
+
|
|
327
|
+
*log* (optional callable(str)) receives descriptive lines (input info,
|
|
328
|
+
output result). *progress* (optional callable(done, total)) is called for
|
|
329
|
+
every rendered page.
|
|
330
|
+
"""
|
|
331
|
+
if crop is not None and crop not in ("global", "page"):
|
|
332
|
+
raise ConversionError(f"crop must be 'global' or 'page', got {crop!r}")
|
|
333
|
+
|
|
334
|
+
def _log(message: str) -> None:
|
|
335
|
+
if log is not None:
|
|
336
|
+
log(message)
|
|
337
|
+
|
|
338
|
+
def _progress(done: int, total: int) -> None:
|
|
339
|
+
if progress is not None:
|
|
340
|
+
progress(done, total)
|
|
341
|
+
|
|
342
|
+
started = time.monotonic()
|
|
343
|
+
|
|
344
|
+
input_path = Path(input_path)
|
|
345
|
+
if not input_path.exists():
|
|
346
|
+
raise ConversionError(f"input file not found: {input_path}")
|
|
347
|
+
if output_path is None:
|
|
348
|
+
output_path = input_path.with_suffix(".epub")
|
|
349
|
+
output_path = Path(output_path)
|
|
350
|
+
|
|
351
|
+
input_size = input_path.stat().st_size
|
|
352
|
+
_log(f"input: {input_path} ({format_size(input_size)})")
|
|
353
|
+
|
|
354
|
+
try:
|
|
355
|
+
doc = pymupdf.open(str(input_path))
|
|
356
|
+
except Exception as exc:
|
|
357
|
+
raise ConversionError(f"could not open PDF: {exc}") from exc
|
|
358
|
+
|
|
359
|
+
try:
|
|
360
|
+
page_count = doc.page_count
|
|
361
|
+
if page_count == 0:
|
|
362
|
+
raise ConversionError(f"PDF has no pages: {input_path}")
|
|
363
|
+
metadata = epub_metadata(doc)
|
|
364
|
+
metadata.setdefault("title", slugify(input_path.stem))
|
|
365
|
+
pages_list = [doc.load_page(i) for i in range(page_count)]
|
|
366
|
+
|
|
367
|
+
first = pages_list[0]
|
|
368
|
+
w, h = first.rect.width, first.rect.height
|
|
369
|
+
uniform = all(
|
|
370
|
+
p.rect.width == w and p.rect.height == h for p in pages_list[1:]
|
|
371
|
+
)
|
|
372
|
+
native = f"{w:.0f} x {h:.0f} px" + ("" if uniform else " (varies)")
|
|
373
|
+
_log(f"pages: {page_count} (native resolution: {native})")
|
|
374
|
+
if crop == "global":
|
|
375
|
+
clips = global_clip_rects(doc)
|
|
376
|
+
elif crop == "page":
|
|
377
|
+
clips = [page_clip_rect(p) for p in pages_list]
|
|
378
|
+
else:
|
|
379
|
+
clips = [None] * page_count
|
|
380
|
+
if resolution is not None:
|
|
381
|
+
_log(f"output: {output_path} ({resolution} px wide)")
|
|
382
|
+
else:
|
|
383
|
+
_log(f"output: {output_path} (native resolution)")
|
|
384
|
+
if crop is not None:
|
|
385
|
+
trimmed = sum(
|
|
386
|
+
1 for p, c in zip(pages_list, clips) if c is not None and c != p.rect
|
|
387
|
+
)
|
|
388
|
+
_log(f"crop: {crop} ({trimmed} of {page_count} pages trimmed)")
|
|
389
|
+
|
|
390
|
+
rendered: list[tuple[str, bytes]] = []
|
|
391
|
+
for i, (page, clip_raw) in enumerate(zip(pages_list, clips)):
|
|
392
|
+
clip = None if (clip_raw is None or clip_raw == page.rect) else clip_raw
|
|
393
|
+
png = render_page(page, resolution, clip)
|
|
394
|
+
rendered.append((f"page-{i + 1:04d}.png", png))
|
|
395
|
+
_progress(i + 1, page_count)
|
|
396
|
+
finally:
|
|
397
|
+
doc.close()
|
|
398
|
+
|
|
399
|
+
pages = rendered
|
|
400
|
+
|
|
401
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
402
|
+
writer = _EpubWriter(output_path, metadata)
|
|
403
|
+
for name, data in pages:
|
|
404
|
+
writer.add_image(name, data)
|
|
405
|
+
writer.build()
|
|
406
|
+
|
|
407
|
+
_log(
|
|
408
|
+
f"wrote: {output_path} "
|
|
409
|
+
f"({format_size(output_path.stat().st_size)}, "
|
|
410
|
+
f"{page_count} pages, {time.monotonic() - started:.1f}s)"
|
|
411
|
+
)
|
|
412
|
+
return output_path
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: epub-pdf-wrap
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: Convert a PDF to an EPUB by wrapping each rendered page in an EPUB page
|
|
5
|
+
Author-email: Andrea Esuli <andrea@esuli.it>
|
|
6
|
+
License-Expression: BSD-3-Clause
|
|
7
|
+
Project-URL: Home, https://github.com/aesuli/epub_pdf_wrap
|
|
8
|
+
Keywords: pdf,epub,conversion,render
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Environment :: Console
|
|
11
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Text Processing :: Markup :: HTML
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: PyMuPDF>=1.24
|
|
19
|
+
Provides-Extra: dev
|
|
20
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
21
|
+
Requires-Dist: ebooklib>=0.18; extra == "dev"
|
|
22
|
+
Dynamic: license-file
|
|
23
|
+
|
|
24
|
+
# EPUB PDF wrap
|
|
25
|
+
|
|
26
|
+
Converts a PDF into an EPUB by rendering and wrapping each page of the PDF in
|
|
27
|
+
an EPUB page. This is specifically aimed at PDF files that cannot be converted
|
|
28
|
+
into EPUB any other way without corrupting their visual rendering (comics,
|
|
29
|
+
scientific papers...).
|
|
30
|
+
|
|
31
|
+
## Installation
|
|
32
|
+
|
|
33
|
+
```
|
|
34
|
+
pip install epub-pdf-wrap
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
From source (with dev dependencies for testing):
|
|
38
|
+
|
|
39
|
+
```
|
|
40
|
+
pip install -e ".[dev]"
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## Usage
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
epub-pdf-wrap <pdf-filename> [-o <epub-filename>]
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
By default the output filename is the input filename with the `pdf` extension
|
|
50
|
+
replaced by the `epub` extension. Running without `pip install` also works via
|
|
51
|
+
`python -m epub_pdf_wrap`.
|
|
52
|
+
|
|
53
|
+
### Options
|
|
54
|
+
|
|
55
|
+
- `-r <num>, --resolution <num>`: target render width in pixels for the pages
|
|
56
|
+
in the output file. By default the pages are rendered at the resolution the
|
|
57
|
+
PDF itself declares.
|
|
58
|
+
- `-c, --crop-global`: trim the white margins around the page content using
|
|
59
|
+
one common inset for all pages (safe: never clips content on any page, all
|
|
60
|
+
pages keep the same size).
|
|
61
|
+
- `--crop-page`: trim the white margins around each page's own content, page
|
|
62
|
+
by page (trims more aggressively but page sizes may vary).
|
|
63
|
+
|
|
64
|
+
`-c/--crop-global` and `--crop-page` are mutually exclusive; with neither
|
|
65
|
+
flag the margins are left as-is.
|
|
66
|
+
|
|
67
|
+
## Metadata
|
|
68
|
+
|
|
69
|
+
Document metadata (title, author, subject, keywords and creation date) is
|
|
70
|
+
taken from the PDF and written into the EPUB. Empty fields are omitted; the
|
|
71
|
+
title falls back to the input filename if the PDF has none.
|
|
72
|
+
|
|
73
|
+
## Examples
|
|
74
|
+
|
|
75
|
+
Convert a paper at a wider resolution and name the output explicitly:
|
|
76
|
+
|
|
77
|
+
```
|
|
78
|
+
epub-pdf-wrap paper.pdf -o paper.epub -r 1400
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Development
|
|
82
|
+
|
|
83
|
+
```
|
|
84
|
+
python -m venv .venv
|
|
85
|
+
.venv\Scripts\activate # Windows (or `source .venv/bin/activate` elsewhere)
|
|
86
|
+
pip install -e ".[dev]"
|
|
87
|
+
pytest
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
`samples/` contains real input PDFs for manually verifying the output.
|
|
91
|
+
|
|
92
|
+
## License
|
|
93
|
+
|
|
94
|
+
Distributed under the BSD 3-Clause License; see `LICENSE`.
|
|
95
|
+
Copyright (c) 2026, Andrea Esuli (andrea@esuli.it).
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
epub_pdf_wrap/__init__.py
|
|
5
|
+
epub_pdf_wrap/__main__.py
|
|
6
|
+
epub_pdf_wrap/core.py
|
|
7
|
+
epub_pdf_wrap.egg-info/PKG-INFO
|
|
8
|
+
epub_pdf_wrap.egg-info/SOURCES.txt
|
|
9
|
+
epub_pdf_wrap.egg-info/dependency_links.txt
|
|
10
|
+
epub_pdf_wrap.egg-info/entry_points.txt
|
|
11
|
+
epub_pdf_wrap.egg-info/requires.txt
|
|
12
|
+
epub_pdf_wrap.egg-info/top_level.txt
|
|
13
|
+
tests/test_core.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
epub_pdf_wrap
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "epub-pdf-wrap"
|
|
7
|
+
version = "0.1.1"
|
|
8
|
+
description = "Convert a PDF to an EPUB by wrapping each rendered page in an EPUB page"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "BSD-3-Clause"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [
|
|
14
|
+
{ name = "Andrea Esuli", email = "andrea@esuli.it" },
|
|
15
|
+
]
|
|
16
|
+
urls = { "Home" = "https://github.com/aesuli/epub_pdf_wrap" }
|
|
17
|
+
keywords = ["pdf", "epub", "conversion", "render"]
|
|
18
|
+
classifiers = [
|
|
19
|
+
"Development Status :: 3 - Alpha",
|
|
20
|
+
"Environment :: Console",
|
|
21
|
+
"Intended Audience :: End Users/Desktop",
|
|
22
|
+
"Operating System :: OS Independent",
|
|
23
|
+
"Programming Language :: Python :: 3",
|
|
24
|
+
"Topic :: Text Processing :: Markup :: HTML",
|
|
25
|
+
]
|
|
26
|
+
dependencies = [
|
|
27
|
+
"PyMuPDF>=1.24",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
[project.optional-dependencies]
|
|
31
|
+
dev = [
|
|
32
|
+
"pytest>=7",
|
|
33
|
+
"ebooklib>=0.18",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
[project.scripts]
|
|
37
|
+
epub-pdf-wrap = "epub_pdf_wrap.__main__:main"
|
|
38
|
+
|
|
39
|
+
[tool.setuptools]
|
|
40
|
+
packages = ["epub_pdf_wrap"]
|
|
@@ -0,0 +1,345 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
import pymupdf
|
|
6
|
+
import pytest
|
|
7
|
+
|
|
8
|
+
from epub_pdf_wrap.core import ConversionError, convert, format_size, page_clip_rect, render_page, slugify
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@pytest.fixture
|
|
12
|
+
def tiny_pdf(tmp_path: Path) -> Path:
|
|
13
|
+
"""Three-page PDF: text page, white-blank page is omitted; use text + image pages."""
|
|
14
|
+
doc = pymupdf.open()
|
|
15
|
+
page = doc.new_page(width=200, height=260)
|
|
16
|
+
page.insert_text((20, 40), "First page")
|
|
17
|
+
page = doc.new_page(width=200, height=260)
|
|
18
|
+
page.insert_text((20, 40), "Second page")
|
|
19
|
+
out = tmp_path / "tiny.pdf"
|
|
20
|
+
doc.save(str(out))
|
|
21
|
+
doc.close()
|
|
22
|
+
return out
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def test_convert_produces_epub(tmp_path: Path, tiny_pdf: Path) -> None:
|
|
26
|
+
out = convert(tiny_pdf, tmp_path / "out.epub")
|
|
27
|
+
assert out.exists()
|
|
28
|
+
import zipfile
|
|
29
|
+
|
|
30
|
+
with zipfile.ZipFile(out) as z:
|
|
31
|
+
names = z.namelist()
|
|
32
|
+
assert names[0] == "mimetype"
|
|
33
|
+
assert z.read("mimetype") == b"application/epub+zip"
|
|
34
|
+
assert "OEBPS/images/page-0001.png" in names
|
|
35
|
+
assert "OEBPS/images/page-0002.png" in names
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def test_epub_is_readable_by_ebooklib(tmp_path: Path, tiny_pdf: Path) -> None:
|
|
39
|
+
"""Reader-grade check: ebooklib must parse metadata, spine and content."""
|
|
40
|
+
import ebooklib
|
|
41
|
+
from ebooklib import epub
|
|
42
|
+
|
|
43
|
+
out = convert(tiny_pdf, tmp_path / "reader.epub")
|
|
44
|
+
book = epub.read_epub(str(out), options={"ignore_ncx": True})
|
|
45
|
+
|
|
46
|
+
# Metadata
|
|
47
|
+
titles = book.get_metadata("DC", "title")
|
|
48
|
+
assert titles and titles[0][0] == "tiny"
|
|
49
|
+
|
|
50
|
+
# Spine (reading order) must resolve to real items
|
|
51
|
+
spine_refs = [ref for ref, _ in book.spine]
|
|
52
|
+
assert spine_refs == ["sec-0000", "sec-0001"]
|
|
53
|
+
for ref in spine_refs:
|
|
54
|
+
item = book.get_item_with_id(ref)
|
|
55
|
+
assert item is not None
|
|
56
|
+
assert "<img" in item.get_content().decode("utf-8")
|
|
57
|
+
|
|
58
|
+
# Every spine image must be present
|
|
59
|
+
images = {i.get_name() for i in book.get_items_of_type(ebooklib.ITEM_IMAGE)}
|
|
60
|
+
assert images == {"images/page-0001.png", "images/page-0002.png"}
|
|
61
|
+
|
|
62
|
+
# A default load must also succeed without the ignore_ncx bypass:
|
|
63
|
+
# the NCX toc id has to resolve in the manifest (id="ncx").
|
|
64
|
+
fresh = epub.read_epub(str(out))
|
|
65
|
+
assert len(list(fresh.get_items_of_type(ebooklib.ITEM_DOCUMENT))) == 2
|
|
66
|
+
# Navigation must provide one TOC entry per page, each resolving to a
|
|
67
|
+
# real item (this is what readers do when a user opens the TOC).
|
|
68
|
+
toc = list(fresh.toc)
|
|
69
|
+
assert len(toc) == 2
|
|
70
|
+
for link in toc:
|
|
71
|
+
item = fresh.get_item_with_href(link.href)
|
|
72
|
+
assert item is not None
|
|
73
|
+
assert "<img" in item.get_content().decode("utf-8")
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def test_convert_default_output_name(tiny_pdf: Path) -> None:
|
|
77
|
+
out = convert(tiny_pdf)
|
|
78
|
+
assert out.name == "tiny.epub"
|
|
79
|
+
out.unlink()
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def test_resolution_flag(tmp_path: Path, tiny_pdf: Path) -> None:
|
|
83
|
+
out = convert(tiny_pdf, tmp_path / "res.epub", resolution=600)
|
|
84
|
+
import zipfile
|
|
85
|
+
|
|
86
|
+
with zipfile.ZipFile(out) as z:
|
|
87
|
+
png = z.read("OEBPS/images/page-0001.png")
|
|
88
|
+
import struct
|
|
89
|
+
|
|
90
|
+
# IHDR: width at bytes 16..20
|
|
91
|
+
width = struct.unpack(">I", png[16:20])[0]
|
|
92
|
+
assert width == 600
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def test_blank_only_pdf_still_produces_epub(tmp_path: Path) -> None:
|
|
96
|
+
doc = pymupdf.open()
|
|
97
|
+
doc.new_page(width=200, height=260)
|
|
98
|
+
blank = tmp_path / "blank.pdf"
|
|
99
|
+
doc.save(str(blank))
|
|
100
|
+
doc.close()
|
|
101
|
+
|
|
102
|
+
out = convert(blank, tmp_path / "blank.epub")
|
|
103
|
+
assert out.exists()
|
|
104
|
+
import zipfile
|
|
105
|
+
|
|
106
|
+
with zipfile.ZipFile(out) as z:
|
|
107
|
+
assert "OEBPS/images/page-0001.png" in z.namelist()
|
|
108
|
+
out.unlink()
|
|
109
|
+
blank.unlink()
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def test_missing_input_raises(tmp_path: Path) -> None:
|
|
113
|
+
with pytest.raises(ConversionError):
|
|
114
|
+
convert(tmp_path / "nope.pdf", tmp_path / "nope.epub")
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def test_bad_resolution_raises(tiny_pdf: Path) -> None:
|
|
118
|
+
doc = pymupdf.open(str(tiny_pdf))
|
|
119
|
+
page = doc.load_page(0)
|
|
120
|
+
try:
|
|
121
|
+
with pytest.raises(ConversionError):
|
|
122
|
+
render_page(page, resolution=0)
|
|
123
|
+
with pytest.raises(ConversionError):
|
|
124
|
+
render_page(page, resolution=-5)
|
|
125
|
+
finally:
|
|
126
|
+
doc.close()
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def test_convert_reports_input_info_and_progress(tmp_path: Path, tiny_pdf: Path) -> None:
|
|
130
|
+
logged: list[str] = []
|
|
131
|
+
steps: list[tuple[int, int]] = []
|
|
132
|
+
out = convert(
|
|
133
|
+
tiny_pdf,
|
|
134
|
+
tmp_path / "report.epub",
|
|
135
|
+
log=logged.append,
|
|
136
|
+
progress=lambda done, total: steps.append((done, total)),
|
|
137
|
+
)
|
|
138
|
+
joined = "\n".join(logged)
|
|
139
|
+
assert str(tiny_pdf) in joined
|
|
140
|
+
assert "pages: 2" in joined
|
|
141
|
+
assert str(out) in joined
|
|
142
|
+
assert "2 pages" in joined
|
|
143
|
+
assert steps[-1] == (2, 2)
|
|
144
|
+
assert len(steps) == 2
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def test_slugify() -> None:
|
|
148
|
+
assert slugify("Some Comic 42") == "Some-Comic-42"
|
|
149
|
+
assert slugify("2406.12128v2") == "2406-12128v2"
|
|
150
|
+
assert slugify(" ") == "document"
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def test_format_size() -> None:
|
|
154
|
+
assert format_size(0) == "0 B"
|
|
155
|
+
assert format_size(999) == "999 B"
|
|
156
|
+
assert format_size(1024) == "1.0 KB"
|
|
157
|
+
assert format_size(2048) == "2.0 KB"
|
|
158
|
+
assert format_size(1536) == "1.5 KB"
|
|
159
|
+
assert format_size(1536 * 1024) == "1.5 MB"
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _png_size(png: bytes) -> tuple[int, int]:
|
|
163
|
+
import struct
|
|
164
|
+
|
|
165
|
+
return struct.unpack(">II", png[16:24])
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
@pytest.fixture
|
|
169
|
+
def inset_pdf(tmp_path: Path) -> Path:
|
|
170
|
+
"""Two pages with content inset in the middle (real margins around it).
|
|
171
|
+
|
|
172
|
+
Page 1: content from (50, 50) to (150, 210).
|
|
173
|
+
Page 2: content from (30, 30) to (170, 230) — wider than page 1, so the
|
|
174
|
+
global (union) clip keeps the wider extent on that side.
|
|
175
|
+
"""
|
|
176
|
+
doc = pymupdf.open()
|
|
177
|
+
p = doc.new_page(width=200, height=260)
|
|
178
|
+
p.insert_text((50, 80), "page one")
|
|
179
|
+
p.draw_rect(pymupdf.Rect(50, 100, 150, 200))
|
|
180
|
+
p = doc.new_page(width=200, height=260)
|
|
181
|
+
p.insert_text((30, 60), "page two")
|
|
182
|
+
p.draw_rect(pymupdf.Rect(30, 80, 170, 220))
|
|
183
|
+
out = tmp_path / "inset.pdf"
|
|
184
|
+
doc.save(str(out))
|
|
185
|
+
doc.close()
|
|
186
|
+
return out
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def test_crop_page_trims_to_content(tmp_path: Path, inset_pdf: Path) -> None:
|
|
190
|
+
out = convert(inset_pdf, tmp_path / "crop.epub", crop="page")
|
|
191
|
+
|
|
192
|
+
# Compare against the no-crop render of the same PDF
|
|
193
|
+
nocrop = convert(inset_pdf, tmp_path / "nocrop.epub")
|
|
194
|
+
import zipfile
|
|
195
|
+
|
|
196
|
+
with zipfile.ZipFile(out) as z:
|
|
197
|
+
cropped = _png_size(z.read("OEBPS/images/page-0001.png"))
|
|
198
|
+
with zipfile.ZipFile(nocrop) as z:
|
|
199
|
+
full = _png_size(z.read("OEBPS/images/page-0001.png"))
|
|
200
|
+
assert full == (200, 260)
|
|
201
|
+
# Page 1 content spans roughly 50..150 wide, 70..200 tall: the crop must
|
|
202
|
+
# clearly remove margins on all four sides.
|
|
203
|
+
assert cropped[0] < full[0] * 0.7
|
|
204
|
+
assert cropped[1] < full[1] * 0.7
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def test_crop_global_uniform_but_safe(tmp_path: Path, inset_pdf: Path) -> None:
|
|
208
|
+
out = convert(inset_pdf, tmp_path / "g.epub", crop="global")
|
|
209
|
+
import zipfile
|
|
210
|
+
|
|
211
|
+
with zipfile.ZipFile(out) as z:
|
|
212
|
+
s1 = _png_size(z.read("OEBPS/images/page-0001.png"))
|
|
213
|
+
s2 = _png_size(z.read("OEBPS/images/page-0002.png"))
|
|
214
|
+
# Union of content boxes is the same size for both pages (each clipped
|
|
215
|
+
# to the same global box), so both page images have equal dimensions.
|
|
216
|
+
assert s1 == s2
|
|
217
|
+
# The global box must never clip content: page 2, which extends
|
|
218
|
+
# furthest (30..170 wide), must not have been cut.
|
|
219
|
+
assert s1[0] >= 170 - 30
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def test_full_bleed_crop_unchanged(tmp_path: Path) -> None:
|
|
223
|
+
doc = pymupdf.open()
|
|
224
|
+
p = doc.new_page(width=200, height=260)
|
|
225
|
+
p.draw_rect(pymupdf.Rect(0, 0, 200, 260)) # content touches all edges
|
|
226
|
+
full = tmp_path / "full.pdf"
|
|
227
|
+
doc.save(str(full))
|
|
228
|
+
doc.close()
|
|
229
|
+
|
|
230
|
+
a = convert(full, tmp_path / "a.epub")
|
|
231
|
+
b = convert(full, tmp_path / "b.epub", crop="global")
|
|
232
|
+
c = convert(full, tmp_path / "c.epub", crop="page")
|
|
233
|
+
import zipfile
|
|
234
|
+
|
|
235
|
+
with zipfile.ZipFile(a) as z:
|
|
236
|
+
base = z.read("OEBPS/images/page-0001.png")
|
|
237
|
+
with zipfile.ZipFile(b) as z:
|
|
238
|
+
assert z.read("OEBPS/images/page-0001.png") == base
|
|
239
|
+
with zipfile.ZipFile(c) as z:
|
|
240
|
+
assert z.read("OEBPS/images/page-0001.png") == base
|
|
241
|
+
full.unlink()
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def test_blank_page_crop_does_not_crash(tmp_path: Path) -> None:
|
|
245
|
+
doc = pymupdf.open()
|
|
246
|
+
doc.new_page(width=200, height=260)
|
|
247
|
+
blank = tmp_path / "blank.pdf"
|
|
248
|
+
doc.save(str(blank))
|
|
249
|
+
doc.close()
|
|
250
|
+
|
|
251
|
+
a = convert(blank, tmp_path / "gb.epub", crop="global")
|
|
252
|
+
b = convert(blank, tmp_path / "pb.epub", crop="page")
|
|
253
|
+
import zipfile
|
|
254
|
+
|
|
255
|
+
for f in (a, b):
|
|
256
|
+
with zipfile.ZipFile(f) as z:
|
|
257
|
+
assert _png_size(z.read("OEBPS/images/page-0001.png")) == (200, 260)
|
|
258
|
+
blank.unlink()
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def test_crop_invalid_value_raises(inset_pdf: Path) -> None:
|
|
262
|
+
with pytest.raises(ConversionError):
|
|
263
|
+
convert(inset_pdf, inset_pdf.with_suffix(".epub"), crop="bogus")
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def test_cli_parser_crop_flags() -> None:
|
|
267
|
+
from epub_pdf_wrap.__main__ import build_parser
|
|
268
|
+
|
|
269
|
+
p = build_parser()
|
|
270
|
+
args = p.parse_args(["in.pdf", "-c"])
|
|
271
|
+
assert args.crop_global is True and args.crop_page is False
|
|
272
|
+
args = p.parse_args(["in.pdf", "--crop-global"])
|
|
273
|
+
assert args.crop_global is True
|
|
274
|
+
args = p.parse_args(["in.pdf", "--crop-page"])
|
|
275
|
+
assert args.crop_global is False and args.crop_page is True
|
|
276
|
+
with pytest.raises(SystemExit) as exc:
|
|
277
|
+
p.parse_args(["in.pdf", "-c", "--crop-page"])
|
|
278
|
+
assert exc.value.code == 2
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
@pytest.fixture
|
|
282
|
+
def metadata_pdf(tmp_path: Path) -> Path:
|
|
283
|
+
doc = pymupdf.open()
|
|
284
|
+
page = doc.new_page(width=200, height=260)
|
|
285
|
+
page.insert_text((20, 40), "content")
|
|
286
|
+
doc.set_metadata(
|
|
287
|
+
{
|
|
288
|
+
"title": "A <Great> Book & Co",
|
|
289
|
+
"author": "Jane Doe, John Roe",
|
|
290
|
+
"subject": "Typesetting with pypdf",
|
|
291
|
+
"keywords": "epub, pdf, wrap",
|
|
292
|
+
"creationDate": "D:20041212120000+01'00'",
|
|
293
|
+
}
|
|
294
|
+
)
|
|
295
|
+
out = tmp_path / "meta.pdf"
|
|
296
|
+
doc.save(str(out))
|
|
297
|
+
doc.close()
|
|
298
|
+
return out
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def test_metadata_is_transferred_to_epub(tmp_path: Path, metadata_pdf: Path) -> None:
|
|
302
|
+
import ebooklib
|
|
303
|
+
from ebooklib import epub
|
|
304
|
+
|
|
305
|
+
out = convert(metadata_pdf, tmp_path / "meta.epub")
|
|
306
|
+
book = epub.read_epub(str(out), options={"ignore_ncx": True})
|
|
307
|
+
|
|
308
|
+
assert book.get_metadata("DC", "title")[0][0] == "A <Great> Book & Co"
|
|
309
|
+
assert book.get_metadata("DC", "creator")[0][0] == "Jane Doe, John Roe"
|
|
310
|
+
assert book.get_metadata("DC", "subject")[0][0] == "Typesetting with pypdf"
|
|
311
|
+
assert book.get_metadata("DC", "date")[0][0] == "2004-12-12T12:00:00"
|
|
312
|
+
# Tooling fields are not transferred
|
|
313
|
+
assert not book.get_metadata("DC", "source")
|
|
314
|
+
|
|
315
|
+
# The OPF XML itself: identifier still present, values escaped in title,
|
|
316
|
+
# keywords present as an EPUB-3 meta element.
|
|
317
|
+
from zipfile import ZipFile
|
|
318
|
+
|
|
319
|
+
with ZipFile(out) as z:
|
|
320
|
+
opf = z.read("OEBPS/content.opf").decode("utf-8")
|
|
321
|
+
assert "A <Great> Book & Co" in opf
|
|
322
|
+
assert '<dc:identifier id="epubid"' in opf
|
|
323
|
+
assert '<meta name="keywords" content="epub, pdf, wrap"/>' in opf
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def test_metadata_fallback_title_when_empty(tmp_path: Path) -> None:
|
|
327
|
+
import pymupdf
|
|
328
|
+
|
|
329
|
+
doc = pymupdf.open()
|
|
330
|
+
doc.new_page(width=200, height=260)
|
|
331
|
+
blank = tmp_path / "no-meta.pdf"
|
|
332
|
+
doc.save(str(blank))
|
|
333
|
+
doc.close()
|
|
334
|
+
|
|
335
|
+
out = convert(blank, tmp_path / "no-meta.epub")
|
|
336
|
+
import zipfile
|
|
337
|
+
|
|
338
|
+
with zipfile.ZipFile(out) as z:
|
|
339
|
+
opf = z.read("OEBPS/content.opf").decode("utf-8")
|
|
340
|
+
assert "<dc:title>no-meta</dc:title>" in opf
|
|
341
|
+
# Only empty fields are omitted, not the required title
|
|
342
|
+
assert "<dc:creator>" not in opf
|
|
343
|
+
assert "<dc:subject>" not in opf
|
|
344
|
+
assert "<dc:date>" not in opf
|
|
345
|
+
blank.unlink()
|