ibook2epub 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ """Convert Apple iBooks epub packages into spec-valid epub files."""
2
+
3
+ __version__ = "2.0.0"
@@ -0,0 +1,8 @@
1
+ """Entry point for ``python -m epubconvert``."""
2
+
3
+ import sys
4
+
5
+ from .run import main
6
+
7
+ if __name__ == "__main__":
8
+ sys.exit(main())
@@ -0,0 +1,128 @@
1
+ """
2
+ Logging setup for the epub conversion tool.
3
+
4
+ Importing this module has no side effects beyond registering a custom TRACE
5
+ level and creating a package logger with a null handler. Handlers are only
6
+ attached when :func:`configure` is called, which keeps the library importable
7
+ from tests without spraying an ``app.log`` into the current directory.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import logging
13
+ from pathlib import Path
14
+ from typing import Any, cast
15
+
16
+ TRACE = 5 # Below DEBUG (10), for very chatty per-file messages.
17
+ logging.addLevelName(TRACE, "TRACE")
18
+
19
+
20
+ class TraceLogger(logging.Logger):
21
+ """A logger with an extra TRACE level below DEBUG."""
22
+
23
+ def trace(self, message: str, *args: Any, **kwargs: Any) -> None:
24
+ """Log a message at the custom TRACE level."""
25
+ if self.isEnabledFor(TRACE):
26
+ self._log(TRACE, message, args, **kwargs)
27
+
28
+
29
+ # Register the subclass only for the duration of our own getLogger call, so
30
+ # that loggers created elsewhere in the process are left alone.
31
+ _previous_class = logging.getLoggerClass()
32
+ logging.setLoggerClass(TraceLogger)
33
+ logger = cast(TraceLogger, logging.getLogger("epubconvert"))
34
+ logging.setLoggerClass(_previous_class)
35
+
36
+ logger.addHandler(logging.NullHandler())
37
+ logger.propagate = False
38
+
39
+ # verbosity 0 = -q, 1 = default, 2 = -v, 3 = -vv (and above)
40
+ _LEVELS = (logging.WARNING, logging.INFO, logging.DEBUG, TRACE)
41
+
42
+ #: A terminal is not a transcript. At default verbosity the message is the
43
+ #: whole point, and a timestamp on every line of an interactive run is noise
44
+ #: the file format already carries for the runs that need it.
45
+ _CONSOLE_PLAIN = "%(message)s"
46
+ _CONSOLE_VERBOSE = "%(asctime)s - %(levelname)s - %(message)s"
47
+ _FILE_FORMAT = "%(asctime)s %(levelname)s %(message)s"
48
+ # ISO 8601 with UTC offset: sorts lexicographically and is unambiguous across
49
+ # timezones, unlike a 12-hour local clock.
50
+ _FILE_DATEFMT = "%Y-%m-%dT%H:%M:%S%z"
51
+
52
+
53
+ def level_for_verbosity(verbosity: int) -> int:
54
+ """
55
+ Map a verbosity count onto a logging level.
56
+
57
+ :param verbosity: 0 quiet, 1 normal, 2 debug, 3+ trace.
58
+
59
+ :return: The corresponding logging level.
60
+ """
61
+ return _LEVELS[max(0, min(verbosity, len(_LEVELS) - 1))]
62
+
63
+
64
+ def file_only(message: str) -> None:
65
+ """
66
+ Record a line in the log file without repeating it on the console.
67
+
68
+ The run summary is printed to stdout for the person watching, and also
69
+ logged so a ``--log-file`` transcript of an interrupted run is not
70
+ indistinguishable from a complete one. Logging it plainly put it on the
71
+ terminal twice.
72
+
73
+ :param message: The line to record.
74
+ """
75
+ for handler in logger.handlers:
76
+ if isinstance(handler, logging.FileHandler):
77
+ handler.handle(
78
+ logger.makeRecord(
79
+ logger.name, logging.INFO, __file__, 0, "%s", (message,), None
80
+ )
81
+ )
82
+
83
+
84
+ def configure(verbosity: int = 1, log_file: Path | None = None) -> TraceLogger:
85
+ """
86
+ Attach handlers to the package logger.
87
+
88
+ Calling this more than once replaces the previously attached handlers, so
89
+ it is safe to use from tests.
90
+
91
+ :param verbosity: 0 quiet, 1 normal, 2 debug, 3+ trace.
92
+ :param log_file: Optional path to also write log records to.
93
+
94
+ :return: The configured package logger.
95
+ """
96
+ for handler in list(logger.handlers):
97
+ logger.removeHandler(handler)
98
+ handler.close()
99
+
100
+ level = level_for_verbosity(verbosity)
101
+ logger.setLevel(level)
102
+
103
+ console_handler = logging.StreamHandler()
104
+ console_handler.setLevel(level)
105
+ console_handler.setFormatter(
106
+ logging.Formatter(_CONSOLE_VERBOSE if verbosity > 1 else _CONSOLE_PLAIN)
107
+ )
108
+ logger.addHandler(console_handler)
109
+
110
+ if log_file is not None:
111
+ # The console handler is already attached, so a failure here can be
112
+ # reported rather than crashing the process. This runs before anything
113
+ # else in main, so an unwritable --log-file used to exit with a raw
114
+ # traceback before any logging existed to explain it.
115
+ log_path = Path(log_file)
116
+ try:
117
+ log_path.parent.mkdir(parents=True, exist_ok=True)
118
+ file_handler = logging.FileHandler(log_path, encoding="utf-8")
119
+ except OSError as exc:
120
+ logger.warning("Not logging to %s: %s", log_path, exc)
121
+ else:
122
+ file_handler.setLevel(level)
123
+ file_handler.setFormatter(
124
+ logging.Formatter(_FILE_FORMAT, datefmt=_FILE_DATEFMT)
125
+ )
126
+ logger.addHandler(file_handler)
127
+
128
+ return logger
epubconvert/archive.py ADDED
@@ -0,0 +1,519 @@
1
+ """
2
+ Finding source packages, and writing one out as an epub archive.
3
+
4
+ The two halves of the mechanical work: locating the ``*.epub/`` directories
5
+ Apple leaves behind, and turning one of them into a zip archive the epub
6
+ specification accepts. Neither half knows anything about runs, reports or
7
+ concurrency -- :mod:`epubconvert.convert` supplies those.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import errno
13
+ import os
14
+ import shutil
15
+ import tempfile
16
+ from collections.abc import Sequence
17
+ from pathlib import Path, PurePosixPath
18
+ from zipfile import ZIP_DEFLATED, ZIP_STORED, ZipFile, ZipInfo
19
+
20
+ from .app_logger import logger
21
+ from .contained import contains, open_contained
22
+ from .display import printable
23
+ from .spec import CONTAINER_PATH, MIMETYPE_CONTENT, MIMETYPE_NAME, PACKAGE_SUFFIX
24
+ from .validate import ArchiveInvalidError, ValidationOptions
25
+
26
+ # Zip cannot represent a timestamp before 1980; using its floor keeps every
27
+ # export byte-identical regardless of when it ran.
28
+ ARCHIVE_TIMESTAMP = (1980, 1, 1, 0, 0, 0)
29
+
30
+ #: Marks a half-written archive. The prefix matters as much as the suffix:
31
+ #: the sweep in :func:`epubconvert.convert.sweep_partials` deletes what it
32
+ #: matches, and a bare ``*.part`` glob also matches a browser's in-progress
33
+ #: download or a user's own file sitting in the output directory.
34
+ PARTIAL_PREFIX = ".ibook2epub-"
35
+ #: Appended to every temporary this tool writes.
36
+ PARTIAL_SUFFIX = ".part"
37
+
38
+ #: Suffixes worth taking along verbatim. A real library holds both forms --
39
+ #: Apple's package directories, and books that arrived already zipped or as
40
+ #: PDFs -- and converting only the first produced a partial shelf whose
41
+ #: summary read complete.
42
+ COPYABLE_SUFFIXES = frozenset({".epub", ".pdf"})
43
+ #: Deflate level for text members. Measured on a 3 MB book: level 9 costs 3.2x
44
+ #: the CPU of level 6 for 0.6% less size, and bare zlib on the same text is
45
+ #: 4.8x for 1.6%. Level 6 is zlib's default and the level worth paying for.
46
+ #: This was inert until :func:`entry` assigned it -- a prebuilt ZipInfo makes
47
+ #: ``ZipFile(compresslevel=)`` a no-op.
48
+ COMPRESS_LEVEL = 6
49
+
50
+ #: Extensions whose bytes are already entropy-coded. Deflating them scans the
51
+ #: data to save nothing: measured 3.8x faster to store, for +0.02% size. The
52
+ #: rule is a pure function of the name, so exports stay byte-identical.
53
+ STORED_SUFFIXES = frozenset(
54
+ {
55
+ ".jpg",
56
+ ".jpeg",
57
+ ".png",
58
+ ".gif",
59
+ ".webp",
60
+ ".avif",
61
+ ".mp3",
62
+ ".m4a",
63
+ ".mp4",
64
+ ".m4v",
65
+ ".ogg",
66
+ ".opus",
67
+ ".woff",
68
+ ".woff2",
69
+ ".otf",
70
+ ".ttf",
71
+ ".zip",
72
+ ".gz",
73
+ }
74
+ )
75
+ #: The permission bits recorded *inside* the zip for each member. Fixed rather
76
+ #: than taken from the umask, because it is archive metadata and re-exports
77
+ #: must stay byte-identical. What the exported file itself gets is a separate
78
+ #: question, answered by :func:`_file_mode` from the user's umask.
79
+ ARCHIVE_MODE = 0o644
80
+
81
+ # Filesystem junk, never book content, so excluded wherever it appears.
82
+ EXCLUDED_ANYWHERE = frozenset({".DS_Store"})
83
+
84
+ # Apple bookkeeping, which only ever sits at the package root. These patterns
85
+ # must NOT be applied deeper: a chapter legitimately named ``bookmarks.xhtml``,
86
+ # a ``.plist`` data asset, or a file called ``mimetype`` inside ``OEBPS/`` are
87
+ # all real content, and dropping them corrupts the book. ``mimetype`` is listed
88
+ # because the root copy is rewritten separately, uncompressed and first, as the
89
+ # epub specification requires.
90
+ EXCLUDED_ROOT_NAMES = frozenset({MIMETYPE_NAME})
91
+ EXCLUDED_ROOT_SUFFIXES = frozenset({".plist"})
92
+ EXCLUDED_ROOT_PREFIXES = ("bookmarks",)
93
+
94
+
95
+ def is_excluded(name: str, *, at_root: bool) -> bool:
96
+ """
97
+ Report whether a package member should be left out of the epub.
98
+
99
+ The Apple bookkeeping patterns apply only at the package root. Applying
100
+ them at every depth silently drops real content — a chapter file named
101
+ ``bookmarks.xhtml`` or a ``.plist`` asset under ``OEBPS/`` — which
102
+ produces an archive that readers reject for a missing spine item.
103
+
104
+ :param name: The bare file name (not a path) to test.
105
+ :param at_root: Whether the file sits directly in the package directory.
106
+
107
+ :return: True if the file must not be copied into the archive.
108
+ """
109
+ if name in EXCLUDED_ANYWHERE:
110
+ return True
111
+ if not at_root:
112
+ return False
113
+ return (
114
+ name in EXCLUDED_ROOT_NAMES
115
+ or Path(name).suffix in EXCLUDED_ROOT_SUFFIXES
116
+ or name.startswith(EXCLUDED_ROOT_PREFIXES)
117
+ )
118
+
119
+
120
+ def collect_copyable(source_dir: Path) -> list[Path]:
121
+ """
122
+ Find files worth copying to the shelf unchanged.
123
+
124
+ A ``*.epub`` **file** rather than a directory is a book that arrived
125
+ already zipped; a ``*.pdf`` is a book this tool has nothing to do to. Both
126
+ belong on the shelf the run produces, and neither needs converting.
127
+
128
+ Only real files: the same trust rule every other reader here uses, so a
129
+ symlink out of the library is not followed.
130
+
131
+ :param source_dir: The directory to search.
132
+
133
+ :return: Files to copy, sorted by path.
134
+ """
135
+ found: list[Path] = []
136
+ resolved = source_dir.resolve()
137
+
138
+ def on_error(exc: OSError) -> None:
139
+ logger.warning("Could not scan %s: %s", exc.filename or source_dir, exc)
140
+
141
+ for root, dirs, files in os.walk(source_dir, onerror=on_error):
142
+ directory = Path(root)
143
+ # A package's own contents are never copy-through candidates.
144
+ dirs[:] = [name for name in dirs if not name.endswith(PACKAGE_SUFFIX)]
145
+ for name in files:
146
+ path = directory / name
147
+ if PurePosixPath(name).suffix.lower() not in COPYABLE_SUFFIXES:
148
+ continue
149
+ if not contains(source_dir, path, resolved_root=resolved):
150
+ logger.warning(
151
+ "Skipped symlink %s in %s", printable(name), source_dir.name
152
+ )
153
+ continue
154
+ found.append(path)
155
+
156
+ found.sort()
157
+ return found
158
+
159
+
160
+ def copy_through(source: Path, target: Path) -> None:
161
+ """
162
+ Put a file on the shelf without touching its bytes.
163
+
164
+ Written to a temporary and moved into place, like every other write here,
165
+ so an interrupted run never leaves a half-copied file that a later run
166
+ mistakes for finished work.
167
+
168
+ :param source: The file to copy.
169
+ :param target: Where it should land.
170
+ """
171
+ handle, partial_name = tempfile.mkstemp(
172
+ dir=target.parent, prefix=PARTIAL_PREFIX, suffix=PARTIAL_SUFFIX
173
+ )
174
+ os.close(handle)
175
+ partial = Path(partial_name)
176
+ try:
177
+ partial.chmod(file_mode())
178
+ with open_contained(source) as reading, partial.open("wb") as writing:
179
+ shutil.copyfileobj(reading, writing)
180
+ partial.replace(target)
181
+ except BaseException:
182
+ partial.unlink(missing_ok=True)
183
+ raise
184
+
185
+
186
+ def count_ignored(source_dir: Path, packages: Sequence[Path]) -> int:
187
+ """
188
+ Count things in the source that were not converted and never mentioned.
189
+
190
+ A real library holds both forms -- Apple's ``*.epub/`` package directories
191
+ and books that were sideloaded already zipped -- plus whatever else lives
192
+ beside them. Anything that is not a package was passed over in silence:
193
+ not skipped, not counted, not listed, so a partial export read as a
194
+ complete one.
195
+
196
+ Counted, not converted. Copying an already-valid archive through is a
197
+ feature with its own decisions to make, not something to do by surprise.
198
+
199
+ :param source_dir: The directory that was searched.
200
+ :param packages: The packages that were found, which are not ignored.
201
+
202
+ :return: How many entries were passed over.
203
+ """
204
+ found = {package.resolve() for package in packages}
205
+ ignored = 0
206
+
207
+ def on_error(exc: OSError) -> None:
208
+ logger.debug("Could not count entries in %s: %s", source_dir, exc)
209
+
210
+ for root, dirs, files in os.walk(source_dir, onerror=on_error):
211
+ directory = Path(root)
212
+ # Not descended into: a package's own contents are not "ignored".
213
+ dirs[:] = [name for name in dirs if (directory / name).resolve() not in found]
214
+ ignored += len(files)
215
+
216
+ return ignored
217
+
218
+
219
+ def collect_package_dirs(source_dir: Path) -> list[Path]:
220
+ """
221
+ Find every ``*.epub/`` package directory beneath the source directory.
222
+
223
+ The walk does not descend into a package once it has been found, so files
224
+ inside a package can never be mistaken for packages themselves. Results are
225
+ full paths, which keeps nested packages addressable; earlier versions
226
+ returned bare directory names and silently broke on anything that was not
227
+ a direct child of the source directory.
228
+
229
+ :param source_dir: The directory to search.
230
+
231
+ :return: Package directories, sorted by path.
232
+ """
233
+ found: list[Path] = []
234
+
235
+ def on_error(exc: OSError) -> None:
236
+ # os.walk swallows scandir failures unless onerror is supplied, so an
237
+ # unreadable directory would otherwise be skipped in total silence.
238
+ logger.warning("Could not scan %s: %s", exc.filename or source_dir, exc)
239
+
240
+ for root, dirs, _files in os.walk(source_dir, onerror=on_error):
241
+ descend = []
242
+ for name in dirs:
243
+ candidate = Path(root) / name
244
+ # A package is a directory the walk owns, never a redirection. A
245
+ # symlink named *.epub was accepted here and its whole target
246
+ # zipped into the shelf. The rule lives in one place; see
247
+ # :mod:`epubconvert.contained` for why it is not restated here.
248
+ if not contains(Path(root), candidate):
249
+ # Only a *.epub link is a skipped book; anything else is a
250
+ # linked directory the walk simply does not follow, and
251
+ # warning about it on a library of symlinked shelves is noise.
252
+ if name.endswith(PACKAGE_SUFFIX):
253
+ logger.warning(
254
+ "Ignoring symlinked package %s", printable(str(candidate))
255
+ )
256
+ else:
257
+ logger.debug("Not following symlinked directory %s", candidate)
258
+ continue
259
+ if name.endswith(PACKAGE_SUFFIX):
260
+ found.append(candidate)
261
+ else:
262
+ descend.append(name)
263
+ dirs[:] = descend
264
+
265
+ found.sort()
266
+ logger.debug("Found %d epub package(s) under %s", len(found), source_dir)
267
+ return found
268
+
269
+
270
+ def entry(arcname: str, compress_type: int) -> ZipInfo:
271
+ """
272
+ Build a zip entry with normalized metadata.
273
+
274
+ Zip members carry a modification time and permission bits, so exporting
275
+ the same book twice would otherwise produce different bytes every time.
276
+ Pinning both makes re-exports byte-identical, which lets backups dedup,
277
+ stops rsync re-copying unchanged books, and allows outputs to be compared
278
+ by hash.
279
+
280
+ :param arcname: Path of the member inside the archive.
281
+ :param compress_type: ``ZIP_STORED`` or ``ZIP_DEFLATED``.
282
+
283
+ :return: The prepared entry.
284
+ """
285
+ member = ZipInfo(arcname, date_time=ARCHIVE_TIMESTAMP)
286
+ member.compress_type = compress_type
287
+ member.external_attr = ARCHIVE_MODE << 16
288
+ # Assigned here rather than on the ZipFile: open() consults the archive's
289
+ # compresslevel only when it builds the ZipInfo itself, so handing it a
290
+ # prebuilt one silently discarded the setting.
291
+ _set_level(member, COMPRESS_LEVEL)
292
+ return member
293
+
294
+
295
+ def _set_level(member: ZipInfo, level: int) -> None:
296
+ """
297
+ Record the deflate level on a member.
298
+
299
+ ``ZipInfo._compresslevel`` is private CPython API, and the only way to
300
+ apply a level to a prebuilt entry: ``ZipFile(compresslevel=)`` is consulted
301
+ only when ``open()`` constructs the ZipInfo itself. The attribute has
302
+ carried this name since 3.7 and is what the public constructor sets.
303
+
304
+ :param member: The entry to annotate.
305
+ :param level: The zlib level to record.
306
+ """
307
+ setattr(member, "_compresslevel", level) # noqa: B010
308
+
309
+
310
+ def level_of(member: ZipInfo) -> int | None:
311
+ """
312
+ Report the deflate level recorded on a member.
313
+
314
+ Exists so tests can assert the level was applied without reaching into
315
+ private CPython API themselves.
316
+
317
+ :param member: The entry to inspect.
318
+
319
+ :return: The recorded level, or None if none was set.
320
+ """
321
+ return getattr(member, "_compresslevel", None)
322
+
323
+
324
+ def compression_for(arcname: str) -> int:
325
+ """
326
+ Choose how a member should be stored.
327
+
328
+ :param arcname: The member's path inside the archive.
329
+
330
+ :return: ``ZIP_STORED`` for already-compressed media, else ``ZIP_DEFLATED``.
331
+ """
332
+ if PurePosixPath(arcname).suffix.lower() in STORED_SUFFIXES:
333
+ return ZIP_STORED
334
+ return ZIP_DEFLATED
335
+
336
+
337
+ def zip_package(
338
+ source_dir: Path,
339
+ target_archive: Path,
340
+ validation: ValidationOptions | None = None,
341
+ ) -> int:
342
+ """
343
+ Write a single package directory out as a spec-valid epub archive.
344
+
345
+ This is a blocking function, intended to be handed to a worker thread. The
346
+ archive is assembled under a temporary name and only moved to
347
+ ``target_archive`` once it is complete.
348
+
349
+ Validation, when asked for, runs against the temporary file *before* the
350
+ move. A book that fails therefore leaves nothing in the output directory
351
+ and will be attempted again on the next run, rather than being recorded as
352
+ finished work.
353
+
354
+ :param source_dir: The ``*.epub/`` package directory to compress.
355
+ :param target_archive: The path of the epub file to create.
356
+ :param validation: Checks to run before the archive is moved into place.
357
+
358
+ :return: The number of package files stored, excluding ``mimetype``.
359
+
360
+ :raises ArchiveInvalidError: If validation was requested and failed.
361
+ """
362
+ # Deriving the temporary name from the target overflows the filesystem's
363
+ # per-component limit when the target is already at it: 255 bytes plus
364
+ # ".part" is 260. Take a short unique name in the same directory instead,
365
+ # which keeps the closing replace atomic and can never be too long.
366
+ handle, partial_name = tempfile.mkstemp(
367
+ dir=target_archive.parent, prefix=PARTIAL_PREFIX, suffix=PARTIAL_SUFFIX
368
+ )
369
+ os.close(handle)
370
+ partial = Path(partial_name)
371
+ # mkstemp creates 0600; exported books should be readable like any other
372
+ # file the user writes -- which means like the umask says, not 0644
373
+ # regardless. A user running with `umask 077` still got world-readable
374
+ # books. ARCHIVE_MODE stays as the zip entry's recorded mode, which is
375
+ # metadata and rightly fixed for byte-identical re-exports.
376
+ file_count = 0
377
+
378
+ try:
379
+ # Inside the guard: a failure here used to leak a .part per book,
380
+ # because the handler that unlinks it starts below.
381
+ partial.chmod(file_mode())
382
+ with ZipFile(
383
+ partial, "w", ZIP_DEFLATED, compresslevel=COMPRESS_LEVEL
384
+ ) as archive:
385
+ # The mimetype entry must come first and must be stored, not deflated.
386
+ archive.writestr(entry(MIMETYPE_NAME, ZIP_STORED), MIMETYPE_CONTENT)
387
+
388
+ stored: set[str] = set()
389
+ for path in _members(source_dir):
390
+ if is_excluded(path.name, at_root=path.parent == source_dir):
391
+ logger.trace("Excluded from archive: %s", path.name)
392
+ continue
393
+ arcname = path.relative_to(source_dir).as_posix()
394
+ member = entry(arcname, compression_for(arcname))
395
+ with (
396
+ open_contained(path) as source,
397
+ archive.open(member, "w") as target,
398
+ ):
399
+ shutil.copyfileobj(source, target)
400
+ stored.add(arcname)
401
+ file_count += 1
402
+
403
+ assert_is_a_book(target_archive.name, stored)
404
+
405
+ if validation is not None:
406
+ problems = validation.check(partial)
407
+ if problems:
408
+ raise ArchiveInvalidError(target_archive.name, problems)
409
+
410
+ partial.replace(target_archive)
411
+ except BaseException:
412
+ # Leave no partial archive behind, so the "already exported" check
413
+ # stays a reliable record of completed work.
414
+ partial.unlink(missing_ok=True)
415
+ raise
416
+
417
+ return file_count
418
+
419
+
420
+ #: The process umask, read once at import. Reading it requires *setting* it --
421
+ #: there is no query-only call -- so doing that per book from up to 64 workers
422
+ #: let one thread observe another's zeroed window and write a world-writable
423
+ #: book, and could leave the process umask at 0 for everything afterwards.
424
+ #: Import happens before any thread exists, and this tool never changes it.
425
+ UMASK = os.umask(0)
426
+ os.umask(UMASK)
427
+
428
+
429
+ def file_mode() -> int:
430
+ """
431
+ Return the mode an exported file should carry, per the user's umask.
432
+
433
+ :return: 0o666 with the umask applied.
434
+ """
435
+ return 0o666 & ~UMASK
436
+
437
+
438
+ def assert_is_a_book(name: str, stored: set[str]) -> None:
439
+ """
440
+ Refuse to hand back an archive that is not a book.
441
+
442
+ **This is the choke point.** The output directory is the tool's only record
443
+ of completed work, so anything that reaches it is recorded as finished and
444
+ no rerun retries it. Every silent-success defect this project has had ended
445
+ here: an unreadable subdirectory that contributed nothing, a package
446
+ deleted between planning and writing, a package that was never downloaded,
447
+ a symlinked directory holding somebody else's files. In each case the run
448
+ wrote a structurally valid zip, reported an export, and permanently
449
+ recorded a book that was wrong or missing.
450
+
451
+ A valid zip is not the bar. The bar is the two things every epub has: the
452
+ container document that says where the package document lives, and at least
453
+ one member besides the ``mimetype`` this function's caller wrote itself.
454
+
455
+ Deliberately cheap and unconditional -- ``--validate`` is the thorough
456
+ check and it is off by default, so this is what protects the invariant on
457
+ an ordinary run.
458
+
459
+ :param name: The archive's name, for the error message.
460
+ :param stored: Arc names written from the package, excluding ``mimetype``.
461
+
462
+ :raises ArchiveInvalidError: If the archive is not a book.
463
+ """
464
+ if not stored:
465
+ raise ArchiveInvalidError(name, ["package holds no files"])
466
+ if CONTAINER_PATH not in stored:
467
+ raise ArchiveInvalidError(name, [f"package has no {CONTAINER_PATH}"])
468
+
469
+
470
+ def _members(source_dir: Path) -> list[Path]:
471
+ """
472
+ List the files to store, in a fixed order, refusing to be misdirected.
473
+
474
+ Two departures from a plain ``rglob``, both of which cost a book its
475
+ integrity when left out:
476
+
477
+ ``os.walk`` is given an ``onerror`` that re-raises, so an unreadable
478
+ subdirectory fails the export instead of contributing nothing. ``rglob``
479
+ swallows that error, and the archive was written without the missing
480
+ content, reported as a success, and recorded as completed work -- so
481
+ repairing the permissions and rerunning skipped the book.
482
+
483
+ A symlink anywhere in the package **fails the export**. Skipping one is
484
+ not safe: ``os.walk`` does not descend into a symlinked directory, so a
485
+ package whose whole content tree is a link contributed nothing at all, and
486
+ the archive -- holding a real ``container.xml`` at the root -- passed
487
+ :func:`assert_is_a_book` and was recorded as finished work. Refusing the
488
+ book is the only answer that cannot silently lose content.
489
+
490
+ :param source_dir: The package directory to enumerate.
491
+
492
+ :return: The files to store, sorted, symlinks excluded.
493
+
494
+ :raises OSError: If any directory under the package cannot be read.
495
+ """
496
+
497
+ def on_error(exc: OSError) -> None:
498
+ raise exc
499
+
500
+ found: list[Path] = []
501
+ resolved = source_dir.resolve()
502
+ for root, dirs, files in os.walk(source_dir, onerror=on_error):
503
+ directory = Path(root)
504
+ # Directories as well as files. os.walk quietly declines to descend a
505
+ # symlinked directory, which is exactly how a package could contribute
506
+ # nothing and still be reported as exported.
507
+ for name in dirs + files:
508
+ path = directory / name
509
+ if not contains(source_dir, path, resolved_root=resolved):
510
+ raise OSError(
511
+ errno.ELOOP,
512
+ f"symlink in package: {printable(name)}",
513
+ str(path),
514
+ )
515
+ if path.is_file():
516
+ found.append(path)
517
+
518
+ found.sort()
519
+ return found