epub-blocks 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- epub_blocks/__init__.py +55 -0
- epub_blocks/cli.py +37 -0
- epub_blocks/errors.py +2 -0
- epub_blocks/extract.py +237 -0
- epub_blocks/models.py +106 -0
- epub_blocks/package.py +179 -0
- epub_blocks/py.typed +0 -0
- epub_blocks/recipe.py +1661 -0
- epub_blocks/safety.py +131 -0
- epub_blocks/schemas/recipe-v1.schema.json +251 -0
- epub_blocks/xhtml.py +209 -0
- epub_blocks/xml.py +247 -0
- epub_blocks-0.2.0.dist-info/METADATA +232 -0
- epub_blocks-0.2.0.dist-info/RECORD +18 -0
- epub_blocks-0.2.0.dist-info/WHEEL +5 -0
- epub_blocks-0.2.0.dist-info/entry_points.txt +2 -0
- epub_blocks-0.2.0.dist-info/licenses/LICENSE +21 -0
- epub_blocks-0.2.0.dist-info/top_level.txt +1 -0
epub_blocks/__init__.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""Public API for deterministic, recipe-driven EPUB text extraction."""
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import version
|
|
4
|
+
|
|
5
|
+
from .errors import EpubBlocksError
|
|
6
|
+
from .extract import extract_blocks, extract_fragments
|
|
7
|
+
from .models import (
|
|
8
|
+
BlockReference,
|
|
9
|
+
CompiledBlock,
|
|
10
|
+
CompiledRecipe,
|
|
11
|
+
EpubPackage,
|
|
12
|
+
ExtractedBlock,
|
|
13
|
+
Fragment,
|
|
14
|
+
NormalizationOptions,
|
|
15
|
+
SpineDocument,
|
|
16
|
+
TextBlock,
|
|
17
|
+
)
|
|
18
|
+
from .package import inspect_epub
|
|
19
|
+
from .recipe import (
|
|
20
|
+
compile_recipe,
|
|
21
|
+
compile_recipe_file,
|
|
22
|
+
compiled_recipe_digest,
|
|
23
|
+
extract_recipe,
|
|
24
|
+
extract_recipe_file,
|
|
25
|
+
load_recipe,
|
|
26
|
+
write_tsv,
|
|
27
|
+
)
|
|
28
|
+
from .safety import SafetyLimits
|
|
29
|
+
|
|
30
|
+
__version__ = version("epub-blocks")
|
|
31
|
+
|
|
32
|
+
__all__ = [
|
|
33
|
+
"BlockReference",
|
|
34
|
+
"CompiledBlock",
|
|
35
|
+
"CompiledRecipe",
|
|
36
|
+
"EpubBlocksError",
|
|
37
|
+
"EpubPackage",
|
|
38
|
+
"ExtractedBlock",
|
|
39
|
+
"Fragment",
|
|
40
|
+
"NormalizationOptions",
|
|
41
|
+
"SafetyLimits",
|
|
42
|
+
"SpineDocument",
|
|
43
|
+
"TextBlock",
|
|
44
|
+
"__version__",
|
|
45
|
+
"compile_recipe",
|
|
46
|
+
"compile_recipe_file",
|
|
47
|
+
"compiled_recipe_digest",
|
|
48
|
+
"extract_blocks",
|
|
49
|
+
"extract_fragments",
|
|
50
|
+
"extract_recipe",
|
|
51
|
+
"extract_recipe_file",
|
|
52
|
+
"inspect_epub",
|
|
53
|
+
"load_recipe",
|
|
54
|
+
"write_tsv",
|
|
55
|
+
]
|
epub_blocks/cli.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import sys
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from . import __version__
|
|
8
|
+
from .errors import EpubBlocksError
|
|
9
|
+
from .recipe import extract_recipe_file, write_tsv
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def main() -> int:
|
|
13
|
+
"""Run the command-line extractor and return a process exit status."""
|
|
14
|
+
|
|
15
|
+
parser = argparse.ArgumentParser(
|
|
16
|
+
description="Apply a recipe to an EPUB and write structured text as TSV."
|
|
17
|
+
)
|
|
18
|
+
parser.add_argument("epub", type=Path)
|
|
19
|
+
parser.add_argument("recipe", type=Path)
|
|
20
|
+
parser.add_argument("output", type=Path)
|
|
21
|
+
parser.add_argument(
|
|
22
|
+
"--version", action="version", version=f"%(prog)s {__version__}"
|
|
23
|
+
)
|
|
24
|
+
args = parser.parse_args()
|
|
25
|
+
|
|
26
|
+
try:
|
|
27
|
+
blocks = extract_recipe_file(args.epub, args.recipe)
|
|
28
|
+
write_tsv(args.output, blocks)
|
|
29
|
+
except (EpubBlocksError, OSError) as error:
|
|
30
|
+
print(f"epub-blocks: error: {error}", file=sys.stderr)
|
|
31
|
+
return 2
|
|
32
|
+
print(f"{len(blocks)} blocks -> {args.output}")
|
|
33
|
+
return 0
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
if __name__ == "__main__":
|
|
37
|
+
raise SystemExit(main())
|
epub_blocks/errors.py
ADDED
epub_blocks/extract.py
ADDED
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import copy
|
|
4
|
+
import hashlib
|
|
5
|
+
import os
|
|
6
|
+
import re
|
|
7
|
+
from collections.abc import Iterator, Sequence
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import BinaryIO
|
|
10
|
+
from zipfile import BadZipFile, ZipFile
|
|
11
|
+
|
|
12
|
+
from .errors import EpubBlocksError
|
|
13
|
+
from .models import EpubPackage, Fragment, NormalizationOptions, TextBlock
|
|
14
|
+
from .package import matches, read_epub_package, select_spine_documents
|
|
15
|
+
from .safety import DEFAULT_SAFETY_LIMITS, EpubArchive, SafetyLimits
|
|
16
|
+
from .xhtml import (
|
|
17
|
+
extract_fragment,
|
|
18
|
+
flatten_text,
|
|
19
|
+
normalize_text,
|
|
20
|
+
read_document_body,
|
|
21
|
+
remove_descendants_by_epub_type,
|
|
22
|
+
)
|
|
23
|
+
from .xml import XmlElement, local_name
|
|
24
|
+
|
|
25
|
+
PRIMARY_BLOCK_TAGS = frozenset({"p", "h1", "h2", "h3", "h4", "h5", "h6", "pre"})
|
|
26
|
+
FALLBACK_BLOCK_TAGS = frozenset({"blockquote", "li"})
|
|
27
|
+
DEFAULT_OMITTED_EPUB_TYPES = frozenset({"noteref", "pagebreak"})
|
|
28
|
+
_DEFAULT_NORMALIZATION = NormalizationOptions()
|
|
29
|
+
_SHA256 = re.compile(r"[0-9a-f]{64}")
|
|
30
|
+
StrPath = str | os.PathLike[str]
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _stream_sha256(source: BinaryIO) -> str:
|
|
34
|
+
digest = hashlib.sha256()
|
|
35
|
+
while block := source.read(1024 * 1024):
|
|
36
|
+
digest.update(block)
|
|
37
|
+
return digest.hexdigest()
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _contains_primary_block(element: XmlElement, primary_tags: frozenset[str]) -> bool:
|
|
41
|
+
return any(
|
|
42
|
+
descendant is not element and local_name(descendant.tag) in primary_tags
|
|
43
|
+
for descendant in element.iter()
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _candidate_elements(
|
|
48
|
+
parent: XmlElement,
|
|
49
|
+
primary_tags: frozenset[str],
|
|
50
|
+
fallback_tags: frozenset[str],
|
|
51
|
+
parent_path: tuple[int, ...] = (),
|
|
52
|
+
) -> Iterator[tuple[XmlElement, tuple[int, ...]]]:
|
|
53
|
+
for index, child in enumerate(list(parent), 1):
|
|
54
|
+
element_path = (*parent_path, index)
|
|
55
|
+
tag = local_name(child.tag)
|
|
56
|
+
if tag in primary_tags or (
|
|
57
|
+
tag in fallback_tags and not _contains_primary_block(child, primary_tags)
|
|
58
|
+
):
|
|
59
|
+
yield child, element_path
|
|
60
|
+
else:
|
|
61
|
+
yield from _candidate_elements(
|
|
62
|
+
child, primary_tags, fallback_tags, element_path
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _extract_blocks_from_epub(
|
|
67
|
+
epub: EpubArchive,
|
|
68
|
+
package: EpubPackage,
|
|
69
|
+
source_path: Path,
|
|
70
|
+
*,
|
|
71
|
+
include_documents: Sequence[str] | None = None,
|
|
72
|
+
exclude_documents: Sequence[str] | None = None,
|
|
73
|
+
exclude_classes: Sequence[str] | None = None,
|
|
74
|
+
include_locators: Sequence[str] | None = None,
|
|
75
|
+
exclude_locators: Sequence[str] | None = None,
|
|
76
|
+
include_non_linear: bool = False,
|
|
77
|
+
primary_tags: frozenset[str] = PRIMARY_BLOCK_TAGS,
|
|
78
|
+
fallback_tags: frozenset[str] = FALLBACK_BLOCK_TAGS,
|
|
79
|
+
omit_epub_types: frozenset[str] = DEFAULT_OMITTED_EPUB_TYPES,
|
|
80
|
+
normalization: NormalizationOptions = _DEFAULT_NORMALIZATION,
|
|
81
|
+
document_cache: dict[str, XmlElement] | None = None,
|
|
82
|
+
) -> list[TextBlock]:
|
|
83
|
+
"""Extract blocks from an already-open, bounded EPUB archive."""
|
|
84
|
+
|
|
85
|
+
blocks: list[TextBlock] = []
|
|
86
|
+
excluded_classes = {value.casefold() for value in (exclude_classes or ())}
|
|
87
|
+
documents = select_spine_documents(
|
|
88
|
+
package,
|
|
89
|
+
include_documents,
|
|
90
|
+
exclude_documents,
|
|
91
|
+
include_non_linear=include_non_linear,
|
|
92
|
+
)
|
|
93
|
+
cache = document_cache if document_cache is not None else {}
|
|
94
|
+
for document in documents:
|
|
95
|
+
body = read_document_body(epub, document.path, cache)
|
|
96
|
+
for element, address in _candidate_elements(body, primary_tags, fallback_tags):
|
|
97
|
+
classes = frozenset(element.get("class", "").split())
|
|
98
|
+
if {value.casefold() for value in classes} & excluded_classes:
|
|
99
|
+
continue
|
|
100
|
+
element_path = ".".join(str(component) for component in address)
|
|
101
|
+
locator = f"{document.path}#{element_path}"
|
|
102
|
+
if include_locators and not matches(locator, include_locators):
|
|
103
|
+
continue
|
|
104
|
+
if exclude_locators and matches(locator, exclude_locators):
|
|
105
|
+
continue
|
|
106
|
+
selected = copy.deepcopy(element)
|
|
107
|
+
remove_descendants_by_epub_type(selected, omit_epub_types)
|
|
108
|
+
text = normalize_text(flatten_text(selected), normalization)
|
|
109
|
+
if not text:
|
|
110
|
+
continue
|
|
111
|
+
blocks.append(
|
|
112
|
+
TextBlock(
|
|
113
|
+
spine_position=document.position,
|
|
114
|
+
document_path=document.path,
|
|
115
|
+
element_path=element_path,
|
|
116
|
+
tag=local_name(element.tag),
|
|
117
|
+
text=text,
|
|
118
|
+
classes=classes,
|
|
119
|
+
)
|
|
120
|
+
)
|
|
121
|
+
if not blocks:
|
|
122
|
+
raise EpubBlocksError(f"{source_path}: no text blocks matched the selection")
|
|
123
|
+
return blocks
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def extract_blocks(
|
|
127
|
+
epub_path: StrPath,
|
|
128
|
+
*,
|
|
129
|
+
expected_identifier: str | None = None,
|
|
130
|
+
include_documents: Sequence[str] | None = None,
|
|
131
|
+
exclude_documents: Sequence[str] | None = None,
|
|
132
|
+
exclude_classes: Sequence[str] | None = None,
|
|
133
|
+
include_locators: Sequence[str] | None = None,
|
|
134
|
+
exclude_locators: Sequence[str] | None = None,
|
|
135
|
+
include_non_linear: bool = False,
|
|
136
|
+
primary_tags: frozenset[str] = PRIMARY_BLOCK_TAGS,
|
|
137
|
+
fallback_tags: frozenset[str] = FALLBACK_BLOCK_TAGS,
|
|
138
|
+
omit_epub_types: frozenset[str] = DEFAULT_OMITTED_EPUB_TYPES,
|
|
139
|
+
normalization: NormalizationOptions = _DEFAULT_NORMALIZATION,
|
|
140
|
+
limits: SafetyLimits = DEFAULT_SAFETY_LIMITS,
|
|
141
|
+
) -> list[TextBlock]:
|
|
142
|
+
"""Extract selected text blocks in EPUB spine and XHTML document order."""
|
|
143
|
+
|
|
144
|
+
path = Path(epub_path)
|
|
145
|
+
try:
|
|
146
|
+
with ZipFile(path) as zip_file:
|
|
147
|
+
epub = EpubArchive(zip_file, limits)
|
|
148
|
+
package = read_epub_package(epub)
|
|
149
|
+
if expected_identifier is not None:
|
|
150
|
+
if not expected_identifier:
|
|
151
|
+
raise EpubBlocksError(
|
|
152
|
+
"expected_identifier must be a non-empty string"
|
|
153
|
+
)
|
|
154
|
+
if expected_identifier not in package.identifiers:
|
|
155
|
+
raise EpubBlocksError(
|
|
156
|
+
f"{path}: expected package identifier "
|
|
157
|
+
f"{expected_identifier!r} not found"
|
|
158
|
+
)
|
|
159
|
+
return _extract_blocks_from_epub(
|
|
160
|
+
epub,
|
|
161
|
+
package,
|
|
162
|
+
path,
|
|
163
|
+
include_documents=include_documents,
|
|
164
|
+
exclude_documents=exclude_documents,
|
|
165
|
+
exclude_classes=exclude_classes,
|
|
166
|
+
include_locators=include_locators,
|
|
167
|
+
exclude_locators=exclude_locators,
|
|
168
|
+
include_non_linear=include_non_linear,
|
|
169
|
+
primary_tags=primary_tags,
|
|
170
|
+
fallback_tags=fallback_tags,
|
|
171
|
+
omit_epub_types=omit_epub_types,
|
|
172
|
+
normalization=normalization,
|
|
173
|
+
)
|
|
174
|
+
except BadZipFile as error:
|
|
175
|
+
raise EpubBlocksError(f"{path}: not a valid ZIP container") from error
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def extract_fragments(
|
|
179
|
+
epub_path: StrPath,
|
|
180
|
+
fragments: Sequence[Fragment],
|
|
181
|
+
*,
|
|
182
|
+
expected_identifier: str | None = None,
|
|
183
|
+
expected_sha256: str | None = None,
|
|
184
|
+
normalization: NormalizationOptions = _DEFAULT_NORMALIZATION,
|
|
185
|
+
omit_epub_types: frozenset[str] = DEFAULT_OMITTED_EPUB_TYPES,
|
|
186
|
+
limits: SafetyLimits = DEFAULT_SAFETY_LIMITS,
|
|
187
|
+
) -> list[str]:
|
|
188
|
+
"""Extract arbitrary XHTML fragments through one validated EPUB read.
|
|
189
|
+
|
|
190
|
+
Each result corresponds to the fragment at the same position. Descendant
|
|
191
|
+
omissions precede normalization; slice offsets index normalized text.
|
|
192
|
+
Callers that need a pinned input can require both a package identifier and
|
|
193
|
+
complete-file SHA-256 without reopening the archive for every fragment.
|
|
194
|
+
"""
|
|
195
|
+
|
|
196
|
+
if expected_identifier is not None and not expected_identifier:
|
|
197
|
+
raise EpubBlocksError("expected_identifier must be a non-empty string")
|
|
198
|
+
if expected_sha256 is not None and _SHA256.fullmatch(expected_sha256) is None:
|
|
199
|
+
raise EpubBlocksError(
|
|
200
|
+
"expected_sha256 must be 64 lowercase hexadecimal characters"
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
path = Path(epub_path)
|
|
204
|
+
with path.open("rb") as source:
|
|
205
|
+
if expected_sha256 is not None:
|
|
206
|
+
if _stream_sha256(source) != expected_sha256:
|
|
207
|
+
raise EpubBlocksError(f"{path}: source SHA-256 does not match")
|
|
208
|
+
source.seek(0)
|
|
209
|
+
try:
|
|
210
|
+
zip_file = ZipFile(source)
|
|
211
|
+
except BadZipFile as error:
|
|
212
|
+
raise EpubBlocksError(f"{path}: not a valid ZIP container") from error
|
|
213
|
+
with zip_file:
|
|
214
|
+
epub = EpubArchive(zip_file, limits)
|
|
215
|
+
package = read_epub_package(epub)
|
|
216
|
+
if (
|
|
217
|
+
expected_identifier is not None
|
|
218
|
+
and expected_identifier not in package.identifiers
|
|
219
|
+
):
|
|
220
|
+
raise EpubBlocksError(
|
|
221
|
+
f"{path}: expected package identifier "
|
|
222
|
+
f"{expected_identifier!r} not found"
|
|
223
|
+
)
|
|
224
|
+
cache: dict[str, XmlElement] = {}
|
|
225
|
+
return [
|
|
226
|
+
normalize_text(
|
|
227
|
+
extract_fragment(
|
|
228
|
+
epub,
|
|
229
|
+
fragment,
|
|
230
|
+
cache=cache,
|
|
231
|
+
normalization=normalization,
|
|
232
|
+
omit_epub_types=omit_epub_types,
|
|
233
|
+
),
|
|
234
|
+
normalization,
|
|
235
|
+
)
|
|
236
|
+
for fragment in fragments
|
|
237
|
+
]
|
epub_blocks/models.py
ADDED
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Literal
|
|
5
|
+
|
|
6
|
+
UnicodeNormalization = Literal["NFC", "NFD", "NFKC", "NFKD", "none"]
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass(frozen=True)
|
|
10
|
+
class SpineDocument:
|
|
11
|
+
"""One XHTML document in EPUB spine order."""
|
|
12
|
+
|
|
13
|
+
position: int
|
|
14
|
+
path: str
|
|
15
|
+
properties: frozenset[str]
|
|
16
|
+
linear: bool = True
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class EpubPackage:
|
|
21
|
+
"""The package information needed for deterministic text extraction."""
|
|
22
|
+
|
|
23
|
+
package_path: str
|
|
24
|
+
identifiers: frozenset[str]
|
|
25
|
+
spine: tuple[SpineDocument, ...]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(frozen=True)
|
|
29
|
+
class NormalizationOptions:
|
|
30
|
+
"""Text normalization applied after XHTML text extraction."""
|
|
31
|
+
|
|
32
|
+
collapse_whitespace: bool = True
|
|
33
|
+
strip: bool = True
|
|
34
|
+
unicode_normalization: UnicodeNormalization = "NFC"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass(frozen=True)
|
|
38
|
+
class Fragment:
|
|
39
|
+
"""A selectable XHTML subtree, with optional omissions and text slice."""
|
|
40
|
+
|
|
41
|
+
document_path: str
|
|
42
|
+
element_path: str = ""
|
|
43
|
+
omit_paths: tuple[str, ...] = ()
|
|
44
|
+
start: int | None = None
|
|
45
|
+
end: int | None = None
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass(frozen=True)
|
|
49
|
+
class BlockReference:
|
|
50
|
+
"""The canonical identifier and type assigned to one output block."""
|
|
51
|
+
|
|
52
|
+
block_id: str
|
|
53
|
+
block_type: str
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass(frozen=True)
|
|
57
|
+
class CompiledBlock:
|
|
58
|
+
"""One output block in a compiled extraction plan."""
|
|
59
|
+
|
|
60
|
+
block_id: str
|
|
61
|
+
block_type: str
|
|
62
|
+
parts: tuple[Fragment, ...]
|
|
63
|
+
separator: str = ""
|
|
64
|
+
consumed_locators: tuple[str, ...] = ()
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass(frozen=True)
|
|
68
|
+
class CompiledRecipe:
|
|
69
|
+
"""The deterministic extraction plan produced from a version 1 recipe."""
|
|
70
|
+
|
|
71
|
+
epub_identifier: str
|
|
72
|
+
epub_sha256: str
|
|
73
|
+
normalization: NormalizationOptions
|
|
74
|
+
omit_epub_types: frozenset[str]
|
|
75
|
+
blocks: tuple[CompiledBlock, ...]
|
|
76
|
+
skipped_locators: tuple[str, ...] = ()
|
|
77
|
+
reserved_locators: tuple[str, ...] = ()
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass(frozen=True)
|
|
81
|
+
class TextBlock:
|
|
82
|
+
"""An extracted XHTML block with a stable source locator."""
|
|
83
|
+
|
|
84
|
+
spine_position: int
|
|
85
|
+
document_path: str
|
|
86
|
+
element_path: str
|
|
87
|
+
tag: str
|
|
88
|
+
text: str
|
|
89
|
+
classes: frozenset[str]
|
|
90
|
+
|
|
91
|
+
@property
|
|
92
|
+
def locator(self) -> str:
|
|
93
|
+
return f"{self.document_path}#{self.element_path}"
|
|
94
|
+
|
|
95
|
+
@property
|
|
96
|
+
def source_locator(self) -> str:
|
|
97
|
+
return f"s{self.spine_position:03d}:{self.locator}"
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@dataclass(frozen=True)
|
|
101
|
+
class ExtractedBlock:
|
|
102
|
+
"""One canonical-reference/type/text record produced by a recipe."""
|
|
103
|
+
|
|
104
|
+
block_id: str
|
|
105
|
+
block_type: str
|
|
106
|
+
text: str
|
epub_blocks/package.py
ADDED
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import fnmatch
|
|
4
|
+
import os
|
|
5
|
+
import posixpath
|
|
6
|
+
from collections.abc import Sequence
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from urllib.parse import unquote
|
|
9
|
+
from xml.etree.ElementTree import Element
|
|
10
|
+
from zipfile import BadZipFile, ZipFile
|
|
11
|
+
|
|
12
|
+
from .errors import EpubBlocksError
|
|
13
|
+
from .models import EpubPackage, SpineDocument
|
|
14
|
+
from .safety import DEFAULT_SAFETY_LIMITS, EpubArchive, SafetyLimits
|
|
15
|
+
from .xml import parse_xml
|
|
16
|
+
|
|
17
|
+
StrPath = str | os.PathLike[str]
|
|
18
|
+
|
|
19
|
+
CONTAINER_NS = "urn:oasis:names:tc:opendocument:xmlns:container"
|
|
20
|
+
OPF_NS = "http://www.idpf.org/2007/opf"
|
|
21
|
+
DC_NS = "http://purl.org/dc/elements/1.1/"
|
|
22
|
+
PACKAGE_MEDIA_TYPE = "application/oebps-package+xml"
|
|
23
|
+
XHTML_MEDIA_TYPE = "application/xhtml+xml"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def archive_path(package_path: str, href: str) -> str:
|
|
27
|
+
"""Resolve a manifest href safely within the EPUB archive."""
|
|
28
|
+
|
|
29
|
+
href_path = unquote(href.split("#", 1)[0])
|
|
30
|
+
if not href_path or "\\" in href_path:
|
|
31
|
+
raise EpubBlocksError(f"invalid EPUB manifest path: {href!r}")
|
|
32
|
+
resolved = posixpath.normpath(
|
|
33
|
+
posixpath.join(posixpath.dirname(package_path), href_path)
|
|
34
|
+
)
|
|
35
|
+
if resolved == ".." or resolved.startswith(("../", "/")):
|
|
36
|
+
raise EpubBlocksError(f"EPUB manifest path escapes the archive root: {href!r}")
|
|
37
|
+
return resolved
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _single_child(parent: Element[str], path: str, label: str) -> Element[str]:
|
|
41
|
+
elements = parent.findall(path)
|
|
42
|
+
if len(elements) != 1:
|
|
43
|
+
raise EpubBlocksError(f"EPUB package must contain exactly one {label}")
|
|
44
|
+
return elements[0]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def read_epub_package(epub: EpubArchive) -> EpubPackage:
|
|
48
|
+
"""Read identifiers and spine metadata from a bounded EPUB reader."""
|
|
49
|
+
|
|
50
|
+
container_path = "META-INF/container.xml"
|
|
51
|
+
container = parse_xml(epub.read(container_path), container_path, epub.limits)
|
|
52
|
+
if container.tag != f"{{{CONTAINER_NS}}}container":
|
|
53
|
+
raise EpubBlocksError("EPUB container has an unexpected root element")
|
|
54
|
+
rootfiles = container.findall(
|
|
55
|
+
f"./{{{CONTAINER_NS}}}rootfiles/{{{CONTAINER_NS}}}rootfile"
|
|
56
|
+
)
|
|
57
|
+
package_rootfiles = [
|
|
58
|
+
rootfile
|
|
59
|
+
for rootfile in rootfiles
|
|
60
|
+
if rootfile.get("media-type") == PACKAGE_MEDIA_TYPE
|
|
61
|
+
]
|
|
62
|
+
if not package_rootfiles:
|
|
63
|
+
raise EpubBlocksError(
|
|
64
|
+
"EPUB container has no package rootfile with the required media type"
|
|
65
|
+
)
|
|
66
|
+
if len(package_rootfiles) > 1:
|
|
67
|
+
raise EpubBlocksError("EPUB container identifies multiple package rootfiles")
|
|
68
|
+
full_path = package_rootfiles[0].get("full-path")
|
|
69
|
+
if not full_path:
|
|
70
|
+
raise EpubBlocksError("EPUB package rootfile has no full-path")
|
|
71
|
+
package_path = archive_path("", full_path)
|
|
72
|
+
|
|
73
|
+
package = parse_xml(epub.read(package_path), package_path, epub.limits)
|
|
74
|
+
if package.tag != f"{{{OPF_NS}}}package":
|
|
75
|
+
raise EpubBlocksError("EPUB package has an unexpected root element")
|
|
76
|
+
|
|
77
|
+
metadata = _single_child(package, f"./{{{OPF_NS}}}metadata", "metadata")
|
|
78
|
+
manifest_element = _single_child(package, f"./{{{OPF_NS}}}manifest", "manifest")
|
|
79
|
+
spine_element = _single_child(package, f"./{{{OPF_NS}}}spine", "spine")
|
|
80
|
+
|
|
81
|
+
identifiers = frozenset(
|
|
82
|
+
(element.text or "").strip()
|
|
83
|
+
for element in metadata.findall(f"./{{{DC_NS}}}identifier")
|
|
84
|
+
if (element.text or "").strip()
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
manifest: dict[str, Element[str]] = {}
|
|
88
|
+
for item in manifest_element.findall(f"./{{{OPF_NS}}}item"):
|
|
89
|
+
item_id = item.get("id")
|
|
90
|
+
if not item_id:
|
|
91
|
+
raise EpubBlocksError("EPUB manifest item has no id")
|
|
92
|
+
if item_id in manifest:
|
|
93
|
+
raise EpubBlocksError(f"EPUB manifest has duplicate id {item_id!r}")
|
|
94
|
+
if not item.get("href"):
|
|
95
|
+
raise EpubBlocksError(f"EPUB manifest item {item_id!r} has no href")
|
|
96
|
+
if not item.get("media-type"):
|
|
97
|
+
raise EpubBlocksError(f"EPUB manifest item {item_id!r} has no media-type")
|
|
98
|
+
manifest[item_id] = item
|
|
99
|
+
|
|
100
|
+
spine: list[SpineDocument] = []
|
|
101
|
+
for position, itemref in enumerate(
|
|
102
|
+
spine_element.findall(f"./{{{OPF_NS}}}itemref"), 1
|
|
103
|
+
):
|
|
104
|
+
item_id = itemref.get("idref")
|
|
105
|
+
if not item_id or item_id not in manifest:
|
|
106
|
+
raise EpubBlocksError(
|
|
107
|
+
f"EPUB spine refers to missing manifest item {item_id!r}"
|
|
108
|
+
)
|
|
109
|
+
linear_value = itemref.get("linear", "yes")
|
|
110
|
+
if linear_value not in {"yes", "no"}:
|
|
111
|
+
raise EpubBlocksError(
|
|
112
|
+
f"EPUB spine item {item_id!r} has invalid linear value {linear_value!r}"
|
|
113
|
+
)
|
|
114
|
+
item = manifest[item_id]
|
|
115
|
+
if item.get("media-type") != XHTML_MEDIA_TYPE:
|
|
116
|
+
continue
|
|
117
|
+
href = item.get("href")
|
|
118
|
+
if href is None: # validated above; narrows the static type
|
|
119
|
+
raise AssertionError("validated manifest href is missing")
|
|
120
|
+
properties = frozenset(
|
|
121
|
+
f"{item.get('properties', '')} {itemref.get('properties', '')}".split()
|
|
122
|
+
)
|
|
123
|
+
spine.append(
|
|
124
|
+
SpineDocument(
|
|
125
|
+
position=position,
|
|
126
|
+
path=archive_path(package_path, href),
|
|
127
|
+
properties=properties,
|
|
128
|
+
linear=linear_value == "yes",
|
|
129
|
+
)
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
if not spine:
|
|
133
|
+
raise EpubBlocksError("EPUB package has no XHTML spine documents")
|
|
134
|
+
return EpubPackage(package_path, identifiers, tuple(spine))
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def inspect_epub(
|
|
138
|
+
epub_path: StrPath,
|
|
139
|
+
*,
|
|
140
|
+
limits: SafetyLimits = DEFAULT_SAFETY_LIMITS,
|
|
141
|
+
) -> EpubPackage:
|
|
142
|
+
"""Open an EPUB and return its validated package information."""
|
|
143
|
+
|
|
144
|
+
path = Path(epub_path)
|
|
145
|
+
try:
|
|
146
|
+
with ZipFile(path) as archive:
|
|
147
|
+
return read_epub_package(EpubArchive(archive, limits))
|
|
148
|
+
except BadZipFile as error:
|
|
149
|
+
raise EpubBlocksError(f"{path}: not a valid ZIP container") from error
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def matches(value: str, patterns: Sequence[str]) -> bool:
|
|
153
|
+
"""Match case-insensitive globs without operating-system path rewriting."""
|
|
154
|
+
|
|
155
|
+
folded = value.casefold()
|
|
156
|
+
return any(fnmatch.fnmatchcase(folded, pattern.casefold()) for pattern in patterns)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def select_spine_documents(
|
|
160
|
+
package: EpubPackage,
|
|
161
|
+
include: Sequence[str] | None = None,
|
|
162
|
+
exclude: Sequence[str] | None = None,
|
|
163
|
+
*,
|
|
164
|
+
include_non_linear: bool = False,
|
|
165
|
+
) -> list[SpineDocument]:
|
|
166
|
+
"""Select spine documents, excluding auxiliary non-linear items by default."""
|
|
167
|
+
|
|
168
|
+
include_patterns = include or ("*",)
|
|
169
|
+
exclude_patterns = exclude or ()
|
|
170
|
+
selected = [
|
|
171
|
+
document
|
|
172
|
+
for document in package.spine
|
|
173
|
+
if (document.linear or include_non_linear)
|
|
174
|
+
and matches(document.path, include_patterns)
|
|
175
|
+
and not matches(document.path, exclude_patterns)
|
|
176
|
+
]
|
|
177
|
+
if not selected:
|
|
178
|
+
raise EpubBlocksError("No EPUB spine documents matched the selection")
|
|
179
|
+
return selected
|
epub_blocks/py.typed
ADDED
|
File without changes
|