pagespring 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pagespring/__init__.py ADDED
@@ -0,0 +1,12 @@
1
+ """pagespring — the lean manual acquisition + normalization layer.
2
+
3
+ Point it at a manual's URL; it recognizes the source type (a "pattern"),
4
+ *acquires* the raw pages, and *normalizes* them into one clean HTML/markdown
5
+ file with absolute asset URLs under ``incoming/<slug>/``. That clean file is the
6
+ deliverable; *conversion* to RAG markdown is a separate concern (**pagespeak**)
7
+ that consumes ``incoming/`` independently — this package neither runs nor imports it.
8
+
9
+ This package stays dependency-light (pf-core[cli] + beautifulsoup4).
10
+ """
11
+
12
+ __version__ = "0.1.0"
pagespring/base.py ADDED
@@ -0,0 +1,57 @@
1
+ """The Pattern contract — the unit that ties acquire + normalize + convert
2
+ together for one source type.
3
+
4
+ A pattern recognizes a family of source URLs (``match``), downloads the raw
5
+ pages (``acquire``), turns them into one clean convertible file with absolute
6
+ asset URLs (``normalize``), and declares the extra ``pagespeak convert`` flags
7
+ its output wants (``convert_recipe``). The conversion engine itself lives in
8
+ pagespeak and is invoked as a subprocess — never imported here.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from dataclasses import dataclass
14
+ from pathlib import Path
15
+ from typing import Literal, Protocol, runtime_checkable
16
+
17
+ # "html" -> hand the clean file to pagespeak to convert (markitdown + pipeline)
18
+ # "markdown" -> already the source's canonical clean form (e.g. GitBook per-page .md)
19
+ # "pdf" -> a downloaded PDF; already a pagespeak input (normalize passthrough)
20
+ SourceKind = Literal["html", "markdown", "pdf"]
21
+
22
+
23
+ @dataclass
24
+ class AcquireResult:
25
+ """What ``acquire`` produced: a local dir of raw pages + how to treat them."""
26
+
27
+ raw_dir: Path # local dir holding the downloaded raw page(s)
28
+ kind: SourceKind # whether normalize emits html (for pagespeak) or markdown
29
+ slug: str # short id for the source; becomes the output dir name
30
+ pages: int | None = None # source units fetched (crawl pages / articles / files); manifest stat
31
+ title: str | None = None # human source title for the deliverable heading (falls back to slug)
32
+
33
+
34
+ @runtime_checkable
35
+ class Pattern(Protocol):
36
+ """One source type's acquire/normalize/convert knowledge.
37
+
38
+ Implementations are instances (see pagespring/patterns/*); the registry holds
39
+ one of each. All *source-specific* knowledge (crawl rules, chrome selectors,
40
+ TOC walking, image-scheme resolution) lives in the pattern — pagespeak stays
41
+ source-agnostic.
42
+ """
43
+
44
+ name: str
45
+ convert_recipe: list[str] # extra flags appended to `pagespeak convert`
46
+
47
+ def match(self, url: str) -> bool:
48
+ """Cheap check (host/path) for whether this pattern handles ``url``."""
49
+ ...
50
+
51
+ def acquire(self, url: str, workdir: Path) -> AcquireResult:
52
+ """Download the source's raw pages into ``workdir``."""
53
+ ...
54
+
55
+ def normalize(self, acq: AcquireResult, workdir: Path) -> Path:
56
+ """Turn the raw pages into ONE clean .html/.md (absolute asset URLs)."""
57
+ ...
pagespring/cli.py ADDED
@@ -0,0 +1,209 @@
1
+ """pagespring command-line interface (Typer, via pf_core.cli).
2
+
3
+ Commands:
4
+ ingest <url> acquire + normalize ("fix") a manual into incoming/<slug>/
5
+ localize <slug> grab an already-ingested deliverable's images (resumable; --all)
6
+ patterns list the registered source patterns
7
+ classify <url> show which pattern handles a URL (no acquisition)
8
+ status list incoming/ deliverables (pattern, pages, size, date, source)
9
+ """
10
+
11
+ from datetime import date
12
+ from pathlib import Path
13
+ from urllib.parse import urlsplit
14
+
15
+ import typer
16
+ from pf_core.cli import create_cli, run_cli
17
+ from pf_core.exceptions import InvalidInputError, PreconditionError
18
+
19
+ from pagespring import manifest
20
+ from pagespring.config import cfg
21
+ from pagespring.orchestrate import (
22
+ AcquireError,
23
+ EmptyOutputError,
24
+ NoPatternError,
25
+ localize_images,
26
+ run_ingest,
27
+ )
28
+ from pagespring.registry import PATTERNS, classify
29
+
30
+ app = create_cli(
31
+ "pagespring",
32
+ help="Find, download, and normalize online software manuals into incoming/.",
33
+ )
34
+
35
+
36
+ @app.command()
37
+ def ingest(
38
+ url: str = typer.Argument(
39
+ ..., help="Manual URL (e.g. a support.apple.com/guide/<app>/ welcome page)."
40
+ ),
41
+ keep_raw: bool = typer.Option(
42
+ False, "--keep-raw", help="Keep the raw crawl alongside the source in incoming/<slug>/raw."
43
+ ),
44
+ download_images: bool = typer.Option(
45
+ False,
46
+ "--download-images",
47
+ help="Download an html/markdown source's images into incoming/<slug>/images/ and re-point refs. No-op for PDFs.",
48
+ ),
49
+ if_changed: bool = typer.Option(
50
+ False,
51
+ "--if-changed",
52
+ help="Skip re-staging when the re-fetch normalizes to byte-identical content (the crawl still runs).",
53
+ ),
54
+ ) -> None:
55
+ """Acquire a manual from URL and normalize it into incoming/<slug>/."""
56
+ try:
57
+ result = run_ingest(
58
+ url, keep_raw=keep_raw, download_images=download_images, if_changed=if_changed
59
+ )
60
+ except NoPatternError:
61
+ typer.echo(
62
+ f"No pattern matched: {url}\n"
63
+ "This source needs a new pattern. Run `bin/run patterns` to see the "
64
+ "registered ones; src/pagespring/patterns/ shows the shape to author one.",
65
+ err=True,
66
+ )
67
+ raise typer.Exit(2) from None
68
+ except InvalidInputError as exc:
69
+ typer.echo(str(exc), err=True)
70
+ raise typer.Exit(2) from None
71
+ except EmptyOutputError:
72
+ typer.echo(
73
+ f"Normalize produced an empty file for {url} — the source may have "
74
+ "changed shape. Nothing was staged; a previous deliverable in "
75
+ "incoming/ is untouched.",
76
+ err=True,
77
+ )
78
+ raise typer.Exit(3) from None
79
+ except AcquireError as exc:
80
+ typer.echo(
81
+ f"Fetch failed during acquire: {exc.detail}\n"
82
+ f"Source: {exc.url}\n"
83
+ "Nothing was staged. The fetch died mid-acquire — check the URL is "
84
+ "reachable; re-run to retry.",
85
+ err=True,
86
+ )
87
+ raise typer.Exit(4) from None
88
+
89
+ typer.echo(f"pattern : {result['pattern']}")
90
+ typer.echo(f"slug : {result['slug']}")
91
+ typer.echo(f"incoming : {result['clean']}")
92
+ if result.get("changed") is False:
93
+ typer.echo(
94
+ "status : unchanged — source matches the existing deliverable, nothing re-staged"
95
+ )
96
+ return
97
+ if result.get("pages") is not None:
98
+ typer.echo(f"pages : {result['pages']}")
99
+ typer.echo(f"size : {_human_size(result['bytes'])}")
100
+ if result.get("images"):
101
+ typer.echo(f"images : {result['images']} downloaded → images/")
102
+
103
+
104
+ @app.command()
105
+ def localize(
106
+ slug: str = typer.Argument(
107
+ None, help="Book slug under incoming/ to localize images for (omit when using --all)."
108
+ ),
109
+ all_books: bool = typer.Option(
110
+ False, "--all", help="Localize images for every incoming/<slug>/."
111
+ ),
112
+ ) -> None:
113
+ """Download an already-ingested deliverable's remote images into images/ and
114
+ re-point refs — no re-crawl. Resumable: re-run until none remain, so a book too
115
+ big to localize in one pass finishes across runs."""
116
+ if all_books:
117
+ incoming = Path(cfg.INCOMING_DIR)
118
+ targets = (
119
+ sorted(p.name for p in incoming.glob("*") if p.is_dir()) if incoming.is_dir() else []
120
+ )
121
+ elif slug:
122
+ targets = [slug]
123
+ else:
124
+ typer.echo("Give a slug or --all.", err=True)
125
+ raise typer.Exit(2)
126
+
127
+ for s in targets:
128
+ try:
129
+ r = localize_images(s)
130
+ except PreconditionError as exc:
131
+ typer.echo(f"skip {s}: {exc}", err=True)
132
+ continue
133
+ tail = "done" if r["remaining"] == 0 else f"{r['remaining']} remaining — re-run to continue"
134
+ typer.echo(f"{s}: +{r['localized']} images (total {r['images_total']}) — {tail}")
135
+
136
+
137
+ @app.command()
138
+ def patterns() -> None:
139
+ """List the registered source patterns and their convert recipes."""
140
+ for p in PATTERNS:
141
+ recipe = " ".join(p.convert_recipe) or "(none)"
142
+ typer.echo(f"{p.name:12} convert-recipe: {recipe}")
143
+
144
+
145
+ @app.command("classify")
146
+ def classify_cmd(
147
+ url: str = typer.Argument(..., help="URL to test against the pattern registry."),
148
+ ) -> None:
149
+ """Show which pattern (if any) handles a URL — no acquisition, no network."""
150
+ p = classify(url)
151
+ typer.echo(p.name if p else "(no pattern matched)")
152
+
153
+
154
+ def _human_size(n: int) -> str:
155
+ size = float(n)
156
+ for unit in ("B", "KB", "MB"):
157
+ if size < 1024:
158
+ return f"{size:.0f} {unit}" if unit == "B" else f"{size:.1f} {unit}"
159
+ size /= 1024
160
+ return f"{size:.1f} GB"
161
+
162
+
163
+ @app.command()
164
+ def status() -> None:
165
+ """One row per incoming/<slug>/, read from its manifest.json: deliverable,
166
+ pattern, pages, size, ingest date, and source host. Legacy (pre-manifest)
167
+ dirs fall back to the deliverable file's own facts. (Conversion into the
168
+ manuals corpus is pagespeak's job — downstream of this tool, out of its view.)"""
169
+ incoming = Path(cfg.INCOMING_DIR)
170
+ slugs = sorted(p for p in incoming.glob("*") if p.is_dir()) if incoming.is_dir() else []
171
+ if not slugs:
172
+ typer.echo("(nothing in incoming/ — run `bin/run ingest <url>`)")
173
+ return
174
+ for d in slugs:
175
+ typer.echo(_status_row(d))
176
+
177
+
178
+ def _status_row(slug_dir: Path) -> str:
179
+ """One status line from the slug's manifest; for legacy (pre-manifest) dirs,
180
+ fall back to the first non-manifest deliverable file's own facts."""
181
+ m = manifest.read_manifest(slug_dir)
182
+ if m is not None:
183
+ deliverable = slug_dir / m["deliverable"]
184
+ size = deliverable.stat().st_size if deliverable.exists() else m["bytes"]
185
+ pages = str(m["pages"]) if m["pages"] is not None else "-"
186
+ host = urlsplit(m["source_url"]).netloc or "-"
187
+ return (
188
+ f"{slug_dir.name:24} {m['deliverable']:32} {m['pattern']:14} "
189
+ f"{pages:>5} {_human_size(size):>9} {m['ingested_at'][:10]} {host}"
190
+ )
191
+ files = sorted(
192
+ p for p in slug_dir.iterdir() if p.is_file() and p.name != manifest.MANIFEST_NAME
193
+ )
194
+ if not files:
195
+ return f"{slug_dir.name:24} {'(no clean file)':32} {'-':14} {'-':>5} {'-':>9} - -"
196
+ f = files[0]
197
+ when = date.fromtimestamp(f.stat().st_mtime).isoformat()
198
+ return (
199
+ f"{slug_dir.name:24} {f.name:32} {'-':14} "
200
+ f"{'-':>5} {_human_size(f.stat().st_size):>9} {when} -"
201
+ )
202
+
203
+
204
+ def main() -> None:
205
+ run_cli(app)
206
+
207
+
208
+ if __name__ == "__main__":
209
+ main()
pagespring/config.py ADDED
@@ -0,0 +1,24 @@
1
+ """pagespring configuration — a pf_core.config.AppConfig subclass.
2
+
3
+ All settings are overridable via environment variables / .env.
4
+ """
5
+
6
+ from pathlib import Path
7
+
8
+ from pf_core.config import AppConfig
9
+
10
+ # src/pagespring/config.py → parents[2] is the project root (editable install).
11
+ _project_root = Path(__file__).resolve().parents[2]
12
+
13
+
14
+ class PagespringConfig(AppConfig):
15
+ """pagespring settings."""
16
+
17
+ APP_NAME: str = "pagespring"
18
+
19
+ # The deliverable: one incoming/<slug>/ per manual — the clean
20
+ # acquired+normalized file. A separate step (pagespeak) consumes these.
21
+ INCOMING_DIR: str = "incoming"
22
+
23
+
24
+ cfg = PagespringConfig(env_file=_project_root / ".env")
pagespring/http.py ADDED
@@ -0,0 +1,93 @@
1
+ """Tiny shared HTTP fetch over the standard library (urllib).
2
+
3
+ Deliberately no httpx dependency — the acquire step only does plain GETs.
4
+ Provides an identifying User-Agent (override via PAGESPRING_UA), a timeout,
5
+ status-aware retries (permanent 4xx fail fast; 429 honors Retry-After;
6
+ 5xx/network errors back off), and a polite inter-request delay for crawls.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import time
12
+ import urllib.error
13
+ import urllib.request
14
+
15
+ from pf_core.utils.env import resolve_str
16
+
17
+ from pagespring import __version__
18
+
19
+ _RETRY_AFTER_CAP = 30.0 # seconds — don't let a server park a crawl for minutes
20
+
21
+ _UA_DEFAULT = f"pagespring/{__version__} (+https://github.com/phierceweb/pagespring)"
22
+ _UA_ENV_VAR = "PAGESPRING_UA"
23
+
24
+
25
+ def _ua() -> str:
26
+ """The identifying default UA, or PAGESPRING_UA for sources that need another."""
27
+ return resolve_str(None, _UA_ENV_VAR, default=_UA_DEFAULT) or _UA_DEFAULT
28
+
29
+
30
+ def _request(url: str) -> urllib.request.Request:
31
+ return urllib.request.Request(
32
+ url,
33
+ headers={
34
+ "User-Agent": _ua(),
35
+ "Accept": "*/*",
36
+ "Accept-Language": "en-US,en;q=0.9",
37
+ },
38
+ )
39
+
40
+
41
+ def _retry_after(exc: urllib.error.HTTPError, attempt: int) -> float:
42
+ """Seconds to wait on a 429 — the server's Retry-After (capped) when sane,
43
+ else the normal backoff."""
44
+ try:
45
+ return min(float(exc.headers.get("Retry-After", "")), _RETRY_AFTER_CAP)
46
+ except ValueError:
47
+ return 0.5 * (attempt + 1)
48
+
49
+
50
+ def _read(url: str, timeout: float, retries: int) -> tuple[str, bytes, str | None]:
51
+ last: Exception | None = None
52
+ for attempt in range(retries + 1):
53
+ try:
54
+ with urllib.request.urlopen(_request(url), timeout=timeout) as r:
55
+ return r.geturl(), r.read(), r.headers.get_content_charset()
56
+ except urllib.error.HTTPError as exc:
57
+ last = exc
58
+ if exc.code == 429:
59
+ if attempt < retries:
60
+ time.sleep(_retry_after(exc, attempt))
61
+ elif 400 <= exc.code < 500 and exc.code != 408:
62
+ raise # permanent client error — retrying can't help
63
+ elif attempt < retries: # 5xx / 408
64
+ time.sleep(0.5 * (attempt + 1))
65
+ except Exception as exc: # URLError, timeout, connection reset, …
66
+ last = exc
67
+ if attempt < retries:
68
+ time.sleep(0.5 * (attempt + 1))
69
+ raise last # type: ignore[misc]
70
+
71
+
72
+ def fetch_text(
73
+ url: str, *, timeout: float = 30, retries: int = 2, encoding: str | None = None
74
+ ) -> tuple[str, str]:
75
+ """Return (final_url, decoded_text) after following redirects.
76
+
77
+ Decodes with ``encoding`` when given, else the response's Content-Type
78
+ charset, else utf-8 — always with replacement, never raising."""
79
+ final_url, raw, charset = _read(url, timeout, retries)
80
+ return final_url, raw.decode(encoding or charset or "utf-8", "replace")
81
+
82
+
83
+ def fetch_bytes(url: str, *, timeout: float = 180, retries: int = 2) -> tuple[str, bytes]:
84
+ """Return (final_url, raw_bytes) — for binary downloads (PDFs, archives,
85
+ images). Longer default timeout than fetch_text: vendor PDFs/doc archives
86
+ can be tens of MB on slow CDNs."""
87
+ final_url, raw, _charset = _read(url, timeout, retries)
88
+ return final_url, raw
89
+
90
+
91
+ def polite_sleep(seconds: float = 0.25) -> None:
92
+ """Sleep between crawl requests to avoid hammering the source."""
93
+ time.sleep(seconds)
pagespring/images.py ADDED
@@ -0,0 +1,125 @@
1
+ """Optional image localizer for HTML/markdown ingests.
2
+
3
+ Downloads a deliverable's remote images into a sibling ``images/`` dir and
4
+ re-points the refs at them — for a self-contained ``incoming/<slug>/``, and to
5
+ capture images behind expiring or tokened URLs (e.g. GitBook's
6
+ ``?alt=media&token=…``) while they still resolve.
7
+
8
+ Opt-in via ``bin/run ingest --download-images`` or ``bin/run localize``.
9
+ Stdlib fetch only (``pagespring.http``).
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import re
15
+ from pathlib import Path
16
+ from urllib.parse import urlparse
17
+
18
+ from pf_core.log import get_logger
19
+
20
+ from pagespring import http
21
+
22
+ log = get_logger(__name__)
23
+
24
+ # Markdown ![alt](http…) and HTML <img … src="http…">.
25
+ _MD_IMG_RE = re.compile(r"!\[[^\]]*\]\((https?://[^)\s]+)\)")
26
+ _HTML_IMG_RE = re.compile(r"""<img\b[^>]*?\bsrc=["'](https?://[^"']+)["']""", re.IGNORECASE)
27
+
28
+ _IMG_EXTS = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".svg", ".bmp", ".tif", ".tiff"}
29
+ _MAGIC = (
30
+ (b"\x89PNG\r\n\x1a\n", ".png"),
31
+ (b"\xff\xd8\xff", ".jpg"),
32
+ (b"GIF87a", ".gif"),
33
+ (b"GIF89a", ".gif"),
34
+ (b"<svg", ".svg"),
35
+ (b"<?xml", ".svg"),
36
+ )
37
+
38
+
39
+ def _ext_for(url: str, data: bytes) -> str:
40
+ suffix = Path(urlparse(url).path).suffix.lower()
41
+ if suffix in _IMG_EXTS:
42
+ return ".jpg" if suffix == ".jpeg" else (".tiff" if suffix == ".tif" else suffix)
43
+ head = data[:16]
44
+ for magic, ext in _MAGIC:
45
+ if head.startswith(magic):
46
+ return ext
47
+ if head[:4] == b"RIFF" and b"WEBP" in data[:16]:
48
+ return ".webp"
49
+ return ".img"
50
+
51
+
52
+ def _name_for(url: str, data: bytes, used: set[str]) -> str:
53
+ base = Path(urlparse(url).path).name
54
+ stem = re.sub(r"\.[A-Za-z0-9]+$", "", base) # drop ext; we set our own
55
+ stem = re.sub(r"[^A-Za-z0-9._-]+", "-", stem).strip("-") or "image"
56
+ ext = _ext_for(url, data)
57
+ name = f"{stem}{ext}"
58
+ i = 2
59
+ while name in used:
60
+ name = f"{stem}-{i}{ext}"
61
+ i += 1
62
+ used.add(name)
63
+ return name
64
+
65
+
66
+ def _remote_image_urls(text: str) -> list[str]:
67
+ """Distinct remote (http/https) image refs in ``text``, in first-seen order."""
68
+ urls: list[str] = []
69
+ seen: set[str] = set()
70
+ for rx in (_MD_IMG_RE, _HTML_IMG_RE):
71
+ for u in rx.findall(text):
72
+ if u not in seen:
73
+ seen.add(u)
74
+ urls.append(u)
75
+ return urls
76
+
77
+
78
+ def count_remote_images(doc_path: Path) -> int:
79
+ """Distinct remote image refs still in ``doc_path`` (0 ⇒ fully localized) — lets
80
+ a caller know whether another ``download_images`` pass is needed."""
81
+ return len(_remote_image_urls(doc_path.read_text(encoding="utf-8")))
82
+
83
+
84
+ def download_images(doc_path: Path, images_dir: Path, *, checkpoint_every: int = 50) -> int:
85
+ """Download the doc's remote images into ``images_dir`` and re-point refs to
86
+ ``images/<name>``. Returns the count downloaded this run; unfetchable refs are
87
+ left untouched (logged).
88
+
89
+ Resumable: each image is re-pointed in the deliverable the moment it lands (the
90
+ file IS the progress ledger — finished refs are ``images/<name>``, pending ones
91
+ stay remote), and the doc is checkpointed every ``checkpoint_every`` images, so
92
+ a run killed partway keeps what it localized. ``used`` names are seeded from
93
+ ``images_dir`` so a resumed run can't clobber a prior run's files. Re-run until
94
+ ``count_remote_images`` returns 0 (how big books beat a per-run time cap).
95
+ """
96
+ text = doc_path.read_text(encoding="utf-8")
97
+ urls = _remote_image_urls(text)
98
+ if not urls:
99
+ return 0
100
+
101
+ images_dir.mkdir(parents=True, exist_ok=True)
102
+ # Seed from a prior run's files so a resumed download can't clobber them.
103
+ used: set[str] = {p.name for p in images_dir.iterdir() if p.is_file()}
104
+ saved = 0
105
+ since_checkpoint = 0
106
+ # Longest URL first so one ref can't be a prefix of another when we re-point.
107
+ for u in sorted(urls, key=len, reverse=True):
108
+ try:
109
+ _f, data = http.fetch_bytes(u)
110
+ except Exception as exc:
111
+ log.warning("images.fetch_error", url=u, error=str(exc))
112
+ continue
113
+ name = _name_for(u, data, used)
114
+ (images_dir / name).write_bytes(data)
115
+ text = text.replace(u, f"images/{name}") # re-point now: the file is the ledger
116
+ saved += 1
117
+ since_checkpoint += 1
118
+ if since_checkpoint >= checkpoint_every:
119
+ doc_path.write_text(text, encoding="utf-8")
120
+ since_checkpoint = 0
121
+ http.polite_sleep()
122
+
123
+ doc_path.write_text(text, encoding="utf-8")
124
+ log.info("images.download", doc=str(doc_path), found=len(urls), saved=saved)
125
+ return saved
pagespring/manifest.py ADDED
@@ -0,0 +1,98 @@
1
+ """The per-slug ``manifest.json`` — the provenance record written beside each
2
+ ``incoming/<slug>/`` deliverable.
3
+
4
+ It records where a manual came from, which pattern acquired it, the downstream
5
+ pagespeak ``convert_recipe`` hint, and a content hash — so the hand-off to
6
+ pagespeak is self-describing, and so ``ingest --if-changed`` can tell whether a
7
+ re-fetch produced anything new. Pure stdlib; no network, no pattern machinery.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import hashlib
13
+ import json
14
+ from pathlib import Path
15
+ from typing import TypedDict
16
+
17
+ from pagespring import __version__
18
+
19
+ MANIFEST_NAME = "manifest.json"
20
+ SCHEMA_VERSION = 1
21
+
22
+
23
+ class Manifest(TypedDict):
24
+ """The on-disk shape of ``incoming/<slug>/manifest.json`` (schema v1)."""
25
+
26
+ schema_version: int
27
+ pagespring_version: str
28
+ source_url: str
29
+ pattern: str
30
+ slug: str
31
+ kind: str
32
+ deliverable: str
33
+ convert_recipe: list[str]
34
+ pages: int | None
35
+ bytes: int
36
+ sha256: str
37
+ images: int
38
+ ingested_at: str
39
+
40
+
41
+ def sha256_file(path: Path) -> str:
42
+ """Hex SHA-256 of ``path``'s bytes (the deliverable's content identity)."""
43
+ return hashlib.sha256(path.read_bytes()).hexdigest()
44
+
45
+
46
+ def build_manifest(
47
+ *,
48
+ source_url: str,
49
+ pattern: str,
50
+ slug: str,
51
+ kind: str,
52
+ deliverable: str,
53
+ convert_recipe: list[str],
54
+ pages: int | None,
55
+ size_bytes: int,
56
+ sha256: str,
57
+ images: int,
58
+ ingested_at: str,
59
+ ) -> Manifest:
60
+ """Assemble a manifest from one ingest's facts (stamps schema + version)."""
61
+ return {
62
+ "schema_version": SCHEMA_VERSION,
63
+ "pagespring_version": __version__,
64
+ "source_url": source_url,
65
+ "pattern": pattern,
66
+ "slug": slug,
67
+ "kind": kind,
68
+ "deliverable": deliverable,
69
+ "convert_recipe": convert_recipe,
70
+ "pages": pages,
71
+ "bytes": size_bytes,
72
+ "sha256": sha256,
73
+ "images": images,
74
+ "ingested_at": ingested_at,
75
+ }
76
+
77
+
78
+ def write_manifest(slug_dir: Path, manifest: Manifest) -> Path:
79
+ """Write ``manifest`` as pretty JSON to ``slug_dir/manifest.json``; return it."""
80
+ path = slug_dir / MANIFEST_NAME
81
+ path.write_text(json.dumps(manifest, indent=2) + "\n", encoding="utf-8")
82
+ return path
83
+
84
+
85
+ def read_manifest(slug_dir: Path) -> Manifest | None:
86
+ """Read ``slug_dir/manifest.json``; ``None`` if absent or unparseable.
87
+
88
+ Tolerant by design: a legacy slug dir (pre-manifest) or a corrupt file must
89
+ not crash ``status`` or ``--if-changed`` — they treat ``None`` as "no record".
90
+ """
91
+ path = slug_dir / MANIFEST_NAME
92
+ if not path.exists():
93
+ return None
94
+ try:
95
+ data: Manifest = json.loads(path.read_text(encoding="utf-8"))
96
+ except (json.JSONDecodeError, OSError):
97
+ return None
98
+ return data