pagespring 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pagespring/__init__.py +12 -0
- pagespring/base.py +57 -0
- pagespring/cli.py +209 -0
- pagespring/config.py +24 -0
- pagespring/http.py +93 -0
- pagespring/images.py +125 -0
- pagespring/manifest.py +98 -0
- pagespring/orchestrate.py +215 -0
- pagespring/patterns/__init__.py +6 -0
- pagespring/patterns/_apple_merge.py +168 -0
- pagespring/patterns/_docusaurus.py +92 -0
- pagespring/patterns/_gitbook.py +121 -0
- pagespring/patterns/_mkdocs.py +70 -0
- pagespring/patterns/_openapi_render.py +156 -0
- pagespring/patterns/_postman_render.py +85 -0
- pagespring/patterns/_site.py +50 -0
- pagespring/patterns/_sphinx.py +112 -0
- pagespring/patterns/api_spec.py +143 -0
- pagespring/patterns/apple_help.py +99 -0
- pagespring/patterns/archive_download.py +88 -0
- pagespring/patterns/docs_probe.py +109 -0
- pagespring/patterns/gitbook.py +89 -0
- pagespring/patterns/github_markdown.py +134 -0
- pagespring/patterns/llms_txt.py +112 -0
- pagespring/patterns/microsoft_support.py +165 -0
- pagespring/patterns/openstax.py +193 -0
- pagespring/patterns/pdf_url.py +65 -0
- pagespring/patterns/readthedocs.py +95 -0
- pagespring/patterns/zendesk_help.py +95 -0
- pagespring/py.typed +0 -0
- pagespring/registry.py +59 -0
- pagespring-0.1.0.dist-info/METADATA +92 -0
- pagespring-0.1.0.dist-info/RECORD +37 -0
- pagespring-0.1.0.dist-info/WHEEL +5 -0
- pagespring-0.1.0.dist-info/entry_points.txt +2 -0
- pagespring-0.1.0.dist-info/licenses/LICENSE +21 -0
- pagespring-0.1.0.dist-info/top_level.txt +1 -0
pagespring/__init__.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""pagespring — the lean manual acquisition + normalization layer.
|
|
2
|
+
|
|
3
|
+
Point it at a manual's URL; it recognizes the source type (a "pattern"),
|
|
4
|
+
*acquires* the raw pages, and *normalizes* them into one clean HTML/markdown
|
|
5
|
+
file with absolute asset URLs under ``incoming/<slug>/``. That clean file is the
|
|
6
|
+
deliverable; *conversion* to RAG markdown is a separate concern (**pagespeak**)
|
|
7
|
+
that consumes ``incoming/`` independently — this package neither runs nor imports it.
|
|
8
|
+
|
|
9
|
+
This package stays dependency-light (pf-core[cli] + beautifulsoup4).
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
__version__ = "0.1.0"
|
pagespring/base.py
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""The Pattern contract — the unit that ties acquire + normalize + convert
|
|
2
|
+
together for one source type.
|
|
3
|
+
|
|
4
|
+
A pattern recognizes a family of source URLs (``match``), downloads the raw
|
|
5
|
+
pages (``acquire``), turns them into one clean convertible file with absolute
|
|
6
|
+
asset URLs (``normalize``), and declares the extra ``pagespeak convert`` flags
|
|
7
|
+
its output wants (``convert_recipe``). The conversion engine itself lives in
|
|
8
|
+
pagespeak and is invoked as a subprocess — never imported here.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Literal, Protocol, runtime_checkable
|
|
16
|
+
|
|
17
|
+
# "html" -> hand the clean file to pagespeak to convert (markitdown + pipeline)
|
|
18
|
+
# "markdown" -> already the source's canonical clean form (e.g. GitBook per-page .md)
|
|
19
|
+
# "pdf" -> a downloaded PDF; already a pagespeak input (normalize passthrough)
|
|
20
|
+
SourceKind = Literal["html", "markdown", "pdf"]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class AcquireResult:
|
|
25
|
+
"""What ``acquire`` produced: a local dir of raw pages + how to treat them."""
|
|
26
|
+
|
|
27
|
+
raw_dir: Path # local dir holding the downloaded raw page(s)
|
|
28
|
+
kind: SourceKind # whether normalize emits html (for pagespeak) or markdown
|
|
29
|
+
slug: str # short id for the source; becomes the output dir name
|
|
30
|
+
pages: int | None = None # source units fetched (crawl pages / articles / files); manifest stat
|
|
31
|
+
title: str | None = None # human source title for the deliverable heading (falls back to slug)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@runtime_checkable
|
|
35
|
+
class Pattern(Protocol):
|
|
36
|
+
"""One source type's acquire/normalize/convert knowledge.
|
|
37
|
+
|
|
38
|
+
Implementations are instances (see pagespring/patterns/*); the registry holds
|
|
39
|
+
one of each. All *source-specific* knowledge (crawl rules, chrome selectors,
|
|
40
|
+
TOC walking, image-scheme resolution) lives in the pattern — pagespeak stays
|
|
41
|
+
source-agnostic.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
name: str
|
|
45
|
+
convert_recipe: list[str] # extra flags appended to `pagespeak convert`
|
|
46
|
+
|
|
47
|
+
def match(self, url: str) -> bool:
|
|
48
|
+
"""Cheap check (host/path) for whether this pattern handles ``url``."""
|
|
49
|
+
...
|
|
50
|
+
|
|
51
|
+
def acquire(self, url: str, workdir: Path) -> AcquireResult:
|
|
52
|
+
"""Download the source's raw pages into ``workdir``."""
|
|
53
|
+
...
|
|
54
|
+
|
|
55
|
+
def normalize(self, acq: AcquireResult, workdir: Path) -> Path:
|
|
56
|
+
"""Turn the raw pages into ONE clean .html/.md (absolute asset URLs)."""
|
|
57
|
+
...
|
pagespring/cli.py
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""pagespring command-line interface (Typer, via pf_core.cli).
|
|
2
|
+
|
|
3
|
+
Commands:
|
|
4
|
+
ingest <url> acquire + normalize ("fix") a manual into incoming/<slug>/
|
|
5
|
+
localize <slug> grab an already-ingested deliverable's images (resumable; --all)
|
|
6
|
+
patterns list the registered source patterns
|
|
7
|
+
classify <url> show which pattern handles a URL (no acquisition)
|
|
8
|
+
status list incoming/ deliverables (pattern, pages, size, date, source)
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from datetime import date
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from urllib.parse import urlsplit
|
|
14
|
+
|
|
15
|
+
import typer
|
|
16
|
+
from pf_core.cli import create_cli, run_cli
|
|
17
|
+
from pf_core.exceptions import InvalidInputError, PreconditionError
|
|
18
|
+
|
|
19
|
+
from pagespring import manifest
|
|
20
|
+
from pagespring.config import cfg
|
|
21
|
+
from pagespring.orchestrate import (
|
|
22
|
+
AcquireError,
|
|
23
|
+
EmptyOutputError,
|
|
24
|
+
NoPatternError,
|
|
25
|
+
localize_images,
|
|
26
|
+
run_ingest,
|
|
27
|
+
)
|
|
28
|
+
from pagespring.registry import PATTERNS, classify
|
|
29
|
+
|
|
30
|
+
app = create_cli(
|
|
31
|
+
"pagespring",
|
|
32
|
+
help="Find, download, and normalize online software manuals into incoming/.",
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@app.command()
|
|
37
|
+
def ingest(
|
|
38
|
+
url: str = typer.Argument(
|
|
39
|
+
..., help="Manual URL (e.g. a support.apple.com/guide/<app>/ welcome page)."
|
|
40
|
+
),
|
|
41
|
+
keep_raw: bool = typer.Option(
|
|
42
|
+
False, "--keep-raw", help="Keep the raw crawl alongside the source in incoming/<slug>/raw."
|
|
43
|
+
),
|
|
44
|
+
download_images: bool = typer.Option(
|
|
45
|
+
False,
|
|
46
|
+
"--download-images",
|
|
47
|
+
help="Download an html/markdown source's images into incoming/<slug>/images/ and re-point refs. No-op for PDFs.",
|
|
48
|
+
),
|
|
49
|
+
if_changed: bool = typer.Option(
|
|
50
|
+
False,
|
|
51
|
+
"--if-changed",
|
|
52
|
+
help="Skip re-staging when the re-fetch normalizes to byte-identical content (the crawl still runs).",
|
|
53
|
+
),
|
|
54
|
+
) -> None:
|
|
55
|
+
"""Acquire a manual from URL and normalize it into incoming/<slug>/."""
|
|
56
|
+
try:
|
|
57
|
+
result = run_ingest(
|
|
58
|
+
url, keep_raw=keep_raw, download_images=download_images, if_changed=if_changed
|
|
59
|
+
)
|
|
60
|
+
except NoPatternError:
|
|
61
|
+
typer.echo(
|
|
62
|
+
f"No pattern matched: {url}\n"
|
|
63
|
+
"This source needs a new pattern. Run `bin/run patterns` to see the "
|
|
64
|
+
"registered ones; src/pagespring/patterns/ shows the shape to author one.",
|
|
65
|
+
err=True,
|
|
66
|
+
)
|
|
67
|
+
raise typer.Exit(2) from None
|
|
68
|
+
except InvalidInputError as exc:
|
|
69
|
+
typer.echo(str(exc), err=True)
|
|
70
|
+
raise typer.Exit(2) from None
|
|
71
|
+
except EmptyOutputError:
|
|
72
|
+
typer.echo(
|
|
73
|
+
f"Normalize produced an empty file for {url} — the source may have "
|
|
74
|
+
"changed shape. Nothing was staged; a previous deliverable in "
|
|
75
|
+
"incoming/ is untouched.",
|
|
76
|
+
err=True,
|
|
77
|
+
)
|
|
78
|
+
raise typer.Exit(3) from None
|
|
79
|
+
except AcquireError as exc:
|
|
80
|
+
typer.echo(
|
|
81
|
+
f"Fetch failed during acquire: {exc.detail}\n"
|
|
82
|
+
f"Source: {exc.url}\n"
|
|
83
|
+
"Nothing was staged. The fetch died mid-acquire — check the URL is "
|
|
84
|
+
"reachable; re-run to retry.",
|
|
85
|
+
err=True,
|
|
86
|
+
)
|
|
87
|
+
raise typer.Exit(4) from None
|
|
88
|
+
|
|
89
|
+
typer.echo(f"pattern : {result['pattern']}")
|
|
90
|
+
typer.echo(f"slug : {result['slug']}")
|
|
91
|
+
typer.echo(f"incoming : {result['clean']}")
|
|
92
|
+
if result.get("changed") is False:
|
|
93
|
+
typer.echo(
|
|
94
|
+
"status : unchanged — source matches the existing deliverable, nothing re-staged"
|
|
95
|
+
)
|
|
96
|
+
return
|
|
97
|
+
if result.get("pages") is not None:
|
|
98
|
+
typer.echo(f"pages : {result['pages']}")
|
|
99
|
+
typer.echo(f"size : {_human_size(result['bytes'])}")
|
|
100
|
+
if result.get("images"):
|
|
101
|
+
typer.echo(f"images : {result['images']} downloaded → images/")
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
@app.command()
|
|
105
|
+
def localize(
|
|
106
|
+
slug: str = typer.Argument(
|
|
107
|
+
None, help="Book slug under incoming/ to localize images for (omit when using --all)."
|
|
108
|
+
),
|
|
109
|
+
all_books: bool = typer.Option(
|
|
110
|
+
False, "--all", help="Localize images for every incoming/<slug>/."
|
|
111
|
+
),
|
|
112
|
+
) -> None:
|
|
113
|
+
"""Download an already-ingested deliverable's remote images into images/ and
|
|
114
|
+
re-point refs — no re-crawl. Resumable: re-run until none remain, so a book too
|
|
115
|
+
big to localize in one pass finishes across runs."""
|
|
116
|
+
if all_books:
|
|
117
|
+
incoming = Path(cfg.INCOMING_DIR)
|
|
118
|
+
targets = (
|
|
119
|
+
sorted(p.name for p in incoming.glob("*") if p.is_dir()) if incoming.is_dir() else []
|
|
120
|
+
)
|
|
121
|
+
elif slug:
|
|
122
|
+
targets = [slug]
|
|
123
|
+
else:
|
|
124
|
+
typer.echo("Give a slug or --all.", err=True)
|
|
125
|
+
raise typer.Exit(2)
|
|
126
|
+
|
|
127
|
+
for s in targets:
|
|
128
|
+
try:
|
|
129
|
+
r = localize_images(s)
|
|
130
|
+
except PreconditionError as exc:
|
|
131
|
+
typer.echo(f"skip {s}: {exc}", err=True)
|
|
132
|
+
continue
|
|
133
|
+
tail = "done" if r["remaining"] == 0 else f"{r['remaining']} remaining — re-run to continue"
|
|
134
|
+
typer.echo(f"{s}: +{r['localized']} images (total {r['images_total']}) — {tail}")
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@app.command()
|
|
138
|
+
def patterns() -> None:
|
|
139
|
+
"""List the registered source patterns and their convert recipes."""
|
|
140
|
+
for p in PATTERNS:
|
|
141
|
+
recipe = " ".join(p.convert_recipe) or "(none)"
|
|
142
|
+
typer.echo(f"{p.name:12} convert-recipe: {recipe}")
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
@app.command("classify")
|
|
146
|
+
def classify_cmd(
|
|
147
|
+
url: str = typer.Argument(..., help="URL to test against the pattern registry."),
|
|
148
|
+
) -> None:
|
|
149
|
+
"""Show which pattern (if any) handles a URL — no acquisition, no network."""
|
|
150
|
+
p = classify(url)
|
|
151
|
+
typer.echo(p.name if p else "(no pattern matched)")
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _human_size(n: int) -> str:
|
|
155
|
+
size = float(n)
|
|
156
|
+
for unit in ("B", "KB", "MB"):
|
|
157
|
+
if size < 1024:
|
|
158
|
+
return f"{size:.0f} {unit}" if unit == "B" else f"{size:.1f} {unit}"
|
|
159
|
+
size /= 1024
|
|
160
|
+
return f"{size:.1f} GB"
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
@app.command()
|
|
164
|
+
def status() -> None:
|
|
165
|
+
"""One row per incoming/<slug>/, read from its manifest.json: deliverable,
|
|
166
|
+
pattern, pages, size, ingest date, and source host. Legacy (pre-manifest)
|
|
167
|
+
dirs fall back to the deliverable file's own facts. (Conversion into the
|
|
168
|
+
manuals corpus is pagespeak's job — downstream of this tool, out of its view.)"""
|
|
169
|
+
incoming = Path(cfg.INCOMING_DIR)
|
|
170
|
+
slugs = sorted(p for p in incoming.glob("*") if p.is_dir()) if incoming.is_dir() else []
|
|
171
|
+
if not slugs:
|
|
172
|
+
typer.echo("(nothing in incoming/ — run `bin/run ingest <url>`)")
|
|
173
|
+
return
|
|
174
|
+
for d in slugs:
|
|
175
|
+
typer.echo(_status_row(d))
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _status_row(slug_dir: Path) -> str:
|
|
179
|
+
"""One status line from the slug's manifest; for legacy (pre-manifest) dirs,
|
|
180
|
+
fall back to the first non-manifest deliverable file's own facts."""
|
|
181
|
+
m = manifest.read_manifest(slug_dir)
|
|
182
|
+
if m is not None:
|
|
183
|
+
deliverable = slug_dir / m["deliverable"]
|
|
184
|
+
size = deliverable.stat().st_size if deliverable.exists() else m["bytes"]
|
|
185
|
+
pages = str(m["pages"]) if m["pages"] is not None else "-"
|
|
186
|
+
host = urlsplit(m["source_url"]).netloc or "-"
|
|
187
|
+
return (
|
|
188
|
+
f"{slug_dir.name:24} {m['deliverable']:32} {m['pattern']:14} "
|
|
189
|
+
f"{pages:>5} {_human_size(size):>9} {m['ingested_at'][:10]} {host}"
|
|
190
|
+
)
|
|
191
|
+
files = sorted(
|
|
192
|
+
p for p in slug_dir.iterdir() if p.is_file() and p.name != manifest.MANIFEST_NAME
|
|
193
|
+
)
|
|
194
|
+
if not files:
|
|
195
|
+
return f"{slug_dir.name:24} {'(no clean file)':32} {'-':14} {'-':>5} {'-':>9} - -"
|
|
196
|
+
f = files[0]
|
|
197
|
+
when = date.fromtimestamp(f.stat().st_mtime).isoformat()
|
|
198
|
+
return (
|
|
199
|
+
f"{slug_dir.name:24} {f.name:32} {'-':14} "
|
|
200
|
+
f"{'-':>5} {_human_size(f.stat().st_size):>9} {when} -"
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def main() -> None:
|
|
205
|
+
run_cli(app)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
if __name__ == "__main__":
|
|
209
|
+
main()
|
pagespring/config.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""pagespring configuration — a pf_core.config.AppConfig subclass.
|
|
2
|
+
|
|
3
|
+
All settings are overridable via environment variables / .env.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from pf_core.config import AppConfig
|
|
9
|
+
|
|
10
|
+
# src/pagespring/config.py → parents[2] is the project root (editable install).
|
|
11
|
+
_project_root = Path(__file__).resolve().parents[2]
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class PagespringConfig(AppConfig):
|
|
15
|
+
"""pagespring settings."""
|
|
16
|
+
|
|
17
|
+
APP_NAME: str = "pagespring"
|
|
18
|
+
|
|
19
|
+
# The deliverable: one incoming/<slug>/ per manual — the clean
|
|
20
|
+
# acquired+normalized file. A separate step (pagespeak) consumes these.
|
|
21
|
+
INCOMING_DIR: str = "incoming"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
cfg = PagespringConfig(env_file=_project_root / ".env")
|
pagespring/http.py
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Tiny shared HTTP fetch over the standard library (urllib).
|
|
2
|
+
|
|
3
|
+
Deliberately no httpx dependency — the acquire step only does plain GETs.
|
|
4
|
+
Provides an identifying User-Agent (override via PAGESPRING_UA), a timeout,
|
|
5
|
+
status-aware retries (permanent 4xx fail fast; 429 honors Retry-After;
|
|
6
|
+
5xx/network errors back off), and a polite inter-request delay for crawls.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import time
|
|
12
|
+
import urllib.error
|
|
13
|
+
import urllib.request
|
|
14
|
+
|
|
15
|
+
from pf_core.utils.env import resolve_str
|
|
16
|
+
|
|
17
|
+
from pagespring import __version__
|
|
18
|
+
|
|
19
|
+
_RETRY_AFTER_CAP = 30.0 # seconds — don't let a server park a crawl for minutes
|
|
20
|
+
|
|
21
|
+
_UA_DEFAULT = f"pagespring/{__version__} (+https://github.com/phierceweb/pagespring)"
|
|
22
|
+
_UA_ENV_VAR = "PAGESPRING_UA"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _ua() -> str:
|
|
26
|
+
"""The identifying default UA, or PAGESPRING_UA for sources that need another."""
|
|
27
|
+
return resolve_str(None, _UA_ENV_VAR, default=_UA_DEFAULT) or _UA_DEFAULT
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _request(url: str) -> urllib.request.Request:
|
|
31
|
+
return urllib.request.Request(
|
|
32
|
+
url,
|
|
33
|
+
headers={
|
|
34
|
+
"User-Agent": _ua(),
|
|
35
|
+
"Accept": "*/*",
|
|
36
|
+
"Accept-Language": "en-US,en;q=0.9",
|
|
37
|
+
},
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _retry_after(exc: urllib.error.HTTPError, attempt: int) -> float:
|
|
42
|
+
"""Seconds to wait on a 429 — the server's Retry-After (capped) when sane,
|
|
43
|
+
else the normal backoff."""
|
|
44
|
+
try:
|
|
45
|
+
return min(float(exc.headers.get("Retry-After", "")), _RETRY_AFTER_CAP)
|
|
46
|
+
except ValueError:
|
|
47
|
+
return 0.5 * (attempt + 1)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _read(url: str, timeout: float, retries: int) -> tuple[str, bytes, str | None]:
|
|
51
|
+
last: Exception | None = None
|
|
52
|
+
for attempt in range(retries + 1):
|
|
53
|
+
try:
|
|
54
|
+
with urllib.request.urlopen(_request(url), timeout=timeout) as r:
|
|
55
|
+
return r.geturl(), r.read(), r.headers.get_content_charset()
|
|
56
|
+
except urllib.error.HTTPError as exc:
|
|
57
|
+
last = exc
|
|
58
|
+
if exc.code == 429:
|
|
59
|
+
if attempt < retries:
|
|
60
|
+
time.sleep(_retry_after(exc, attempt))
|
|
61
|
+
elif 400 <= exc.code < 500 and exc.code != 408:
|
|
62
|
+
raise # permanent client error — retrying can't help
|
|
63
|
+
elif attempt < retries: # 5xx / 408
|
|
64
|
+
time.sleep(0.5 * (attempt + 1))
|
|
65
|
+
except Exception as exc: # URLError, timeout, connection reset, …
|
|
66
|
+
last = exc
|
|
67
|
+
if attempt < retries:
|
|
68
|
+
time.sleep(0.5 * (attempt + 1))
|
|
69
|
+
raise last # type: ignore[misc]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def fetch_text(
|
|
73
|
+
url: str, *, timeout: float = 30, retries: int = 2, encoding: str | None = None
|
|
74
|
+
) -> tuple[str, str]:
|
|
75
|
+
"""Return (final_url, decoded_text) after following redirects.
|
|
76
|
+
|
|
77
|
+
Decodes with ``encoding`` when given, else the response's Content-Type
|
|
78
|
+
charset, else utf-8 — always with replacement, never raising."""
|
|
79
|
+
final_url, raw, charset = _read(url, timeout, retries)
|
|
80
|
+
return final_url, raw.decode(encoding or charset or "utf-8", "replace")
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def fetch_bytes(url: str, *, timeout: float = 180, retries: int = 2) -> tuple[str, bytes]:
|
|
84
|
+
"""Return (final_url, raw_bytes) — for binary downloads (PDFs, archives,
|
|
85
|
+
images). Longer default timeout than fetch_text: vendor PDFs/doc archives
|
|
86
|
+
can be tens of MB on slow CDNs."""
|
|
87
|
+
final_url, raw, _charset = _read(url, timeout, retries)
|
|
88
|
+
return final_url, raw
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def polite_sleep(seconds: float = 0.25) -> None:
|
|
92
|
+
"""Sleep between crawl requests to avoid hammering the source."""
|
|
93
|
+
time.sleep(seconds)
|
pagespring/images.py
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""Optional image localizer for HTML/markdown ingests.
|
|
2
|
+
|
|
3
|
+
Downloads a deliverable's remote images into a sibling ``images/`` dir and
|
|
4
|
+
re-points the refs at them — for a self-contained ``incoming/<slug>/``, and to
|
|
5
|
+
capture images behind expiring or tokened URLs (e.g. GitBook's
|
|
6
|
+
``?alt=media&token=…``) while they still resolve.
|
|
7
|
+
|
|
8
|
+
Opt-in via ``bin/run ingest --download-images`` or ``bin/run localize``.
|
|
9
|
+
Stdlib fetch only (``pagespring.http``).
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import re
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from urllib.parse import urlparse
|
|
17
|
+
|
|
18
|
+
from pf_core.log import get_logger
|
|
19
|
+
|
|
20
|
+
from pagespring import http
|
|
21
|
+
|
|
22
|
+
log = get_logger(__name__)
|
|
23
|
+
|
|
24
|
+
# Markdown  and HTML <img … src="http…">.
|
|
25
|
+
_MD_IMG_RE = re.compile(r"!\[[^\]]*\]\((https?://[^)\s]+)\)")
|
|
26
|
+
_HTML_IMG_RE = re.compile(r"""<img\b[^>]*?\bsrc=["'](https?://[^"']+)["']""", re.IGNORECASE)
|
|
27
|
+
|
|
28
|
+
_IMG_EXTS = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".svg", ".bmp", ".tif", ".tiff"}
|
|
29
|
+
_MAGIC = (
|
|
30
|
+
(b"\x89PNG\r\n\x1a\n", ".png"),
|
|
31
|
+
(b"\xff\xd8\xff", ".jpg"),
|
|
32
|
+
(b"GIF87a", ".gif"),
|
|
33
|
+
(b"GIF89a", ".gif"),
|
|
34
|
+
(b"<svg", ".svg"),
|
|
35
|
+
(b"<?xml", ".svg"),
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _ext_for(url: str, data: bytes) -> str:
|
|
40
|
+
suffix = Path(urlparse(url).path).suffix.lower()
|
|
41
|
+
if suffix in _IMG_EXTS:
|
|
42
|
+
return ".jpg" if suffix == ".jpeg" else (".tiff" if suffix == ".tif" else suffix)
|
|
43
|
+
head = data[:16]
|
|
44
|
+
for magic, ext in _MAGIC:
|
|
45
|
+
if head.startswith(magic):
|
|
46
|
+
return ext
|
|
47
|
+
if head[:4] == b"RIFF" and b"WEBP" in data[:16]:
|
|
48
|
+
return ".webp"
|
|
49
|
+
return ".img"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _name_for(url: str, data: bytes, used: set[str]) -> str:
|
|
53
|
+
base = Path(urlparse(url).path).name
|
|
54
|
+
stem = re.sub(r"\.[A-Za-z0-9]+$", "", base) # drop ext; we set our own
|
|
55
|
+
stem = re.sub(r"[^A-Za-z0-9._-]+", "-", stem).strip("-") or "image"
|
|
56
|
+
ext = _ext_for(url, data)
|
|
57
|
+
name = f"{stem}{ext}"
|
|
58
|
+
i = 2
|
|
59
|
+
while name in used:
|
|
60
|
+
name = f"{stem}-{i}{ext}"
|
|
61
|
+
i += 1
|
|
62
|
+
used.add(name)
|
|
63
|
+
return name
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _remote_image_urls(text: str) -> list[str]:
|
|
67
|
+
"""Distinct remote (http/https) image refs in ``text``, in first-seen order."""
|
|
68
|
+
urls: list[str] = []
|
|
69
|
+
seen: set[str] = set()
|
|
70
|
+
for rx in (_MD_IMG_RE, _HTML_IMG_RE):
|
|
71
|
+
for u in rx.findall(text):
|
|
72
|
+
if u not in seen:
|
|
73
|
+
seen.add(u)
|
|
74
|
+
urls.append(u)
|
|
75
|
+
return urls
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def count_remote_images(doc_path: Path) -> int:
|
|
79
|
+
"""Distinct remote image refs still in ``doc_path`` (0 ⇒ fully localized) — lets
|
|
80
|
+
a caller know whether another ``download_images`` pass is needed."""
|
|
81
|
+
return len(_remote_image_urls(doc_path.read_text(encoding="utf-8")))
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def download_images(doc_path: Path, images_dir: Path, *, checkpoint_every: int = 50) -> int:
|
|
85
|
+
"""Download the doc's remote images into ``images_dir`` and re-point refs to
|
|
86
|
+
``images/<name>``. Returns the count downloaded this run; unfetchable refs are
|
|
87
|
+
left untouched (logged).
|
|
88
|
+
|
|
89
|
+
Resumable: each image is re-pointed in the deliverable the moment it lands (the
|
|
90
|
+
file IS the progress ledger — finished refs are ``images/<name>``, pending ones
|
|
91
|
+
stay remote), and the doc is checkpointed every ``checkpoint_every`` images, so
|
|
92
|
+
a run killed partway keeps what it localized. ``used`` names are seeded from
|
|
93
|
+
``images_dir`` so a resumed run can't clobber a prior run's files. Re-run until
|
|
94
|
+
``count_remote_images`` returns 0 (how big books beat a per-run time cap).
|
|
95
|
+
"""
|
|
96
|
+
text = doc_path.read_text(encoding="utf-8")
|
|
97
|
+
urls = _remote_image_urls(text)
|
|
98
|
+
if not urls:
|
|
99
|
+
return 0
|
|
100
|
+
|
|
101
|
+
images_dir.mkdir(parents=True, exist_ok=True)
|
|
102
|
+
# Seed from a prior run's files so a resumed download can't clobber them.
|
|
103
|
+
used: set[str] = {p.name for p in images_dir.iterdir() if p.is_file()}
|
|
104
|
+
saved = 0
|
|
105
|
+
since_checkpoint = 0
|
|
106
|
+
# Longest URL first so one ref can't be a prefix of another when we re-point.
|
|
107
|
+
for u in sorted(urls, key=len, reverse=True):
|
|
108
|
+
try:
|
|
109
|
+
_f, data = http.fetch_bytes(u)
|
|
110
|
+
except Exception as exc:
|
|
111
|
+
log.warning("images.fetch_error", url=u, error=str(exc))
|
|
112
|
+
continue
|
|
113
|
+
name = _name_for(u, data, used)
|
|
114
|
+
(images_dir / name).write_bytes(data)
|
|
115
|
+
text = text.replace(u, f"images/{name}") # re-point now: the file is the ledger
|
|
116
|
+
saved += 1
|
|
117
|
+
since_checkpoint += 1
|
|
118
|
+
if since_checkpoint >= checkpoint_every:
|
|
119
|
+
doc_path.write_text(text, encoding="utf-8")
|
|
120
|
+
since_checkpoint = 0
|
|
121
|
+
http.polite_sleep()
|
|
122
|
+
|
|
123
|
+
doc_path.write_text(text, encoding="utf-8")
|
|
124
|
+
log.info("images.download", doc=str(doc_path), found=len(urls), saved=saved)
|
|
125
|
+
return saved
|
pagespring/manifest.py
ADDED
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""The per-slug ``manifest.json`` — the provenance record written beside each
|
|
2
|
+
``incoming/<slug>/`` deliverable.
|
|
3
|
+
|
|
4
|
+
It records where a manual came from, which pattern acquired it, the downstream
|
|
5
|
+
pagespeak ``convert_recipe`` hint, and a content hash — so the hand-off to
|
|
6
|
+
pagespeak is self-describing, and so ``ingest --if-changed`` can tell whether a
|
|
7
|
+
re-fetch produced anything new. Pure stdlib; no network, no pattern machinery.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import hashlib
|
|
13
|
+
import json
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import TypedDict
|
|
16
|
+
|
|
17
|
+
from pagespring import __version__
|
|
18
|
+
|
|
19
|
+
MANIFEST_NAME = "manifest.json"
|
|
20
|
+
SCHEMA_VERSION = 1
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class Manifest(TypedDict):
|
|
24
|
+
"""The on-disk shape of ``incoming/<slug>/manifest.json`` (schema v1)."""
|
|
25
|
+
|
|
26
|
+
schema_version: int
|
|
27
|
+
pagespring_version: str
|
|
28
|
+
source_url: str
|
|
29
|
+
pattern: str
|
|
30
|
+
slug: str
|
|
31
|
+
kind: str
|
|
32
|
+
deliverable: str
|
|
33
|
+
convert_recipe: list[str]
|
|
34
|
+
pages: int | None
|
|
35
|
+
bytes: int
|
|
36
|
+
sha256: str
|
|
37
|
+
images: int
|
|
38
|
+
ingested_at: str
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def sha256_file(path: Path) -> str:
|
|
42
|
+
"""Hex SHA-256 of ``path``'s bytes (the deliverable's content identity)."""
|
|
43
|
+
return hashlib.sha256(path.read_bytes()).hexdigest()
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def build_manifest(
|
|
47
|
+
*,
|
|
48
|
+
source_url: str,
|
|
49
|
+
pattern: str,
|
|
50
|
+
slug: str,
|
|
51
|
+
kind: str,
|
|
52
|
+
deliverable: str,
|
|
53
|
+
convert_recipe: list[str],
|
|
54
|
+
pages: int | None,
|
|
55
|
+
size_bytes: int,
|
|
56
|
+
sha256: str,
|
|
57
|
+
images: int,
|
|
58
|
+
ingested_at: str,
|
|
59
|
+
) -> Manifest:
|
|
60
|
+
"""Assemble a manifest from one ingest's facts (stamps schema + version)."""
|
|
61
|
+
return {
|
|
62
|
+
"schema_version": SCHEMA_VERSION,
|
|
63
|
+
"pagespring_version": __version__,
|
|
64
|
+
"source_url": source_url,
|
|
65
|
+
"pattern": pattern,
|
|
66
|
+
"slug": slug,
|
|
67
|
+
"kind": kind,
|
|
68
|
+
"deliverable": deliverable,
|
|
69
|
+
"convert_recipe": convert_recipe,
|
|
70
|
+
"pages": pages,
|
|
71
|
+
"bytes": size_bytes,
|
|
72
|
+
"sha256": sha256,
|
|
73
|
+
"images": images,
|
|
74
|
+
"ingested_at": ingested_at,
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def write_manifest(slug_dir: Path, manifest: Manifest) -> Path:
|
|
79
|
+
"""Write ``manifest`` as pretty JSON to ``slug_dir/manifest.json``; return it."""
|
|
80
|
+
path = slug_dir / MANIFEST_NAME
|
|
81
|
+
path.write_text(json.dumps(manifest, indent=2) + "\n", encoding="utf-8")
|
|
82
|
+
return path
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def read_manifest(slug_dir: Path) -> Manifest | None:
|
|
86
|
+
"""Read ``slug_dir/manifest.json``; ``None`` if absent or unparseable.
|
|
87
|
+
|
|
88
|
+
Tolerant by design: a legacy slug dir (pre-manifest) or a corrupt file must
|
|
89
|
+
not crash ``status`` or ``--if-changed`` — they treat ``None`` as "no record".
|
|
90
|
+
"""
|
|
91
|
+
path = slug_dir / MANIFEST_NAME
|
|
92
|
+
if not path.exists():
|
|
93
|
+
return None
|
|
94
|
+
try:
|
|
95
|
+
data: Manifest = json.loads(path.read_text(encoding="utf-8"))
|
|
96
|
+
except (json.JSONDecodeError, OSError):
|
|
97
|
+
return None
|
|
98
|
+
return data
|