pagespring 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. pagespring-0.1.0/LICENSE +21 -0
  2. pagespring-0.1.0/PKG-INFO +92 -0
  3. pagespring-0.1.0/README.md +61 -0
  4. pagespring-0.1.0/pyproject.toml +75 -0
  5. pagespring-0.1.0/setup.cfg +4 -0
  6. pagespring-0.1.0/src/pagespring/__init__.py +12 -0
  7. pagespring-0.1.0/src/pagespring/base.py +57 -0
  8. pagespring-0.1.0/src/pagespring/cli.py +209 -0
  9. pagespring-0.1.0/src/pagespring/config.py +24 -0
  10. pagespring-0.1.0/src/pagespring/http.py +93 -0
  11. pagespring-0.1.0/src/pagespring/images.py +125 -0
  12. pagespring-0.1.0/src/pagespring/manifest.py +98 -0
  13. pagespring-0.1.0/src/pagespring/orchestrate.py +215 -0
  14. pagespring-0.1.0/src/pagespring/patterns/__init__.py +6 -0
  15. pagespring-0.1.0/src/pagespring/patterns/_apple_merge.py +168 -0
  16. pagespring-0.1.0/src/pagespring/patterns/_docusaurus.py +92 -0
  17. pagespring-0.1.0/src/pagespring/patterns/_gitbook.py +121 -0
  18. pagespring-0.1.0/src/pagespring/patterns/_mkdocs.py +70 -0
  19. pagespring-0.1.0/src/pagespring/patterns/_openapi_render.py +156 -0
  20. pagespring-0.1.0/src/pagespring/patterns/_postman_render.py +85 -0
  21. pagespring-0.1.0/src/pagespring/patterns/_site.py +50 -0
  22. pagespring-0.1.0/src/pagespring/patterns/_sphinx.py +112 -0
  23. pagespring-0.1.0/src/pagespring/patterns/api_spec.py +143 -0
  24. pagespring-0.1.0/src/pagespring/patterns/apple_help.py +99 -0
  25. pagespring-0.1.0/src/pagespring/patterns/archive_download.py +88 -0
  26. pagespring-0.1.0/src/pagespring/patterns/docs_probe.py +109 -0
  27. pagespring-0.1.0/src/pagespring/patterns/gitbook.py +89 -0
  28. pagespring-0.1.0/src/pagespring/patterns/github_markdown.py +134 -0
  29. pagespring-0.1.0/src/pagespring/patterns/llms_txt.py +112 -0
  30. pagespring-0.1.0/src/pagespring/patterns/microsoft_support.py +165 -0
  31. pagespring-0.1.0/src/pagespring/patterns/openstax.py +193 -0
  32. pagespring-0.1.0/src/pagespring/patterns/pdf_url.py +65 -0
  33. pagespring-0.1.0/src/pagespring/patterns/readthedocs.py +95 -0
  34. pagespring-0.1.0/src/pagespring/patterns/zendesk_help.py +95 -0
  35. pagespring-0.1.0/src/pagespring/py.typed +0 -0
  36. pagespring-0.1.0/src/pagespring/registry.py +59 -0
  37. pagespring-0.1.0/src/pagespring.egg-info/PKG-INFO +92 -0
  38. pagespring-0.1.0/src/pagespring.egg-info/SOURCES.txt +63 -0
  39. pagespring-0.1.0/src/pagespring.egg-info/dependency_links.txt +1 -0
  40. pagespring-0.1.0/src/pagespring.egg-info/entry_points.txt +2 -0
  41. pagespring-0.1.0/src/pagespring.egg-info/requires.txt +9 -0
  42. pagespring-0.1.0/src/pagespring.egg-info/top_level.txt +1 -0
  43. pagespring-0.1.0/tests/test_api_spec.py +365 -0
  44. pagespring-0.1.0/tests/test_apple_help.py +102 -0
  45. pagespring-0.1.0/tests/test_archive_download.py +42 -0
  46. pagespring-0.1.0/tests/test_cli.py +227 -0
  47. pagespring-0.1.0/tests/test_docs_probe.py +123 -0
  48. pagespring-0.1.0/tests/test_docusaurus.py +74 -0
  49. pagespring-0.1.0/tests/test_gitbook.py +137 -0
  50. pagespring-0.1.0/tests/test_github_markdown.py +67 -0
  51. pagespring-0.1.0/tests/test_http.py +125 -0
  52. pagespring-0.1.0/tests/test_images.py +109 -0
  53. pagespring-0.1.0/tests/test_llms_txt.py +60 -0
  54. pagespring-0.1.0/tests/test_manifest.py +67 -0
  55. pagespring-0.1.0/tests/test_microsoft_support.py +205 -0
  56. pagespring-0.1.0/tests/test_mkdocs.py +73 -0
  57. pagespring-0.1.0/tests/test_openstax.py +180 -0
  58. pagespring-0.1.0/tests/test_orchestrate.py +282 -0
  59. pagespring-0.1.0/tests/test_pdf_url.py +47 -0
  60. pagespring-0.1.0/tests/test_readthedocs.py +92 -0
  61. pagespring-0.1.0/tests/test_registry.py +61 -0
  62. pagespring-0.1.0/tests/test_scaffold.py +21 -0
  63. pagespring-0.1.0/tests/test_site.py +44 -0
  64. pagespring-0.1.0/tests/test_sphinx.py +134 -0
  65. pagespring-0.1.0/tests/test_zendesk_help.py +90 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Mike Farr
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,92 @@
1
+ Metadata-Version: 2.4
2
+ Name: pagespring
3
+ Version: 0.1.0
4
+ Summary: Acquire and normalize online software manuals into clean, convertible source files — the acquisition front-end to pagespeak.
5
+ Author: Mike Farr
6
+ License-Expression: MIT
7
+ Project-URL: Repository, https://github.com/phierceweb/pagespring
8
+ Project-URL: Changelog, https://github.com/phierceweb/pagespring/blob/main/CHANGELOG.md
9
+ Project-URL: Issues, https://github.com/phierceweb/pagespring/issues
10
+ Keywords: manuals,documentation,ingest,acquisition,rag,markdown
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.11
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Topic :: Text Processing :: Markup :: Markdown
18
+ Classifier: Typing :: Typed
19
+ Requires-Python: >=3.11
20
+ Description-Content-Type: text/markdown
21
+ License-File: LICENSE
22
+ Requires-Dist: beautifulsoup4>=4.12
23
+ Requires-Dist: pyyaml>=6.0
24
+ Requires-Dist: pf-core[cli]~=0.4.1
25
+ Provides-Extra: dev
26
+ Requires-Dist: pytest>=8.0; extra == "dev"
27
+ Requires-Dist: ruff>=0.6; extra == "dev"
28
+ Requires-Dist: mypy>=1.10; extra == "dev"
29
+ Requires-Dist: pre-commit>=3.7; extra == "dev"
30
+ Dynamic: license-file
31
+
32
+ # pagespring
33
+
34
+ Acquire and normalize online software manuals into clean, convertible source
35
+ files — the **acquisition front-end to
36
+ [pagespeak](https://github.com/phierceweb/pagespeak)**.
37
+
38
+ Point it at a manual's URL. A *pattern* recognizes the source type, *acquires*
39
+ the raw pages (stdlib `urllib`), and *normalizes* them into ONE clean
40
+ HTML/markdown file with absolute asset URLs under `incoming/<slug>/`. That clean
41
+ file is the deliverable; converting it into the finished RAG corpus is a separate
42
+ step (pagespeak) that consumes `incoming/` on its own — pagespring never runs it.
43
+
44
+ Lean by design: `pf-core[cli]` + `beautifulsoup4`, stdlib fetch, no ML stack.
45
+
46
+ ## Intended use
47
+
48
+ pagespring is for **publicly available documentation** — vendor manuals, help
49
+ centers, open textbooks, API specs. It fetches only what the source serves to
50
+ any reader: there is no login/session handling, no paywall traversal, and no
51
+ bot-detection evasion. It is a **polite client**: it identifies itself with a
52
+ `pagespring/<version>` User-Agent (see `.env.example` to override), honors
53
+ `429 Retry-After`, backs off on server errors, paces crawl requests, and caps
54
+ crawl sizes.
55
+
56
+ It is a *user-invoked, one-manual-at-a-time* archiver — closer to "Save Page
57
+ As" than to an autonomous crawler — so it does not consult `robots.txt`
58
+ (which governs bots that discover URLs on their own; you supply the URL).
59
+ Before mirroring a site, check its terms of use. What you may do with the
60
+ acquired copy (personal RAG corpus, internal search, redistribution) is
61
+ governed by the source's license — the deliverable under `incoming/` stays on
62
+ your machine, and nothing is re-published by this tool.
63
+
64
+ ## Install
65
+
66
+ ```bash
67
+ pip install pagespring
68
+ ```
69
+
70
+ ## Quick start
71
+
72
+ ```bash
73
+ pagespring ingest https://docs.tableplus.com # acquire + normalize → incoming/tableplus/
74
+ pagespring localize <slug> # pull a deliverable's images later (resumable; --all)
75
+ pagespring patterns # list the source patterns
76
+ pagespring classify <url> # which pattern handles a URL (no fetch)
77
+ pagespring status # what's been acquired
78
+ ```
79
+
80
+ Deliverables land in `./incoming/<slug>/` under the directory you run from.
81
+
82
+ ## Dev
83
+
84
+ ```bash
85
+ bin/setup # clone → venv + editable install with dev extras
86
+ bin/test # pytest
87
+ bin/lint # ruff check + ruff format --check + mypy (strict)
88
+ ```
89
+
90
+ See [docs/usage.md](docs/usage.md) for the full command set and
91
+ [docs/architecture.md](docs/architecture.md) for the acquire → normalize flow
92
+ and how to add a new source pattern.
@@ -0,0 +1,61 @@
1
+ # pagespring
2
+
3
+ Acquire and normalize online software manuals into clean, convertible source
4
+ files — the **acquisition front-end to
5
+ [pagespeak](https://github.com/phierceweb/pagespeak)**.
6
+
7
+ Point it at a manual's URL. A *pattern* recognizes the source type, *acquires*
8
+ the raw pages (stdlib `urllib`), and *normalizes* them into ONE clean
9
+ HTML/markdown file with absolute asset URLs under `incoming/<slug>/`. That clean
10
+ file is the deliverable; converting it into the finished RAG corpus is a separate
11
+ step (pagespeak) that consumes `incoming/` on its own — pagespring never runs it.
12
+
13
+ Lean by design: `pf-core[cli]` + `beautifulsoup4`, stdlib fetch, no ML stack.
14
+
15
+ ## Intended use
16
+
17
+ pagespring is for **publicly available documentation** — vendor manuals, help
18
+ centers, open textbooks, API specs. It fetches only what the source serves to
19
+ any reader: there is no login/session handling, no paywall traversal, and no
20
+ bot-detection evasion. It is a **polite client**: it identifies itself with a
21
+ `pagespring/<version>` User-Agent (see `.env.example` to override), honors
22
+ `429 Retry-After`, backs off on server errors, paces crawl requests, and caps
23
+ crawl sizes.
24
+
25
+ It is a *user-invoked, one-manual-at-a-time* archiver — closer to "Save Page
26
+ As" than to an autonomous crawler — so it does not consult `robots.txt`
27
+ (which governs bots that discover URLs on their own; you supply the URL).
28
+ Before mirroring a site, check its terms of use. What you may do with the
29
+ acquired copy (personal RAG corpus, internal search, redistribution) is
30
+ governed by the source's license — the deliverable under `incoming/` stays on
31
+ your machine, and nothing is re-published by this tool.
32
+
33
+ ## Install
34
+
35
+ ```bash
36
+ pip install pagespring
37
+ ```
38
+
39
+ ## Quick start
40
+
41
+ ```bash
42
+ pagespring ingest https://docs.tableplus.com # acquire + normalize → incoming/tableplus/
43
+ pagespring localize <slug> # pull a deliverable's images later (resumable; --all)
44
+ pagespring patterns # list the source patterns
45
+ pagespring classify <url> # which pattern handles a URL (no fetch)
46
+ pagespring status # what's been acquired
47
+ ```
48
+
49
+ Deliverables land in `./incoming/<slug>/` under the directory you run from.
50
+
51
+ ## Dev
52
+
53
+ ```bash
54
+ bin/setup # clone → venv + editable install with dev extras
55
+ bin/test # pytest
56
+ bin/lint # ruff check + ruff format --check + mypy (strict)
57
+ ```
58
+
59
+ See [docs/usage.md](docs/usage.md) for the full command set and
60
+ [docs/architecture.md](docs/architecture.md) for the acquire → normalize flow
61
+ and how to add a new source pattern.
@@ -0,0 +1,75 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "pagespring"
7
+ version = "0.1.0"
8
+ description = "Acquire and normalize online software manuals into clean, convertible source files — the acquisition front-end to pagespeak."
9
+ readme = "README.md"
10
+ requires-python = ">=3.11"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Mike Farr" }]
14
+ keywords = ["manuals", "documentation", "ingest", "acquisition", "rag", "markdown"]
15
+ classifiers = [
16
+ "Development Status :: 4 - Beta",
17
+ "Intended Audience :: Developers",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3.11",
20
+ "Programming Language :: Python :: 3.12",
21
+ "Operating System :: OS Independent",
22
+ "Topic :: Text Processing :: Markup :: Markdown",
23
+ "Typing :: Typed",
24
+ ]
25
+ dependencies = [
26
+ # Lean by design: stdlib urllib fetches, BeautifulSoup parses, pf-core[cli]
27
+ # gives the Typer scaffolding + AppConfig + structured logging. No ML stack.
28
+ "beautifulsoup4>=4.12",
29
+ # YAML OpenAPI specs (JSON is stdlib); pure-data parser, no transport.
30
+ "pyyaml>=6.0",
31
+ "pf-core[cli]~=0.4.1",
32
+ ]
33
+
34
+ [project.optional-dependencies]
35
+ dev = [
36
+ "pytest>=8.0",
37
+ "ruff>=0.6",
38
+ "mypy>=1.10",
39
+ "pre-commit>=3.7",
40
+ ]
41
+
42
+ [project.scripts]
43
+ pagespring = "pagespring.cli:main"
44
+
45
+ [project.urls]
46
+ Repository = "https://github.com/phierceweb/pagespring"
47
+ Changelog = "https://github.com/phierceweb/pagespring/blob/main/CHANGELOG.md"
48
+ Issues = "https://github.com/phierceweb/pagespring/issues"
49
+
50
+ [tool.setuptools.packages.find]
51
+ where = ["src"]
52
+
53
+ [tool.setuptools.package-data]
54
+ pagespring = ["py.typed"]
55
+
56
+ [tool.ruff]
57
+ line-length = 100
58
+ target-version = "py311"
59
+
60
+ [tool.ruff.lint]
61
+ select = ["E", "F", "W", "I", "B", "UP", "SIM"]
62
+ ignore = ["E501"]
63
+
64
+ [tool.ruff.lint.per-file-ignores]
65
+ "src/pagespring/cli.py" = ["B008"] # Typer uses function calls in argument defaults
66
+ "tests/*" = ["B011"]
67
+
68
+ [tool.mypy]
69
+ python_version = "3.11"
70
+ strict = true
71
+ ignore_missing_imports = true
72
+
73
+ [tool.pytest.ini_options]
74
+ testpaths = ["tests"]
75
+ addopts = "-ra"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,12 @@
1
+ """pagespring — the lean manual acquisition + normalization layer.
2
+
3
+ Point it at a manual's URL; it recognizes the source type (a "pattern"),
4
+ *acquires* the raw pages, and *normalizes* them into one clean HTML/markdown
5
+ file with absolute asset URLs under ``incoming/<slug>/``. That clean file is the
6
+ deliverable; *conversion* to RAG markdown is a separate concern (**pagespeak**)
7
+ that consumes ``incoming/`` independently — this package neither runs nor imports it.
8
+
9
+ This package stays dependency-light (pf-core[cli] + beautifulsoup4).
10
+ """
11
+
12
+ __version__ = "0.1.0"
@@ -0,0 +1,57 @@
1
+ """The Pattern contract — the unit that ties acquire + normalize + convert
2
+ together for one source type.
3
+
4
+ A pattern recognizes a family of source URLs (``match``), downloads the raw
5
+ pages (``acquire``), turns them into one clean convertible file with absolute
6
+ asset URLs (``normalize``), and declares the extra ``pagespeak convert`` flags
7
+ its output wants (``convert_recipe``). The conversion engine itself lives in
8
+ pagespeak and is invoked as a subprocess — never imported here.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from dataclasses import dataclass
14
+ from pathlib import Path
15
+ from typing import Literal, Protocol, runtime_checkable
16
+
17
+ # "html" -> hand the clean file to pagespeak to convert (markitdown + pipeline)
18
+ # "markdown" -> already the source's canonical clean form (e.g. GitBook per-page .md)
19
+ # "pdf" -> a downloaded PDF; already a pagespeak input (normalize passthrough)
20
+ SourceKind = Literal["html", "markdown", "pdf"]
21
+
22
+
23
+ @dataclass
24
+ class AcquireResult:
25
+ """What ``acquire`` produced: a local dir of raw pages + how to treat them."""
26
+
27
+ raw_dir: Path # local dir holding the downloaded raw page(s)
28
+ kind: SourceKind # whether normalize emits html (for pagespeak) or markdown
29
+ slug: str # short id for the source; becomes the output dir name
30
+ pages: int | None = None # source units fetched (crawl pages / articles / files); manifest stat
31
+ title: str | None = None # human source title for the deliverable heading (falls back to slug)
32
+
33
+
34
+ @runtime_checkable
35
+ class Pattern(Protocol):
36
+ """One source type's acquire/normalize/convert knowledge.
37
+
38
+ Implementations are instances (see pagespring/patterns/*); the registry holds
39
+ one of each. All *source-specific* knowledge (crawl rules, chrome selectors,
40
+ TOC walking, image-scheme resolution) lives in the pattern — pagespeak stays
41
+ source-agnostic.
42
+ """
43
+
44
+ name: str
45
+ convert_recipe: list[str] # extra flags appended to `pagespeak convert`
46
+
47
+ def match(self, url: str) -> bool:
48
+ """Cheap check (host/path) for whether this pattern handles ``url``."""
49
+ ...
50
+
51
+ def acquire(self, url: str, workdir: Path) -> AcquireResult:
52
+ """Download the source's raw pages into ``workdir``."""
53
+ ...
54
+
55
+ def normalize(self, acq: AcquireResult, workdir: Path) -> Path:
56
+ """Turn the raw pages into ONE clean .html/.md (absolute asset URLs)."""
57
+ ...
@@ -0,0 +1,209 @@
1
+ """pagespring command-line interface (Typer, via pf_core.cli).
2
+
3
+ Commands:
4
+ ingest <url> acquire + normalize ("fix") a manual into incoming/<slug>/
5
+ localize <slug> grab an already-ingested deliverable's images (resumable; --all)
6
+ patterns list the registered source patterns
7
+ classify <url> show which pattern handles a URL (no acquisition)
8
+ status list incoming/ deliverables (pattern, pages, size, date, source)
9
+ """
10
+
11
+ from datetime import date
12
+ from pathlib import Path
13
+ from urllib.parse import urlsplit
14
+
15
+ import typer
16
+ from pf_core.cli import create_cli, run_cli
17
+ from pf_core.exceptions import InvalidInputError, PreconditionError
18
+
19
+ from pagespring import manifest
20
+ from pagespring.config import cfg
21
+ from pagespring.orchestrate import (
22
+ AcquireError,
23
+ EmptyOutputError,
24
+ NoPatternError,
25
+ localize_images,
26
+ run_ingest,
27
+ )
28
+ from pagespring.registry import PATTERNS, classify
29
+
30
+ app = create_cli(
31
+ "pagespring",
32
+ help="Find, download, and normalize online software manuals into incoming/.",
33
+ )
34
+
35
+
36
+ @app.command()
37
+ def ingest(
38
+ url: str = typer.Argument(
39
+ ..., help="Manual URL (e.g. a support.apple.com/guide/<app>/ welcome page)."
40
+ ),
41
+ keep_raw: bool = typer.Option(
42
+ False, "--keep-raw", help="Keep the raw crawl alongside the source in incoming/<slug>/raw."
43
+ ),
44
+ download_images: bool = typer.Option(
45
+ False,
46
+ "--download-images",
47
+ help="Download an html/markdown source's images into incoming/<slug>/images/ and re-point refs. No-op for PDFs.",
48
+ ),
49
+ if_changed: bool = typer.Option(
50
+ False,
51
+ "--if-changed",
52
+ help="Skip re-staging when the re-fetch normalizes to byte-identical content (the crawl still runs).",
53
+ ),
54
+ ) -> None:
55
+ """Acquire a manual from URL and normalize it into incoming/<slug>/."""
56
+ try:
57
+ result = run_ingest(
58
+ url, keep_raw=keep_raw, download_images=download_images, if_changed=if_changed
59
+ )
60
+ except NoPatternError:
61
+ typer.echo(
62
+ f"No pattern matched: {url}\n"
63
+ "This source needs a new pattern. Run `bin/run patterns` to see the "
64
+ "registered ones; src/pagespring/patterns/ shows the shape to author one.",
65
+ err=True,
66
+ )
67
+ raise typer.Exit(2) from None
68
+ except InvalidInputError as exc:
69
+ typer.echo(str(exc), err=True)
70
+ raise typer.Exit(2) from None
71
+ except EmptyOutputError:
72
+ typer.echo(
73
+ f"Normalize produced an empty file for {url} — the source may have "
74
+ "changed shape. Nothing was staged; a previous deliverable in "
75
+ "incoming/ is untouched.",
76
+ err=True,
77
+ )
78
+ raise typer.Exit(3) from None
79
+ except AcquireError as exc:
80
+ typer.echo(
81
+ f"Fetch failed during acquire: {exc.detail}\n"
82
+ f"Source: {exc.url}\n"
83
+ "Nothing was staged. The fetch died mid-acquire — check the URL is "
84
+ "reachable; re-run to retry.",
85
+ err=True,
86
+ )
87
+ raise typer.Exit(4) from None
88
+
89
+ typer.echo(f"pattern : {result['pattern']}")
90
+ typer.echo(f"slug : {result['slug']}")
91
+ typer.echo(f"incoming : {result['clean']}")
92
+ if result.get("changed") is False:
93
+ typer.echo(
94
+ "status : unchanged — source matches the existing deliverable, nothing re-staged"
95
+ )
96
+ return
97
+ if result.get("pages") is not None:
98
+ typer.echo(f"pages : {result['pages']}")
99
+ typer.echo(f"size : {_human_size(result['bytes'])}")
100
+ if result.get("images"):
101
+ typer.echo(f"images : {result['images']} downloaded → images/")
102
+
103
+
104
+ @app.command()
105
+ def localize(
106
+ slug: str = typer.Argument(
107
+ None, help="Book slug under incoming/ to localize images for (omit when using --all)."
108
+ ),
109
+ all_books: bool = typer.Option(
110
+ False, "--all", help="Localize images for every incoming/<slug>/."
111
+ ),
112
+ ) -> None:
113
+ """Download an already-ingested deliverable's remote images into images/ and
114
+ re-point refs — no re-crawl. Resumable: re-run until none remain, so a book too
115
+ big to localize in one pass finishes across runs."""
116
+ if all_books:
117
+ incoming = Path(cfg.INCOMING_DIR)
118
+ targets = (
119
+ sorted(p.name for p in incoming.glob("*") if p.is_dir()) if incoming.is_dir() else []
120
+ )
121
+ elif slug:
122
+ targets = [slug]
123
+ else:
124
+ typer.echo("Give a slug or --all.", err=True)
125
+ raise typer.Exit(2)
126
+
127
+ for s in targets:
128
+ try:
129
+ r = localize_images(s)
130
+ except PreconditionError as exc:
131
+ typer.echo(f"skip {s}: {exc}", err=True)
132
+ continue
133
+ tail = "done" if r["remaining"] == 0 else f"{r['remaining']} remaining — re-run to continue"
134
+ typer.echo(f"{s}: +{r['localized']} images (total {r['images_total']}) — {tail}")
135
+
136
+
137
+ @app.command()
138
+ def patterns() -> None:
139
+ """List the registered source patterns and their convert recipes."""
140
+ for p in PATTERNS:
141
+ recipe = " ".join(p.convert_recipe) or "(none)"
142
+ typer.echo(f"{p.name:12} convert-recipe: {recipe}")
143
+
144
+
145
+ @app.command("classify")
146
+ def classify_cmd(
147
+ url: str = typer.Argument(..., help="URL to test against the pattern registry."),
148
+ ) -> None:
149
+ """Show which pattern (if any) handles a URL — no acquisition, no network."""
150
+ p = classify(url)
151
+ typer.echo(p.name if p else "(no pattern matched)")
152
+
153
+
154
+ def _human_size(n: int) -> str:
155
+ size = float(n)
156
+ for unit in ("B", "KB", "MB"):
157
+ if size < 1024:
158
+ return f"{size:.0f} {unit}" if unit == "B" else f"{size:.1f} {unit}"
159
+ size /= 1024
160
+ return f"{size:.1f} GB"
161
+
162
+
163
+ @app.command()
164
+ def status() -> None:
165
+ """One row per incoming/<slug>/, read from its manifest.json: deliverable,
166
+ pattern, pages, size, ingest date, and source host. Legacy (pre-manifest)
167
+ dirs fall back to the deliverable file's own facts. (Conversion into the
168
+ manuals corpus is pagespeak's job — downstream of this tool, out of its view.)"""
169
+ incoming = Path(cfg.INCOMING_DIR)
170
+ slugs = sorted(p for p in incoming.glob("*") if p.is_dir()) if incoming.is_dir() else []
171
+ if not slugs:
172
+ typer.echo("(nothing in incoming/ — run `bin/run ingest <url>`)")
173
+ return
174
+ for d in slugs:
175
+ typer.echo(_status_row(d))
176
+
177
+
178
+ def _status_row(slug_dir: Path) -> str:
179
+ """One status line from the slug's manifest; for legacy (pre-manifest) dirs,
180
+ fall back to the first non-manifest deliverable file's own facts."""
181
+ m = manifest.read_manifest(slug_dir)
182
+ if m is not None:
183
+ deliverable = slug_dir / m["deliverable"]
184
+ size = deliverable.stat().st_size if deliverable.exists() else m["bytes"]
185
+ pages = str(m["pages"]) if m["pages"] is not None else "-"
186
+ host = urlsplit(m["source_url"]).netloc or "-"
187
+ return (
188
+ f"{slug_dir.name:24} {m['deliverable']:32} {m['pattern']:14} "
189
+ f"{pages:>5} {_human_size(size):>9} {m['ingested_at'][:10]} {host}"
190
+ )
191
+ files = sorted(
192
+ p for p in slug_dir.iterdir() if p.is_file() and p.name != manifest.MANIFEST_NAME
193
+ )
194
+ if not files:
195
+ return f"{slug_dir.name:24} {'(no clean file)':32} {'-':14} {'-':>5} {'-':>9} - -"
196
+ f = files[0]
197
+ when = date.fromtimestamp(f.stat().st_mtime).isoformat()
198
+ return (
199
+ f"{slug_dir.name:24} {f.name:32} {'-':14} "
200
+ f"{'-':>5} {_human_size(f.stat().st_size):>9} {when} -"
201
+ )
202
+
203
+
204
+ def main() -> None:
205
+ run_cli(app)
206
+
207
+
208
+ if __name__ == "__main__":
209
+ main()
@@ -0,0 +1,24 @@
1
+ """pagespring configuration — a pf_core.config.AppConfig subclass.
2
+
3
+ All settings are overridable via environment variables / .env.
4
+ """
5
+
6
+ from pathlib import Path
7
+
8
+ from pf_core.config import AppConfig
9
+
10
+ # src/pagespring/config.py → parents[2] is the project root (editable install).
11
+ _project_root = Path(__file__).resolve().parents[2]
12
+
13
+
14
+ class PagespringConfig(AppConfig):
15
+ """pagespring settings."""
16
+
17
+ APP_NAME: str = "pagespring"
18
+
19
+ # The deliverable: one incoming/<slug>/ per manual — the clean
20
+ # acquired+normalized file. A separate step (pagespeak) consumes these.
21
+ INCOMING_DIR: str = "incoming"
22
+
23
+
24
+ cfg = PagespringConfig(env_file=_project_root / ".env")
@@ -0,0 +1,93 @@
1
+ """Tiny shared HTTP fetch over the standard library (urllib).
2
+
3
+ Deliberately no httpx dependency — the acquire step only does plain GETs.
4
+ Provides an identifying User-Agent (override via PAGESPRING_UA), a timeout,
5
+ status-aware retries (permanent 4xx fail fast; 429 honors Retry-After;
6
+ 5xx/network errors back off), and a polite inter-request delay for crawls.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import time
12
+ import urllib.error
13
+ import urllib.request
14
+
15
+ from pf_core.utils.env import resolve_str
16
+
17
+ from pagespring import __version__
18
+
19
+ _RETRY_AFTER_CAP = 30.0 # seconds — don't let a server park a crawl for minutes
20
+
21
+ _UA_DEFAULT = f"pagespring/{__version__} (+https://github.com/phierceweb/pagespring)"
22
+ _UA_ENV_VAR = "PAGESPRING_UA"
23
+
24
+
25
+ def _ua() -> str:
26
+ """The identifying default UA, or PAGESPRING_UA for sources that need another."""
27
+ return resolve_str(None, _UA_ENV_VAR, default=_UA_DEFAULT) or _UA_DEFAULT
28
+
29
+
30
+ def _request(url: str) -> urllib.request.Request:
31
+ return urllib.request.Request(
32
+ url,
33
+ headers={
34
+ "User-Agent": _ua(),
35
+ "Accept": "*/*",
36
+ "Accept-Language": "en-US,en;q=0.9",
37
+ },
38
+ )
39
+
40
+
41
+ def _retry_after(exc: urllib.error.HTTPError, attempt: int) -> float:
42
+ """Seconds to wait on a 429 — the server's Retry-After (capped) when sane,
43
+ else the normal backoff."""
44
+ try:
45
+ return min(float(exc.headers.get("Retry-After", "")), _RETRY_AFTER_CAP)
46
+ except ValueError:
47
+ return 0.5 * (attempt + 1)
48
+
49
+
50
+ def _read(url: str, timeout: float, retries: int) -> tuple[str, bytes, str | None]:
51
+ last: Exception | None = None
52
+ for attempt in range(retries + 1):
53
+ try:
54
+ with urllib.request.urlopen(_request(url), timeout=timeout) as r:
55
+ return r.geturl(), r.read(), r.headers.get_content_charset()
56
+ except urllib.error.HTTPError as exc:
57
+ last = exc
58
+ if exc.code == 429:
59
+ if attempt < retries:
60
+ time.sleep(_retry_after(exc, attempt))
61
+ elif 400 <= exc.code < 500 and exc.code != 408:
62
+ raise # permanent client error — retrying can't help
63
+ elif attempt < retries: # 5xx / 408
64
+ time.sleep(0.5 * (attempt + 1))
65
+ except Exception as exc: # URLError, timeout, connection reset, …
66
+ last = exc
67
+ if attempt < retries:
68
+ time.sleep(0.5 * (attempt + 1))
69
+ raise last # type: ignore[misc]
70
+
71
+
72
+ def fetch_text(
73
+ url: str, *, timeout: float = 30, retries: int = 2, encoding: str | None = None
74
+ ) -> tuple[str, str]:
75
+ """Return (final_url, decoded_text) after following redirects.
76
+
77
+ Decodes with ``encoding`` when given, else the response's Content-Type
78
+ charset, else utf-8 — always with replacement, never raising."""
79
+ final_url, raw, charset = _read(url, timeout, retries)
80
+ return final_url, raw.decode(encoding or charset or "utf-8", "replace")
81
+
82
+
83
+ def fetch_bytes(url: str, *, timeout: float = 180, retries: int = 2) -> tuple[str, bytes]:
84
+ """Return (final_url, raw_bytes) — for binary downloads (PDFs, archives,
85
+ images). Longer default timeout than fetch_text: vendor PDFs/doc archives
86
+ can be tens of MB on slow CDNs."""
87
+ final_url, raw, _charset = _read(url, timeout, retries)
88
+ return final_url, raw
89
+
90
+
91
+ def polite_sleep(seconds: float = 0.25) -> None:
92
+ """Sleep between crawl requests to avoid hammering the source."""
93
+ time.sleep(seconds)