pagespring 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pagespring-0.1.0/LICENSE +21 -0
- pagespring-0.1.0/PKG-INFO +92 -0
- pagespring-0.1.0/README.md +61 -0
- pagespring-0.1.0/pyproject.toml +75 -0
- pagespring-0.1.0/setup.cfg +4 -0
- pagespring-0.1.0/src/pagespring/__init__.py +12 -0
- pagespring-0.1.0/src/pagespring/base.py +57 -0
- pagespring-0.1.0/src/pagespring/cli.py +209 -0
- pagespring-0.1.0/src/pagespring/config.py +24 -0
- pagespring-0.1.0/src/pagespring/http.py +93 -0
- pagespring-0.1.0/src/pagespring/images.py +125 -0
- pagespring-0.1.0/src/pagespring/manifest.py +98 -0
- pagespring-0.1.0/src/pagespring/orchestrate.py +215 -0
- pagespring-0.1.0/src/pagespring/patterns/__init__.py +6 -0
- pagespring-0.1.0/src/pagespring/patterns/_apple_merge.py +168 -0
- pagespring-0.1.0/src/pagespring/patterns/_docusaurus.py +92 -0
- pagespring-0.1.0/src/pagespring/patterns/_gitbook.py +121 -0
- pagespring-0.1.0/src/pagespring/patterns/_mkdocs.py +70 -0
- pagespring-0.1.0/src/pagespring/patterns/_openapi_render.py +156 -0
- pagespring-0.1.0/src/pagespring/patterns/_postman_render.py +85 -0
- pagespring-0.1.0/src/pagespring/patterns/_site.py +50 -0
- pagespring-0.1.0/src/pagespring/patterns/_sphinx.py +112 -0
- pagespring-0.1.0/src/pagespring/patterns/api_spec.py +143 -0
- pagespring-0.1.0/src/pagespring/patterns/apple_help.py +99 -0
- pagespring-0.1.0/src/pagespring/patterns/archive_download.py +88 -0
- pagespring-0.1.0/src/pagespring/patterns/docs_probe.py +109 -0
- pagespring-0.1.0/src/pagespring/patterns/gitbook.py +89 -0
- pagespring-0.1.0/src/pagespring/patterns/github_markdown.py +134 -0
- pagespring-0.1.0/src/pagespring/patterns/llms_txt.py +112 -0
- pagespring-0.1.0/src/pagespring/patterns/microsoft_support.py +165 -0
- pagespring-0.1.0/src/pagespring/patterns/openstax.py +193 -0
- pagespring-0.1.0/src/pagespring/patterns/pdf_url.py +65 -0
- pagespring-0.1.0/src/pagespring/patterns/readthedocs.py +95 -0
- pagespring-0.1.0/src/pagespring/patterns/zendesk_help.py +95 -0
- pagespring-0.1.0/src/pagespring/py.typed +0 -0
- pagespring-0.1.0/src/pagespring/registry.py +59 -0
- pagespring-0.1.0/src/pagespring.egg-info/PKG-INFO +92 -0
- pagespring-0.1.0/src/pagespring.egg-info/SOURCES.txt +63 -0
- pagespring-0.1.0/src/pagespring.egg-info/dependency_links.txt +1 -0
- pagespring-0.1.0/src/pagespring.egg-info/entry_points.txt +2 -0
- pagespring-0.1.0/src/pagespring.egg-info/requires.txt +9 -0
- pagespring-0.1.0/src/pagespring.egg-info/top_level.txt +1 -0
- pagespring-0.1.0/tests/test_api_spec.py +365 -0
- pagespring-0.1.0/tests/test_apple_help.py +102 -0
- pagespring-0.1.0/tests/test_archive_download.py +42 -0
- pagespring-0.1.0/tests/test_cli.py +227 -0
- pagespring-0.1.0/tests/test_docs_probe.py +123 -0
- pagespring-0.1.0/tests/test_docusaurus.py +74 -0
- pagespring-0.1.0/tests/test_gitbook.py +137 -0
- pagespring-0.1.0/tests/test_github_markdown.py +67 -0
- pagespring-0.1.0/tests/test_http.py +125 -0
- pagespring-0.1.0/tests/test_images.py +109 -0
- pagespring-0.1.0/tests/test_llms_txt.py +60 -0
- pagespring-0.1.0/tests/test_manifest.py +67 -0
- pagespring-0.1.0/tests/test_microsoft_support.py +205 -0
- pagespring-0.1.0/tests/test_mkdocs.py +73 -0
- pagespring-0.1.0/tests/test_openstax.py +180 -0
- pagespring-0.1.0/tests/test_orchestrate.py +282 -0
- pagespring-0.1.0/tests/test_pdf_url.py +47 -0
- pagespring-0.1.0/tests/test_readthedocs.py +92 -0
- pagespring-0.1.0/tests/test_registry.py +61 -0
- pagespring-0.1.0/tests/test_scaffold.py +21 -0
- pagespring-0.1.0/tests/test_site.py +44 -0
- pagespring-0.1.0/tests/test_sphinx.py +134 -0
- pagespring-0.1.0/tests/test_zendesk_help.py +90 -0
pagespring-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Mike Farr
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pagespring
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Acquire and normalize online software manuals into clean, convertible source files — the acquisition front-end to pagespeak.
|
|
5
|
+
Author: Mike Farr
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Repository, https://github.com/phierceweb/pagespring
|
|
8
|
+
Project-URL: Changelog, https://github.com/phierceweb/pagespring/blob/main/CHANGELOG.md
|
|
9
|
+
Project-URL: Issues, https://github.com/phierceweb/pagespring/issues
|
|
10
|
+
Keywords: manuals,documentation,ingest,acquisition,rag,markdown
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Topic :: Text Processing :: Markup :: Markdown
|
|
18
|
+
Classifier: Typing :: Typed
|
|
19
|
+
Requires-Python: >=3.11
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Requires-Dist: beautifulsoup4>=4.12
|
|
23
|
+
Requires-Dist: pyyaml>=6.0
|
|
24
|
+
Requires-Dist: pf-core[cli]~=0.4.1
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
27
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
28
|
+
Requires-Dist: mypy>=1.10; extra == "dev"
|
|
29
|
+
Requires-Dist: pre-commit>=3.7; extra == "dev"
|
|
30
|
+
Dynamic: license-file
|
|
31
|
+
|
|
32
|
+
# pagespring
|
|
33
|
+
|
|
34
|
+
Acquire and normalize online software manuals into clean, convertible source
|
|
35
|
+
files — the **acquisition front-end to
|
|
36
|
+
[pagespeak](https://github.com/phierceweb/pagespeak)**.
|
|
37
|
+
|
|
38
|
+
Point it at a manual's URL. A *pattern* recognizes the source type, *acquires*
|
|
39
|
+
the raw pages (stdlib `urllib`), and *normalizes* them into ONE clean
|
|
40
|
+
HTML/markdown file with absolute asset URLs under `incoming/<slug>/`. That clean
|
|
41
|
+
file is the deliverable; converting it into the finished RAG corpus is a separate
|
|
42
|
+
step (pagespeak) that consumes `incoming/` on its own — pagespring never runs it.
|
|
43
|
+
|
|
44
|
+
Lean by design: `pf-core[cli]` + `beautifulsoup4`, stdlib fetch, no ML stack.
|
|
45
|
+
|
|
46
|
+
## Intended use
|
|
47
|
+
|
|
48
|
+
pagespring is for **publicly available documentation** — vendor manuals, help
|
|
49
|
+
centers, open textbooks, API specs. It fetches only what the source serves to
|
|
50
|
+
any reader: there is no login/session handling, no paywall traversal, and no
|
|
51
|
+
bot-detection evasion. It is a **polite client**: it identifies itself with a
|
|
52
|
+
`pagespring/<version>` User-Agent (see `.env.example` to override), honors
|
|
53
|
+
`429 Retry-After`, backs off on server errors, paces crawl requests, and caps
|
|
54
|
+
crawl sizes.
|
|
55
|
+
|
|
56
|
+
It is a *user-invoked, one-manual-at-a-time* archiver — closer to "Save Page
|
|
57
|
+
As" than to an autonomous crawler — so it does not consult `robots.txt`
|
|
58
|
+
(which governs bots that discover URLs on their own; you supply the URL).
|
|
59
|
+
Before mirroring a site, check its terms of use. What you may do with the
|
|
60
|
+
acquired copy (personal RAG corpus, internal search, redistribution) is
|
|
61
|
+
governed by the source's license — the deliverable under `incoming/` stays on
|
|
62
|
+
your machine, and nothing is re-published by this tool.
|
|
63
|
+
|
|
64
|
+
## Install
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
pip install pagespring
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
## Quick start
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
pagespring ingest https://docs.tableplus.com # acquire + normalize → incoming/tableplus/
|
|
74
|
+
pagespring localize <slug> # pull a deliverable's images later (resumable; --all)
|
|
75
|
+
pagespring patterns # list the source patterns
|
|
76
|
+
pagespring classify <url> # which pattern handles a URL (no fetch)
|
|
77
|
+
pagespring status # what's been acquired
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Deliverables land in `./incoming/<slug>/` under the directory you run from.
|
|
81
|
+
|
|
82
|
+
## Dev
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
bin/setup # clone → venv + editable install with dev extras
|
|
86
|
+
bin/test # pytest
|
|
87
|
+
bin/lint # ruff check + ruff format --check + mypy (strict)
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
See [docs/usage.md](docs/usage.md) for the full command set and
|
|
91
|
+
[docs/architecture.md](docs/architecture.md) for the acquire → normalize flow
|
|
92
|
+
and how to add a new source pattern.
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
# pagespring
|
|
2
|
+
|
|
3
|
+
Acquire and normalize online software manuals into clean, convertible source
|
|
4
|
+
files — the **acquisition front-end to
|
|
5
|
+
[pagespeak](https://github.com/phierceweb/pagespeak)**.
|
|
6
|
+
|
|
7
|
+
Point it at a manual's URL. A *pattern* recognizes the source type, *acquires*
|
|
8
|
+
the raw pages (stdlib `urllib`), and *normalizes* them into ONE clean
|
|
9
|
+
HTML/markdown file with absolute asset URLs under `incoming/<slug>/`. That clean
|
|
10
|
+
file is the deliverable; converting it into the finished RAG corpus is a separate
|
|
11
|
+
step (pagespeak) that consumes `incoming/` on its own — pagespring never runs it.
|
|
12
|
+
|
|
13
|
+
Lean by design: `pf-core[cli]` + `beautifulsoup4`, stdlib fetch, no ML stack.
|
|
14
|
+
|
|
15
|
+
## Intended use
|
|
16
|
+
|
|
17
|
+
pagespring is for **publicly available documentation** — vendor manuals, help
|
|
18
|
+
centers, open textbooks, API specs. It fetches only what the source serves to
|
|
19
|
+
any reader: there is no login/session handling, no paywall traversal, and no
|
|
20
|
+
bot-detection evasion. It is a **polite client**: it identifies itself with a
|
|
21
|
+
`pagespring/<version>` User-Agent (see `.env.example` to override), honors
|
|
22
|
+
`429 Retry-After`, backs off on server errors, paces crawl requests, and caps
|
|
23
|
+
crawl sizes.
|
|
24
|
+
|
|
25
|
+
It is a *user-invoked, one-manual-at-a-time* archiver — closer to "Save Page
|
|
26
|
+
As" than to an autonomous crawler — so it does not consult `robots.txt`
|
|
27
|
+
(which governs bots that discover URLs on their own; you supply the URL).
|
|
28
|
+
Before mirroring a site, check its terms of use. What you may do with the
|
|
29
|
+
acquired copy (personal RAG corpus, internal search, redistribution) is
|
|
30
|
+
governed by the source's license — the deliverable under `incoming/` stays on
|
|
31
|
+
your machine, and nothing is re-published by this tool.
|
|
32
|
+
|
|
33
|
+
## Install
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
pip install pagespring
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Quick start
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
pagespring ingest https://docs.tableplus.com # acquire + normalize → incoming/tableplus/
|
|
43
|
+
pagespring localize <slug> # pull a deliverable's images later (resumable; --all)
|
|
44
|
+
pagespring patterns # list the source patterns
|
|
45
|
+
pagespring classify <url> # which pattern handles a URL (no fetch)
|
|
46
|
+
pagespring status # what's been acquired
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Deliverables land in `./incoming/<slug>/` under the directory you run from.
|
|
50
|
+
|
|
51
|
+
## Dev
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
bin/setup # clone → venv + editable install with dev extras
|
|
55
|
+
bin/test # pytest
|
|
56
|
+
bin/lint # ruff check + ruff format --check + mypy (strict)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
See [docs/usage.md](docs/usage.md) for the full command set and
|
|
60
|
+
[docs/architecture.md](docs/architecture.md) for the acquire → normalize flow
|
|
61
|
+
and how to add a new source pattern.
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "pagespring"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Acquire and normalize online software manuals into clean, convertible source files — the acquisition front-end to pagespeak."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Mike Farr" }]
|
|
14
|
+
keywords = ["manuals", "documentation", "ingest", "acquisition", "rag", "markdown"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 4 - Beta",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3.11",
|
|
20
|
+
"Programming Language :: Python :: 3.12",
|
|
21
|
+
"Operating System :: OS Independent",
|
|
22
|
+
"Topic :: Text Processing :: Markup :: Markdown",
|
|
23
|
+
"Typing :: Typed",
|
|
24
|
+
]
|
|
25
|
+
dependencies = [
|
|
26
|
+
# Lean by design: stdlib urllib fetches, BeautifulSoup parses, pf-core[cli]
|
|
27
|
+
# gives the Typer scaffolding + AppConfig + structured logging. No ML stack.
|
|
28
|
+
"beautifulsoup4>=4.12",
|
|
29
|
+
# YAML OpenAPI specs (JSON is stdlib); pure-data parser, no transport.
|
|
30
|
+
"pyyaml>=6.0",
|
|
31
|
+
"pf-core[cli]~=0.4.1",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
[project.optional-dependencies]
|
|
35
|
+
dev = [
|
|
36
|
+
"pytest>=8.0",
|
|
37
|
+
"ruff>=0.6",
|
|
38
|
+
"mypy>=1.10",
|
|
39
|
+
"pre-commit>=3.7",
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
[project.scripts]
|
|
43
|
+
pagespring = "pagespring.cli:main"
|
|
44
|
+
|
|
45
|
+
[project.urls]
|
|
46
|
+
Repository = "https://github.com/phierceweb/pagespring"
|
|
47
|
+
Changelog = "https://github.com/phierceweb/pagespring/blob/main/CHANGELOG.md"
|
|
48
|
+
Issues = "https://github.com/phierceweb/pagespring/issues"
|
|
49
|
+
|
|
50
|
+
[tool.setuptools.packages.find]
|
|
51
|
+
where = ["src"]
|
|
52
|
+
|
|
53
|
+
[tool.setuptools.package-data]
|
|
54
|
+
pagespring = ["py.typed"]
|
|
55
|
+
|
|
56
|
+
[tool.ruff]
|
|
57
|
+
line-length = 100
|
|
58
|
+
target-version = "py311"
|
|
59
|
+
|
|
60
|
+
[tool.ruff.lint]
|
|
61
|
+
select = ["E", "F", "W", "I", "B", "UP", "SIM"]
|
|
62
|
+
ignore = ["E501"]
|
|
63
|
+
|
|
64
|
+
[tool.ruff.lint.per-file-ignores]
|
|
65
|
+
"src/pagespring/cli.py" = ["B008"] # Typer uses function calls in argument defaults
|
|
66
|
+
"tests/*" = ["B011"]
|
|
67
|
+
|
|
68
|
+
[tool.mypy]
|
|
69
|
+
python_version = "3.11"
|
|
70
|
+
strict = true
|
|
71
|
+
ignore_missing_imports = true
|
|
72
|
+
|
|
73
|
+
[tool.pytest.ini_options]
|
|
74
|
+
testpaths = ["tests"]
|
|
75
|
+
addopts = "-ra"
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""pagespring — the lean manual acquisition + normalization layer.
|
|
2
|
+
|
|
3
|
+
Point it at a manual's URL; it recognizes the source type (a "pattern"),
|
|
4
|
+
*acquires* the raw pages, and *normalizes* them into one clean HTML/markdown
|
|
5
|
+
file with absolute asset URLs under ``incoming/<slug>/``. That clean file is the
|
|
6
|
+
deliverable; *conversion* to RAG markdown is a separate concern (**pagespeak**)
|
|
7
|
+
that consumes ``incoming/`` independently — this package neither runs nor imports it.
|
|
8
|
+
|
|
9
|
+
This package stays dependency-light (pf-core[cli] + beautifulsoup4).
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""The Pattern contract — the unit that ties acquire + normalize + convert
|
|
2
|
+
together for one source type.
|
|
3
|
+
|
|
4
|
+
A pattern recognizes a family of source URLs (``match``), downloads the raw
|
|
5
|
+
pages (``acquire``), turns them into one clean convertible file with absolute
|
|
6
|
+
asset URLs (``normalize``), and declares the extra ``pagespeak convert`` flags
|
|
7
|
+
its output wants (``convert_recipe``). The conversion engine itself lives in
|
|
8
|
+
pagespeak and is invoked as a subprocess — never imported here.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Literal, Protocol, runtime_checkable
|
|
16
|
+
|
|
17
|
+
# "html" -> hand the clean file to pagespeak to convert (markitdown + pipeline)
|
|
18
|
+
# "markdown" -> already the source's canonical clean form (e.g. GitBook per-page .md)
|
|
19
|
+
# "pdf" -> a downloaded PDF; already a pagespeak input (normalize passthrough)
|
|
20
|
+
SourceKind = Literal["html", "markdown", "pdf"]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class AcquireResult:
|
|
25
|
+
"""What ``acquire`` produced: a local dir of raw pages + how to treat them."""
|
|
26
|
+
|
|
27
|
+
raw_dir: Path # local dir holding the downloaded raw page(s)
|
|
28
|
+
kind: SourceKind # whether normalize emits html (for pagespeak) or markdown
|
|
29
|
+
slug: str # short id for the source; becomes the output dir name
|
|
30
|
+
pages: int | None = None # source units fetched (crawl pages / articles / files); manifest stat
|
|
31
|
+
title: str | None = None # human source title for the deliverable heading (falls back to slug)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@runtime_checkable
|
|
35
|
+
class Pattern(Protocol):
|
|
36
|
+
"""One source type's acquire/normalize/convert knowledge.
|
|
37
|
+
|
|
38
|
+
Implementations are instances (see pagespring/patterns/*); the registry holds
|
|
39
|
+
one of each. All *source-specific* knowledge (crawl rules, chrome selectors,
|
|
40
|
+
TOC walking, image-scheme resolution) lives in the pattern — pagespeak stays
|
|
41
|
+
source-agnostic.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
name: str
|
|
45
|
+
convert_recipe: list[str] # extra flags appended to `pagespeak convert`
|
|
46
|
+
|
|
47
|
+
def match(self, url: str) -> bool:
|
|
48
|
+
"""Cheap check (host/path) for whether this pattern handles ``url``."""
|
|
49
|
+
...
|
|
50
|
+
|
|
51
|
+
def acquire(self, url: str, workdir: Path) -> AcquireResult:
|
|
52
|
+
"""Download the source's raw pages into ``workdir``."""
|
|
53
|
+
...
|
|
54
|
+
|
|
55
|
+
def normalize(self, acq: AcquireResult, workdir: Path) -> Path:
|
|
56
|
+
"""Turn the raw pages into ONE clean .html/.md (absolute asset URLs)."""
|
|
57
|
+
...
|
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""pagespring command-line interface (Typer, via pf_core.cli).
|
|
2
|
+
|
|
3
|
+
Commands:
|
|
4
|
+
ingest <url> acquire + normalize ("fix") a manual into incoming/<slug>/
|
|
5
|
+
localize <slug> grab an already-ingested deliverable's images (resumable; --all)
|
|
6
|
+
patterns list the registered source patterns
|
|
7
|
+
classify <url> show which pattern handles a URL (no acquisition)
|
|
8
|
+
status list incoming/ deliverables (pattern, pages, size, date, source)
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from datetime import date
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from urllib.parse import urlsplit
|
|
14
|
+
|
|
15
|
+
import typer
|
|
16
|
+
from pf_core.cli import create_cli, run_cli
|
|
17
|
+
from pf_core.exceptions import InvalidInputError, PreconditionError
|
|
18
|
+
|
|
19
|
+
from pagespring import manifest
|
|
20
|
+
from pagespring.config import cfg
|
|
21
|
+
from pagespring.orchestrate import (
|
|
22
|
+
AcquireError,
|
|
23
|
+
EmptyOutputError,
|
|
24
|
+
NoPatternError,
|
|
25
|
+
localize_images,
|
|
26
|
+
run_ingest,
|
|
27
|
+
)
|
|
28
|
+
from pagespring.registry import PATTERNS, classify
|
|
29
|
+
|
|
30
|
+
app = create_cli(
|
|
31
|
+
"pagespring",
|
|
32
|
+
help="Find, download, and normalize online software manuals into incoming/.",
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@app.command()
|
|
37
|
+
def ingest(
|
|
38
|
+
url: str = typer.Argument(
|
|
39
|
+
..., help="Manual URL (e.g. a support.apple.com/guide/<app>/ welcome page)."
|
|
40
|
+
),
|
|
41
|
+
keep_raw: bool = typer.Option(
|
|
42
|
+
False, "--keep-raw", help="Keep the raw crawl alongside the source in incoming/<slug>/raw."
|
|
43
|
+
),
|
|
44
|
+
download_images: bool = typer.Option(
|
|
45
|
+
False,
|
|
46
|
+
"--download-images",
|
|
47
|
+
help="Download an html/markdown source's images into incoming/<slug>/images/ and re-point refs. No-op for PDFs.",
|
|
48
|
+
),
|
|
49
|
+
if_changed: bool = typer.Option(
|
|
50
|
+
False,
|
|
51
|
+
"--if-changed",
|
|
52
|
+
help="Skip re-staging when the re-fetch normalizes to byte-identical content (the crawl still runs).",
|
|
53
|
+
),
|
|
54
|
+
) -> None:
|
|
55
|
+
"""Acquire a manual from URL and normalize it into incoming/<slug>/."""
|
|
56
|
+
try:
|
|
57
|
+
result = run_ingest(
|
|
58
|
+
url, keep_raw=keep_raw, download_images=download_images, if_changed=if_changed
|
|
59
|
+
)
|
|
60
|
+
except NoPatternError:
|
|
61
|
+
typer.echo(
|
|
62
|
+
f"No pattern matched: {url}\n"
|
|
63
|
+
"This source needs a new pattern. Run `bin/run patterns` to see the "
|
|
64
|
+
"registered ones; src/pagespring/patterns/ shows the shape to author one.",
|
|
65
|
+
err=True,
|
|
66
|
+
)
|
|
67
|
+
raise typer.Exit(2) from None
|
|
68
|
+
except InvalidInputError as exc:
|
|
69
|
+
typer.echo(str(exc), err=True)
|
|
70
|
+
raise typer.Exit(2) from None
|
|
71
|
+
except EmptyOutputError:
|
|
72
|
+
typer.echo(
|
|
73
|
+
f"Normalize produced an empty file for {url} — the source may have "
|
|
74
|
+
"changed shape. Nothing was staged; a previous deliverable in "
|
|
75
|
+
"incoming/ is untouched.",
|
|
76
|
+
err=True,
|
|
77
|
+
)
|
|
78
|
+
raise typer.Exit(3) from None
|
|
79
|
+
except AcquireError as exc:
|
|
80
|
+
typer.echo(
|
|
81
|
+
f"Fetch failed during acquire: {exc.detail}\n"
|
|
82
|
+
f"Source: {exc.url}\n"
|
|
83
|
+
"Nothing was staged. The fetch died mid-acquire — check the URL is "
|
|
84
|
+
"reachable; re-run to retry.",
|
|
85
|
+
err=True,
|
|
86
|
+
)
|
|
87
|
+
raise typer.Exit(4) from None
|
|
88
|
+
|
|
89
|
+
typer.echo(f"pattern : {result['pattern']}")
|
|
90
|
+
typer.echo(f"slug : {result['slug']}")
|
|
91
|
+
typer.echo(f"incoming : {result['clean']}")
|
|
92
|
+
if result.get("changed") is False:
|
|
93
|
+
typer.echo(
|
|
94
|
+
"status : unchanged — source matches the existing deliverable, nothing re-staged"
|
|
95
|
+
)
|
|
96
|
+
return
|
|
97
|
+
if result.get("pages") is not None:
|
|
98
|
+
typer.echo(f"pages : {result['pages']}")
|
|
99
|
+
typer.echo(f"size : {_human_size(result['bytes'])}")
|
|
100
|
+
if result.get("images"):
|
|
101
|
+
typer.echo(f"images : {result['images']} downloaded → images/")
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
@app.command()
|
|
105
|
+
def localize(
|
|
106
|
+
slug: str = typer.Argument(
|
|
107
|
+
None, help="Book slug under incoming/ to localize images for (omit when using --all)."
|
|
108
|
+
),
|
|
109
|
+
all_books: bool = typer.Option(
|
|
110
|
+
False, "--all", help="Localize images for every incoming/<slug>/."
|
|
111
|
+
),
|
|
112
|
+
) -> None:
|
|
113
|
+
"""Download an already-ingested deliverable's remote images into images/ and
|
|
114
|
+
re-point refs — no re-crawl. Resumable: re-run until none remain, so a book too
|
|
115
|
+
big to localize in one pass finishes across runs."""
|
|
116
|
+
if all_books:
|
|
117
|
+
incoming = Path(cfg.INCOMING_DIR)
|
|
118
|
+
targets = (
|
|
119
|
+
sorted(p.name for p in incoming.glob("*") if p.is_dir()) if incoming.is_dir() else []
|
|
120
|
+
)
|
|
121
|
+
elif slug:
|
|
122
|
+
targets = [slug]
|
|
123
|
+
else:
|
|
124
|
+
typer.echo("Give a slug or --all.", err=True)
|
|
125
|
+
raise typer.Exit(2)
|
|
126
|
+
|
|
127
|
+
for s in targets:
|
|
128
|
+
try:
|
|
129
|
+
r = localize_images(s)
|
|
130
|
+
except PreconditionError as exc:
|
|
131
|
+
typer.echo(f"skip {s}: {exc}", err=True)
|
|
132
|
+
continue
|
|
133
|
+
tail = "done" if r["remaining"] == 0 else f"{r['remaining']} remaining — re-run to continue"
|
|
134
|
+
typer.echo(f"{s}: +{r['localized']} images (total {r['images_total']}) — {tail}")
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@app.command()
|
|
138
|
+
def patterns() -> None:
|
|
139
|
+
"""List the registered source patterns and their convert recipes."""
|
|
140
|
+
for p in PATTERNS:
|
|
141
|
+
recipe = " ".join(p.convert_recipe) or "(none)"
|
|
142
|
+
typer.echo(f"{p.name:12} convert-recipe: {recipe}")
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
@app.command("classify")
|
|
146
|
+
def classify_cmd(
|
|
147
|
+
url: str = typer.Argument(..., help="URL to test against the pattern registry."),
|
|
148
|
+
) -> None:
|
|
149
|
+
"""Show which pattern (if any) handles a URL — no acquisition, no network."""
|
|
150
|
+
p = classify(url)
|
|
151
|
+
typer.echo(p.name if p else "(no pattern matched)")
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _human_size(n: int) -> str:
|
|
155
|
+
size = float(n)
|
|
156
|
+
for unit in ("B", "KB", "MB"):
|
|
157
|
+
if size < 1024:
|
|
158
|
+
return f"{size:.0f} {unit}" if unit == "B" else f"{size:.1f} {unit}"
|
|
159
|
+
size /= 1024
|
|
160
|
+
return f"{size:.1f} GB"
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
@app.command()
|
|
164
|
+
def status() -> None:
|
|
165
|
+
"""One row per incoming/<slug>/, read from its manifest.json: deliverable,
|
|
166
|
+
pattern, pages, size, ingest date, and source host. Legacy (pre-manifest)
|
|
167
|
+
dirs fall back to the deliverable file's own facts. (Conversion into the
|
|
168
|
+
manuals corpus is pagespeak's job — downstream of this tool, out of its view.)"""
|
|
169
|
+
incoming = Path(cfg.INCOMING_DIR)
|
|
170
|
+
slugs = sorted(p for p in incoming.glob("*") if p.is_dir()) if incoming.is_dir() else []
|
|
171
|
+
if not slugs:
|
|
172
|
+
typer.echo("(nothing in incoming/ — run `bin/run ingest <url>`)")
|
|
173
|
+
return
|
|
174
|
+
for d in slugs:
|
|
175
|
+
typer.echo(_status_row(d))
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _status_row(slug_dir: Path) -> str:
|
|
179
|
+
"""One status line from the slug's manifest; for legacy (pre-manifest) dirs,
|
|
180
|
+
fall back to the first non-manifest deliverable file's own facts."""
|
|
181
|
+
m = manifest.read_manifest(slug_dir)
|
|
182
|
+
if m is not None:
|
|
183
|
+
deliverable = slug_dir / m["deliverable"]
|
|
184
|
+
size = deliverable.stat().st_size if deliverable.exists() else m["bytes"]
|
|
185
|
+
pages = str(m["pages"]) if m["pages"] is not None else "-"
|
|
186
|
+
host = urlsplit(m["source_url"]).netloc or "-"
|
|
187
|
+
return (
|
|
188
|
+
f"{slug_dir.name:24} {m['deliverable']:32} {m['pattern']:14} "
|
|
189
|
+
f"{pages:>5} {_human_size(size):>9} {m['ingested_at'][:10]} {host}"
|
|
190
|
+
)
|
|
191
|
+
files = sorted(
|
|
192
|
+
p for p in slug_dir.iterdir() if p.is_file() and p.name != manifest.MANIFEST_NAME
|
|
193
|
+
)
|
|
194
|
+
if not files:
|
|
195
|
+
return f"{slug_dir.name:24} {'(no clean file)':32} {'-':14} {'-':>5} {'-':>9} - -"
|
|
196
|
+
f = files[0]
|
|
197
|
+
when = date.fromtimestamp(f.stat().st_mtime).isoformat()
|
|
198
|
+
return (
|
|
199
|
+
f"{slug_dir.name:24} {f.name:32} {'-':14} "
|
|
200
|
+
f"{'-':>5} {_human_size(f.stat().st_size):>9} {when} -"
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def main() -> None:
|
|
205
|
+
run_cli(app)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
if __name__ == "__main__":
|
|
209
|
+
main()
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""pagespring configuration — a pf_core.config.AppConfig subclass.
|
|
2
|
+
|
|
3
|
+
All settings are overridable via environment variables / .env.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from pf_core.config import AppConfig
|
|
9
|
+
|
|
10
|
+
# src/pagespring/config.py → parents[2] is the project root (editable install).
|
|
11
|
+
_project_root = Path(__file__).resolve().parents[2]
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class PagespringConfig(AppConfig):
|
|
15
|
+
"""pagespring settings."""
|
|
16
|
+
|
|
17
|
+
APP_NAME: str = "pagespring"
|
|
18
|
+
|
|
19
|
+
# The deliverable: one incoming/<slug>/ per manual — the clean
|
|
20
|
+
# acquired+normalized file. A separate step (pagespeak) consumes these.
|
|
21
|
+
INCOMING_DIR: str = "incoming"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
cfg = PagespringConfig(env_file=_project_root / ".env")
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Tiny shared HTTP fetch over the standard library (urllib).
|
|
2
|
+
|
|
3
|
+
Deliberately no httpx dependency — the acquire step only does plain GETs.
|
|
4
|
+
Provides an identifying User-Agent (override via PAGESPRING_UA), a timeout,
|
|
5
|
+
status-aware retries (permanent 4xx fail fast; 429 honors Retry-After;
|
|
6
|
+
5xx/network errors back off), and a polite inter-request delay for crawls.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import time
|
|
12
|
+
import urllib.error
|
|
13
|
+
import urllib.request
|
|
14
|
+
|
|
15
|
+
from pf_core.utils.env import resolve_str
|
|
16
|
+
|
|
17
|
+
from pagespring import __version__
|
|
18
|
+
|
|
19
|
+
_RETRY_AFTER_CAP = 30.0 # seconds — don't let a server park a crawl for minutes
|
|
20
|
+
|
|
21
|
+
_UA_DEFAULT = f"pagespring/{__version__} (+https://github.com/phierceweb/pagespring)"
|
|
22
|
+
_UA_ENV_VAR = "PAGESPRING_UA"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _ua() -> str:
|
|
26
|
+
"""The identifying default UA, or PAGESPRING_UA for sources that need another."""
|
|
27
|
+
return resolve_str(None, _UA_ENV_VAR, default=_UA_DEFAULT) or _UA_DEFAULT
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _request(url: str) -> urllib.request.Request:
|
|
31
|
+
return urllib.request.Request(
|
|
32
|
+
url,
|
|
33
|
+
headers={
|
|
34
|
+
"User-Agent": _ua(),
|
|
35
|
+
"Accept": "*/*",
|
|
36
|
+
"Accept-Language": "en-US,en;q=0.9",
|
|
37
|
+
},
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _retry_after(exc: urllib.error.HTTPError, attempt: int) -> float:
|
|
42
|
+
"""Seconds to wait on a 429 — the server's Retry-After (capped) when sane,
|
|
43
|
+
else the normal backoff."""
|
|
44
|
+
try:
|
|
45
|
+
return min(float(exc.headers.get("Retry-After", "")), _RETRY_AFTER_CAP)
|
|
46
|
+
except ValueError:
|
|
47
|
+
return 0.5 * (attempt + 1)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _read(url: str, timeout: float, retries: int) -> tuple[str, bytes, str | None]:
|
|
51
|
+
last: Exception | None = None
|
|
52
|
+
for attempt in range(retries + 1):
|
|
53
|
+
try:
|
|
54
|
+
with urllib.request.urlopen(_request(url), timeout=timeout) as r:
|
|
55
|
+
return r.geturl(), r.read(), r.headers.get_content_charset()
|
|
56
|
+
except urllib.error.HTTPError as exc:
|
|
57
|
+
last = exc
|
|
58
|
+
if exc.code == 429:
|
|
59
|
+
if attempt < retries:
|
|
60
|
+
time.sleep(_retry_after(exc, attempt))
|
|
61
|
+
elif 400 <= exc.code < 500 and exc.code != 408:
|
|
62
|
+
raise # permanent client error — retrying can't help
|
|
63
|
+
elif attempt < retries: # 5xx / 408
|
|
64
|
+
time.sleep(0.5 * (attempt + 1))
|
|
65
|
+
except Exception as exc: # URLError, timeout, connection reset, …
|
|
66
|
+
last = exc
|
|
67
|
+
if attempt < retries:
|
|
68
|
+
time.sleep(0.5 * (attempt + 1))
|
|
69
|
+
raise last # type: ignore[misc]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def fetch_text(
|
|
73
|
+
url: str, *, timeout: float = 30, retries: int = 2, encoding: str | None = None
|
|
74
|
+
) -> tuple[str, str]:
|
|
75
|
+
"""Return (final_url, decoded_text) after following redirects.
|
|
76
|
+
|
|
77
|
+
Decodes with ``encoding`` when given, else the response's Content-Type
|
|
78
|
+
charset, else utf-8 — always with replacement, never raising."""
|
|
79
|
+
final_url, raw, charset = _read(url, timeout, retries)
|
|
80
|
+
return final_url, raw.decode(encoding or charset or "utf-8", "replace")
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def fetch_bytes(url: str, *, timeout: float = 180, retries: int = 2) -> tuple[str, bytes]:
|
|
84
|
+
"""Return (final_url, raw_bytes) — for binary downloads (PDFs, archives,
|
|
85
|
+
images). Longer default timeout than fetch_text: vendor PDFs/doc archives
|
|
86
|
+
can be tens of MB on slow CDNs."""
|
|
87
|
+
final_url, raw, _charset = _read(url, timeout, retries)
|
|
88
|
+
return final_url, raw
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def polite_sleep(seconds: float = 0.25) -> None:
|
|
92
|
+
"""Sleep between crawl requests to avoid hammering the source."""
|
|
93
|
+
time.sleep(seconds)
|