docs-site 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docs_site-0.1.1/LICENSE +21 -0
- docs_site-0.1.1/PKG-INFO +95 -0
- docs_site-0.1.1/README.md +65 -0
- docs_site-0.1.1/pyproject.toml +73 -0
- docs_site-0.1.1/setup.cfg +4 -0
- docs_site-0.1.1/src/docs_site/__init__.py +5 -0
- docs_site-0.1.1/src/docs_site/__main__.py +4 -0
- docs_site-0.1.1/src/docs_site/cli.py +59 -0
- docs_site-0.1.1/src/docs_site/html_processing.py +123 -0
- docs_site-0.1.1/src/docs_site/manifest.py +111 -0
- docs_site-0.1.1/src/docs_site/pdf_processing.py +69 -0
- docs_site-0.1.1/src/docs_site/py.typed +0 -0
- docs_site-0.1.1/src/docs_site/site_builder.py +73 -0
- docs_site-0.1.1/src/docs_site/static/css/content.css +91 -0
- docs_site-0.1.1/src/docs_site/static/css/shell.css +279 -0
- docs_site-0.1.1/src/docs_site/static/js/content-dark-listener.js +17 -0
- docs_site-0.1.1/src/docs_site/static/js/shell.js +333 -0
- docs_site-0.1.1/src/docs_site/templates/shell.html.jinja +35 -0
- docs_site-0.1.1/src/docs_site/templates/welcome.html.jinja +22 -0
- docs_site-0.1.1/src/docs_site/types.py +17 -0
- docs_site-0.1.1/src/docs_site/utils.py +27 -0
- docs_site-0.1.1/src/docs_site.egg-info/PKG-INFO +95 -0
- docs_site-0.1.1/src/docs_site.egg-info/SOURCES.txt +29 -0
- docs_site-0.1.1/src/docs_site.egg-info/dependency_links.txt +1 -0
- docs_site-0.1.1/src/docs_site.egg-info/entry_points.txt +2 -0
- docs_site-0.1.1/src/docs_site.egg-info/requires.txt +3 -0
- docs_site-0.1.1/src/docs_site.egg-info/top_level.txt +1 -0
- docs_site-0.1.1/tests/test_html_processing.py +52 -0
- docs_site-0.1.1/tests/test_pdf_processing.py +39 -0
- docs_site-0.1.1/tests/test_site_builder.py +36 -0
- docs_site-0.1.1/tests/test_utils.py +42 -0
docs_site-0.1.1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Lucas Rollin Ferreira
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
docs_site-0.1.1/PKG-INFO
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: docs-site
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: Turn a folder of dumped HTML + PDF files into a browsable, searchable local documentation site.
|
|
5
|
+
Author-email: Lucas Rollin Ferreira <lucasrollinferreira@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/lucas-rollin/docs-site
|
|
8
|
+
Project-URL: Repository, https://github.com/lucas-rollin/docs-site
|
|
9
|
+
Project-URL: Issues, https://github.com/lucas-rollin/docs-site/issues
|
|
10
|
+
Keywords: documentation,pdf,html,search,static-site,reader
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Intended Audience :: Education
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
21
|
+
Classifier: Topic :: Documentation
|
|
22
|
+
Classifier: Topic :: Text Processing :: Markup :: HTML
|
|
23
|
+
Requires-Python: >=3.11
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
License-File: LICENSE
|
|
26
|
+
Requires-Dist: beautifulsoup4>=4.12
|
|
27
|
+
Requires-Dist: jinja2>=3.1
|
|
28
|
+
Requires-Dist: pypdf>=4.0
|
|
29
|
+
Dynamic: license-file
|
|
30
|
+
|
|
31
|
+
# docs-site
|
|
32
|
+
|
|
33
|
+
[](LICENSE)
|
|
34
|
+
[](https://www.python.org/downloads/)
|
|
35
|
+
|
|
36
|
+
Turn a folder of dumped HTML and PDF files into a clean, searchable, dark-mode-ready static documentation portal.
|
|
37
|
+
|
|
38
|
+
## Features
|
|
39
|
+
|
|
40
|
+
- **Static & Offline**: Generates self-contained static HTML/CSS/JS with zero runtime server required.
|
|
41
|
+
- **Fast Client-side Search**: Full-text search across all HTML and PDF content, with highlight previews and keyboard navigation (`↑`/`↓`/`Enter`/`Esc`).
|
|
42
|
+
- **Automatic Table of Contents**: Extracts heading outlines from HTML tags (`<h1>`–`<h6>`) and bookmark outlines from PDF files into a side navigation panel.
|
|
43
|
+
- **Dark Mode**: Built-in dark/light theme toggle, including synchronized dark theme styling for framed HTML documents.
|
|
44
|
+
- **Zero Configuration**: Point it at any nested folder and it recursively scans and indexes supported files.
|
|
45
|
+
|
|
46
|
+
## Installation
|
|
47
|
+
|
|
48
|
+
Install globally as a CLI tool using [uv](https://docs.astral.sh/uv/):
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
uv tool install docs-site
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Alternatively, install with `pipx`:
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
pipx install docs-site
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
## Usage
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
# Build and open the site in your default browser
|
|
64
|
+
docs-site ~/notes/linux-learning
|
|
65
|
+
|
|
66
|
+
# Specify a custom output directory (default is <folder>/_site)
|
|
67
|
+
docs-site ~/notes/linux-learning -o ~/public_html/docs
|
|
68
|
+
|
|
69
|
+
# Build without automatically opening the browser
|
|
70
|
+
docs-site ~/notes/linux-learning --no-open
|
|
71
|
+
|
|
72
|
+
# View all options
|
|
73
|
+
docs-site --help
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Re-running the command rebuilds the site from scratch.
|
|
77
|
+
|
|
78
|
+
## Project Scope & Constraints
|
|
79
|
+
|
|
80
|
+
- **Supported Formats**: `.html`, `.htm`, and `.pdf` files.
|
|
81
|
+
- **HTML Assets**: HTML files are wrapped and rendered inside an iframe. Relative links to external assets (e.g. images or local stylesheets) should be self-contained or absolute.
|
|
82
|
+
- **PDF Viewing**: PDFs are displayed directly using your browser's native PDF reader.
|
|
83
|
+
- **Indexing Cap**: Indexed full text is capped at 400,000 characters per file to keep the client-side search index responsive and lightweight.
|
|
84
|
+
|
|
85
|
+
## Contributing
|
|
86
|
+
|
|
87
|
+
Looking to work on `docs-site` itself? See [CONTRIBUTING.md](CONTRIBUTING.md) for the development setup, testing, and architecture overview.
|
|
88
|
+
|
|
89
|
+
## Project Status
|
|
90
|
+
|
|
91
|
+
This tool was initially built to solve personal study and document organization workflows. Feedback, bug reports, and contributions are welcome.
|
|
92
|
+
|
|
93
|
+
## License
|
|
94
|
+
|
|
95
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# docs-site
|
|
2
|
+
|
|
3
|
+
[](LICENSE)
|
|
4
|
+
[](https://www.python.org/downloads/)
|
|
5
|
+
|
|
6
|
+
Turn a folder of dumped HTML and PDF files into a clean, searchable, dark-mode-ready static documentation portal.
|
|
7
|
+
|
|
8
|
+
## Features
|
|
9
|
+
|
|
10
|
+
- **Static & Offline**: Generates self-contained static HTML/CSS/JS with zero runtime server required.
|
|
11
|
+
- **Fast Client-side Search**: Full-text search across all HTML and PDF content, with highlight previews and keyboard navigation (`↑`/`↓`/`Enter`/`Esc`).
|
|
12
|
+
- **Automatic Table of Contents**: Extracts heading outlines from HTML tags (`<h1>`–`<h6>`) and bookmark outlines from PDF files into a side navigation panel.
|
|
13
|
+
- **Dark Mode**: Built-in dark/light theme toggle, including synchronized dark theme styling for framed HTML documents.
|
|
14
|
+
- **Zero Configuration**: Point it at any nested folder and it recursively scans and indexes supported files.
|
|
15
|
+
|
|
16
|
+
## Installation
|
|
17
|
+
|
|
18
|
+
Install globally as a CLI tool using [uv](https://docs.astral.sh/uv/):
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
uv tool install docs-site
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
Alternatively, install with `pipx`:
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
pipx install docs-site
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
## Usage
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
# Build and open the site in your default browser
|
|
34
|
+
docs-site ~/notes/linux-learning
|
|
35
|
+
|
|
36
|
+
# Specify a custom output directory (default is <folder>/_site)
|
|
37
|
+
docs-site ~/notes/linux-learning -o ~/public_html/docs
|
|
38
|
+
|
|
39
|
+
# Build without automatically opening the browser
|
|
40
|
+
docs-site ~/notes/linux-learning --no-open
|
|
41
|
+
|
|
42
|
+
# View all options
|
|
43
|
+
docs-site --help
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Re-running the command rebuilds the site from scratch.
|
|
47
|
+
|
|
48
|
+
## Project Scope & Constraints
|
|
49
|
+
|
|
50
|
+
- **Supported Formats**: `.html`, `.htm`, and `.pdf` files.
|
|
51
|
+
- **HTML Assets**: HTML files are wrapped and rendered inside an iframe. Relative links to external assets (e.g. images or local stylesheets) should be self-contained or absolute.
|
|
52
|
+
- **PDF Viewing**: PDFs are displayed directly using your browser's native PDF reader.
|
|
53
|
+
- **Indexing Cap**: Indexed full text is capped at 400,000 characters per file to keep the client-side search index responsive and lightweight.
|
|
54
|
+
|
|
55
|
+
## Contributing
|
|
56
|
+
|
|
57
|
+
Looking to work on `docs-site` itself? See [CONTRIBUTING.md](CONTRIBUTING.md) for the development setup, testing, and architecture overview.
|
|
58
|
+
|
|
59
|
+
## Project Status
|
|
60
|
+
|
|
61
|
+
This tool was initially built to solve personal study and document organization workflows. Feedback, bug reports, and contributions are welcome.
|
|
62
|
+
|
|
63
|
+
## License
|
|
64
|
+
|
|
65
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "docs-site"
|
|
7
|
+
version = "0.1.1"
|
|
8
|
+
description = "Turn a folder of dumped HTML + PDF files into a browsable, searchable local documentation site."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [
|
|
13
|
+
{ name = "Lucas Rollin Ferreira", email = "lucasrollinferreira@gmail.com" }
|
|
14
|
+
]
|
|
15
|
+
keywords = ["documentation", "pdf", "html", "search", "static-site", "reader"]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Development Status :: 4 - Beta",
|
|
18
|
+
"Environment :: Console",
|
|
19
|
+
"Intended Audience :: Developers",
|
|
20
|
+
"Intended Audience :: Education",
|
|
21
|
+
"License :: OSI Approved :: MIT License",
|
|
22
|
+
"Programming Language :: Python :: 3",
|
|
23
|
+
"Programming Language :: Python :: 3.11",
|
|
24
|
+
"Programming Language :: Python :: 3.12",
|
|
25
|
+
"Programming Language :: Python :: 3.13",
|
|
26
|
+
"Programming Language :: Python :: 3.14",
|
|
27
|
+
"Topic :: Documentation",
|
|
28
|
+
"Topic :: Text Processing :: Markup :: HTML",
|
|
29
|
+
]
|
|
30
|
+
dependencies = [
|
|
31
|
+
"beautifulsoup4>=4.12",
|
|
32
|
+
"jinja2>=3.1",
|
|
33
|
+
"pypdf>=4.0",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
[project.urls]
|
|
37
|
+
Homepage = "https://github.com/lucas-rollin/docs-site"
|
|
38
|
+
Repository = "https://github.com/lucas-rollin/docs-site"
|
|
39
|
+
Issues = "https://github.com/lucas-rollin/docs-site/issues"
|
|
40
|
+
|
|
41
|
+
[dependency-groups]
|
|
42
|
+
dev = [
|
|
43
|
+
"djlint>=1.46.2",
|
|
44
|
+
"mypy>=1.10",
|
|
45
|
+
"pytest>=8.0",
|
|
46
|
+
"ruff>=0.16.9",
|
|
47
|
+
]
|
|
48
|
+
|
|
49
|
+
[project.scripts]
|
|
50
|
+
docs-site = "docs_site.cli:main"
|
|
51
|
+
|
|
52
|
+
[tool.setuptools.packages.find]
|
|
53
|
+
where = ["src"]
|
|
54
|
+
|
|
55
|
+
[tool.setuptools.package-data]
|
|
56
|
+
docs_site = ["py.typed", "templates/*", "static/css/*", "static/js/*"]
|
|
57
|
+
|
|
58
|
+
[tool.mypy]
|
|
59
|
+
mypy_path = "src"
|
|
60
|
+
warn_return_any = true
|
|
61
|
+
warn_unused_configs = true
|
|
62
|
+
warn_redundant_casts = true
|
|
63
|
+
|
|
64
|
+
[[tool.mypy.overrides]]
|
|
65
|
+
module = ["bs4.*"]
|
|
66
|
+
ignore_missing_imports = true
|
|
67
|
+
|
|
68
|
+
[tool.djlint]
|
|
69
|
+
profile = "jinja"
|
|
70
|
+
extension = "html.jinja"
|
|
71
|
+
indent = 2
|
|
72
|
+
exclude = ".venv"
|
|
73
|
+
ignore = "H030,J004"
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""Command-line entry point: `docs-site /path/to/folder`."""
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import sys
|
|
5
|
+
import webbrowser
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from .site_builder import build_site
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def main(argv=None) -> int:
|
|
12
|
+
parser = argparse.ArgumentParser(
|
|
13
|
+
prog="docs-site",
|
|
14
|
+
description="Build a searchable local doc site from a folder of HTML + PDF files.",
|
|
15
|
+
)
|
|
16
|
+
parser.add_argument(
|
|
17
|
+
"folder",
|
|
18
|
+
type=Path,
|
|
19
|
+
help="Folder to scan (recursively) for .html/.htm/.pdf files",
|
|
20
|
+
)
|
|
21
|
+
parser.add_argument(
|
|
22
|
+
"-o",
|
|
23
|
+
"--output",
|
|
24
|
+
type=Path,
|
|
25
|
+
default=None,
|
|
26
|
+
help="Output directory (default: <folder>/_site)",
|
|
27
|
+
)
|
|
28
|
+
parser.add_argument(
|
|
29
|
+
"--no-open",
|
|
30
|
+
action="store_true",
|
|
31
|
+
help="Don't open the site in a browser after building",
|
|
32
|
+
)
|
|
33
|
+
args = parser.parse_args(argv)
|
|
34
|
+
|
|
35
|
+
root = args.folder.resolve()
|
|
36
|
+
if not root.is_dir():
|
|
37
|
+
print(f"error: not a folder: {root}", file=sys.stderr)
|
|
38
|
+
return 1
|
|
39
|
+
|
|
40
|
+
output_dir = (args.output or root / "_site").resolve()
|
|
41
|
+
|
|
42
|
+
manifest = build_site(root, output_dir, site_title=root.name)
|
|
43
|
+
|
|
44
|
+
n_html = sum(1 for f in manifest["files"] if f["type"] == "html")
|
|
45
|
+
n_pdf = sum(1 for f in manifest["files"] if f["type"] == "pdf")
|
|
46
|
+
print(f"Indexed {n_html} HTML file(s) and {n_pdf} PDF file(s).")
|
|
47
|
+
print(f"Site written to: {output_dir}")
|
|
48
|
+
|
|
49
|
+
if not manifest["files"]:
|
|
50
|
+
print(f"(No .html/.htm/.pdf files found under {root})")
|
|
51
|
+
|
|
52
|
+
if not args.no_open:
|
|
53
|
+
webbrowser.open((output_dir / "index.html").as_uri())
|
|
54
|
+
|
|
55
|
+
return 0
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
if __name__ == "__main__":
|
|
59
|
+
sys.exit(main())
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Everything needed to turn one arbitrary HTML file into a processed content page."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import TypedDict
|
|
6
|
+
|
|
7
|
+
from bs4 import BeautifulSoup
|
|
8
|
+
|
|
9
|
+
from .types import Heading
|
|
10
|
+
from .utils import slugify
|
|
11
|
+
|
|
12
|
+
HTML_TAG_RE = re.compile(r"<html[\s>]", re.IGNORECASE)
|
|
13
|
+
HEAD_TAG_RE = re.compile(r"<head[\s>]", re.IGNORECASE)
|
|
14
|
+
BODY_TAG_RE = re.compile(r"<body[\s>]", re.IGNORECASE)
|
|
15
|
+
HEAD_CLOSE_RE = re.compile(r"</head>", re.IGNORECASE)
|
|
16
|
+
HTML_OPEN_RE = re.compile(r"(<html[^>]*>)", re.IGNORECASE)
|
|
17
|
+
HTML_CLOSE_RE = re.compile(r"(</html>)", re.IGNORECASE)
|
|
18
|
+
HEADING_TAG_RE = re.compile(r"^h[1-6]$")
|
|
19
|
+
|
|
20
|
+
# Relative path (from output/pages/*.html) to the shared frontend static assets.
|
|
21
|
+
CONTENT_CSS_HREF = "../static/css/content.css"
|
|
22
|
+
CONTENT_DARK_LISTENER_SRC = "../static/js/content-dark-listener.js"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class HtmlPageResult(TypedDict):
|
|
26
|
+
html: str
|
|
27
|
+
title: str
|
|
28
|
+
headings: list[Heading]
|
|
29
|
+
text: str
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def ensure_full_document(document: str, title: str) -> str:
|
|
33
|
+
"""Wrap a bare HTML fragment into a full <html><head><body> document.
|
|
34
|
+
|
|
35
|
+
Leaves already-complete documents untouched; fills in only whichever of
|
|
36
|
+
<html>/<head>/<body> is missing.
|
|
37
|
+
"""
|
|
38
|
+
if not HTML_TAG_RE.search(document):
|
|
39
|
+
return (
|
|
40
|
+
"<!DOCTYPE html>\n<html>\n<head>\n"
|
|
41
|
+
'<meta charset="utf-8">\n'
|
|
42
|
+
f"<title>{title}</title>\n"
|
|
43
|
+
"</head>\n<body>\n"
|
|
44
|
+
f"{document}\n"
|
|
45
|
+
"</body>\n</html>"
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
if not HEAD_TAG_RE.search(document):
|
|
49
|
+
document = HTML_OPEN_RE.sub(
|
|
50
|
+
r"\1\n<head>\n"
|
|
51
|
+
'<meta charset="utf-8">\n'
|
|
52
|
+
f"<title>{title}</title>\n"
|
|
53
|
+
"</head>",
|
|
54
|
+
document,
|
|
55
|
+
count=1,
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
if not BODY_TAG_RE.search(document):
|
|
59
|
+
if HEAD_CLOSE_RE.search(document):
|
|
60
|
+
document = HEAD_CLOSE_RE.sub(r"\g<0>\n<body>", document, count=1)
|
|
61
|
+
else:
|
|
62
|
+
document = HTML_OPEN_RE.sub(r"\1\n<body>", document, count=1)
|
|
63
|
+
document = HTML_CLOSE_RE.sub("</body>\n\\1", document, count=1)
|
|
64
|
+
|
|
65
|
+
return document
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def process_html_file(path: Path) -> HtmlPageResult:
|
|
69
|
+
"""Parse, wrap-if-needed, id-tag headings, and link in the reader stylesheet/script."""
|
|
70
|
+
raw = path.read_text(encoding="utf-8", errors="replace")
|
|
71
|
+
raw = ensure_full_document(raw, title=path.stem)
|
|
72
|
+
|
|
73
|
+
soup = BeautifulSoup(raw, "html.parser")
|
|
74
|
+
html_tag = soup.html
|
|
75
|
+
if html_tag is None:
|
|
76
|
+
# Extremely malformed input; fall back to a minimal shell around it.
|
|
77
|
+
soup = BeautifulSoup(ensure_full_document(str(soup), path.stem), "html.parser")
|
|
78
|
+
html_tag = soup.html
|
|
79
|
+
if html_tag is None:
|
|
80
|
+
# ensure_full_document() always produces a top-level <html>
|
|
81
|
+
raise ValueError(f"could not construct a valid <html> root for {path}")
|
|
82
|
+
|
|
83
|
+
head_tag = soup.head
|
|
84
|
+
if head_tag is None:
|
|
85
|
+
head_tag = soup.new_tag("head")
|
|
86
|
+
html_tag.insert(0, head_tag)
|
|
87
|
+
|
|
88
|
+
body_tag = soup.body
|
|
89
|
+
if body_tag is None:
|
|
90
|
+
body_tag = soup.new_tag("body")
|
|
91
|
+
html_tag.append(body_tag)
|
|
92
|
+
|
|
93
|
+
title = path.stem
|
|
94
|
+
if soup.title and soup.title.string and soup.title.string.strip():
|
|
95
|
+
title = soup.title.string.strip()
|
|
96
|
+
|
|
97
|
+
used_ids = {tag.get("id") for tag in soup.find_all(id=True) if tag.get("id")}
|
|
98
|
+
headings: list[Heading] = []
|
|
99
|
+
for tag in soup.find_all(HEADING_TAG_RE):
|
|
100
|
+
text = tag.get_text(strip=True)
|
|
101
|
+
if not text:
|
|
102
|
+
continue
|
|
103
|
+
hid = tag.get("id")
|
|
104
|
+
if not hid:
|
|
105
|
+
hid = slugify(text, used_ids)
|
|
106
|
+
tag["id"] = hid
|
|
107
|
+
else:
|
|
108
|
+
used_ids.add(hid)
|
|
109
|
+
headings.append({"level": int(tag.name[1]), "id": str(hid), "text": text})
|
|
110
|
+
|
|
111
|
+
plain_text = soup.get_text(separator=" ", strip=True)
|
|
112
|
+
|
|
113
|
+
link_tag = soup.new_tag("link", rel="stylesheet", href=CONTENT_CSS_HREF)
|
|
114
|
+
head_tag.append(link_tag)
|
|
115
|
+
script_tag = soup.new_tag("script", src=CONTENT_DARK_LISTENER_SRC)
|
|
116
|
+
body_tag.append(script_tag)
|
|
117
|
+
|
|
118
|
+
return {
|
|
119
|
+
"html": str(soup),
|
|
120
|
+
"title": title,
|
|
121
|
+
"headings": headings,
|
|
122
|
+
"text": plain_text,
|
|
123
|
+
}
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""Scan a folder for HTML/PDF files and build the JSON manifest the frontend reads."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import asdict, dataclass, field
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any, Literal, TypedDict
|
|
6
|
+
|
|
7
|
+
from .html_processing import process_html_file
|
|
8
|
+
from .pdf_processing import process_pdf_file
|
|
9
|
+
from .types import Heading
|
|
10
|
+
from .utils import safe_id
|
|
11
|
+
|
|
12
|
+
INDEXABLE_SUFFIXES = (".html", ".htm", ".pdf")
|
|
13
|
+
|
|
14
|
+
# Per-file cap on indexed text so the manifest can't explode on huge PDFs.
|
|
15
|
+
MAX_INDEXED_CHARS = 400_000
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass
|
|
19
|
+
class ManifestEntry:
|
|
20
|
+
"""One indexed file, in the shape the frontend's JS expects (see
|
|
21
|
+
static/js/shell.js). `page_count` is only meaningful for PDFs but is
|
|
22
|
+
always present so the JSON shape is uniform across entries.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
id: str
|
|
26
|
+
type: Literal["html", "pdf"]
|
|
27
|
+
title: str
|
|
28
|
+
relpath: str
|
|
29
|
+
src: str
|
|
30
|
+
headings: list[Heading] = field(default_factory=list)
|
|
31
|
+
text: str = ""
|
|
32
|
+
page_count: int | None = None
|
|
33
|
+
|
|
34
|
+
def to_dict(self) -> dict[str, Any]:
|
|
35
|
+
return asdict(self)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class Manifest(TypedDict):
|
|
39
|
+
files: list[dict[str, Any]]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def scan_folder(root: Path, output_dir: Path) -> list[Path]:
|
|
43
|
+
"""Recursively find all indexable files under root, skipping the output dir itself."""
|
|
44
|
+
files = []
|
|
45
|
+
for p in sorted(root.rglob("*")):
|
|
46
|
+
if not p.is_file():
|
|
47
|
+
continue
|
|
48
|
+
try:
|
|
49
|
+
p.relative_to(output_dir)
|
|
50
|
+
continue # already inside the generated site — don't re-index it
|
|
51
|
+
except ValueError:
|
|
52
|
+
pass
|
|
53
|
+
if p.suffix.lower() in INDEXABLE_SUFFIXES:
|
|
54
|
+
files.append(p)
|
|
55
|
+
return files
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def build_manifest(root: Path, files: list[Path], pages_dir: Path) -> Manifest:
|
|
59
|
+
"""Process every file, write out HTML content pages, and return the site manifest."""
|
|
60
|
+
manifest_entries: list[ManifestEntry] = []
|
|
61
|
+
used_ids: set[str] = set()
|
|
62
|
+
|
|
63
|
+
for path in files:
|
|
64
|
+
relpath = path.relative_to(root).as_posix()
|
|
65
|
+
fid = safe_id(relpath, used_ids)
|
|
66
|
+
suffix = path.suffix.lower()
|
|
67
|
+
entry: ManifestEntry
|
|
68
|
+
|
|
69
|
+
if suffix in (".html", ".htm"):
|
|
70
|
+
try:
|
|
71
|
+
html_result = process_html_file(path)
|
|
72
|
+
except Exception as exc: # noqa: BLE001 - keep building the rest of the site
|
|
73
|
+
print(f"warning: skipping {relpath} ({exc})")
|
|
74
|
+
continue
|
|
75
|
+
out_path = pages_dir / f"{fid}.html"
|
|
76
|
+
out_path.write_text(html_result["html"], encoding="utf-8")
|
|
77
|
+
entry = ManifestEntry(
|
|
78
|
+
id=fid,
|
|
79
|
+
type="html",
|
|
80
|
+
title=html_result["title"],
|
|
81
|
+
relpath=relpath,
|
|
82
|
+
src=f"pages/{out_path.name}",
|
|
83
|
+
headings=html_result["headings"],
|
|
84
|
+
text=html_result["text"][:MAX_INDEXED_CHARS],
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
elif suffix == ".pdf":
|
|
88
|
+
try:
|
|
89
|
+
pdf_result = process_pdf_file(path)
|
|
90
|
+
except Exception as exc: # noqa: BLE001
|
|
91
|
+
print(f"warning: skipping {relpath} ({exc})")
|
|
92
|
+
continue
|
|
93
|
+
# Reference the original PDF in place rather than duplicating it.
|
|
94
|
+
src = (Path("..") / path.relative_to(root)).as_posix()
|
|
95
|
+
entry = ManifestEntry(
|
|
96
|
+
id=fid,
|
|
97
|
+
type="pdf",
|
|
98
|
+
title=pdf_result["title"],
|
|
99
|
+
relpath=relpath,
|
|
100
|
+
src=src,
|
|
101
|
+
headings=pdf_result["headings"],
|
|
102
|
+
text=pdf_result["text"][:MAX_INDEXED_CHARS],
|
|
103
|
+
page_count=pdf_result["page_count"],
|
|
104
|
+
)
|
|
105
|
+
else:
|
|
106
|
+
continue
|
|
107
|
+
|
|
108
|
+
manifest_entries.append(entry)
|
|
109
|
+
|
|
110
|
+
manifest_entries.sort(key=lambda e: e.relpath.lower())
|
|
111
|
+
return {"files": [e.to_dict() for e in manifest_entries]}
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Everything needed to index one PDF file (no rendering, the browser's
|
|
2
|
+
native PDF viewer handles that inside the iframe)."""
|
|
3
|
+
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any, TypedDict
|
|
6
|
+
|
|
7
|
+
from pypdf import PdfReader
|
|
8
|
+
from pypdf.errors import PyPdfError
|
|
9
|
+
|
|
10
|
+
from .types import Heading
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class PdfResult(TypedDict):
|
|
14
|
+
title: str
|
|
15
|
+
headings: list[Heading]
|
|
16
|
+
text: str
|
|
17
|
+
page_count: int
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _extract_headings(
|
|
21
|
+
reader: PdfReader, outline: list[Any], level: int = 1
|
|
22
|
+
) -> list[Heading]:
|
|
23
|
+
headings: list[Heading] = []
|
|
24
|
+
for item in outline:
|
|
25
|
+
if isinstance(item, list):
|
|
26
|
+
headings.extend(_extract_headings(reader, item, level + 1))
|
|
27
|
+
else:
|
|
28
|
+
try:
|
|
29
|
+
page_num = reader.get_destination_page_number(item)
|
|
30
|
+
title = getattr(item, "title", "") or ""
|
|
31
|
+
if page_num is not None and title.strip():
|
|
32
|
+
headings.append(
|
|
33
|
+
{"level": level, "page": page_num + 1, "text": title.strip()}
|
|
34
|
+
)
|
|
35
|
+
except (PyPdfError, AttributeError, TypeError, ValueError):
|
|
36
|
+
continue
|
|
37
|
+
return headings
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def process_pdf_file(path: Path) -> PdfResult:
|
|
41
|
+
reader = PdfReader(str(path))
|
|
42
|
+
metadata = reader.metadata
|
|
43
|
+
raw_title = metadata.title if metadata else ""
|
|
44
|
+
title = raw_title.strip() if raw_title else path.stem
|
|
45
|
+
|
|
46
|
+
try:
|
|
47
|
+
outline = reader.outline or []
|
|
48
|
+
headings = _extract_headings(reader, outline)
|
|
49
|
+
except (PyPdfError, AttributeError, TypeError, ValueError):
|
|
50
|
+
headings = []
|
|
51
|
+
|
|
52
|
+
page_count = len(reader.pages)
|
|
53
|
+
text_chunks: list[str] = []
|
|
54
|
+
for page in reader.pages:
|
|
55
|
+
try:
|
|
56
|
+
page_text = page.extract_text()
|
|
57
|
+
if page_text:
|
|
58
|
+
text_chunks.append(page_text)
|
|
59
|
+
except (PyPdfError, AttributeError, TypeError, ValueError):
|
|
60
|
+
continue
|
|
61
|
+
|
|
62
|
+
plain_text = " ".join(text_chunks)
|
|
63
|
+
|
|
64
|
+
return {
|
|
65
|
+
"title": title,
|
|
66
|
+
"headings": headings,
|
|
67
|
+
"text": plain_text,
|
|
68
|
+
"page_count": page_count,
|
|
69
|
+
}
|
|
File without changes
|