docs-site 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. docs_site-0.1.1/LICENSE +21 -0
  2. docs_site-0.1.1/PKG-INFO +95 -0
  3. docs_site-0.1.1/README.md +65 -0
  4. docs_site-0.1.1/pyproject.toml +73 -0
  5. docs_site-0.1.1/setup.cfg +4 -0
  6. docs_site-0.1.1/src/docs_site/__init__.py +5 -0
  7. docs_site-0.1.1/src/docs_site/__main__.py +4 -0
  8. docs_site-0.1.1/src/docs_site/cli.py +59 -0
  9. docs_site-0.1.1/src/docs_site/html_processing.py +123 -0
  10. docs_site-0.1.1/src/docs_site/manifest.py +111 -0
  11. docs_site-0.1.1/src/docs_site/pdf_processing.py +69 -0
  12. docs_site-0.1.1/src/docs_site/py.typed +0 -0
  13. docs_site-0.1.1/src/docs_site/site_builder.py +73 -0
  14. docs_site-0.1.1/src/docs_site/static/css/content.css +91 -0
  15. docs_site-0.1.1/src/docs_site/static/css/shell.css +279 -0
  16. docs_site-0.1.1/src/docs_site/static/js/content-dark-listener.js +17 -0
  17. docs_site-0.1.1/src/docs_site/static/js/shell.js +333 -0
  18. docs_site-0.1.1/src/docs_site/templates/shell.html.jinja +35 -0
  19. docs_site-0.1.1/src/docs_site/templates/welcome.html.jinja +22 -0
  20. docs_site-0.1.1/src/docs_site/types.py +17 -0
  21. docs_site-0.1.1/src/docs_site/utils.py +27 -0
  22. docs_site-0.1.1/src/docs_site.egg-info/PKG-INFO +95 -0
  23. docs_site-0.1.1/src/docs_site.egg-info/SOURCES.txt +29 -0
  24. docs_site-0.1.1/src/docs_site.egg-info/dependency_links.txt +1 -0
  25. docs_site-0.1.1/src/docs_site.egg-info/entry_points.txt +2 -0
  26. docs_site-0.1.1/src/docs_site.egg-info/requires.txt +3 -0
  27. docs_site-0.1.1/src/docs_site.egg-info/top_level.txt +1 -0
  28. docs_site-0.1.1/tests/test_html_processing.py +52 -0
  29. docs_site-0.1.1/tests/test_pdf_processing.py +39 -0
  30. docs_site-0.1.1/tests/test_site_builder.py +36 -0
  31. docs_site-0.1.1/tests/test_utils.py +42 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Lucas Rollin Ferreira
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,95 @@
1
+ Metadata-Version: 2.4
2
+ Name: docs-site
3
+ Version: 0.1.1
4
+ Summary: Turn a folder of dumped HTML + PDF files into a browsable, searchable local documentation site.
5
+ Author-email: Lucas Rollin Ferreira <lucasrollinferreira@gmail.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/lucas-rollin/docs-site
8
+ Project-URL: Repository, https://github.com/lucas-rollin/docs-site
9
+ Project-URL: Issues, https://github.com/lucas-rollin/docs-site/issues
10
+ Keywords: documentation,pdf,html,search,static-site,reader
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Intended Audience :: Education
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Programming Language :: Python :: 3.14
21
+ Classifier: Topic :: Documentation
22
+ Classifier: Topic :: Text Processing :: Markup :: HTML
23
+ Requires-Python: >=3.11
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE
26
+ Requires-Dist: beautifulsoup4>=4.12
27
+ Requires-Dist: jinja2>=3.1
28
+ Requires-Dist: pypdf>=4.0
29
+ Dynamic: license-file
30
+
31
+ # docs-site
32
+
33
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
34
+ [![Python: 3.11+](https://img.shields.io/badge/python-3.11+-blue.svg)](https://www.python.org/downloads/)
35
+
36
+ Turn a folder of dumped HTML and PDF files into a clean, searchable, dark-mode-ready static documentation portal.
37
+
38
+ ## Features
39
+
40
+ - **Static & Offline**: Generates self-contained static HTML/CSS/JS with zero runtime server required.
41
+ - **Fast Client-side Search**: Full-text search across all HTML and PDF content, with highlight previews and keyboard navigation (`↑`/`↓`/`Enter`/`Esc`).
42
+ - **Automatic Table of Contents**: Extracts heading outlines from HTML tags (`<h1>`–`<h6>`) and bookmark outlines from PDF files into a side navigation panel.
43
+ - **Dark Mode**: Built-in dark/light theme toggle, including synchronized dark theme styling for framed HTML documents.
44
+ - **Zero Configuration**: Point it at any nested folder and it recursively scans and indexes supported files.
45
+
46
+ ## Installation
47
+
48
+ Install globally as a CLI tool using [uv](https://docs.astral.sh/uv/):
49
+
50
+ ```bash
51
+ uv tool install docs-site
52
+ ```
53
+
54
+ Alternatively, install with `pipx`:
55
+
56
+ ```bash
57
+ pipx install docs-site
58
+ ```
59
+
60
+ ## Usage
61
+
62
+ ```bash
63
+ # Build and open the site in your default browser
64
+ docs-site ~/notes/linux-learning
65
+
66
+ # Specify a custom output directory (default is <folder>/_site)
67
+ docs-site ~/notes/linux-learning -o ~/public_html/docs
68
+
69
+ # Build without automatically opening the browser
70
+ docs-site ~/notes/linux-learning --no-open
71
+
72
+ # View all options
73
+ docs-site --help
74
+ ```
75
+
76
+ Re-running the command rebuilds the site from scratch.
77
+
78
+ ## Project Scope & Constraints
79
+
80
+ - **Supported Formats**: `.html`, `.htm`, and `.pdf` files.
81
+ - **HTML Assets**: HTML files are wrapped and rendered inside an iframe. Relative links to external assets (e.g. images or local stylesheets) should be self-contained or absolute.
82
+ - **PDF Viewing**: PDFs are displayed directly using your browser's native PDF reader.
83
+ - **Indexing Cap**: Indexed full text is capped at 400,000 characters per file to keep the client-side search index responsive and lightweight.
84
+
85
+ ## Contributing
86
+
87
+ Looking to work on `docs-site` itself? See [CONTRIBUTING.md](CONTRIBUTING.md) for the development setup, testing, and architecture overview.
88
+
89
+ ## Project Status
90
+
91
+ This tool was initially built to solve personal study and document organization workflows. Feedback, bug reports, and contributions are welcome.
92
+
93
+ ## License
94
+
95
+ This project is licensed under the [MIT License](LICENSE).
@@ -0,0 +1,65 @@
1
+ # docs-site
2
+
3
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
4
+ [![Python: 3.11+](https://img.shields.io/badge/python-3.11+-blue.svg)](https://www.python.org/downloads/)
5
+
6
+ Turn a folder of dumped HTML and PDF files into a clean, searchable, dark-mode-ready static documentation portal.
7
+
8
+ ## Features
9
+
10
+ - **Static & Offline**: Generates self-contained static HTML/CSS/JS with zero runtime server required.
11
+ - **Fast Client-side Search**: Full-text search across all HTML and PDF content, with highlight previews and keyboard navigation (`↑`/`↓`/`Enter`/`Esc`).
12
+ - **Automatic Table of Contents**: Extracts heading outlines from HTML tags (`<h1>`–`<h6>`) and bookmark outlines from PDF files into a side navigation panel.
13
+ - **Dark Mode**: Built-in dark/light theme toggle, including synchronized dark theme styling for framed HTML documents.
14
+ - **Zero Configuration**: Point it at any nested folder and it recursively scans and indexes supported files.
15
+
16
+ ## Installation
17
+
18
+ Install globally as a CLI tool using [uv](https://docs.astral.sh/uv/):
19
+
20
+ ```bash
21
+ uv tool install docs-site
22
+ ```
23
+
24
+ Alternatively, install with `pipx`:
25
+
26
+ ```bash
27
+ pipx install docs-site
28
+ ```
29
+
30
+ ## Usage
31
+
32
+ ```bash
33
+ # Build and open the site in your default browser
34
+ docs-site ~/notes/linux-learning
35
+
36
+ # Specify a custom output directory (default is <folder>/_site)
37
+ docs-site ~/notes/linux-learning -o ~/public_html/docs
38
+
39
+ # Build without automatically opening the browser
40
+ docs-site ~/notes/linux-learning --no-open
41
+
42
+ # View all options
43
+ docs-site --help
44
+ ```
45
+
46
+ Re-running the command rebuilds the site from scratch.
47
+
48
+ ## Project Scope & Constraints
49
+
50
+ - **Supported Formats**: `.html`, `.htm`, and `.pdf` files.
51
+ - **HTML Assets**: HTML files are wrapped and rendered inside an iframe. Relative links to external assets (e.g. images or local stylesheets) should be self-contained or absolute.
52
+ - **PDF Viewing**: PDFs are displayed directly using your browser's native PDF reader.
53
+ - **Indexing Cap**: Indexed full text is capped at 400,000 characters per file to keep the client-side search index responsive and lightweight.
54
+
55
+ ## Contributing
56
+
57
+ Looking to work on `docs-site` itself? See [CONTRIBUTING.md](CONTRIBUTING.md) for the development setup, testing, and architecture overview.
58
+
59
+ ## Project Status
60
+
61
+ This tool was initially built to solve personal study and document organization workflows. Feedback, bug reports, and contributions are welcome.
62
+
63
+ ## License
64
+
65
+ This project is licensed under the [MIT License](LICENSE).
@@ -0,0 +1,73 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "docs-site"
7
+ version = "0.1.1"
8
+ description = "Turn a folder of dumped HTML + PDF files into a browsable, searchable local documentation site."
9
+ readme = "README.md"
10
+ requires-python = ">=3.11"
11
+ license = { text = "MIT" }
12
+ authors = [
13
+ { name = "Lucas Rollin Ferreira", email = "lucasrollinferreira@gmail.com" }
14
+ ]
15
+ keywords = ["documentation", "pdf", "html", "search", "static-site", "reader"]
16
+ classifiers = [
17
+ "Development Status :: 4 - Beta",
18
+ "Environment :: Console",
19
+ "Intended Audience :: Developers",
20
+ "Intended Audience :: Education",
21
+ "License :: OSI Approved :: MIT License",
22
+ "Programming Language :: Python :: 3",
23
+ "Programming Language :: Python :: 3.11",
24
+ "Programming Language :: Python :: 3.12",
25
+ "Programming Language :: Python :: 3.13",
26
+ "Programming Language :: Python :: 3.14",
27
+ "Topic :: Documentation",
28
+ "Topic :: Text Processing :: Markup :: HTML",
29
+ ]
30
+ dependencies = [
31
+ "beautifulsoup4>=4.12",
32
+ "jinja2>=3.1",
33
+ "pypdf>=4.0",
34
+ ]
35
+
36
+ [project.urls]
37
+ Homepage = "https://github.com/lucas-rollin/docs-site"
38
+ Repository = "https://github.com/lucas-rollin/docs-site"
39
+ Issues = "https://github.com/lucas-rollin/docs-site/issues"
40
+
41
+ [dependency-groups]
42
+ dev = [
43
+ "djlint>=1.46.2",
44
+ "mypy>=1.10",
45
+ "pytest>=8.0",
46
+ "ruff>=0.16.9",
47
+ ]
48
+
49
+ [project.scripts]
50
+ docs-site = "docs_site.cli:main"
51
+
52
+ [tool.setuptools.packages.find]
53
+ where = ["src"]
54
+
55
+ [tool.setuptools.package-data]
56
+ docs_site = ["py.typed", "templates/*", "static/css/*", "static/js/*"]
57
+
58
+ [tool.mypy]
59
+ mypy_path = "src"
60
+ warn_return_any = true
61
+ warn_unused_configs = true
62
+ warn_redundant_casts = true
63
+
64
+ [[tool.mypy.overrides]]
65
+ module = ["bs4.*"]
66
+ ignore_missing_imports = true
67
+
68
+ [tool.djlint]
69
+ profile = "jinja"
70
+ extension = "html.jinja"
71
+ indent = 2
72
+ exclude = ".venv"
73
+ ignore = "H030,J004"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,5 @@
1
+ """docs_site, turn a folder of HTML + PDF files into a searchable local doc site."""
2
+
3
+ from .site_builder import build_site
4
+
5
+ __all__ = ["build_site"]
@@ -0,0 +1,4 @@
1
+ from .cli import main
2
+
3
+ if __name__ == "__main__":
4
+ raise SystemExit(main())
@@ -0,0 +1,59 @@
1
+ """Command-line entry point: `docs-site /path/to/folder`."""
2
+
3
+ import argparse
4
+ import sys
5
+ import webbrowser
6
+ from pathlib import Path
7
+
8
+ from .site_builder import build_site
9
+
10
+
11
+ def main(argv=None) -> int:
12
+ parser = argparse.ArgumentParser(
13
+ prog="docs-site",
14
+ description="Build a searchable local doc site from a folder of HTML + PDF files.",
15
+ )
16
+ parser.add_argument(
17
+ "folder",
18
+ type=Path,
19
+ help="Folder to scan (recursively) for .html/.htm/.pdf files",
20
+ )
21
+ parser.add_argument(
22
+ "-o",
23
+ "--output",
24
+ type=Path,
25
+ default=None,
26
+ help="Output directory (default: <folder>/_site)",
27
+ )
28
+ parser.add_argument(
29
+ "--no-open",
30
+ action="store_true",
31
+ help="Don't open the site in a browser after building",
32
+ )
33
+ args = parser.parse_args(argv)
34
+
35
+ root = args.folder.resolve()
36
+ if not root.is_dir():
37
+ print(f"error: not a folder: {root}", file=sys.stderr)
38
+ return 1
39
+
40
+ output_dir = (args.output or root / "_site").resolve()
41
+
42
+ manifest = build_site(root, output_dir, site_title=root.name)
43
+
44
+ n_html = sum(1 for f in manifest["files"] if f["type"] == "html")
45
+ n_pdf = sum(1 for f in manifest["files"] if f["type"] == "pdf")
46
+ print(f"Indexed {n_html} HTML file(s) and {n_pdf} PDF file(s).")
47
+ print(f"Site written to: {output_dir}")
48
+
49
+ if not manifest["files"]:
50
+ print(f"(No .html/.htm/.pdf files found under {root})")
51
+
52
+ if not args.no_open:
53
+ webbrowser.open((output_dir / "index.html").as_uri())
54
+
55
+ return 0
56
+
57
+
58
+ if __name__ == "__main__":
59
+ sys.exit(main())
@@ -0,0 +1,123 @@
1
+ """Everything needed to turn one arbitrary HTML file into a processed content page."""
2
+
3
+ import re
4
+ from pathlib import Path
5
+ from typing import TypedDict
6
+
7
+ from bs4 import BeautifulSoup
8
+
9
+ from .types import Heading
10
+ from .utils import slugify
11
+
12
+ HTML_TAG_RE = re.compile(r"<html[\s>]", re.IGNORECASE)
13
+ HEAD_TAG_RE = re.compile(r"<head[\s>]", re.IGNORECASE)
14
+ BODY_TAG_RE = re.compile(r"<body[\s>]", re.IGNORECASE)
15
+ HEAD_CLOSE_RE = re.compile(r"</head>", re.IGNORECASE)
16
+ HTML_OPEN_RE = re.compile(r"(<html[^>]*>)", re.IGNORECASE)
17
+ HTML_CLOSE_RE = re.compile(r"(</html>)", re.IGNORECASE)
18
+ HEADING_TAG_RE = re.compile(r"^h[1-6]$")
19
+
20
+ # Relative path (from output/pages/*.html) to the shared frontend static assets.
21
+ CONTENT_CSS_HREF = "../static/css/content.css"
22
+ CONTENT_DARK_LISTENER_SRC = "../static/js/content-dark-listener.js"
23
+
24
+
25
+ class HtmlPageResult(TypedDict):
26
+ html: str
27
+ title: str
28
+ headings: list[Heading]
29
+ text: str
30
+
31
+
32
+ def ensure_full_document(document: str, title: str) -> str:
33
+ """Wrap a bare HTML fragment into a full <html><head><body> document.
34
+
35
+ Leaves already-complete documents untouched; fills in only whichever of
36
+ <html>/<head>/<body> is missing.
37
+ """
38
+ if not HTML_TAG_RE.search(document):
39
+ return (
40
+ "<!DOCTYPE html>\n<html>\n<head>\n"
41
+ '<meta charset="utf-8">\n'
42
+ f"<title>{title}</title>\n"
43
+ "</head>\n<body>\n"
44
+ f"{document}\n"
45
+ "</body>\n</html>"
46
+ )
47
+
48
+ if not HEAD_TAG_RE.search(document):
49
+ document = HTML_OPEN_RE.sub(
50
+ r"\1\n<head>\n"
51
+ '<meta charset="utf-8">\n'
52
+ f"<title>{title}</title>\n"
53
+ "</head>",
54
+ document,
55
+ count=1,
56
+ )
57
+
58
+ if not BODY_TAG_RE.search(document):
59
+ if HEAD_CLOSE_RE.search(document):
60
+ document = HEAD_CLOSE_RE.sub(r"\g<0>\n<body>", document, count=1)
61
+ else:
62
+ document = HTML_OPEN_RE.sub(r"\1\n<body>", document, count=1)
63
+ document = HTML_CLOSE_RE.sub("</body>\n\\1", document, count=1)
64
+
65
+ return document
66
+
67
+
68
+ def process_html_file(path: Path) -> HtmlPageResult:
69
+ """Parse, wrap-if-needed, id-tag headings, and link in the reader stylesheet/script."""
70
+ raw = path.read_text(encoding="utf-8", errors="replace")
71
+ raw = ensure_full_document(raw, title=path.stem)
72
+
73
+ soup = BeautifulSoup(raw, "html.parser")
74
+ html_tag = soup.html
75
+ if html_tag is None:
76
+ # Extremely malformed input; fall back to a minimal shell around it.
77
+ soup = BeautifulSoup(ensure_full_document(str(soup), path.stem), "html.parser")
78
+ html_tag = soup.html
79
+ if html_tag is None:
80
+ # ensure_full_document() always produces a top-level <html>
81
+ raise ValueError(f"could not construct a valid <html> root for {path}")
82
+
83
+ head_tag = soup.head
84
+ if head_tag is None:
85
+ head_tag = soup.new_tag("head")
86
+ html_tag.insert(0, head_tag)
87
+
88
+ body_tag = soup.body
89
+ if body_tag is None:
90
+ body_tag = soup.new_tag("body")
91
+ html_tag.append(body_tag)
92
+
93
+ title = path.stem
94
+ if soup.title and soup.title.string and soup.title.string.strip():
95
+ title = soup.title.string.strip()
96
+
97
+ used_ids = {tag.get("id") for tag in soup.find_all(id=True) if tag.get("id")}
98
+ headings: list[Heading] = []
99
+ for tag in soup.find_all(HEADING_TAG_RE):
100
+ text = tag.get_text(strip=True)
101
+ if not text:
102
+ continue
103
+ hid = tag.get("id")
104
+ if not hid:
105
+ hid = slugify(text, used_ids)
106
+ tag["id"] = hid
107
+ else:
108
+ used_ids.add(hid)
109
+ headings.append({"level": int(tag.name[1]), "id": str(hid), "text": text})
110
+
111
+ plain_text = soup.get_text(separator=" ", strip=True)
112
+
113
+ link_tag = soup.new_tag("link", rel="stylesheet", href=CONTENT_CSS_HREF)
114
+ head_tag.append(link_tag)
115
+ script_tag = soup.new_tag("script", src=CONTENT_DARK_LISTENER_SRC)
116
+ body_tag.append(script_tag)
117
+
118
+ return {
119
+ "html": str(soup),
120
+ "title": title,
121
+ "headings": headings,
122
+ "text": plain_text,
123
+ }
@@ -0,0 +1,111 @@
1
+ """Scan a folder for HTML/PDF files and build the JSON manifest the frontend reads."""
2
+
3
+ from dataclasses import asdict, dataclass, field
4
+ from pathlib import Path
5
+ from typing import Any, Literal, TypedDict
6
+
7
+ from .html_processing import process_html_file
8
+ from .pdf_processing import process_pdf_file
9
+ from .types import Heading
10
+ from .utils import safe_id
11
+
12
+ INDEXABLE_SUFFIXES = (".html", ".htm", ".pdf")
13
+
14
+ # Per-file cap on indexed text so the manifest can't explode on huge PDFs.
15
+ MAX_INDEXED_CHARS = 400_000
16
+
17
+
18
+ @dataclass
19
+ class ManifestEntry:
20
+ """One indexed file, in the shape the frontend's JS expects (see
21
+ static/js/shell.js). `page_count` is only meaningful for PDFs but is
22
+ always present so the JSON shape is uniform across entries.
23
+ """
24
+
25
+ id: str
26
+ type: Literal["html", "pdf"]
27
+ title: str
28
+ relpath: str
29
+ src: str
30
+ headings: list[Heading] = field(default_factory=list)
31
+ text: str = ""
32
+ page_count: int | None = None
33
+
34
+ def to_dict(self) -> dict[str, Any]:
35
+ return asdict(self)
36
+
37
+
38
+ class Manifest(TypedDict):
39
+ files: list[dict[str, Any]]
40
+
41
+
42
+ def scan_folder(root: Path, output_dir: Path) -> list[Path]:
43
+ """Recursively find all indexable files under root, skipping the output dir itself."""
44
+ files = []
45
+ for p in sorted(root.rglob("*")):
46
+ if not p.is_file():
47
+ continue
48
+ try:
49
+ p.relative_to(output_dir)
50
+ continue # already inside the generated site — don't re-index it
51
+ except ValueError:
52
+ pass
53
+ if p.suffix.lower() in INDEXABLE_SUFFIXES:
54
+ files.append(p)
55
+ return files
56
+
57
+
58
+ def build_manifest(root: Path, files: list[Path], pages_dir: Path) -> Manifest:
59
+ """Process every file, write out HTML content pages, and return the site manifest."""
60
+ manifest_entries: list[ManifestEntry] = []
61
+ used_ids: set[str] = set()
62
+
63
+ for path in files:
64
+ relpath = path.relative_to(root).as_posix()
65
+ fid = safe_id(relpath, used_ids)
66
+ suffix = path.suffix.lower()
67
+ entry: ManifestEntry
68
+
69
+ if suffix in (".html", ".htm"):
70
+ try:
71
+ html_result = process_html_file(path)
72
+ except Exception as exc: # noqa: BLE001 - keep building the rest of the site
73
+ print(f"warning: skipping {relpath} ({exc})")
74
+ continue
75
+ out_path = pages_dir / f"{fid}.html"
76
+ out_path.write_text(html_result["html"], encoding="utf-8")
77
+ entry = ManifestEntry(
78
+ id=fid,
79
+ type="html",
80
+ title=html_result["title"],
81
+ relpath=relpath,
82
+ src=f"pages/{out_path.name}",
83
+ headings=html_result["headings"],
84
+ text=html_result["text"][:MAX_INDEXED_CHARS],
85
+ )
86
+
87
+ elif suffix == ".pdf":
88
+ try:
89
+ pdf_result = process_pdf_file(path)
90
+ except Exception as exc: # noqa: BLE001
91
+ print(f"warning: skipping {relpath} ({exc})")
92
+ continue
93
+ # Reference the original PDF in place rather than duplicating it.
94
+ src = (Path("..") / path.relative_to(root)).as_posix()
95
+ entry = ManifestEntry(
96
+ id=fid,
97
+ type="pdf",
98
+ title=pdf_result["title"],
99
+ relpath=relpath,
100
+ src=src,
101
+ headings=pdf_result["headings"],
102
+ text=pdf_result["text"][:MAX_INDEXED_CHARS],
103
+ page_count=pdf_result["page_count"],
104
+ )
105
+ else:
106
+ continue
107
+
108
+ manifest_entries.append(entry)
109
+
110
+ manifest_entries.sort(key=lambda e: e.relpath.lower())
111
+ return {"files": [e.to_dict() for e in manifest_entries]}
@@ -0,0 +1,69 @@
1
+ """Everything needed to index one PDF file (no rendering, the browser's
2
+ native PDF viewer handles that inside the iframe)."""
3
+
4
+ from pathlib import Path
5
+ from typing import Any, TypedDict
6
+
7
+ from pypdf import PdfReader
8
+ from pypdf.errors import PyPdfError
9
+
10
+ from .types import Heading
11
+
12
+
13
+ class PdfResult(TypedDict):
14
+ title: str
15
+ headings: list[Heading]
16
+ text: str
17
+ page_count: int
18
+
19
+
20
+ def _extract_headings(
21
+ reader: PdfReader, outline: list[Any], level: int = 1
22
+ ) -> list[Heading]:
23
+ headings: list[Heading] = []
24
+ for item in outline:
25
+ if isinstance(item, list):
26
+ headings.extend(_extract_headings(reader, item, level + 1))
27
+ else:
28
+ try:
29
+ page_num = reader.get_destination_page_number(item)
30
+ title = getattr(item, "title", "") or ""
31
+ if page_num is not None and title.strip():
32
+ headings.append(
33
+ {"level": level, "page": page_num + 1, "text": title.strip()}
34
+ )
35
+ except (PyPdfError, AttributeError, TypeError, ValueError):
36
+ continue
37
+ return headings
38
+
39
+
40
+ def process_pdf_file(path: Path) -> PdfResult:
41
+ reader = PdfReader(str(path))
42
+ metadata = reader.metadata
43
+ raw_title = metadata.title if metadata else ""
44
+ title = raw_title.strip() if raw_title else path.stem
45
+
46
+ try:
47
+ outline = reader.outline or []
48
+ headings = _extract_headings(reader, outline)
49
+ except (PyPdfError, AttributeError, TypeError, ValueError):
50
+ headings = []
51
+
52
+ page_count = len(reader.pages)
53
+ text_chunks: list[str] = []
54
+ for page in reader.pages:
55
+ try:
56
+ page_text = page.extract_text()
57
+ if page_text:
58
+ text_chunks.append(page_text)
59
+ except (PyPdfError, AttributeError, TypeError, ValueError):
60
+ continue
61
+
62
+ plain_text = " ".join(text_chunks)
63
+
64
+ return {
65
+ "title": title,
66
+ "headings": headings,
67
+ "text": plain_text,
68
+ "page_count": page_count,
69
+ }
File without changes