docs-site 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
docs_site/__init__.py ADDED
@@ -0,0 +1,5 @@
1
+ """docs_site, turn a folder of HTML + PDF files into a searchable local doc site."""
2
+
3
+ from .site_builder import build_site
4
+
5
+ __all__ = ["build_site"]
docs_site/__main__.py ADDED
@@ -0,0 +1,4 @@
1
+ from .cli import main
2
+
3
+ if __name__ == "__main__":
4
+ raise SystemExit(main())
docs_site/cli.py ADDED
@@ -0,0 +1,59 @@
1
+ """Command-line entry point: `docs-site /path/to/folder`."""
2
+
3
+ import argparse
4
+ import sys
5
+ import webbrowser
6
+ from pathlib import Path
7
+
8
+ from .site_builder import build_site
9
+
10
+
11
+ def main(argv=None) -> int:
12
+ parser = argparse.ArgumentParser(
13
+ prog="docs-site",
14
+ description="Build a searchable local doc site from a folder of HTML + PDF files.",
15
+ )
16
+ parser.add_argument(
17
+ "folder",
18
+ type=Path,
19
+ help="Folder to scan (recursively) for .html/.htm/.pdf files",
20
+ )
21
+ parser.add_argument(
22
+ "-o",
23
+ "--output",
24
+ type=Path,
25
+ default=None,
26
+ help="Output directory (default: <folder>/_site)",
27
+ )
28
+ parser.add_argument(
29
+ "--no-open",
30
+ action="store_true",
31
+ help="Don't open the site in a browser after building",
32
+ )
33
+ args = parser.parse_args(argv)
34
+
35
+ root = args.folder.resolve()
36
+ if not root.is_dir():
37
+ print(f"error: not a folder: {root}", file=sys.stderr)
38
+ return 1
39
+
40
+ output_dir = (args.output or root / "_site").resolve()
41
+
42
+ manifest = build_site(root, output_dir, site_title=root.name)
43
+
44
+ n_html = sum(1 for f in manifest["files"] if f["type"] == "html")
45
+ n_pdf = sum(1 for f in manifest["files"] if f["type"] == "pdf")
46
+ print(f"Indexed {n_html} HTML file(s) and {n_pdf} PDF file(s).")
47
+ print(f"Site written to: {output_dir}")
48
+
49
+ if not manifest["files"]:
50
+ print(f"(No .html/.htm/.pdf files found under {root})")
51
+
52
+ if not args.no_open:
53
+ webbrowser.open((output_dir / "index.html").as_uri())
54
+
55
+ return 0
56
+
57
+
58
+ if __name__ == "__main__":
59
+ sys.exit(main())
@@ -0,0 +1,123 @@
1
+ """Everything needed to turn one arbitrary HTML file into a processed content page."""
2
+
3
+ import re
4
+ from pathlib import Path
5
+ from typing import TypedDict
6
+
7
+ from bs4 import BeautifulSoup
8
+
9
+ from .types import Heading
10
+ from .utils import slugify
11
+
12
+ HTML_TAG_RE = re.compile(r"<html[\s>]", re.IGNORECASE)
13
+ HEAD_TAG_RE = re.compile(r"<head[\s>]", re.IGNORECASE)
14
+ BODY_TAG_RE = re.compile(r"<body[\s>]", re.IGNORECASE)
15
+ HEAD_CLOSE_RE = re.compile(r"</head>", re.IGNORECASE)
16
+ HTML_OPEN_RE = re.compile(r"(<html[^>]*>)", re.IGNORECASE)
17
+ HTML_CLOSE_RE = re.compile(r"(</html>)", re.IGNORECASE)
18
+ HEADING_TAG_RE = re.compile(r"^h[1-6]$")
19
+
20
+ # Relative path (from output/pages/*.html) to the shared frontend static assets.
21
+ CONTENT_CSS_HREF = "../static/css/content.css"
22
+ CONTENT_DARK_LISTENER_SRC = "../static/js/content-dark-listener.js"
23
+
24
+
25
+ class HtmlPageResult(TypedDict):
26
+ html: str
27
+ title: str
28
+ headings: list[Heading]
29
+ text: str
30
+
31
+
32
+ def ensure_full_document(document: str, title: str) -> str:
33
+ """Wrap a bare HTML fragment into a full <html><head><body> document.
34
+
35
+ Leaves already-complete documents untouched; fills in only whichever of
36
+ <html>/<head>/<body> is missing.
37
+ """
38
+ if not HTML_TAG_RE.search(document):
39
+ return (
40
+ "<!DOCTYPE html>\n<html>\n<head>\n"
41
+ '<meta charset="utf-8">\n'
42
+ f"<title>{title}</title>\n"
43
+ "</head>\n<body>\n"
44
+ f"{document}\n"
45
+ "</body>\n</html>"
46
+ )
47
+
48
+ if not HEAD_TAG_RE.search(document):
49
+ document = HTML_OPEN_RE.sub(
50
+ r"\1\n<head>\n"
51
+ '<meta charset="utf-8">\n'
52
+ f"<title>{title}</title>\n"
53
+ "</head>",
54
+ document,
55
+ count=1,
56
+ )
57
+
58
+ if not BODY_TAG_RE.search(document):
59
+ if HEAD_CLOSE_RE.search(document):
60
+ document = HEAD_CLOSE_RE.sub(r"\g<0>\n<body>", document, count=1)
61
+ else:
62
+ document = HTML_OPEN_RE.sub(r"\1\n<body>", document, count=1)
63
+ document = HTML_CLOSE_RE.sub("</body>\n\\1", document, count=1)
64
+
65
+ return document
66
+
67
+
68
+ def process_html_file(path: Path) -> HtmlPageResult:
69
+ """Parse, wrap-if-needed, id-tag headings, and link in the reader stylesheet/script."""
70
+ raw = path.read_text(encoding="utf-8", errors="replace")
71
+ raw = ensure_full_document(raw, title=path.stem)
72
+
73
+ soup = BeautifulSoup(raw, "html.parser")
74
+ html_tag = soup.html
75
+ if html_tag is None:
76
+ # Extremely malformed input; fall back to a minimal shell around it.
77
+ soup = BeautifulSoup(ensure_full_document(str(soup), path.stem), "html.parser")
78
+ html_tag = soup.html
79
+ if html_tag is None:
80
+ # ensure_full_document() always produces a top-level <html>
81
+ raise ValueError(f"could not construct a valid <html> root for {path}")
82
+
83
+ head_tag = soup.head
84
+ if head_tag is None:
85
+ head_tag = soup.new_tag("head")
86
+ html_tag.insert(0, head_tag)
87
+
88
+ body_tag = soup.body
89
+ if body_tag is None:
90
+ body_tag = soup.new_tag("body")
91
+ html_tag.append(body_tag)
92
+
93
+ title = path.stem
94
+ if soup.title and soup.title.string and soup.title.string.strip():
95
+ title = soup.title.string.strip()
96
+
97
+ used_ids = {tag.get("id") for tag in soup.find_all(id=True) if tag.get("id")}
98
+ headings: list[Heading] = []
99
+ for tag in soup.find_all(HEADING_TAG_RE):
100
+ text = tag.get_text(strip=True)
101
+ if not text:
102
+ continue
103
+ hid = tag.get("id")
104
+ if not hid:
105
+ hid = slugify(text, used_ids)
106
+ tag["id"] = hid
107
+ else:
108
+ used_ids.add(hid)
109
+ headings.append({"level": int(tag.name[1]), "id": str(hid), "text": text})
110
+
111
+ plain_text = soup.get_text(separator=" ", strip=True)
112
+
113
+ link_tag = soup.new_tag("link", rel="stylesheet", href=CONTENT_CSS_HREF)
114
+ head_tag.append(link_tag)
115
+ script_tag = soup.new_tag("script", src=CONTENT_DARK_LISTENER_SRC)
116
+ body_tag.append(script_tag)
117
+
118
+ return {
119
+ "html": str(soup),
120
+ "title": title,
121
+ "headings": headings,
122
+ "text": plain_text,
123
+ }
docs_site/manifest.py ADDED
@@ -0,0 +1,111 @@
1
+ """Scan a folder for HTML/PDF files and build the JSON manifest the frontend reads."""
2
+
3
+ from dataclasses import asdict, dataclass, field
4
+ from pathlib import Path
5
+ from typing import Any, Literal, TypedDict
6
+
7
+ from .html_processing import process_html_file
8
+ from .pdf_processing import process_pdf_file
9
+ from .types import Heading
10
+ from .utils import safe_id
11
+
12
+ INDEXABLE_SUFFIXES = (".html", ".htm", ".pdf")
13
+
14
+ # Per-file cap on indexed text so the manifest can't explode on huge PDFs.
15
+ MAX_INDEXED_CHARS = 400_000
16
+
17
+
18
+ @dataclass
19
+ class ManifestEntry:
20
+ """One indexed file, in the shape the frontend's JS expects (see
21
+ static/js/shell.js). `page_count` is only meaningful for PDFs but is
22
+ always present so the JSON shape is uniform across entries.
23
+ """
24
+
25
+ id: str
26
+ type: Literal["html", "pdf"]
27
+ title: str
28
+ relpath: str
29
+ src: str
30
+ headings: list[Heading] = field(default_factory=list)
31
+ text: str = ""
32
+ page_count: int | None = None
33
+
34
+ def to_dict(self) -> dict[str, Any]:
35
+ return asdict(self)
36
+
37
+
38
+ class Manifest(TypedDict):
39
+ files: list[dict[str, Any]]
40
+
41
+
42
+ def scan_folder(root: Path, output_dir: Path) -> list[Path]:
43
+ """Recursively find all indexable files under root, skipping the output dir itself."""
44
+ files = []
45
+ for p in sorted(root.rglob("*")):
46
+ if not p.is_file():
47
+ continue
48
+ try:
49
+ p.relative_to(output_dir)
50
+ continue # already inside the generated site — don't re-index it
51
+ except ValueError:
52
+ pass
53
+ if p.suffix.lower() in INDEXABLE_SUFFIXES:
54
+ files.append(p)
55
+ return files
56
+
57
+
58
+ def build_manifest(root: Path, files: list[Path], pages_dir: Path) -> Manifest:
59
+ """Process every file, write out HTML content pages, and return the site manifest."""
60
+ manifest_entries: list[ManifestEntry] = []
61
+ used_ids: set[str] = set()
62
+
63
+ for path in files:
64
+ relpath = path.relative_to(root).as_posix()
65
+ fid = safe_id(relpath, used_ids)
66
+ suffix = path.suffix.lower()
67
+ entry: ManifestEntry
68
+
69
+ if suffix in (".html", ".htm"):
70
+ try:
71
+ html_result = process_html_file(path)
72
+ except Exception as exc: # noqa: BLE001 - keep building the rest of the site
73
+ print(f"warning: skipping {relpath} ({exc})")
74
+ continue
75
+ out_path = pages_dir / f"{fid}.html"
76
+ out_path.write_text(html_result["html"], encoding="utf-8")
77
+ entry = ManifestEntry(
78
+ id=fid,
79
+ type="html",
80
+ title=html_result["title"],
81
+ relpath=relpath,
82
+ src=f"pages/{out_path.name}",
83
+ headings=html_result["headings"],
84
+ text=html_result["text"][:MAX_INDEXED_CHARS],
85
+ )
86
+
87
+ elif suffix == ".pdf":
88
+ try:
89
+ pdf_result = process_pdf_file(path)
90
+ except Exception as exc: # noqa: BLE001
91
+ print(f"warning: skipping {relpath} ({exc})")
92
+ continue
93
+ # Reference the original PDF in place rather than duplicating it.
94
+ src = (Path("..") / path.relative_to(root)).as_posix()
95
+ entry = ManifestEntry(
96
+ id=fid,
97
+ type="pdf",
98
+ title=pdf_result["title"],
99
+ relpath=relpath,
100
+ src=src,
101
+ headings=pdf_result["headings"],
102
+ text=pdf_result["text"][:MAX_INDEXED_CHARS],
103
+ page_count=pdf_result["page_count"],
104
+ )
105
+ else:
106
+ continue
107
+
108
+ manifest_entries.append(entry)
109
+
110
+ manifest_entries.sort(key=lambda e: e.relpath.lower())
111
+ return {"files": [e.to_dict() for e in manifest_entries]}
@@ -0,0 +1,69 @@
1
+ """Everything needed to index one PDF file (no rendering, the browser's
2
+ native PDF viewer handles that inside the iframe)."""
3
+
4
+ from pathlib import Path
5
+ from typing import Any, TypedDict
6
+
7
+ from pypdf import PdfReader
8
+ from pypdf.errors import PyPdfError
9
+
10
+ from .types import Heading
11
+
12
+
13
+ class PdfResult(TypedDict):
14
+ title: str
15
+ headings: list[Heading]
16
+ text: str
17
+ page_count: int
18
+
19
+
20
+ def _extract_headings(
21
+ reader: PdfReader, outline: list[Any], level: int = 1
22
+ ) -> list[Heading]:
23
+ headings: list[Heading] = []
24
+ for item in outline:
25
+ if isinstance(item, list):
26
+ headings.extend(_extract_headings(reader, item, level + 1))
27
+ else:
28
+ try:
29
+ page_num = reader.get_destination_page_number(item)
30
+ title = getattr(item, "title", "") or ""
31
+ if page_num is not None and title.strip():
32
+ headings.append(
33
+ {"level": level, "page": page_num + 1, "text": title.strip()}
34
+ )
35
+ except (PyPdfError, AttributeError, TypeError, ValueError):
36
+ continue
37
+ return headings
38
+
39
+
40
+ def process_pdf_file(path: Path) -> PdfResult:
41
+ reader = PdfReader(str(path))
42
+ metadata = reader.metadata
43
+ raw_title = metadata.title if metadata else ""
44
+ title = raw_title.strip() if raw_title else path.stem
45
+
46
+ try:
47
+ outline = reader.outline or []
48
+ headings = _extract_headings(reader, outline)
49
+ except (PyPdfError, AttributeError, TypeError, ValueError):
50
+ headings = []
51
+
52
+ page_count = len(reader.pages)
53
+ text_chunks: list[str] = []
54
+ for page in reader.pages:
55
+ try:
56
+ page_text = page.extract_text()
57
+ if page_text:
58
+ text_chunks.append(page_text)
59
+ except (PyPdfError, AttributeError, TypeError, ValueError):
60
+ continue
61
+
62
+ plain_text = " ".join(text_chunks)
63
+
64
+ return {
65
+ "title": title,
66
+ "headings": headings,
67
+ "text": plain_text,
68
+ "page_count": page_count,
69
+ }
docs_site/py.typed ADDED
File without changes
@@ -0,0 +1,73 @@
1
+ """Ties scanning/processing (manifest.py) together with the frontend
2
+ templates/static assets to produce a finished, browsable site directory.
3
+
4
+ This is the only module that knows about output paths and templates,
5
+ manifest.py and the processors know nothing about how the result gets
6
+ rendered, and the templates/static/ files know nothing about Python.
7
+ """
8
+
9
+ import json
10
+ import shutil
11
+ from importlib import resources
12
+ from pathlib import Path
13
+
14
+ from jinja2 import Environment, FileSystemLoader
15
+
16
+ from .manifest import Manifest, build_manifest, scan_folder
17
+
18
+ try: # Python 3.11+
19
+ from importlib.resources.abc import Traversable
20
+ except ImportError: # Python 3.9 / 3.10
21
+ from importlib.abc import Traversable # type: ignore[no-redef,attr-defined]
22
+
23
+
24
+ def _package_path(*parts: str) -> Traversable:
25
+ return resources.files("docs_site").joinpath(*parts)
26
+
27
+
28
+ def _copy_static_assets(output_dir: Path) -> None:
29
+ static_src = _package_path("static")
30
+ with resources.as_file(static_src) as static_path:
31
+ shutil.copytree(static_path, output_dir / "static", dirs_exist_ok=True)
32
+
33
+
34
+ def _jinja_env() -> Environment:
35
+ templates_src = _package_path("templates")
36
+ with resources.as_file(templates_src) as templates_path:
37
+ # FileSystemLoader needs a real, stable directory; as_file guarantees
38
+ # one exists for the lifetime of this call even from a zipped install.
39
+ return Environment(
40
+ loader=FileSystemLoader(str(templates_path)),
41
+ autoescape=False, # we control every template; manifest JSON must stay raw
42
+ )
43
+
44
+
45
+ def build_site(root: Path, output_dir: Path, site_title: str) -> Manifest:
46
+ """Scan `root`, build the manifest, render the site into `output_dir`.
47
+
48
+ Returns the manifest (handy for callers that want summary stats).
49
+ """
50
+ pages_dir = output_dir / "pages"
51
+ pages_dir.mkdir(parents=True, exist_ok=True)
52
+
53
+ files = scan_folder(root, output_dir)
54
+ manifest = build_manifest(root, files, pages_dir)
55
+
56
+ _copy_static_assets(output_dir)
57
+
58
+ env = _jinja_env()
59
+
60
+ welcome_html = env.get_template("welcome.html.jinja").render(
61
+ count_files=len(manifest["files"])
62
+ )
63
+ (pages_dir / "_welcome.html").write_text(welcome_html, encoding="utf-8")
64
+
65
+ manifest_json = json.dumps(manifest, ensure_ascii=False).replace("</", "<\\/")
66
+ shell_html = env.get_template("shell.html.jinja").render(
67
+ site_title=site_title,
68
+ welcome_file="_welcome.html",
69
+ manifest_json=manifest_json,
70
+ )
71
+ (output_dir / "index.html").write_text(shell_html, encoding="utf-8")
72
+
73
+ return manifest
@@ -0,0 +1,91 @@
1
+ :root {
2
+ --bg: #ffffff;
3
+ --fg: #1a1a1a;
4
+ --muted: #6b6b6b;
5
+ --accent: #2563eb;
6
+ --border: #e2e2e2;
7
+ --code-bg: #f2f2f3;
8
+ }
9
+ html.reader-dark {
10
+ --bg: #1a1a1a;
11
+ --fg: #e6e6e6;
12
+ --muted: #a0a0a0;
13
+ --accent: #60a5fa;
14
+ --border: #3a3a3a;
15
+ --code-bg: #2a2a2a;
16
+ }
17
+ html,
18
+ body {
19
+ background: var(--bg);
20
+ color: var(--fg);
21
+ }
22
+ body {
23
+ box-sizing: border-box;
24
+ margin: 0 auto;
25
+ max-width: 760px;
26
+ padding: 2.5rem 2rem 6rem;
27
+ font-family: Georgia, "Times New Roman", serif;
28
+ line-height: 1.7;
29
+ font-size: 18px;
30
+ }
31
+ h1,
32
+ h2,
33
+ h3,
34
+ h4,
35
+ h5,
36
+ h6 {
37
+ line-height: 1.3;
38
+ scroll-margin-top: 1.5rem;
39
+ }
40
+ p,
41
+ ul,
42
+ ol,
43
+ blockquote {
44
+ margin: 0 0 1.1em 0;
45
+ }
46
+ a {
47
+ color: var(--accent);
48
+ }
49
+ img {
50
+ max-width: 100%;
51
+ height: auto;
52
+ }
53
+ pre,
54
+ code {
55
+ background: var(--code-bg);
56
+ border-radius: 4px;
57
+ font-family: "Fira Code", Consolas, monospace;
58
+ font-size: 15px;
59
+ }
60
+ pre {
61
+ padding: 1em;
62
+ overflow-x: auto;
63
+ }
64
+ code {
65
+ padding: 0.15em 0.35em;
66
+ }
67
+ pre code {
68
+ padding: 0;
69
+ background: none;
70
+ }
71
+ blockquote {
72
+ border-left: 3px solid var(--border);
73
+ margin-left: 0;
74
+ padding-left: 1em;
75
+ color: var(--muted);
76
+ }
77
+ table {
78
+ border-collapse: collapse;
79
+ width: 100%;
80
+ }
81
+ th,
82
+ td {
83
+ border: 1px solid var(--border);
84
+ padding: 0.5em 0.75em;
85
+ text-align: left;
86
+ }
87
+ mark {
88
+ background: #fde68a;
89
+ color: #1a1a1a;
90
+ border-radius: 2px;
91
+ }