brief-spec-renderer-pdf 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,13 @@
1
+ .DS_Store
2
+ .briefspec/
3
+ .coverage
4
+ .pytest_cache/
5
+ .ruff_cache/
6
+ .venv/
7
+ __pycache__/
8
+ *.egg-info/
9
+ *.py[cod]
10
+ build/
11
+ dist/
12
+ coverage.xml
13
+ htmlcov/
@@ -0,0 +1,17 @@
1
+ Metadata-Version: 2.5
2
+ Name: brief-spec-renderer-pdf
3
+ Version: 0.5.0
4
+ Summary: Verified Playwright PDF rendering for Brief-Spec deliveries.
5
+ Author: Luan Moreno Maciel
6
+ License-Expression: MIT
7
+ Requires-Python: >=3.11
8
+ Requires-Dist: brief-spec<0.6,>=0.5
9
+ Requires-Dist: playwright>=1.54
10
+ Description-Content-Type: text/markdown
11
+
12
+ # Brief-Spec PDF renderer
13
+
14
+ Optional Playwright/Chromium renderer for verified Brief-Spec PDF downloads.
15
+
16
+ Install it into the Brief-Spec tool environment, then run `playwright install chromium` or
17
+ `brief-spec doctor all --fix` before the first render.
@@ -0,0 +1,6 @@
1
+ # Brief-Spec PDF renderer
2
+
3
+ Optional Playwright/Chromium renderer for verified Brief-Spec PDF downloads.
4
+
5
+ Install it into the Brief-Spec tool environment, then run `playwright install chromium` or
6
+ `brief-spec doctor all --fix` before the first render.
@@ -0,0 +1,25 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.27"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "brief-spec-renderer-pdf"
7
+ version = "0.5.0"
8
+ description = "Verified Playwright PDF rendering for Brief-Spec deliveries."
9
+ readme = "README.md"
10
+ requires-python = ">=3.11"
11
+ license = "MIT"
12
+ authors = [{ name = "Luan Moreno Maciel" }]
13
+ dependencies = [
14
+ "brief-spec>=0.5,<0.6",
15
+ "playwright>=1.54",
16
+ ]
17
+
18
+ [project.entry-points."brief_spec.renderers"]
19
+ pdf = "briefspec_renderer_pdf:PDFRenderer"
20
+
21
+ [project.entry-points."briefspec.renderers"]
22
+ pdf = "briefspec_renderer_pdf:PDFRenderer"
23
+
24
+ [tool.hatch.build.targets.wheel]
25
+ packages = ["src/briefspec_renderer_pdf"]
@@ -0,0 +1,262 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ import shutil
5
+ import subprocess
6
+ import sys
7
+ import tempfile
8
+ from datetime import UTC, datetime
9
+ from pathlib import Path
10
+ from typing import Any
11
+
12
+ from briefspec.delivery import render_html, sha256_bytes
13
+ from briefspec.state import atomic_write_public
14
+
15
+ __version__ = "0.5.0"
16
+
17
+ _PDF_FIELD = re.compile(r"^(?P<name>[A-Za-z ]+):\s*(?P<value>.+)$", re.MULTILINE)
18
+ _PAGE = re.compile(r'<page\s+width="(?P<width>[\d.]+)"\s+height="(?P<height>[\d.]+)">')
19
+ _WORD = re.compile(
20
+ r'<word\s+xMin="(?P<x_min>[\d.]+)"\s+yMin="(?P<y_min>[\d.]+)"\s+'
21
+ r'xMax="(?P<x_max>[\d.]+)"\s+yMax="(?P<y_max>[\d.]+)"'
22
+ )
23
+ _PDF_TIMESTAMP = re.compile(rb"/(?P<field>CreationDate|ModDate) \(D:[^)]+\)")
24
+
25
+
26
+ def _canonicalize_pdf_timestamps(content: bytes, created_at: str) -> bytes:
27
+ """Replace Chromium wall-clock metadata with the canonical delivery timestamp."""
28
+ value = datetime.fromisoformat(created_at.replace("Z", "+00:00")).astimezone(UTC)
29
+ timestamp = value.strftime("D:%Y%m%d%H%M%S+00'00'").encode("ascii")
30
+
31
+ def replacement(match: re.Match[bytes]) -> bytes:
32
+ return b"/" + match.group("field") + b" (" + timestamp + b")"
33
+
34
+ normalized, count = _PDF_TIMESTAMP.subn(replacement, content)
35
+ if count != 2 or len(normalized) != len(content):
36
+ raise RuntimeError("Chromium PDF timestamps could not be normalized deterministically")
37
+ return normalized
38
+
39
+
40
+ def _pdf_fields(output: str) -> dict[str, str]:
41
+ return {
42
+ match.group("name").strip().lower().replace(" ", "_"): match.group("value").strip()
43
+ for match in _PDF_FIELD.finditer(output)
44
+ }
45
+
46
+
47
+ def render_html_document(
48
+ html_content: bytes,
49
+ output: Path,
50
+ *,
51
+ created_at: str,
52
+ title: str,
53
+ page_format: str = "A4",
54
+ force: bool = False,
55
+ ) -> dict[str, Any]:
56
+ """Render any canonical, self-contained Brief-Spec HTML document to PDF."""
57
+ if page_format not in {"A4", "Letter"}:
58
+ raise ValueError("PDF page format must be A4 or Letter")
59
+ if not created_at:
60
+ raise ValueError("PDF rendering requires a canonical created_at")
61
+ if output.exists() and not force:
62
+ raise FileExistsError(f"Refusing to overwrite existing output: {output}")
63
+ from playwright.sync_api import sync_playwright
64
+
65
+ with tempfile.TemporaryDirectory(prefix="briefspec-pdf-") as temporary:
66
+ root = Path(temporary)
67
+ source = root / "brief.html"
68
+ rendered = root / "brief.pdf"
69
+ source.write_bytes(html_content)
70
+ with sync_playwright() as playwright:
71
+ browser = playwright.chromium.launch()
72
+ browser_version = browser.version
73
+ page = browser.new_page()
74
+ page.goto(source.as_uri(), wait_until="load")
75
+ page.pdf(
76
+ path=str(rendered),
77
+ format=page_format,
78
+ print_background=True,
79
+ prefer_css_page_size=False,
80
+ )
81
+ browser.close()
82
+ content = _canonicalize_pdf_timestamps(rendered.read_bytes(), created_at)
83
+ atomic_write_public(output, content, mode=0o644)
84
+ visual_sha256 = None
85
+ pdftoppm = shutil.which("pdftoppm")
86
+ if pdftoppm:
87
+ with tempfile.TemporaryDirectory(prefix="briefspec-pdf-visual-") as temporary:
88
+ page = Path(temporary) / "page"
89
+ result = subprocess.run(
90
+ [pdftoppm, "-f", "1", "-singlefile", "-png", str(output), str(page)],
91
+ text=True,
92
+ capture_output=True,
93
+ timeout=60,
94
+ check=False,
95
+ )
96
+ rendered_page = page.with_suffix(".png")
97
+ if result.returncode == 0 and rendered_page.is_file():
98
+ visual_sha256 = sha256_bytes(rendered_page.read_bytes())
99
+ metadata = {
100
+ "renderer": "pdf",
101
+ "renderer_version": __version__,
102
+ "source_html_sha256": sha256_bytes(html_content),
103
+ "chromium_version": browser_version,
104
+ "page_format": page_format,
105
+ "canonical_created_at": created_at,
106
+ "title": title,
107
+ "visual_sha256": visual_sha256,
108
+ }
109
+ return {
110
+ "format": "pdf",
111
+ "path": output.name,
112
+ "media_type": "application/pdf",
113
+ "size_bytes": len(content),
114
+ "sha256": sha256_bytes(content),
115
+ "renderer_version": __version__,
116
+ "metadata": metadata,
117
+ }
118
+
119
+
120
+ class PDFRenderer:
121
+ name = "pdf"
122
+ media_type = "application/pdf"
123
+ filename = "brief.pdf"
124
+
125
+ def capabilities(self) -> dict[str, Any]:
126
+ try:
127
+ import playwright # noqa: F401
128
+ except ImportError:
129
+ python_api = False
130
+ else:
131
+ python_api = True
132
+ tools = {
133
+ name: shutil.which(name) for name in ("pdfinfo", "pdftotext", "pdffonts", "pdftoppm")
134
+ }
135
+ return {
136
+ "renderer_version": __version__,
137
+ "media_type": self.media_type,
138
+ "playwright": python_api,
139
+ "verification_tools": tools,
140
+ "ready": python_api and all(tools.values()),
141
+ }
142
+
143
+ def setup(self, *, dry_run: bool = False) -> dict[str, Any]:
144
+ command = [sys.executable, "-m", "playwright", "install", "chromium"]
145
+ if dry_run:
146
+ return {"status": "DRY-RUN", "command": command}
147
+ result = subprocess.run(
148
+ command,
149
+ text=True,
150
+ capture_output=True,
151
+ timeout=600,
152
+ check=False,
153
+ )
154
+ if result.returncode != 0:
155
+ raise RuntimeError(result.stderr.strip() or "Chromium installation failed")
156
+ return {"status": "PASS", "command": command}
157
+
158
+ def render(
159
+ self,
160
+ delivery: dict[str, Any],
161
+ output: Path,
162
+ options: dict[str, Any],
163
+ ) -> dict[str, Any]:
164
+ page_format = str(options.get("page_format", "A4"))
165
+ if page_format not in {"A4", "Letter"}:
166
+ raise ValueError("PDF page format must be A4 or Letter")
167
+ created_at = str(delivery.get("source", {}).get("created_at") or "")
168
+ if not created_at:
169
+ raise ValueError("PDF rendering requires source.created_at")
170
+ html_content = render_html(delivery).encode("utf-8")
171
+ record = render_html_document(
172
+ html_content,
173
+ output,
174
+ created_at=created_at,
175
+ title="Brief-Spec delivery",
176
+ page_format=page_format,
177
+ )
178
+ record["path"] = str(output)
179
+ return record
180
+
181
+ def verify(self, artifact: Path) -> dict[str, Any]:
182
+ missing = [
183
+ name
184
+ for name in ("pdfinfo", "pdftotext", "pdffonts", "pdftoppm")
185
+ if shutil.which(name) is None
186
+ ]
187
+ if missing:
188
+ return {"status": "FAIL", "detail": f"missing PDF verifier(s): {', '.join(missing)}"}
189
+ with tempfile.TemporaryDirectory(prefix="briefspec-pdf-verify-") as temporary:
190
+ root = Path(temporary)
191
+ commands = [
192
+ ["pdfinfo", str(artifact)],
193
+ ["pdftotext", str(artifact), str(root / "brief.txt")],
194
+ ["pdffonts", str(artifact)],
195
+ ["pdftotext", "-bbox-layout", str(artifact), str(root / "bbox.html")],
196
+ [
197
+ "pdftoppm",
198
+ "-f",
199
+ "1",
200
+ "-singlefile",
201
+ "-png",
202
+ str(artifact),
203
+ str(root / "page"),
204
+ ],
205
+ ]
206
+ outputs = []
207
+ for command in commands:
208
+ result = subprocess.run(
209
+ command,
210
+ text=True,
211
+ capture_output=True,
212
+ timeout=60,
213
+ check=False,
214
+ )
215
+ if result.returncode != 0:
216
+ return {
217
+ "status": "FAIL",
218
+ "detail": result.stderr.strip() or f"failed: {command[0]}",
219
+ }
220
+ outputs.append(result.stdout)
221
+ text = (root / "brief.txt").read_text(encoding="utf-8", errors="replace")
222
+ if not text.strip() or not (root / "page.png").is_file():
223
+ return {"status": "FAIL", "detail": "PDF text or first-page render is empty"}
224
+ fields = _pdf_fields(outputs[0])
225
+ try:
226
+ pages = int(fields["pages"])
227
+ except (KeyError, ValueError):
228
+ return {"status": "FAIL", "detail": "PDF page count is missing or invalid"}
229
+ if pages < 1 or fields.get("encrypted", "").lower() != "no":
230
+ return {"status": "FAIL", "detail": "PDF page count or encryption is invalid"}
231
+ if not fields.get("title") or not fields.get("page_size"):
232
+ return {"status": "FAIL", "detail": "PDF title or page size metadata is missing"}
233
+ font_lines = [line for line in outputs[2].splitlines() if line.strip()]
234
+ if len(font_lines) < 3:
235
+ return {"status": "FAIL", "detail": "PDF contains no inspectable embedded font"}
236
+ bbox = (root / "bbox.html").read_text(encoding="utf-8", errors="replace")
237
+ page_match = _PAGE.search(bbox)
238
+ words = list(_WORD.finditer(bbox))
239
+ if page_match is None or not words:
240
+ return {"status": "FAIL", "detail": "PDF text geometry is unavailable"}
241
+ width = float(page_match.group("width"))
242
+ height = float(page_match.group("height"))
243
+ clipped = any(
244
+ float(word.group("x_min")) < 0
245
+ or float(word.group("y_min")) < 0
246
+ or float(word.group("x_max")) > width
247
+ or float(word.group("y_max")) > height
248
+ for word in words
249
+ )
250
+ if clipped:
251
+ return {"status": "FAIL", "detail": "PDF contains text outside the page bounds"}
252
+ visual_sha256 = sha256_bytes((root / "page.png").read_bytes())
253
+ return {
254
+ "status": "PASS",
255
+ "detail": (
256
+ f"{pages} page(s); selectable text, fonts, metadata, geometry, and page render "
257
+ "verified"
258
+ ),
259
+ "page_count": pages,
260
+ "visual_sha256": visual_sha256,
261
+ "page_size": fields["page_size"],
262
+ }
@@ -0,0 +1,56 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+
5
+ import briefspec_renderer_pdf as pdf
6
+ import pytest
7
+
8
+
9
+ def test_pdf_timestamp_normalization_is_deterministic() -> None:
10
+ first = b"/CreationDate (D:20260812192135+00'00') /ModDate (D:20260812192135+00'00')"
11
+ second = b"/CreationDate (D:20260812192137+00'00') /ModDate (D:20260812192137+00'00')"
12
+ expected = b"/CreationDate (D:20260811120000+00'00') /ModDate (D:20260811120000+00'00')"
13
+ assert pdf._canonicalize_pdf_timestamps(first, "2026-08-11T12:00:00Z") == expected
14
+ assert pdf._canonicalize_pdf_timestamps(second, "2026-08-11T12:00:00Z") == expected
15
+
16
+
17
+ def test_setup_dry_run_uses_current_python() -> None:
18
+ result = pdf.PDFRenderer().setup(dry_run=True)
19
+ assert result["status"] == "DRY-RUN"
20
+ assert result["command"][1:] == ["-m", "playwright", "install", "chromium"]
21
+
22
+
23
+ def test_verify_reports_missing_tools(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None:
24
+ monkeypatch.setattr(pdf.shutil, "which", lambda _name: None)
25
+ result = pdf.PDFRenderer().verify(tmp_path / "missing.pdf")
26
+ assert result["status"] == "FAIL"
27
+ assert "missing PDF verifier" in result["detail"]
28
+
29
+
30
+ def test_invalid_page_format_fails_before_browser_launch(tmp_path: Path) -> None:
31
+ with pytest.raises(ValueError, match="A4 or Letter"):
32
+ pdf.PDFRenderer().render({}, tmp_path / "brief.pdf", {"page_format": "Legal"})
33
+
34
+
35
+ def test_missing_canonical_timestamp_fails_before_browser_launch(tmp_path: Path) -> None:
36
+ with pytest.raises(ValueError, match="source.created_at"):
37
+ pdf.PDFRenderer().render({}, tmp_path / "brief.pdf", {})
38
+
39
+
40
+ def test_generic_html_helper_fails_closed_before_browser_launch(tmp_path: Path) -> None:
41
+ target = tmp_path / "chronicle.pdf"
42
+ target.write_bytes(b"existing")
43
+ with pytest.raises(FileExistsError, match="Refusing to overwrite"):
44
+ pdf.render_html_document(
45
+ b"<!doctype html><title>Chronicle</title>",
46
+ target,
47
+ created_at="2026-08-14T12:00:00+00:00",
48
+ title="Chronicle",
49
+ )
50
+ with pytest.raises(ValueError, match="canonical created_at"):
51
+ pdf.render_html_document(
52
+ b"<!doctype html><title>Chronicle</title>",
53
+ tmp_path / "new.pdf",
54
+ created_at="",
55
+ title="Chronicle",
56
+ )