unaltraweb 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. checksums.yaml +4 -4
  2. data/Makefile +5 -5
  3. data/README.md +19 -7
  4. data/_plugins/figure_captions.rb +47 -10
  5. data/_sass/_documentation.scss +7 -5
  6. data/_sass/_manual.scss +7 -0
  7. data/docs/_documentation/en/02-tools.md +4 -4
  8. data/docs/_documentation/en/03-usage.md +61 -0
  9. data/docs/_documentation/en/13-unaltremanual.md +1 -1
  10. data/docs/_documentation/en/20-syntax.md +14 -0
  11. data/docs/_documentation/en/25-caption-credits.md +120 -0
  12. data/docs/_documentation/en/26-image-backgrounds.md +103 -0
  13. data/docs/_documentation/en/31-template.md +1 -1
  14. data/docs/_documentation/en/32-development.md +1 -1
  15. data/docs/_documentation/en/40-distribution.md +25 -5
  16. data/docs/_documentation/en/42-docker-image.md +7 -7
  17. data/docs/_documentation/en/43-workspace-path-policies.md +232 -0
  18. data/docs/_documentation/en/44-editorial-review.md +237 -0
  19. data/docs/agents/action-prompts/00-start-site-session.txt +10 -5
  20. data/docs/agents/action-prompts/22-manual-style-audit.txt +3 -1
  21. data/docs/agents/manual-authoring-components.md +38 -0
  22. data/docs/agents/mcp-contract.md +102 -10
  23. data/docs/agents/visual-companions-0.4.0.md +74 -0
  24. data/docs/assets/img/caption-credits-demo.svg +19 -0
  25. data/scripts/editorial_check.py +12 -0
  26. data/scripts/image_background_check.py +12 -0
  27. data/scripts/manual/build_pdf.py +94 -16
  28. data/scripts/manual/filters/figure-captions.lua +65 -10
  29. data/scripts/manual/templates/manual.tex +2 -1
  30. data/scripts/test_gem_build.py +32 -2
  31. data/scripts/test_reproducible_jekyll_build.py +1 -1
  32. data/scripts/test_wheel_install.py +28 -0
  33. data/scripts/unaltraweb-mcp-bootstrap.sh +1 -1
  34. data/scripts/validate_distribution.py +18 -3
  35. data/src/unaltraweb_mcp/component-contract.json +28 -28
  36. data/src/unaltraweb_mcp/editorial.py +495 -0
  37. data/src/unaltraweb_mcp/editorial_sources.py +504 -0
  38. data/src/unaltraweb_mcp/image_backgrounds.py +334 -0
  39. data/src/unaltraweb_mcp/image_probe.py +149 -0
  40. data/src/unaltraweb_mcp/processes.py +146 -0
  41. metadata +15 -2
@@ -0,0 +1,334 @@
1
+ """Advisory, offline background checks for referenced publication images."""
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import hashlib
6
+ import json
7
+ import os
8
+ import posixpath
9
+ import re
10
+ import stat
11
+ import sys
12
+ import time
13
+ import urllib.parse
14
+ from html.parser import HTMLParser
15
+ from pathlib import Path
16
+ from typing import Any
17
+
18
+ from .editorial_sources import Reader, corpus, default_language, relative_path, strict_json, yaml_mapping, yaml_value
19
+ from .processes import run_process
20
+
21
+ IMAGE_SUFFIXES = (".svg", ".svgz", ".png", ".jpg", ".jpeg", ".webp", ".gif", ".bmp", ".tif", ".tiff", ".avif", ".ico")
22
+ DIAGRAM_SUFFIXES = (".mmd", ".mermaid", ".puml", ".plantuml", ".uml")
23
+ CAPTURE_SUFFIXES = (".capture.yml", ".capture.yaml")
24
+ VEGA_SUFFIXES = (".vl.json", ".vg.json")
25
+ COMPUTE_SUFFIXES = (".qmd", ".rmd", ".r", ".py", ".ipynb")
26
+ VISUAL_SUFFIXES = tuple(sorted({*IMAGE_SUFFIXES, *DIAGRAM_SUFFIXES, *CAPTURE_SUFFIXES, *VEGA_SUFFIXES,
27
+ *COMPUTE_SUFFIXES, ".edited.svg", *(suffix + ".svg" for suffix in DIAGRAM_SUFFIXES)}, key=len, reverse=True))
28
+ MAX_IMAGE_BYTES = 32 * 1024 * 1024
29
+ MAX_IMAGE_TOTAL_BYTES = 128 * 1024 * 1024
30
+ MAX_IMAGES = 128
31
+ MAX_REFERENCES = 2000
32
+ CHECK_SECONDS = 30.0
33
+ IMAGE_FIELDS = {"image", "img", "cover", "cover_image", "photo", "avatar", "logo", "logo_inverse", "logo_cafe", "thumbnail", "preview"}
34
+
35
+
36
+ class ImageReferences(HTMLParser):
37
+ def __init__(self):
38
+ super().__init__(convert_charrefs=True)
39
+ self.references: list[tuple[int, str]] = []
40
+
41
+ def handle_starttag(self, tag, attrs):
42
+ values = dict(attrs)
43
+ names = {"img": ("src",), "source": (), "video": ("poster",), "object": ("data",), "embed": ("src",)}.get(tag, ())
44
+ if tag == "link" and "icon" in (values.get("rel") or ""):
45
+ names = ("href",)
46
+ for name in names:
47
+ if values.get(name):
48
+ self.references.append((self.getpos()[0], values[name]))
49
+ if tag in {"img", "source"} and values.get("srcset"):
50
+ srcset = values["srcset"]
51
+ if srcset.lstrip().startswith("data:"):
52
+ self.references.append((self.getpos()[0], "data:"))
53
+ else:
54
+ self.references.extend((self.getpos()[0], item.strip().split()[0]) for item in srcset.split(",") if item.strip())
55
+ if len(self.references) > MAX_REFERENCES:
56
+ raise ValueError("Image reference budget exceeded.")
57
+
58
+
59
+ def _metadata_images(value: Any, *, parent_image: bool = False, depth: int = 0):
60
+ if depth > 20:
61
+ raise ValueError("Image metadata nesting limit exceeded.")
62
+ if isinstance(value, dict):
63
+ for key, child in value.items():
64
+ image_field = isinstance(key, str) and (key in IMAGE_FIELDS or (parent_image and key in {"path", "src", "file"}))
65
+ if image_field and isinstance(child, str) and child.strip():
66
+ yield child
67
+ elif isinstance(child, (dict, list)):
68
+ yield from _metadata_images(child, parent_image=image_field, depth=depth + 1)
69
+ elif isinstance(value, list):
70
+ for child in value:
71
+ if parent_image and isinstance(child, str) and child.strip():
72
+ yield child
73
+ else:
74
+ yield from _metadata_images(child, parent_image=parent_image, depth=depth + 1)
75
+
76
+
77
+ def _source_references(text: str) -> list[tuple[int, str]]:
78
+ # Resolve only literal supported URL sugar, never evaluate Liquid or code.
79
+ text = re.sub(r"\{\{\s*site\.baseurl\s*\}\}", "", text)
80
+ text = re.sub(r'''\{\{\s*(['"])(.*?)\1\s*\|\s*(?:relative_url|absolute_url)\s*\}\}''', lambda m: m[2], text)
81
+ from .editorial_sources import blank
82
+
83
+ text = re.sub(r"<!--.*?(?:-->|\Z)|{%\s*comment\s*%}.*?(?:{%\s*endcomment\s*%}|\Z)", lambda m: blank(m[0]), text, flags=re.S)
84
+ visible, fence = [], ""
85
+ for line in text.splitlines(keepends=True):
86
+ match = re.match(r"^\s*(`{3,}|~{3,})", line)
87
+ if match:
88
+ marker = match[1]
89
+ if not fence:
90
+ fence = marker
91
+ elif marker[0] == fence[0] and len(marker) >= len(fence):
92
+ fence = ""
93
+ visible.append(blank(line))
94
+ elif fence or line.startswith((" ", "\t")):
95
+ visible.append(blank(line))
96
+ else:
97
+ visible.append(re.sub(r"(`+)[^`\n]*?\1", lambda m: blank(m[0]), line))
98
+ text = "".join(visible)
99
+ references = []
100
+ # Escaped and ordinary alt-text characters must be disjoint so malformed
101
+ # labels with many backslashes cannot trigger exponential backtracking.
102
+ for match in re.finditer(r"!\[(?:\\.|[^\\\]\n])*\]\(\s*(?:<([^>\n]+)>|([^\s)]+))", text):
103
+ references.append((text[:match.start()].count("\n") + 1, match[1] or match[2]))
104
+ parser = ImageReferences()
105
+ parser.feed(text)
106
+ references.extend(parser.references)
107
+ return references
108
+
109
+
110
+ def _collect(reader: Reader, config: dict, source: str, output_folder: str) -> list[dict]:
111
+ language = default_language(config)
112
+ if source and source.lower().endswith(VISUAL_SUFFIXES):
113
+ relative_path(source)
114
+ return [{"path": source, "line": 0, "reference": source, "language": language, "direct": True}]
115
+ references = []
116
+ if output_folder:
117
+ paths = reader.walk(output_folder)
118
+ if not paths or paths == [output_folder]:
119
+ raise ValueError("Choose a non-empty rendered output directory.")
120
+ for path in paths:
121
+ if path.lower().endswith(".html"):
122
+ parser = ImageReferences()
123
+ parser.feed(reader.text(path))
124
+ references.extend({"path": path, "line": line, "reference": url, "language": language} for line, url in parser.references)
125
+ else:
126
+ uw = config.get("unaltraweb") or {}
127
+ if not isinstance(uw, dict):
128
+ raise ValueError("unaltraweb configuration must be a mapping.")
129
+ selected = corpus(reader, config, str(uw.get("site_profile") or ""), source)
130
+ for path, record in selected["sources"].items():
131
+ text = record["text"]
132
+ if path == "_config.yml" or path.startswith("_data/"):
133
+ data = strict_json(text) if path.endswith(".json") else yaml_value(text)
134
+ references.extend({"path": path, "line": 0, "reference": url, "language": language, "metadata": True}
135
+ for url in _metadata_images(data))
136
+ else:
137
+ front = re.match(r"\A---\s*\n(.*?)\n---[^\S\n]*(?:\n|\Z)", text, re.S)
138
+ front_data = yaml_mapping(front[1]) if front else {}
139
+ document_language = str(front_data.get("lang") or next((part for part in Path(path).parts if part in config.get("languages", [])), language))
140
+ references.extend({"path": path, "line": 0, "reference": url, "language": document_language, "metadata": True}
141
+ for url in _metadata_images(front_data))
142
+ body = ("\n" * text[:front.end()].count("\n") + text[front.end():]) if front else text
143
+ references.extend({"path": path, "line": line, "reference": url, "language": document_language}
144
+ for line, url in _source_references(body))
145
+ if len(references) > MAX_REFERENCES:
146
+ raise ValueError("Image reference budget exceeded; select a smaller source.")
147
+ if len(references) > MAX_REFERENCES:
148
+ raise ValueError("Image reference budget exceeded; select a smaller source.")
149
+ return references
150
+
151
+
152
+ def _local_path(item: dict, config: dict, output_folder: str) -> str:
153
+ raw = item["reference"].strip()
154
+ if any(token in raw for token in ("{{", "{%")):
155
+ raise ValueError("Dynamic image references need the rendered-output check.")
156
+ parsed = urllib.parse.urlsplit(raw)
157
+ if parsed.scheme or parsed.netloc:
158
+ site = urllib.parse.urlsplit(str(config.get("url") or ""))
159
+ if parsed.scheme not in {"http", "https"} or not site.netloc or parsed.netloc != site.netloc:
160
+ raise ValueError("Remote/data images are not fetched; inspect a local published asset.")
161
+ if parsed.fragment:
162
+ raise ValueError("Fragment-selected images need review of that exact rendered view.")
163
+ path = urllib.parse.unquote(parsed.path)
164
+ if not path or "\\" in path or any(ord(c) < 32 for c in path):
165
+ raise ValueError("Invalid local image reference.")
166
+ baseurl = "/" + str(config.get("baseurl") or "").strip("/")
167
+ if baseurl != "/" and path.startswith(baseurl + "/"):
168
+ path = path[len(baseurl):]
169
+ if output_folder:
170
+ base = output_folder if path.startswith("/") else posixpath.dirname(item["path"])
171
+ result = posixpath.normpath(posixpath.join(base, path.lstrip("/")))
172
+ if not result.startswith(output_folder + "/"):
173
+ raise ValueError("Image URL escapes the rendered output directory.")
174
+ elif path.startswith("/") or path.startswith("assets/") or item.get("metadata") or item.get("direct"):
175
+ result = posixpath.normpath(path.lstrip("/"))
176
+ else:
177
+ result = posixpath.normpath(posixpath.join(posixpath.dirname(item["path"]), path))
178
+ relative_path(result)
179
+ if result.split("/")[0] in {"context", ".cache", ".unaltraweb"}:
180
+ raise ValueError("Private editorial/runtime paths are not image assets.")
181
+ return result
182
+
183
+
184
+ def _exists(reader: Reader, path: str) -> bool:
185
+ try:
186
+ with reader.parent(path) as (parent, name):
187
+ info = os.stat(name, dir_fd=parent, follow_symlinks=False)
188
+ if not stat.S_ISREG(info.st_mode):
189
+ raise ValueError(f"Image input must be a regular, non-symlink file: {path}")
190
+ return True
191
+ except FileNotFoundError:
192
+ return False
193
+
194
+
195
+ def _resolve(reader: Reader, path: str, language: str, config: dict) -> tuple[str, str]:
196
+ original = path
197
+ default = default_language(config)
198
+ suffix = next((suffix for suffix in VISUAL_SUFFIXES if path.lower().endswith(suffix)), "")
199
+ if language != default and suffix:
200
+ stem = path[:-len(suffix)]
201
+ codes = {*config.get("languages", []), language, default}
202
+ if not any(stem.lower().endswith("." + str(code).lower()) for code in codes):
203
+ variant = f"{stem}.{language}{path[-len(suffix):]}"
204
+ if _exists(reader, variant):
205
+ path = variant
206
+ if not _exists(reader, path):
207
+ raise ValueError(f"Missing image/source: {path}")
208
+ lower = path.lower()
209
+ if lower.endswith(DIAGRAM_SUFFIXES):
210
+ candidates = [path + ".edited.svg", path + ".svg"]
211
+ elif lower.endswith(CAPTURE_SUFFIXES):
212
+ base = path.rsplit(".", 1)[0]
213
+ candidates = [base + ".edited.svg", base + ".svg"]
214
+ elif lower.endswith(VEGA_SUFFIXES):
215
+ manifest = yaml_mapping(reader.text(".vegavisuals.yml"))
216
+ entries = manifest.get("visualizations") or []
217
+ matches = [item for item in entries if isinstance(item, dict) and item.get("source") == path]
218
+ if len(matches) != 1 or not isinstance(matches[0].get("output"), str):
219
+ raise ValueError("Vega image needs exactly one declared output.")
220
+ candidates = [relative_path(matches[0]["output"])]
221
+ elif lower.endswith(COMPUTE_SUFFIXES):
222
+ text = reader.text(path)
223
+ if lower.endswith(".ipynb"):
224
+ metadata = strict_json(text).get("metadata", {}).get("unaltraweb_front_matter", {})
225
+ else:
226
+ if lower.endswith((".r", ".py")):
227
+ text = "\n".join(re.sub(r"^#'? ?", "", line) for line in text.splitlines() if line.startswith("#"))
228
+ front = re.search(r"(?:\A|\n)---\s*\n(.*?)\n---", text, re.S)
229
+ metadata = yaml_mapping(front[1]) if front else {}
230
+ compute = metadata.get("unaltraweb_compute") or {}
231
+ outputs = compute.get("outputs") or ([compute["output"]] if compute.get("output") else [])
232
+ if compute.get("mode") != "figure" or not isinstance(outputs, list) or not outputs or not isinstance(outputs[0], str):
233
+ raise ValueError("Computed image needs a declared mode: figure output.")
234
+ output = relative_path(outputs[0])
235
+ candidates = [str(Path(output).with_suffix(".edited.svg")), output]
236
+ else:
237
+ return path, path if path != original else original
238
+ for output in candidates:
239
+ if _exists(reader, output):
240
+ return output, path
241
+ raise ValueError(f"Render the missing image output from {path} before inspecting it.")
242
+
243
+
244
+ def _inspect(data: bytes, timeout: float) -> dict:
245
+ env = {**os.environ, "HOME": "/proc/unaltraweb-image-check", "XDG_CACHE_HOME": "/proc/unaltraweb-image-check"}
246
+ result = run_process([sys.executable, "-I", "-B", str(Path(__file__).with_name("image_probe.py"))],
247
+ input_data=data, env=env, timeout_seconds=max(.1, min(10.0, timeout)), output_limit=8192)
248
+ if result.returncode or result.stdout_truncated or result.stderr_truncated:
249
+ return {"state": "unverifiable", "reason": "Image decoder exceeded its limits or could not complete."}
250
+ value = strict_json(result.stdout)
251
+ if not isinstance(value, dict) or value.get("state") not in {"opaque", "transparent", "unverifiable"}:
252
+ raise ValueError("Invalid image decoder result.")
253
+ return value
254
+
255
+
256
+ def image_background_check(project: Path, source: str = "", output_folder: str = "") -> dict[str, Any]:
257
+ images: dict[str, dict] = {}
258
+ warnings = []
259
+ deadline = time.monotonic() + CHECK_SECONDS
260
+ try:
261
+ if source and output_folder:
262
+ raise ValueError("Choose a source or a rendered output directory, not both.")
263
+ with Reader(project) as reader, Reader(project, max_bytes=MAX_IMAGE_BYTES, max_total_bytes=MAX_IMAGE_TOTAL_BYTES) as binary:
264
+ config = yaml_mapping(reader.text("_config.yml", optional=True))
265
+ references = _collect(reader, config, source, output_folder)
266
+ cache: dict[str, dict] = {}
267
+ for item in references:
268
+ path, owner = "", ""
269
+ try:
270
+ path = _local_path(item, config, output_folder)
271
+ if not output_folder:
272
+ path, owner = _resolve(reader, path, item["language"], config)
273
+ key = path
274
+ if key in images:
275
+ images[key]["reference_count"] += 1
276
+ if len(images[key]["references"]) < 10:
277
+ images[key]["references"].append({"path": item["path"], "line": item["line"]})
278
+ continue
279
+ if len(images) >= MAX_IMAGES or time.monotonic() >= deadline:
280
+ raise ValueError("Image inspection budget reached; inspect a smaller source explicitly.")
281
+ data = binary.read(path)
282
+ sha256 = hashlib.sha256(data).hexdigest()
283
+ if sha256 not in cache:
284
+ cache[sha256] = _inspect(data, deadline - time.monotonic())
285
+ result = {**cache[sha256], "sha256": sha256}
286
+ except (OSError, ValueError, TypeError, AttributeError, RecursionError) as exc:
287
+ key = path or f"unresolved:{item['path']}:{item['line']}:{hashlib.sha256(item['reference'].encode()).hexdigest()}"
288
+ result = {"state": "unverifiable", "reason": str(exc)[:400]}
289
+ if key in images:
290
+ images[key]["reference_count"] += 1
291
+ if len(images[key]["references"]) < 10:
292
+ images[key]["references"].append({"path": item["path"], "line": item["line"]})
293
+ continue
294
+ record = {**result, "path": path, "owner": owner, "references": [{"path": item["path"], "line": item["line"]}], "reference_count": 1}
295
+ images[key] = record
296
+ if result["state"] != "opaque":
297
+ transparent = result["state"] == "transparent"
298
+ action = f"Choose an opaque background colour in {owner or path or item['path']} and export/render again."
299
+ if not transparent:
300
+ action = "Resolve the inspection reason; use supported local, self-contained images and the Pillow/CairoSVG inspection dependencies."
301
+ if path.endswith(".edited.svg"):
302
+ action = f"Review the author-owned {path} before editing its background; preserve the edited override."
303
+ warnings.append({"severity": "warning", "code": "UW-IMAGE-TRANSPARENT" if transparent else "UW-IMAGE-UNVERIFIABLE",
304
+ "path": path or item["path"], "source": item["path"], "line": item["line"],
305
+ "message": "Image has transparent or partially transparent pixels." if transparent else result.get("reason", "Image background could not be verified."),
306
+ "remediation": action})
307
+ return {"ok": True, "offline": True, "read_only": True, "source": source, "output_folder": output_folder,
308
+ "images": list(images.values()), "image_count": len(images), "reference_count": len(references), "warnings": warnings,
309
+ "policy": "Advisory: any opaque colour is valid; no files are changed and white is not imposed.",
310
+ "coverage": "Local referenced raster frames and bounded self-contained SVG samples; remote/data/fragment references and unsupported SVG features are reported as unverifiable."}
311
+ except (OSError, ValueError, TypeError, AttributeError, RecursionError) as exc:
312
+ return {"ok": False, "offline": True, "read_only": True, "images": [], "image_count": 0, "warnings": warnings,
313
+ "error": str(exc), "source": source, "output_folder": output_folder}
314
+
315
+
316
+ def print_warnings(result: dict) -> None:
317
+ for finding in result.get("warnings", []):
318
+ print(f"{finding['code']}: {finding['path']}: {finding['message']} {finding['remediation']}", file=sys.stderr)
319
+
320
+
321
+ def main(argv=None) -> int:
322
+ parser = argparse.ArgumentParser(description=__doc__)
323
+ parser.add_argument("--project", type=Path, required=True)
324
+ parser.add_argument("--source", default="")
325
+ parser.add_argument("--output-folder", default="")
326
+ args = parser.parse_args(argv)
327
+ result = image_background_check(args.project, args.source, args.output_folder)
328
+ print_warnings(result)
329
+ print(json.dumps(result, indent=2, ensure_ascii=False, allow_nan=False))
330
+ return 0 if result["ok"] else 1
331
+
332
+
333
+ if __name__ == "__main__":
334
+ raise SystemExit(main())
@@ -0,0 +1,149 @@
1
+ """Bounded, file-free image decoding worker. Input is bytes on stdin, not a path."""
2
+ from __future__ import annotations
3
+
4
+ import base64
5
+ import gzip
6
+ import io
7
+ import json
8
+ import math
9
+ import re
10
+ import sys
11
+ import warnings
12
+ import xml.etree.ElementTree as ET
13
+
14
+ MAX_BYTES = 32 * 1024 * 1024
15
+ MAX_SVG_BYTES = 8 * 1024 * 1024
16
+ MAX_PIXELS = 16_000_000
17
+ MAX_TOTAL_PIXELS = 64_000_000
18
+ MAX_FRAMES = 32
19
+ MAX_SVG_SIDE = 1536
20
+ RASTER_FORMATS = ["PNG", "JPEG", "GIF", "WEBP", "TIFF", "BMP", "AVIF"]
21
+
22
+
23
+ def bounded_image(image) -> None:
24
+ width, height = image.size
25
+ if width <= 0 or height <= 0 or width * height > MAX_PIXELS:
26
+ raise ValueError("Image exceeds the 16-megapixel decoding budget.")
27
+
28
+
29
+ def svg_resource(url: str, resource_type: str) -> bytes:
30
+ """Only bounded embedded PNG/JPEG data; never open a URL or filesystem path."""
31
+ from PIL import Image
32
+
33
+ match = re.fullmatch(r"data:image/(?:png|jpeg);base64,([A-Za-z0-9+/=\s]+)", url)
34
+ if match is None or len(url) > MAX_SVG_BYTES:
35
+ raise ValueError("External or unsupported embedded SVG resources cannot be inspected.")
36
+ data = base64.b64decode(re.sub(r"\s", "", match[1]), validate=True)
37
+ with Image.open(io.BytesIO(data), formats=["PNG", "JPEG"]) as image:
38
+ bounded_image(image)
39
+ if getattr(image, "is_animated", False):
40
+ raise ValueError("Animated images embedded in SVG require separate review.")
41
+ return data
42
+
43
+
44
+ def svg_dimension(value: str, fallback: float) -> float:
45
+ match = re.fullmatch(r"\s*([0-9]+(?:\.[0-9]+)?)\s*(px|pt|pc|in|cm|mm)?\s*", value)
46
+ if not match:
47
+ return fallback
48
+ scale = {None: 1, "px": 1, "pt": 96 / 72, "pc": 16, "in": 96, "cm": 96 / 2.54, "mm": 96 / 25.4}
49
+ return float(match[1]) * scale[match[2]]
50
+
51
+
52
+ def inspect_svg(data: bytes) -> dict:
53
+ from PIL import Image
54
+ from cairosvg.surface import PNGSurface
55
+
56
+ if len(data) > MAX_SVG_BYTES:
57
+ raise ValueError("SVG exceeds the 8 MiB text budget.")
58
+ text = data.decode("utf-8")
59
+ if re.search(r"<!\s*(?:DOCTYPE|ENTITY)|<\?xml-stylesheet", text, re.I):
60
+ raise ValueError("SVG declarations and external stylesheets are not allowed in inspection.")
61
+ root = ET.fromstring(text)
62
+ if root.tag.rsplit("}", 1)[-1] != "svg":
63
+ raise ValueError("XML input is not an SVG image.")
64
+ for count, node in enumerate(root.iter(), 1):
65
+ if count > 20000:
66
+ raise ValueError("SVG node budget exceeded.")
67
+ if node.tag.rsplit("}", 1)[-1] in {"script", "foreignObject", "animate", "animateMotion", "animateTransform", "set", "filter", "mask"}:
68
+ raise ValueError("Dynamic SVG, masks and filters require rendered human review.")
69
+ viewbox = re.split(r"[\s,]+", root.get("viewBox", "").strip())
70
+ box = [float(value) for value in viewbox] if len(viewbox) == 4 else [0, 0, 300, 150]
71
+ width = svg_dimension(root.get("width", ""), box[2])
72
+ height = svg_dimension(root.get("height", ""), box[3])
73
+ if not all(math.isfinite(value) and value > 0 for value in (width, height)):
74
+ raise ValueError("SVG needs a finite positive viewport.")
75
+ scale = min(1.0, MAX_SVG_SIDE / max(width, height))
76
+ size = [max(1, math.ceil(width * scale)), max(1, math.ceil(height * scale))]
77
+ # No background_color is supplied: painting one here would hide the defect.
78
+ rendered = PNGSurface.convert(bytestring=data, output_width=size[0], output_height=size[1],
79
+ unsafe=False, url_fetcher=svg_resource)
80
+ with Image.open(io.BytesIO(rendered), formats=["PNG"]) as image:
81
+ alpha = image.convert("RGBA").getchannel("A").getextrema()
82
+ return {"state": "transparent" if alpha[0] < 255 else "opaque", "format": "SVG",
83
+ "method": "svg-raster-sample", "sample_size": size, "alpha_min": alpha[0],
84
+ "note": "Opacity is sampled at this bounded viewport, not certified at every possible scale."}
85
+
86
+
87
+ def inspect_raster(data: bytes) -> dict:
88
+ from PIL import Image
89
+
90
+ with Image.open(io.BytesIO(data), formats=RASTER_FORMATS) as image:
91
+ detected = image.format
92
+ total = 0
93
+ for index in range(MAX_FRAMES + 1):
94
+ try:
95
+ image.seek(index)
96
+ except EOFError:
97
+ return {"state": "opaque", "format": detected, "method": "decoded-frames", "frames_checked": index, "alpha_min": 255}
98
+ if index == MAX_FRAMES:
99
+ raise ValueError("Image exceeds the 32-frame inspection budget.")
100
+ bounded_image(image)
101
+ total += image.width * image.height
102
+ if total > MAX_TOTAL_PIXELS:
103
+ raise ValueError("Image exceeds the total frame-pixel budget.")
104
+ alpha = image.convert("RGBA").getchannel("A").getextrema()
105
+ if alpha[0] < 255:
106
+ return {"state": "transparent", "format": detected, "method": "decoded-frames",
107
+ "frames_checked": index + 1, "transparent_frame": index, "alpha_min": alpha[0]}
108
+ raise ValueError("No image frame was decoded.")
109
+
110
+
111
+ def probe(data: bytes) -> dict:
112
+ from PIL import Image
113
+
114
+ if not data or len(data) > MAX_BYTES:
115
+ raise ValueError("Image must be non-empty and at most 32 MiB.")
116
+ with warnings.catch_warnings():
117
+ warnings.simplefilter("error", Image.DecompressionBombWarning)
118
+ if data.startswith(b"\x1f\x8b"):
119
+ with gzip.GzipFile(fileobj=io.BytesIO(data)) as stream:
120
+ data = stream.read(MAX_SVG_BYTES + 1)
121
+ return inspect_svg(data)
122
+ if data.lstrip().startswith((b"<", b"\xef\xbb\xbf")):
123
+ return inspect_svg(data)
124
+ return inspect_raster(data)
125
+
126
+
127
+ def main() -> int:
128
+ sys.dont_write_bytecode = True
129
+ try:
130
+ import resource
131
+
132
+ resource.setrlimit(resource.RLIMIT_AS, (768 * 1024 * 1024, 768 * 1024 * 1024))
133
+ resource.setrlimit(resource.RLIMIT_CPU, (5, 5))
134
+ data = sys.stdin.buffer.read(MAX_BYTES + 1)
135
+ # Load trusted codecs before prohibiting regular-file growth: native
136
+ # library discovery may probe an OS temporary directory during import.
137
+ from PIL import Image
138
+ if data.lstrip().startswith((b"<", b"\xef\xbb\xbf", b"\x1f\x8b")):
139
+ from cairosvg.surface import PNGSurface
140
+ resource.setrlimit(resource.RLIMIT_FSIZE, (0, 0))
141
+ result = probe(data)
142
+ except Exception as exc:
143
+ result = {"state": "unverifiable", "reason": f"{type(exc).__name__}: {str(exc)[:300]}"}
144
+ print(json.dumps(result, ensure_ascii=True, allow_nan=False))
145
+ return 0
146
+
147
+
148
+ if __name__ == "__main__":
149
+ raise SystemExit(main())
@@ -0,0 +1,146 @@
1
+ from __future__ import annotations
2
+
3
+ import os
4
+ import selectors
5
+ import signal
6
+ import subprocess
7
+ import time
8
+ from dataclasses import dataclass
9
+ from pathlib import Path
10
+
11
+
12
+ DEFAULT_OUTPUT_LIMIT = 128 * 1024
13
+
14
+
15
+ @dataclass(frozen=True)
16
+ class ProcessResult:
17
+ args: list[str]
18
+ returncode: int
19
+ stdout: str
20
+ stderr: str
21
+ timed_out: bool
22
+ stdout_truncated: bool
23
+ stderr_truncated: bool
24
+
25
+
26
+ def run_process(
27
+ command: list[str],
28
+ *,
29
+ cwd: Path | str | None = None,
30
+ env: dict[str, str] | None = None,
31
+ timeout_seconds: float,
32
+ output_limit: int = DEFAULT_OUTPUT_LIMIT,
33
+ input_data: bytes | None = None,
34
+ ) -> ProcessResult:
35
+ """Run a bounded child process and terminate its process group on timeout."""
36
+ if timeout_seconds <= 0:
37
+ raise ValueError("Process timeout must be positive.")
38
+ if output_limit <= 0:
39
+ raise ValueError("Process output limit must be positive.")
40
+ if input_data is not None and not isinstance(input_data, bytes):
41
+ raise ValueError("Process input must be bytes.")
42
+
43
+ process = subprocess.Popen(
44
+ command,
45
+ cwd=str(cwd) if cwd is not None else None,
46
+ env=env,
47
+ stdin=subprocess.PIPE if input_data is not None else subprocess.DEVNULL,
48
+ stdout=subprocess.PIPE,
49
+ stderr=subprocess.PIPE,
50
+ start_new_session=True,
51
+ )
52
+ assert process.stdout is not None
53
+ assert process.stderr is not None
54
+
55
+ selector = selectors.DefaultSelector()
56
+ selector.register(process.stdout, selectors.EVENT_READ, "stdout")
57
+ selector.register(process.stderr, selectors.EVENT_READ, "stderr")
58
+ pending_input = memoryview(input_data or b"")
59
+ if process.stdin is not None:
60
+ if pending_input:
61
+ os.set_blocking(process.stdin.fileno(), False)
62
+ selector.register(process.stdin, selectors.EVENT_WRITE, "stdin")
63
+ else:
64
+ process.stdin.close()
65
+ output = {"stdout": bytearray(), "stderr": bytearray()}
66
+ truncated = {"stdout": False, "stderr": False}
67
+ deadline = time.monotonic() + timeout_seconds
68
+ timed_out = False
69
+ terminate_deadline = 0.0
70
+ drain_deadline = float("inf")
71
+ killed = False
72
+
73
+ try:
74
+ while selector.get_map() or process.poll() is None:
75
+ now = time.monotonic()
76
+ if not timed_out and now >= deadline:
77
+ timed_out = True
78
+ terminate_deadline = now + 1.0
79
+ drain_deadline = now + 2.0
80
+ try:
81
+ os.killpg(process.pid, signal.SIGTERM)
82
+ except ProcessLookupError:
83
+ pass
84
+ elif timed_out and not killed and now >= terminate_deadline:
85
+ try:
86
+ os.killpg(process.pid, signal.SIGKILL)
87
+ except ProcessLookupError:
88
+ pass
89
+ if process.poll() is None:
90
+ process.kill()
91
+ killed = True
92
+ drain_deadline = now + 1.0
93
+ elif timed_out and now >= drain_deadline and process.poll() is not None:
94
+ for key in list(selector.get_map().values()):
95
+ selector.unregister(key.fileobj)
96
+ break
97
+
98
+ events = selector.select(0.05)
99
+ for key, _ in events:
100
+ stream = str(key.data)
101
+ if stream == "stdin":
102
+ try:
103
+ written = os.write(key.fd, pending_input[:65536])
104
+ pending_input = pending_input[written:]
105
+ except BlockingIOError:
106
+ continue
107
+ except BrokenPipeError:
108
+ pending_input = memoryview(b"")
109
+ if not pending_input:
110
+ selector.unregister(key.fileobj)
111
+ key.fileobj.close()
112
+ continue
113
+ try:
114
+ chunk = os.read(key.fd, 65536)
115
+ except BlockingIOError:
116
+ continue
117
+ if not chunk:
118
+ selector.unregister(key.fileobj)
119
+ continue
120
+ remaining = output_limit - len(output[stream])
121
+ if remaining > 0:
122
+ output[stream].extend(chunk[:remaining])
123
+ if len(chunk) > remaining:
124
+ truncated[stream] = True
125
+
126
+ try:
127
+ process.wait(timeout=1.0 if timed_out else None)
128
+ except subprocess.TimeoutExpired:
129
+ process.kill()
130
+ process.wait()
131
+ finally:
132
+ selector.close()
133
+ process.stdout.close()
134
+ process.stderr.close()
135
+ if process.stdin is not None and not process.stdin.closed:
136
+ process.stdin.close()
137
+
138
+ return ProcessResult(
139
+ args=list(command),
140
+ returncode=124 if timed_out else process.returncode,
141
+ stdout=output["stdout"].decode("utf-8", errors="replace"),
142
+ stderr=output["stderr"].decode("utf-8", errors="replace"),
143
+ timed_out=timed_out,
144
+ stdout_truncated=truncated["stdout"],
145
+ stderr_truncated=truncated["stderr"],
146
+ )