unaltraweb 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/Makefile +5 -5
- data/README.md +19 -7
- data/_plugins/figure_captions.rb +47 -10
- data/_sass/_documentation.scss +7 -5
- data/_sass/_manual.scss +7 -0
- data/docs/_documentation/en/02-tools.md +4 -4
- data/docs/_documentation/en/03-usage.md +61 -0
- data/docs/_documentation/en/13-unaltremanual.md +1 -1
- data/docs/_documentation/en/20-syntax.md +14 -0
- data/docs/_documentation/en/25-caption-credits.md +120 -0
- data/docs/_documentation/en/26-image-backgrounds.md +103 -0
- data/docs/_documentation/en/31-template.md +1 -1
- data/docs/_documentation/en/32-development.md +1 -1
- data/docs/_documentation/en/40-distribution.md +25 -5
- data/docs/_documentation/en/42-docker-image.md +7 -7
- data/docs/_documentation/en/43-workspace-path-policies.md +232 -0
- data/docs/_documentation/en/44-editorial-review.md +237 -0
- data/docs/agents/action-prompts/00-start-site-session.txt +10 -5
- data/docs/agents/action-prompts/22-manual-style-audit.txt +3 -1
- data/docs/agents/manual-authoring-components.md +38 -0
- data/docs/agents/mcp-contract.md +102 -10
- data/docs/agents/visual-companions-0.4.0.md +74 -0
- data/docs/assets/img/caption-credits-demo.svg +19 -0
- data/scripts/editorial_check.py +12 -0
- data/scripts/image_background_check.py +12 -0
- data/scripts/manual/build_pdf.py +94 -16
- data/scripts/manual/filters/figure-captions.lua +65 -10
- data/scripts/manual/templates/manual.tex +2 -1
- data/scripts/test_gem_build.py +32 -2
- data/scripts/test_reproducible_jekyll_build.py +1 -1
- data/scripts/test_wheel_install.py +28 -0
- data/scripts/unaltraweb-mcp-bootstrap.sh +1 -1
- data/scripts/validate_distribution.py +18 -3
- data/src/unaltraweb_mcp/component-contract.json +28 -28
- data/src/unaltraweb_mcp/editorial.py +495 -0
- data/src/unaltraweb_mcp/editorial_sources.py +504 -0
- data/src/unaltraweb_mcp/image_backgrounds.py +334 -0
- data/src/unaltraweb_mcp/image_probe.py +149 -0
- data/src/unaltraweb_mcp/processes.py +146 -0
- metadata +15 -2
|
@@ -0,0 +1,334 @@
|
|
|
1
|
+
"""Advisory, offline background checks for referenced publication images."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import os
|
|
8
|
+
import posixpath
|
|
9
|
+
import re
|
|
10
|
+
import stat
|
|
11
|
+
import sys
|
|
12
|
+
import time
|
|
13
|
+
import urllib.parse
|
|
14
|
+
from html.parser import HTMLParser
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from .editorial_sources import Reader, corpus, default_language, relative_path, strict_json, yaml_mapping, yaml_value
|
|
19
|
+
from .processes import run_process
|
|
20
|
+
|
|
21
|
+
IMAGE_SUFFIXES = (".svg", ".svgz", ".png", ".jpg", ".jpeg", ".webp", ".gif", ".bmp", ".tif", ".tiff", ".avif", ".ico")
|
|
22
|
+
DIAGRAM_SUFFIXES = (".mmd", ".mermaid", ".puml", ".plantuml", ".uml")
|
|
23
|
+
CAPTURE_SUFFIXES = (".capture.yml", ".capture.yaml")
|
|
24
|
+
VEGA_SUFFIXES = (".vl.json", ".vg.json")
|
|
25
|
+
COMPUTE_SUFFIXES = (".qmd", ".rmd", ".r", ".py", ".ipynb")
|
|
26
|
+
VISUAL_SUFFIXES = tuple(sorted({*IMAGE_SUFFIXES, *DIAGRAM_SUFFIXES, *CAPTURE_SUFFIXES, *VEGA_SUFFIXES,
|
|
27
|
+
*COMPUTE_SUFFIXES, ".edited.svg", *(suffix + ".svg" for suffix in DIAGRAM_SUFFIXES)}, key=len, reverse=True))
|
|
28
|
+
MAX_IMAGE_BYTES = 32 * 1024 * 1024
|
|
29
|
+
MAX_IMAGE_TOTAL_BYTES = 128 * 1024 * 1024
|
|
30
|
+
MAX_IMAGES = 128
|
|
31
|
+
MAX_REFERENCES = 2000
|
|
32
|
+
CHECK_SECONDS = 30.0
|
|
33
|
+
IMAGE_FIELDS = {"image", "img", "cover", "cover_image", "photo", "avatar", "logo", "logo_inverse", "logo_cafe", "thumbnail", "preview"}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class ImageReferences(HTMLParser):
|
|
37
|
+
def __init__(self):
|
|
38
|
+
super().__init__(convert_charrefs=True)
|
|
39
|
+
self.references: list[tuple[int, str]] = []
|
|
40
|
+
|
|
41
|
+
def handle_starttag(self, tag, attrs):
|
|
42
|
+
values = dict(attrs)
|
|
43
|
+
names = {"img": ("src",), "source": (), "video": ("poster",), "object": ("data",), "embed": ("src",)}.get(tag, ())
|
|
44
|
+
if tag == "link" and "icon" in (values.get("rel") or ""):
|
|
45
|
+
names = ("href",)
|
|
46
|
+
for name in names:
|
|
47
|
+
if values.get(name):
|
|
48
|
+
self.references.append((self.getpos()[0], values[name]))
|
|
49
|
+
if tag in {"img", "source"} and values.get("srcset"):
|
|
50
|
+
srcset = values["srcset"]
|
|
51
|
+
if srcset.lstrip().startswith("data:"):
|
|
52
|
+
self.references.append((self.getpos()[0], "data:"))
|
|
53
|
+
else:
|
|
54
|
+
self.references.extend((self.getpos()[0], item.strip().split()[0]) for item in srcset.split(",") if item.strip())
|
|
55
|
+
if len(self.references) > MAX_REFERENCES:
|
|
56
|
+
raise ValueError("Image reference budget exceeded.")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _metadata_images(value: Any, *, parent_image: bool = False, depth: int = 0):
|
|
60
|
+
if depth > 20:
|
|
61
|
+
raise ValueError("Image metadata nesting limit exceeded.")
|
|
62
|
+
if isinstance(value, dict):
|
|
63
|
+
for key, child in value.items():
|
|
64
|
+
image_field = isinstance(key, str) and (key in IMAGE_FIELDS or (parent_image and key in {"path", "src", "file"}))
|
|
65
|
+
if image_field and isinstance(child, str) and child.strip():
|
|
66
|
+
yield child
|
|
67
|
+
elif isinstance(child, (dict, list)):
|
|
68
|
+
yield from _metadata_images(child, parent_image=image_field, depth=depth + 1)
|
|
69
|
+
elif isinstance(value, list):
|
|
70
|
+
for child in value:
|
|
71
|
+
if parent_image and isinstance(child, str) and child.strip():
|
|
72
|
+
yield child
|
|
73
|
+
else:
|
|
74
|
+
yield from _metadata_images(child, parent_image=parent_image, depth=depth + 1)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _source_references(text: str) -> list[tuple[int, str]]:
|
|
78
|
+
# Resolve only literal supported URL sugar, never evaluate Liquid or code.
|
|
79
|
+
text = re.sub(r"\{\{\s*site\.baseurl\s*\}\}", "", text)
|
|
80
|
+
text = re.sub(r'''\{\{\s*(['"])(.*?)\1\s*\|\s*(?:relative_url|absolute_url)\s*\}\}''', lambda m: m[2], text)
|
|
81
|
+
from .editorial_sources import blank
|
|
82
|
+
|
|
83
|
+
text = re.sub(r"<!--.*?(?:-->|\Z)|{%\s*comment\s*%}.*?(?:{%\s*endcomment\s*%}|\Z)", lambda m: blank(m[0]), text, flags=re.S)
|
|
84
|
+
visible, fence = [], ""
|
|
85
|
+
for line in text.splitlines(keepends=True):
|
|
86
|
+
match = re.match(r"^\s*(`{3,}|~{3,})", line)
|
|
87
|
+
if match:
|
|
88
|
+
marker = match[1]
|
|
89
|
+
if not fence:
|
|
90
|
+
fence = marker
|
|
91
|
+
elif marker[0] == fence[0] and len(marker) >= len(fence):
|
|
92
|
+
fence = ""
|
|
93
|
+
visible.append(blank(line))
|
|
94
|
+
elif fence or line.startswith((" ", "\t")):
|
|
95
|
+
visible.append(blank(line))
|
|
96
|
+
else:
|
|
97
|
+
visible.append(re.sub(r"(`+)[^`\n]*?\1", lambda m: blank(m[0]), line))
|
|
98
|
+
text = "".join(visible)
|
|
99
|
+
references = []
|
|
100
|
+
# Escaped and ordinary alt-text characters must be disjoint so malformed
|
|
101
|
+
# labels with many backslashes cannot trigger exponential backtracking.
|
|
102
|
+
for match in re.finditer(r"!\[(?:\\.|[^\\\]\n])*\]\(\s*(?:<([^>\n]+)>|([^\s)]+))", text):
|
|
103
|
+
references.append((text[:match.start()].count("\n") + 1, match[1] or match[2]))
|
|
104
|
+
parser = ImageReferences()
|
|
105
|
+
parser.feed(text)
|
|
106
|
+
references.extend(parser.references)
|
|
107
|
+
return references
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _collect(reader: Reader, config: dict, source: str, output_folder: str) -> list[dict]:
|
|
111
|
+
language = default_language(config)
|
|
112
|
+
if source and source.lower().endswith(VISUAL_SUFFIXES):
|
|
113
|
+
relative_path(source)
|
|
114
|
+
return [{"path": source, "line": 0, "reference": source, "language": language, "direct": True}]
|
|
115
|
+
references = []
|
|
116
|
+
if output_folder:
|
|
117
|
+
paths = reader.walk(output_folder)
|
|
118
|
+
if not paths or paths == [output_folder]:
|
|
119
|
+
raise ValueError("Choose a non-empty rendered output directory.")
|
|
120
|
+
for path in paths:
|
|
121
|
+
if path.lower().endswith(".html"):
|
|
122
|
+
parser = ImageReferences()
|
|
123
|
+
parser.feed(reader.text(path))
|
|
124
|
+
references.extend({"path": path, "line": line, "reference": url, "language": language} for line, url in parser.references)
|
|
125
|
+
else:
|
|
126
|
+
uw = config.get("unaltraweb") or {}
|
|
127
|
+
if not isinstance(uw, dict):
|
|
128
|
+
raise ValueError("unaltraweb configuration must be a mapping.")
|
|
129
|
+
selected = corpus(reader, config, str(uw.get("site_profile") or ""), source)
|
|
130
|
+
for path, record in selected["sources"].items():
|
|
131
|
+
text = record["text"]
|
|
132
|
+
if path == "_config.yml" or path.startswith("_data/"):
|
|
133
|
+
data = strict_json(text) if path.endswith(".json") else yaml_value(text)
|
|
134
|
+
references.extend({"path": path, "line": 0, "reference": url, "language": language, "metadata": True}
|
|
135
|
+
for url in _metadata_images(data))
|
|
136
|
+
else:
|
|
137
|
+
front = re.match(r"\A---\s*\n(.*?)\n---[^\S\n]*(?:\n|\Z)", text, re.S)
|
|
138
|
+
front_data = yaml_mapping(front[1]) if front else {}
|
|
139
|
+
document_language = str(front_data.get("lang") or next((part for part in Path(path).parts if part in config.get("languages", [])), language))
|
|
140
|
+
references.extend({"path": path, "line": 0, "reference": url, "language": document_language, "metadata": True}
|
|
141
|
+
for url in _metadata_images(front_data))
|
|
142
|
+
body = ("\n" * text[:front.end()].count("\n") + text[front.end():]) if front else text
|
|
143
|
+
references.extend({"path": path, "line": line, "reference": url, "language": document_language}
|
|
144
|
+
for line, url in _source_references(body))
|
|
145
|
+
if len(references) > MAX_REFERENCES:
|
|
146
|
+
raise ValueError("Image reference budget exceeded; select a smaller source.")
|
|
147
|
+
if len(references) > MAX_REFERENCES:
|
|
148
|
+
raise ValueError("Image reference budget exceeded; select a smaller source.")
|
|
149
|
+
return references
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _local_path(item: dict, config: dict, output_folder: str) -> str:
|
|
153
|
+
raw = item["reference"].strip()
|
|
154
|
+
if any(token in raw for token in ("{{", "{%")):
|
|
155
|
+
raise ValueError("Dynamic image references need the rendered-output check.")
|
|
156
|
+
parsed = urllib.parse.urlsplit(raw)
|
|
157
|
+
if parsed.scheme or parsed.netloc:
|
|
158
|
+
site = urllib.parse.urlsplit(str(config.get("url") or ""))
|
|
159
|
+
if parsed.scheme not in {"http", "https"} or not site.netloc or parsed.netloc != site.netloc:
|
|
160
|
+
raise ValueError("Remote/data images are not fetched; inspect a local published asset.")
|
|
161
|
+
if parsed.fragment:
|
|
162
|
+
raise ValueError("Fragment-selected images need review of that exact rendered view.")
|
|
163
|
+
path = urllib.parse.unquote(parsed.path)
|
|
164
|
+
if not path or "\\" in path or any(ord(c) < 32 for c in path):
|
|
165
|
+
raise ValueError("Invalid local image reference.")
|
|
166
|
+
baseurl = "/" + str(config.get("baseurl") or "").strip("/")
|
|
167
|
+
if baseurl != "/" and path.startswith(baseurl + "/"):
|
|
168
|
+
path = path[len(baseurl):]
|
|
169
|
+
if output_folder:
|
|
170
|
+
base = output_folder if path.startswith("/") else posixpath.dirname(item["path"])
|
|
171
|
+
result = posixpath.normpath(posixpath.join(base, path.lstrip("/")))
|
|
172
|
+
if not result.startswith(output_folder + "/"):
|
|
173
|
+
raise ValueError("Image URL escapes the rendered output directory.")
|
|
174
|
+
elif path.startswith("/") or path.startswith("assets/") or item.get("metadata") or item.get("direct"):
|
|
175
|
+
result = posixpath.normpath(path.lstrip("/"))
|
|
176
|
+
else:
|
|
177
|
+
result = posixpath.normpath(posixpath.join(posixpath.dirname(item["path"]), path))
|
|
178
|
+
relative_path(result)
|
|
179
|
+
if result.split("/")[0] in {"context", ".cache", ".unaltraweb"}:
|
|
180
|
+
raise ValueError("Private editorial/runtime paths are not image assets.")
|
|
181
|
+
return result
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _exists(reader: Reader, path: str) -> bool:
|
|
185
|
+
try:
|
|
186
|
+
with reader.parent(path) as (parent, name):
|
|
187
|
+
info = os.stat(name, dir_fd=parent, follow_symlinks=False)
|
|
188
|
+
if not stat.S_ISREG(info.st_mode):
|
|
189
|
+
raise ValueError(f"Image input must be a regular, non-symlink file: {path}")
|
|
190
|
+
return True
|
|
191
|
+
except FileNotFoundError:
|
|
192
|
+
return False
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _resolve(reader: Reader, path: str, language: str, config: dict) -> tuple[str, str]:
|
|
196
|
+
original = path
|
|
197
|
+
default = default_language(config)
|
|
198
|
+
suffix = next((suffix for suffix in VISUAL_SUFFIXES if path.lower().endswith(suffix)), "")
|
|
199
|
+
if language != default and suffix:
|
|
200
|
+
stem = path[:-len(suffix)]
|
|
201
|
+
codes = {*config.get("languages", []), language, default}
|
|
202
|
+
if not any(stem.lower().endswith("." + str(code).lower()) for code in codes):
|
|
203
|
+
variant = f"{stem}.{language}{path[-len(suffix):]}"
|
|
204
|
+
if _exists(reader, variant):
|
|
205
|
+
path = variant
|
|
206
|
+
if not _exists(reader, path):
|
|
207
|
+
raise ValueError(f"Missing image/source: {path}")
|
|
208
|
+
lower = path.lower()
|
|
209
|
+
if lower.endswith(DIAGRAM_SUFFIXES):
|
|
210
|
+
candidates = [path + ".edited.svg", path + ".svg"]
|
|
211
|
+
elif lower.endswith(CAPTURE_SUFFIXES):
|
|
212
|
+
base = path.rsplit(".", 1)[0]
|
|
213
|
+
candidates = [base + ".edited.svg", base + ".svg"]
|
|
214
|
+
elif lower.endswith(VEGA_SUFFIXES):
|
|
215
|
+
manifest = yaml_mapping(reader.text(".vegavisuals.yml"))
|
|
216
|
+
entries = manifest.get("visualizations") or []
|
|
217
|
+
matches = [item for item in entries if isinstance(item, dict) and item.get("source") == path]
|
|
218
|
+
if len(matches) != 1 or not isinstance(matches[0].get("output"), str):
|
|
219
|
+
raise ValueError("Vega image needs exactly one declared output.")
|
|
220
|
+
candidates = [relative_path(matches[0]["output"])]
|
|
221
|
+
elif lower.endswith(COMPUTE_SUFFIXES):
|
|
222
|
+
text = reader.text(path)
|
|
223
|
+
if lower.endswith(".ipynb"):
|
|
224
|
+
metadata = strict_json(text).get("metadata", {}).get("unaltraweb_front_matter", {})
|
|
225
|
+
else:
|
|
226
|
+
if lower.endswith((".r", ".py")):
|
|
227
|
+
text = "\n".join(re.sub(r"^#'? ?", "", line) for line in text.splitlines() if line.startswith("#"))
|
|
228
|
+
front = re.search(r"(?:\A|\n)---\s*\n(.*?)\n---", text, re.S)
|
|
229
|
+
metadata = yaml_mapping(front[1]) if front else {}
|
|
230
|
+
compute = metadata.get("unaltraweb_compute") or {}
|
|
231
|
+
outputs = compute.get("outputs") or ([compute["output"]] if compute.get("output") else [])
|
|
232
|
+
if compute.get("mode") != "figure" or not isinstance(outputs, list) or not outputs or not isinstance(outputs[0], str):
|
|
233
|
+
raise ValueError("Computed image needs a declared mode: figure output.")
|
|
234
|
+
output = relative_path(outputs[0])
|
|
235
|
+
candidates = [str(Path(output).with_suffix(".edited.svg")), output]
|
|
236
|
+
else:
|
|
237
|
+
return path, path if path != original else original
|
|
238
|
+
for output in candidates:
|
|
239
|
+
if _exists(reader, output):
|
|
240
|
+
return output, path
|
|
241
|
+
raise ValueError(f"Render the missing image output from {path} before inspecting it.")
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _inspect(data: bytes, timeout: float) -> dict:
|
|
245
|
+
env = {**os.environ, "HOME": "/proc/unaltraweb-image-check", "XDG_CACHE_HOME": "/proc/unaltraweb-image-check"}
|
|
246
|
+
result = run_process([sys.executable, "-I", "-B", str(Path(__file__).with_name("image_probe.py"))],
|
|
247
|
+
input_data=data, env=env, timeout_seconds=max(.1, min(10.0, timeout)), output_limit=8192)
|
|
248
|
+
if result.returncode or result.stdout_truncated or result.stderr_truncated:
|
|
249
|
+
return {"state": "unverifiable", "reason": "Image decoder exceeded its limits or could not complete."}
|
|
250
|
+
value = strict_json(result.stdout)
|
|
251
|
+
if not isinstance(value, dict) or value.get("state") not in {"opaque", "transparent", "unverifiable"}:
|
|
252
|
+
raise ValueError("Invalid image decoder result.")
|
|
253
|
+
return value
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def image_background_check(project: Path, source: str = "", output_folder: str = "") -> dict[str, Any]:
|
|
257
|
+
images: dict[str, dict] = {}
|
|
258
|
+
warnings = []
|
|
259
|
+
deadline = time.monotonic() + CHECK_SECONDS
|
|
260
|
+
try:
|
|
261
|
+
if source and output_folder:
|
|
262
|
+
raise ValueError("Choose a source or a rendered output directory, not both.")
|
|
263
|
+
with Reader(project) as reader, Reader(project, max_bytes=MAX_IMAGE_BYTES, max_total_bytes=MAX_IMAGE_TOTAL_BYTES) as binary:
|
|
264
|
+
config = yaml_mapping(reader.text("_config.yml", optional=True))
|
|
265
|
+
references = _collect(reader, config, source, output_folder)
|
|
266
|
+
cache: dict[str, dict] = {}
|
|
267
|
+
for item in references:
|
|
268
|
+
path, owner = "", ""
|
|
269
|
+
try:
|
|
270
|
+
path = _local_path(item, config, output_folder)
|
|
271
|
+
if not output_folder:
|
|
272
|
+
path, owner = _resolve(reader, path, item["language"], config)
|
|
273
|
+
key = path
|
|
274
|
+
if key in images:
|
|
275
|
+
images[key]["reference_count"] += 1
|
|
276
|
+
if len(images[key]["references"]) < 10:
|
|
277
|
+
images[key]["references"].append({"path": item["path"], "line": item["line"]})
|
|
278
|
+
continue
|
|
279
|
+
if len(images) >= MAX_IMAGES or time.monotonic() >= deadline:
|
|
280
|
+
raise ValueError("Image inspection budget reached; inspect a smaller source explicitly.")
|
|
281
|
+
data = binary.read(path)
|
|
282
|
+
sha256 = hashlib.sha256(data).hexdigest()
|
|
283
|
+
if sha256 not in cache:
|
|
284
|
+
cache[sha256] = _inspect(data, deadline - time.monotonic())
|
|
285
|
+
result = {**cache[sha256], "sha256": sha256}
|
|
286
|
+
except (OSError, ValueError, TypeError, AttributeError, RecursionError) as exc:
|
|
287
|
+
key = path or f"unresolved:{item['path']}:{item['line']}:{hashlib.sha256(item['reference'].encode()).hexdigest()}"
|
|
288
|
+
result = {"state": "unverifiable", "reason": str(exc)[:400]}
|
|
289
|
+
if key in images:
|
|
290
|
+
images[key]["reference_count"] += 1
|
|
291
|
+
if len(images[key]["references"]) < 10:
|
|
292
|
+
images[key]["references"].append({"path": item["path"], "line": item["line"]})
|
|
293
|
+
continue
|
|
294
|
+
record = {**result, "path": path, "owner": owner, "references": [{"path": item["path"], "line": item["line"]}], "reference_count": 1}
|
|
295
|
+
images[key] = record
|
|
296
|
+
if result["state"] != "opaque":
|
|
297
|
+
transparent = result["state"] == "transparent"
|
|
298
|
+
action = f"Choose an opaque background colour in {owner or path or item['path']} and export/render again."
|
|
299
|
+
if not transparent:
|
|
300
|
+
action = "Resolve the inspection reason; use supported local, self-contained images and the Pillow/CairoSVG inspection dependencies."
|
|
301
|
+
if path.endswith(".edited.svg"):
|
|
302
|
+
action = f"Review the author-owned {path} before editing its background; preserve the edited override."
|
|
303
|
+
warnings.append({"severity": "warning", "code": "UW-IMAGE-TRANSPARENT" if transparent else "UW-IMAGE-UNVERIFIABLE",
|
|
304
|
+
"path": path or item["path"], "source": item["path"], "line": item["line"],
|
|
305
|
+
"message": "Image has transparent or partially transparent pixels." if transparent else result.get("reason", "Image background could not be verified."),
|
|
306
|
+
"remediation": action})
|
|
307
|
+
return {"ok": True, "offline": True, "read_only": True, "source": source, "output_folder": output_folder,
|
|
308
|
+
"images": list(images.values()), "image_count": len(images), "reference_count": len(references), "warnings": warnings,
|
|
309
|
+
"policy": "Advisory: any opaque colour is valid; no files are changed and white is not imposed.",
|
|
310
|
+
"coverage": "Local referenced raster frames and bounded self-contained SVG samples; remote/data/fragment references and unsupported SVG features are reported as unverifiable."}
|
|
311
|
+
except (OSError, ValueError, TypeError, AttributeError, RecursionError) as exc:
|
|
312
|
+
return {"ok": False, "offline": True, "read_only": True, "images": [], "image_count": 0, "warnings": warnings,
|
|
313
|
+
"error": str(exc), "source": source, "output_folder": output_folder}
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def print_warnings(result: dict) -> None:
|
|
317
|
+
for finding in result.get("warnings", []):
|
|
318
|
+
print(f"{finding['code']}: {finding['path']}: {finding['message']} {finding['remediation']}", file=sys.stderr)
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def main(argv=None) -> int:
|
|
322
|
+
parser = argparse.ArgumentParser(description=__doc__)
|
|
323
|
+
parser.add_argument("--project", type=Path, required=True)
|
|
324
|
+
parser.add_argument("--source", default="")
|
|
325
|
+
parser.add_argument("--output-folder", default="")
|
|
326
|
+
args = parser.parse_args(argv)
|
|
327
|
+
result = image_background_check(args.project, args.source, args.output_folder)
|
|
328
|
+
print_warnings(result)
|
|
329
|
+
print(json.dumps(result, indent=2, ensure_ascii=False, allow_nan=False))
|
|
330
|
+
return 0 if result["ok"] else 1
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
if __name__ == "__main__":
|
|
334
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""Bounded, file-free image decoding worker. Input is bytes on stdin, not a path."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import base64
|
|
5
|
+
import gzip
|
|
6
|
+
import io
|
|
7
|
+
import json
|
|
8
|
+
import math
|
|
9
|
+
import re
|
|
10
|
+
import sys
|
|
11
|
+
import warnings
|
|
12
|
+
import xml.etree.ElementTree as ET
|
|
13
|
+
|
|
14
|
+
MAX_BYTES = 32 * 1024 * 1024
|
|
15
|
+
MAX_SVG_BYTES = 8 * 1024 * 1024
|
|
16
|
+
MAX_PIXELS = 16_000_000
|
|
17
|
+
MAX_TOTAL_PIXELS = 64_000_000
|
|
18
|
+
MAX_FRAMES = 32
|
|
19
|
+
MAX_SVG_SIDE = 1536
|
|
20
|
+
RASTER_FORMATS = ["PNG", "JPEG", "GIF", "WEBP", "TIFF", "BMP", "AVIF"]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def bounded_image(image) -> None:
|
|
24
|
+
width, height = image.size
|
|
25
|
+
if width <= 0 or height <= 0 or width * height > MAX_PIXELS:
|
|
26
|
+
raise ValueError("Image exceeds the 16-megapixel decoding budget.")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def svg_resource(url: str, resource_type: str) -> bytes:
|
|
30
|
+
"""Only bounded embedded PNG/JPEG data; never open a URL or filesystem path."""
|
|
31
|
+
from PIL import Image
|
|
32
|
+
|
|
33
|
+
match = re.fullmatch(r"data:image/(?:png|jpeg);base64,([A-Za-z0-9+/=\s]+)", url)
|
|
34
|
+
if match is None or len(url) > MAX_SVG_BYTES:
|
|
35
|
+
raise ValueError("External or unsupported embedded SVG resources cannot be inspected.")
|
|
36
|
+
data = base64.b64decode(re.sub(r"\s", "", match[1]), validate=True)
|
|
37
|
+
with Image.open(io.BytesIO(data), formats=["PNG", "JPEG"]) as image:
|
|
38
|
+
bounded_image(image)
|
|
39
|
+
if getattr(image, "is_animated", False):
|
|
40
|
+
raise ValueError("Animated images embedded in SVG require separate review.")
|
|
41
|
+
return data
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def svg_dimension(value: str, fallback: float) -> float:
|
|
45
|
+
match = re.fullmatch(r"\s*([0-9]+(?:\.[0-9]+)?)\s*(px|pt|pc|in|cm|mm)?\s*", value)
|
|
46
|
+
if not match:
|
|
47
|
+
return fallback
|
|
48
|
+
scale = {None: 1, "px": 1, "pt": 96 / 72, "pc": 16, "in": 96, "cm": 96 / 2.54, "mm": 96 / 25.4}
|
|
49
|
+
return float(match[1]) * scale[match[2]]
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def inspect_svg(data: bytes) -> dict:
|
|
53
|
+
from PIL import Image
|
|
54
|
+
from cairosvg.surface import PNGSurface
|
|
55
|
+
|
|
56
|
+
if len(data) > MAX_SVG_BYTES:
|
|
57
|
+
raise ValueError("SVG exceeds the 8 MiB text budget.")
|
|
58
|
+
text = data.decode("utf-8")
|
|
59
|
+
if re.search(r"<!\s*(?:DOCTYPE|ENTITY)|<\?xml-stylesheet", text, re.I):
|
|
60
|
+
raise ValueError("SVG declarations and external stylesheets are not allowed in inspection.")
|
|
61
|
+
root = ET.fromstring(text)
|
|
62
|
+
if root.tag.rsplit("}", 1)[-1] != "svg":
|
|
63
|
+
raise ValueError("XML input is not an SVG image.")
|
|
64
|
+
for count, node in enumerate(root.iter(), 1):
|
|
65
|
+
if count > 20000:
|
|
66
|
+
raise ValueError("SVG node budget exceeded.")
|
|
67
|
+
if node.tag.rsplit("}", 1)[-1] in {"script", "foreignObject", "animate", "animateMotion", "animateTransform", "set", "filter", "mask"}:
|
|
68
|
+
raise ValueError("Dynamic SVG, masks and filters require rendered human review.")
|
|
69
|
+
viewbox = re.split(r"[\s,]+", root.get("viewBox", "").strip())
|
|
70
|
+
box = [float(value) for value in viewbox] if len(viewbox) == 4 else [0, 0, 300, 150]
|
|
71
|
+
width = svg_dimension(root.get("width", ""), box[2])
|
|
72
|
+
height = svg_dimension(root.get("height", ""), box[3])
|
|
73
|
+
if not all(math.isfinite(value) and value > 0 for value in (width, height)):
|
|
74
|
+
raise ValueError("SVG needs a finite positive viewport.")
|
|
75
|
+
scale = min(1.0, MAX_SVG_SIDE / max(width, height))
|
|
76
|
+
size = [max(1, math.ceil(width * scale)), max(1, math.ceil(height * scale))]
|
|
77
|
+
# No background_color is supplied: painting one here would hide the defect.
|
|
78
|
+
rendered = PNGSurface.convert(bytestring=data, output_width=size[0], output_height=size[1],
|
|
79
|
+
unsafe=False, url_fetcher=svg_resource)
|
|
80
|
+
with Image.open(io.BytesIO(rendered), formats=["PNG"]) as image:
|
|
81
|
+
alpha = image.convert("RGBA").getchannel("A").getextrema()
|
|
82
|
+
return {"state": "transparent" if alpha[0] < 255 else "opaque", "format": "SVG",
|
|
83
|
+
"method": "svg-raster-sample", "sample_size": size, "alpha_min": alpha[0],
|
|
84
|
+
"note": "Opacity is sampled at this bounded viewport, not certified at every possible scale."}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def inspect_raster(data: bytes) -> dict:
|
|
88
|
+
from PIL import Image
|
|
89
|
+
|
|
90
|
+
with Image.open(io.BytesIO(data), formats=RASTER_FORMATS) as image:
|
|
91
|
+
detected = image.format
|
|
92
|
+
total = 0
|
|
93
|
+
for index in range(MAX_FRAMES + 1):
|
|
94
|
+
try:
|
|
95
|
+
image.seek(index)
|
|
96
|
+
except EOFError:
|
|
97
|
+
return {"state": "opaque", "format": detected, "method": "decoded-frames", "frames_checked": index, "alpha_min": 255}
|
|
98
|
+
if index == MAX_FRAMES:
|
|
99
|
+
raise ValueError("Image exceeds the 32-frame inspection budget.")
|
|
100
|
+
bounded_image(image)
|
|
101
|
+
total += image.width * image.height
|
|
102
|
+
if total > MAX_TOTAL_PIXELS:
|
|
103
|
+
raise ValueError("Image exceeds the total frame-pixel budget.")
|
|
104
|
+
alpha = image.convert("RGBA").getchannel("A").getextrema()
|
|
105
|
+
if alpha[0] < 255:
|
|
106
|
+
return {"state": "transparent", "format": detected, "method": "decoded-frames",
|
|
107
|
+
"frames_checked": index + 1, "transparent_frame": index, "alpha_min": alpha[0]}
|
|
108
|
+
raise ValueError("No image frame was decoded.")
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def probe(data: bytes) -> dict:
|
|
112
|
+
from PIL import Image
|
|
113
|
+
|
|
114
|
+
if not data or len(data) > MAX_BYTES:
|
|
115
|
+
raise ValueError("Image must be non-empty and at most 32 MiB.")
|
|
116
|
+
with warnings.catch_warnings():
|
|
117
|
+
warnings.simplefilter("error", Image.DecompressionBombWarning)
|
|
118
|
+
if data.startswith(b"\x1f\x8b"):
|
|
119
|
+
with gzip.GzipFile(fileobj=io.BytesIO(data)) as stream:
|
|
120
|
+
data = stream.read(MAX_SVG_BYTES + 1)
|
|
121
|
+
return inspect_svg(data)
|
|
122
|
+
if data.lstrip().startswith((b"<", b"\xef\xbb\xbf")):
|
|
123
|
+
return inspect_svg(data)
|
|
124
|
+
return inspect_raster(data)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def main() -> int:
|
|
128
|
+
sys.dont_write_bytecode = True
|
|
129
|
+
try:
|
|
130
|
+
import resource
|
|
131
|
+
|
|
132
|
+
resource.setrlimit(resource.RLIMIT_AS, (768 * 1024 * 1024, 768 * 1024 * 1024))
|
|
133
|
+
resource.setrlimit(resource.RLIMIT_CPU, (5, 5))
|
|
134
|
+
data = sys.stdin.buffer.read(MAX_BYTES + 1)
|
|
135
|
+
# Load trusted codecs before prohibiting regular-file growth: native
|
|
136
|
+
# library discovery may probe an OS temporary directory during import.
|
|
137
|
+
from PIL import Image
|
|
138
|
+
if data.lstrip().startswith((b"<", b"\xef\xbb\xbf", b"\x1f\x8b")):
|
|
139
|
+
from cairosvg.surface import PNGSurface
|
|
140
|
+
resource.setrlimit(resource.RLIMIT_FSIZE, (0, 0))
|
|
141
|
+
result = probe(data)
|
|
142
|
+
except Exception as exc:
|
|
143
|
+
result = {"state": "unverifiable", "reason": f"{type(exc).__name__}: {str(exc)[:300]}"}
|
|
144
|
+
print(json.dumps(result, ensure_ascii=True, allow_nan=False))
|
|
145
|
+
return 0
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
if __name__ == "__main__":
|
|
149
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import selectors
|
|
5
|
+
import signal
|
|
6
|
+
import subprocess
|
|
7
|
+
import time
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
DEFAULT_OUTPUT_LIMIT = 128 * 1024
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class ProcessResult:
|
|
17
|
+
args: list[str]
|
|
18
|
+
returncode: int
|
|
19
|
+
stdout: str
|
|
20
|
+
stderr: str
|
|
21
|
+
timed_out: bool
|
|
22
|
+
stdout_truncated: bool
|
|
23
|
+
stderr_truncated: bool
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def run_process(
|
|
27
|
+
command: list[str],
|
|
28
|
+
*,
|
|
29
|
+
cwd: Path | str | None = None,
|
|
30
|
+
env: dict[str, str] | None = None,
|
|
31
|
+
timeout_seconds: float,
|
|
32
|
+
output_limit: int = DEFAULT_OUTPUT_LIMIT,
|
|
33
|
+
input_data: bytes | None = None,
|
|
34
|
+
) -> ProcessResult:
|
|
35
|
+
"""Run a bounded child process and terminate its process group on timeout."""
|
|
36
|
+
if timeout_seconds <= 0:
|
|
37
|
+
raise ValueError("Process timeout must be positive.")
|
|
38
|
+
if output_limit <= 0:
|
|
39
|
+
raise ValueError("Process output limit must be positive.")
|
|
40
|
+
if input_data is not None and not isinstance(input_data, bytes):
|
|
41
|
+
raise ValueError("Process input must be bytes.")
|
|
42
|
+
|
|
43
|
+
process = subprocess.Popen(
|
|
44
|
+
command,
|
|
45
|
+
cwd=str(cwd) if cwd is not None else None,
|
|
46
|
+
env=env,
|
|
47
|
+
stdin=subprocess.PIPE if input_data is not None else subprocess.DEVNULL,
|
|
48
|
+
stdout=subprocess.PIPE,
|
|
49
|
+
stderr=subprocess.PIPE,
|
|
50
|
+
start_new_session=True,
|
|
51
|
+
)
|
|
52
|
+
assert process.stdout is not None
|
|
53
|
+
assert process.stderr is not None
|
|
54
|
+
|
|
55
|
+
selector = selectors.DefaultSelector()
|
|
56
|
+
selector.register(process.stdout, selectors.EVENT_READ, "stdout")
|
|
57
|
+
selector.register(process.stderr, selectors.EVENT_READ, "stderr")
|
|
58
|
+
pending_input = memoryview(input_data or b"")
|
|
59
|
+
if process.stdin is not None:
|
|
60
|
+
if pending_input:
|
|
61
|
+
os.set_blocking(process.stdin.fileno(), False)
|
|
62
|
+
selector.register(process.stdin, selectors.EVENT_WRITE, "stdin")
|
|
63
|
+
else:
|
|
64
|
+
process.stdin.close()
|
|
65
|
+
output = {"stdout": bytearray(), "stderr": bytearray()}
|
|
66
|
+
truncated = {"stdout": False, "stderr": False}
|
|
67
|
+
deadline = time.monotonic() + timeout_seconds
|
|
68
|
+
timed_out = False
|
|
69
|
+
terminate_deadline = 0.0
|
|
70
|
+
drain_deadline = float("inf")
|
|
71
|
+
killed = False
|
|
72
|
+
|
|
73
|
+
try:
|
|
74
|
+
while selector.get_map() or process.poll() is None:
|
|
75
|
+
now = time.monotonic()
|
|
76
|
+
if not timed_out and now >= deadline:
|
|
77
|
+
timed_out = True
|
|
78
|
+
terminate_deadline = now + 1.0
|
|
79
|
+
drain_deadline = now + 2.0
|
|
80
|
+
try:
|
|
81
|
+
os.killpg(process.pid, signal.SIGTERM)
|
|
82
|
+
except ProcessLookupError:
|
|
83
|
+
pass
|
|
84
|
+
elif timed_out and not killed and now >= terminate_deadline:
|
|
85
|
+
try:
|
|
86
|
+
os.killpg(process.pid, signal.SIGKILL)
|
|
87
|
+
except ProcessLookupError:
|
|
88
|
+
pass
|
|
89
|
+
if process.poll() is None:
|
|
90
|
+
process.kill()
|
|
91
|
+
killed = True
|
|
92
|
+
drain_deadline = now + 1.0
|
|
93
|
+
elif timed_out and now >= drain_deadline and process.poll() is not None:
|
|
94
|
+
for key in list(selector.get_map().values()):
|
|
95
|
+
selector.unregister(key.fileobj)
|
|
96
|
+
break
|
|
97
|
+
|
|
98
|
+
events = selector.select(0.05)
|
|
99
|
+
for key, _ in events:
|
|
100
|
+
stream = str(key.data)
|
|
101
|
+
if stream == "stdin":
|
|
102
|
+
try:
|
|
103
|
+
written = os.write(key.fd, pending_input[:65536])
|
|
104
|
+
pending_input = pending_input[written:]
|
|
105
|
+
except BlockingIOError:
|
|
106
|
+
continue
|
|
107
|
+
except BrokenPipeError:
|
|
108
|
+
pending_input = memoryview(b"")
|
|
109
|
+
if not pending_input:
|
|
110
|
+
selector.unregister(key.fileobj)
|
|
111
|
+
key.fileobj.close()
|
|
112
|
+
continue
|
|
113
|
+
try:
|
|
114
|
+
chunk = os.read(key.fd, 65536)
|
|
115
|
+
except BlockingIOError:
|
|
116
|
+
continue
|
|
117
|
+
if not chunk:
|
|
118
|
+
selector.unregister(key.fileobj)
|
|
119
|
+
continue
|
|
120
|
+
remaining = output_limit - len(output[stream])
|
|
121
|
+
if remaining > 0:
|
|
122
|
+
output[stream].extend(chunk[:remaining])
|
|
123
|
+
if len(chunk) > remaining:
|
|
124
|
+
truncated[stream] = True
|
|
125
|
+
|
|
126
|
+
try:
|
|
127
|
+
process.wait(timeout=1.0 if timed_out else None)
|
|
128
|
+
except subprocess.TimeoutExpired:
|
|
129
|
+
process.kill()
|
|
130
|
+
process.wait()
|
|
131
|
+
finally:
|
|
132
|
+
selector.close()
|
|
133
|
+
process.stdout.close()
|
|
134
|
+
process.stderr.close()
|
|
135
|
+
if process.stdin is not None and not process.stdin.closed:
|
|
136
|
+
process.stdin.close()
|
|
137
|
+
|
|
138
|
+
return ProcessResult(
|
|
139
|
+
args=list(command),
|
|
140
|
+
returncode=124 if timed_out else process.returncode,
|
|
141
|
+
stdout=output["stdout"].decode("utf-8", errors="replace"),
|
|
142
|
+
stderr=output["stderr"].decode("utf-8", errors="replace"),
|
|
143
|
+
timed_out=timed_out,
|
|
144
|
+
stdout_truncated=truncated["stdout"],
|
|
145
|
+
stderr_truncated=truncated["stderr"],
|
|
146
|
+
)
|