vulnly 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
vulnly/__init__.py ADDED
@@ -0,0 +1,1385 @@
1
+ __version__ = "1.0.0"
2
+
3
+ import base64
4
+ import json
5
+ import html
6
+ import mimetypes
7
+ import os
8
+ import re
9
+ import struct
10
+ import datetime
11
+ import argparse
12
+ import shutil
13
+ import sys
14
+ import urllib.request
15
+ import urllib.error
16
+ from pathlib import Path
17
+ from typing import Any, overload
18
+
19
+ SEVERITY_ORDER = {"critical": 0, "high": 1, "medium": 2, "low": 3, "unknown": 4}
20
+
21
+ TEMPLATE_DIR = Path(__file__).parent
22
+ # One theme-aware template per report type. The theme is applied as a
23
+ # data-theme attribute on the root element rather than by picking a file.
24
+ REPORT_TEMPLATE = TEMPLATE_DIR / "report_template.html"
25
+ REPO_SUMMARY_TEMPLATE = TEMPLATE_DIR / "repo_summary_template.html"
26
+ THEMES = ("dark", "light")
27
+ # Pure-path SVG: no embedded text, so it needs no font and stays crisp at
28
+ # any size. The wordmark beside it is HTML, styled per theme.
29
+ DEFAULT_LOGO = TEMPLATE_DIR / "vulnly-icon.svg"
30
+ # Chart.js is bundled rather than loaded from a CDN so reports render on hosts
31
+ # with no outbound network access.
32
+ CHART_JS = TEMPLATE_DIR / "chart.umd.min.js"
33
+ # Inter, embedded for the same reason. Variable font covering weights 400-800.
34
+ FONT_WOFF2 = TEMPLATE_DIR / "inter-var.woff2"
35
+ # Browser-tab icon, embedded so a report saved to disk still shows it.
36
+ FAVICON = TEMPLATE_DIR / "vulnly-favicon.png"
37
+
38
+
39
+ def _read_chart_js() -> str:
40
+ """Return the bundled Chart.js source, or '' if it is missing."""
41
+ try:
42
+ return CHART_JS.read_text(encoding="utf-8")
43
+ except OSError:
44
+ return ""
45
+
46
+
47
+ def _read_favicon_b64() -> str:
48
+ """Return the bundled favicon as base64, or '' if it is missing."""
49
+ try:
50
+ return base64.b64encode(FAVICON.read_bytes()).decode("ascii")
51
+ except OSError:
52
+ return ""
53
+
54
+
55
+ def _read_font_b64() -> str:
56
+ """Return the bundled font as base64, or '' if it is missing.
57
+
58
+ An empty value leaves the @font-face src unresolvable, and the template's
59
+ fallback stack takes over — degraded typography rather than a broken page.
60
+ """
61
+ try:
62
+ return base64.b64encode(FONT_WOFF2.read_bytes()).decode("ascii")
63
+ except OSError:
64
+ return ""
65
+
66
+
67
+ def _apply_replacements(template: str, replacements: dict) -> str:
68
+ """Substitute placeholders, skipping the template's documentation comment.
69
+
70
+ That comment lists every placeholder by name, so a plain replace across
71
+ the whole file also substituted there — embedding a copy of the logo data
72
+ URI, and now the Chart.js bundle, into a comment no reader ever sees.
73
+ """
74
+ head, sep, body = template.partition("-->")
75
+ if not sep:
76
+ head, sep, body = "", "", template
77
+ for placeholder, value in replacements.items():
78
+ body = body.replace(placeholder, value)
79
+ return head + sep + body
80
+
81
+
82
+ SUPPORTED_SOURCES = {"cloudsmith", "trivy", "grype", "snyk"}
83
+
84
+
85
+ def is_cloudsmith_cli_format(data: dict) -> bool:
86
+ """Return True if the JSON looks like raw Cloudsmith CLI output."""
87
+ return isinstance(data.get("data"), dict) and "scans" in data["data"]
88
+
89
+
90
+ def is_cloudsmith_repo_summary_format(data: dict) -> bool:
91
+ """Return True if the JSON looks like a Cloudsmith repo-level summary."""
92
+ d = data.get("data")
93
+ return isinstance(d, dict) and "packages" in d and "repository" in d and "scans" not in d
94
+
95
+
96
+ def is_trivy_format(data: dict) -> bool:
97
+ """Return True if the JSON looks like Trivy JSON output."""
98
+ return "Results" in data and isinstance(data.get("Results"), list)
99
+
100
+
101
+ def is_grype_format(data: dict) -> bool:
102
+ """Return True if the JSON looks like Grype JSON output."""
103
+ return "matches" in data and isinstance(data.get("matches"), list) and "descriptor" in data
104
+
105
+
106
+ def is_snyk_format(data: dict) -> bool:
107
+ """Return True if the JSON looks like Snyk CLI container test output."""
108
+ return (
109
+ "vulnerabilities" in data
110
+ and isinstance(data.get("vulnerabilities"), list)
111
+ and "packageManager" in data
112
+ and "projectName" in data
113
+ )
114
+
115
+
116
+
117
+ def normalize_snyk(raw: dict) -> dict:
118
+ """Transform Snyk CLI JSON into the internal report format."""
119
+ project_name = raw.get("projectName", "Unknown")
120
+ # projectName is typically "docker-image|nginx" — extract the image name
121
+ if "|" in project_name:
122
+ artifact_name = project_name.split("|", 1)[1]
123
+ else:
124
+ artifact_name = project_name
125
+
126
+ docker_info = raw.get("docker", {})
127
+ os_info = docker_info.get("os", {})
128
+ scan_target = os_info.get("prettyName", raw.get("platform", "Unknown"))
129
+ package_format = raw.get("packageManager", "Unknown")
130
+ path_info = raw.get("path", "")
131
+
132
+ # Deduplicate vulnerabilities by (id, name, version)
133
+ seen = set()
134
+ vulnerabilities = []
135
+ for v in raw.get("vulnerabilities", []):
136
+ vuln_id = v.get("id", "")
137
+ pkg_name = v.get("name", v.get("packageName", "N/A"))
138
+ pkg_version = v.get("version", "N/A")
139
+ dedup_key = (vuln_id, pkg_name, pkg_version)
140
+ if dedup_key in seen:
141
+ continue
142
+ seen.add(dedup_key)
143
+
144
+ # Prefer CVE identifier if available, fall back to Snyk ID
145
+ identifiers = v.get("identifiers", {})
146
+ cves = identifiers.get("CVE", [])
147
+ identifier = cves[0] if cves else vuln_id
148
+
149
+ # CVSS score
150
+ cvss = v.get("cvssScore")
151
+
152
+ # Fixed version
153
+ fixed_in = v.get("fixedIn", [])
154
+ fixed_version = ", ".join(fixed_in) if fixed_in else None
155
+
156
+ vuln = {
157
+ "severity": (v.get("severity") or "unknown").lower(),
158
+ "identifier": identifier,
159
+ "package": pkg_name,
160
+ "affected_version": pkg_version,
161
+ "fixed_version": fixed_version,
162
+ "title": v.get("title", ""),
163
+ "cvss": cvss,
164
+ }
165
+ vulnerabilities.append(vuln)
166
+
167
+ return {
168
+ "scan_date": "",
169
+ "repository": artifact_name,
170
+ "package_name": artifact_name,
171
+ "package_version": path_info.split(":")[-1] if ":" in path_info else "latest",
172
+ "package_format": package_format,
173
+ "scan_target": scan_target,
174
+ "scan_id": raw.get("projectId", "N/A"),
175
+ "vulnerabilities": vulnerabilities,
176
+ }
177
+
178
+
179
+ def normalize_grype(raw: dict) -> dict:
180
+ """Transform Grype JSON into the internal report format."""
181
+ source = raw.get("source", {})
182
+ target = source.get("target", {})
183
+ distro = raw.get("distro", {})
184
+ descriptor = raw.get("descriptor", {})
185
+
186
+ artifact_name = target.get("userInput", "Unknown")
187
+ image_id = target.get("imageID", "")
188
+ tags = target.get("tags", [])
189
+ scan_target = tags[0] if tags else artifact_name
190
+
191
+ distro_name = distro.get("name", "")
192
+ distro_version = distro.get("version", "")
193
+ package_format = f"{distro_name} {distro_version}".strip() if distro_name else source.get("type", "Unknown")
194
+
195
+ vulnerabilities = []
196
+ for match in raw.get("matches", []):
197
+ v = match.get("vulnerability", {})
198
+ artifact = match.get("artifact", {})
199
+
200
+ # Extract best CVSS v3 score
201
+ cvss = None
202
+ for entry in v.get("cvss", []):
203
+ if isinstance(entry, dict) and entry.get("version") == "3.1":
204
+ metrics = entry.get("metrics", {})
205
+ score = metrics.get("baseScore")
206
+ if score is not None:
207
+ if cvss is None or score > cvss:
208
+ cvss = score
209
+
210
+ # Extract fixed version
211
+ fix = v.get("fix", {})
212
+ fix_versions = fix.get("versions", [])
213
+ fixed_version = fix_versions[0] if fix_versions else None
214
+
215
+ vuln = {
216
+ "severity": (v.get("severity") or "unknown").lower(),
217
+ "identifier": v.get("id", "N/A"),
218
+ "package": artifact.get("name", "N/A"),
219
+ "affected_version": artifact.get("version", "N/A"),
220
+ "fixed_version": fixed_version,
221
+ "title": v.get("description", "")[:120] or None,
222
+ "cvss": cvss,
223
+ }
224
+ vulnerabilities.append(vuln)
225
+
226
+ return {
227
+ "scan_date": "",
228
+ "repository": artifact_name,
229
+ "package_name": artifact_name,
230
+ "package_version": image_id.replace("sha256:", "")[:12] if image_id else "latest",
231
+ "package_format": package_format,
232
+ "scan_target": scan_target,
233
+ "scan_id": "N/A",
234
+ "vulnerabilities": vulnerabilities,
235
+ }
236
+
237
+
238
+ def normalize_trivy(raw: dict) -> dict:
239
+ """Transform Trivy JSON into the internal report format."""
240
+ artifact_name = raw.get("ArtifactName", "Unknown")
241
+ artifact_type = raw.get("ArtifactType", "Unknown")
242
+ created_at = raw.get("CreatedAt", "")
243
+ scan_date = created_at[:10] if created_at else ""
244
+ report_id = raw.get("ReportID", "N/A")
245
+
246
+ metadata = raw.get("Metadata", {})
247
+ os_info = metadata.get("OS", {})
248
+ os_family = os_info.get("Family", "")
249
+ os_name = os_info.get("Name", "")
250
+ package_format = f"{os_family} {os_name}".strip() if os_family else artifact_type
251
+
252
+ # Collect scan targets from Results
253
+ targets = []
254
+ for result in raw.get("Results", []):
255
+ t = result.get("Target")
256
+ if t:
257
+ targets.append(t)
258
+ scan_target = ", ".join(dict.fromkeys(targets)) if targets else artifact_name
259
+
260
+ # Flatten all vulnerabilities across all Results
261
+ vulnerabilities = []
262
+ for result in raw.get("Results", []):
263
+ for v in result.get("Vulnerabilities", []):
264
+ # Extract best CVSS v3 score
265
+ cvss = None
266
+ cvss_data = v.get("CVSS", {})
267
+ if isinstance(cvss_data, dict):
268
+ for vendor_scores in cvss_data.values():
269
+ if isinstance(vendor_scores, dict):
270
+ score = vendor_scores.get("V3Score")
271
+ if score is not None:
272
+ if cvss is None or score > cvss:
273
+ cvss = score
274
+
275
+ fixed_version = v.get("FixedVersion") or None
276
+
277
+ vuln = {
278
+ "severity": (v.get("Severity") or "unknown").lower(),
279
+ "identifier": v.get("VulnerabilityID", "N/A"),
280
+ "package": v.get("PkgName", "N/A"),
281
+ "affected_version": v.get("InstalledVersion", "N/A"),
282
+ "fixed_version": fixed_version,
283
+ "title": v.get("Title") or (v.get("Description", "")[:120] or None),
284
+ "cvss": cvss,
285
+ }
286
+ vulnerabilities.append(vuln)
287
+
288
+ return {
289
+ "scan_date": scan_date,
290
+ "repository": artifact_name,
291
+ "package_name": artifact_name,
292
+ "package_version": metadata.get("ImageID", "latest").replace("sha256:", "")[:12] if metadata.get("ImageID") else "latest",
293
+ "package_format": package_format,
294
+ "scan_target": scan_target,
295
+ "scan_id": report_id,
296
+ "vulnerabilities": vulnerabilities,
297
+ }
298
+
299
+
300
+ def normalize_cloudsmith_cli(raw: dict) -> dict:
301
+ """Transform Cloudsmith CLI JSON into the internal report format."""
302
+ data = raw["data"]
303
+ pkg = data.get("package", {})
304
+
305
+ # Extract namespace and repository from the API URL
306
+ # e.g. https://api.cloudsmith.io/v1/packages/colinmoynes-test-org/java/…
307
+ url = pkg.get("url", "")
308
+ namespace, repository = "unknown", "unknown"
309
+ parts = url.rstrip("/").split("/")
310
+ # Find "packages" in the URL and grab the next two segments
311
+ if "packages" in parts:
312
+ idx = parts.index("packages")
313
+ if idx + 2 < len(parts):
314
+ namespace = parts[idx + 1]
315
+ repository = parts[idx + 2]
316
+
317
+ scan_date = data.get("created_at", "")[:10] # "2025-06-09T…" → "2025-06-09"
318
+
319
+ # Determine scan target and format from scans metadata
320
+ scans = data.get("scans", [])
321
+ scan_targets = []
322
+ scan_types = set()
323
+ for scan in scans:
324
+ if scan.get("target"):
325
+ scan_targets.append(scan["target"])
326
+ if scan.get("type"):
327
+ scan_types.add(scan["type"])
328
+ scan_target = ", ".join(dict.fromkeys(scan_targets)) if scan_targets else "Unknown"
329
+ package_format = ", ".join(sorted(scan_types)) if scan_types else "Unknown"
330
+
331
+ # Flatten all scan results into a single vulnerabilities list
332
+ vulnerabilities = []
333
+ for scan in scans:
334
+ for result in scan.get("results", []):
335
+ # Extract version strings from nested objects
336
+ affected_ver = result.get("affected_version")
337
+ if isinstance(affected_ver, dict):
338
+ affected_version = affected_ver.get("version") or affected_ver.get("raw_version", "N/A")
339
+ else:
340
+ affected_version = str(affected_ver) if affected_ver else "N/A"
341
+
342
+ fixed_ver = result.get("fixed_version")
343
+ if isinstance(fixed_ver, dict):
344
+ fixed_version = fixed_ver.get("version") or fixed_ver.get("raw_version")
345
+ else:
346
+ fixed_version = str(fixed_ver) if fixed_ver else None
347
+
348
+ if not fixed_version:
349
+ fixed_version = None
350
+
351
+ # Extract CVSS score from cvss_scores
352
+ cvss = None
353
+ cvss_scores = result.get("cvss_scores")
354
+ if isinstance(cvss_scores, list) and cvss_scores:
355
+ # Use the first available score
356
+ first = cvss_scores[0]
357
+ if isinstance(first, dict):
358
+ cvss = first.get("score") or first.get("base_score")
359
+ elif isinstance(first, (int, float)):
360
+ cvss = first
361
+ elif isinstance(cvss_scores, (int, float)):
362
+ cvss = cvss_scores
363
+
364
+ vuln = {
365
+ "severity": result.get("severity", "unknown"),
366
+ "identifier": result.get("vulnerability_id", "N/A"),
367
+ "package": result.get("package_name", "N/A"),
368
+ "affected_version": affected_version,
369
+ "fixed_version": fixed_version,
370
+ "title": result.get("title") or result.get("description", "")[:120] or None,
371
+ "cvss": cvss,
372
+ }
373
+ vulnerabilities.append(vuln)
374
+
375
+ return {
376
+ "scan_date": scan_date,
377
+ "repository": f"{namespace}/{repository}",
378
+ "package_name": pkg.get("name", "Unknown"),
379
+ "package_version": pkg.get("version", "Unknown"),
380
+ "package_format": package_format,
381
+ "scan_target": scan_target,
382
+ "scan_id": data.get("identifier", "N/A"),
383
+ "vulnerabilities": vulnerabilities,
384
+ }
385
+
386
+
387
+ def normalize_cloudsmith_repo_summary(raw: dict) -> dict:
388
+ """Transform Cloudsmith repo-level summary JSON into a repo summary format."""
389
+ data = raw["data"]
390
+ owner = data.get("owner", "unknown")
391
+ repository = data.get("repository", "unknown")
392
+ packages = data.get("packages", [])
393
+
394
+ # Aggregate totals
395
+ total_critical = 0
396
+ total_high = 0
397
+ total_medium = 0
398
+ total_low = 0
399
+ total_unknown = 0
400
+ status_counts = {"vulnerable": 0, "no_issues_found": 0, "no_scan": 0}
401
+
402
+ normalized_packages = []
403
+ for pkg in packages:
404
+ vulns = pkg.get("vulnerabilities", {})
405
+ if not isinstance(vulns, dict):
406
+ vulns = {}
407
+ critical = _as_int(vulns.get("critical"))
408
+ high = _as_int(vulns.get("high"))
409
+ medium = _as_int(vulns.get("medium"))
410
+ low = _as_int(vulns.get("low"))
411
+ unknown = _as_int(vulns.get("unknown"))
412
+
413
+ total_critical += critical
414
+ total_high += high
415
+ total_medium += medium
416
+ total_low += low
417
+ total_unknown += unknown
418
+
419
+ status = pkg.get("status", "unknown")
420
+ if status in status_counts:
421
+ status_counts[status] += 1
422
+
423
+ pkg_total = critical + high + medium + low + unknown
424
+
425
+ normalized_packages.append({
426
+ "package": pkg.get("package", "Unknown"),
427
+ "slug_perm": pkg.get("slug_perm", ""),
428
+ "status": status,
429
+ "critical": critical,
430
+ "high": high,
431
+ "medium": medium,
432
+ "low": low,
433
+ "unknown": unknown,
434
+ "total": pkg_total,
435
+ })
436
+
437
+ return {
438
+ "report_type": "repo_summary",
439
+ "owner": owner,
440
+ "repository": repository,
441
+ "total_packages": len(packages),
442
+ "status_counts": status_counts,
443
+ "total_critical": total_critical,
444
+ "total_high": total_high,
445
+ "total_medium": total_medium,
446
+ "total_low": total_low,
447
+ "total_unknown": total_unknown,
448
+ "total_vulnerabilities": total_critical + total_high + total_medium + total_low + total_unknown,
449
+ "packages": normalized_packages,
450
+ }
451
+
452
+
453
+ def load_vulnerability_data(source: str) -> dict:
454
+ """Load vulnerability data from a JSON file or stdin (when source is '-')."""
455
+ if source == "-":
456
+ return json.load(sys.stdin)
457
+ with open(source, "r") as f:
458
+ return json.load(f)
459
+
460
+
461
+ def validate_data(data: dict) -> list[str]:
462
+ """Validate the structure of the input JSON. Returns a list of warnings."""
463
+ warnings = []
464
+ if not isinstance(data, dict):
465
+ raise SystemExit("Error: Input JSON must be a top-level object.")
466
+ if "vulnerabilities" not in data:
467
+ raise SystemExit(
468
+ "Error: Input JSON is missing the 'vulnerabilities' key. "
469
+ "If using Cloudsmith CLI output, ensure the JSON contains a 'data' object with 'scans'."
470
+ )
471
+ if not isinstance(data["vulnerabilities"], list):
472
+ raise SystemExit("Error: 'vulnerabilities' must be an array.")
473
+
474
+ required_fields = {"severity", "identifier", "package"}
475
+ for i, vuln in enumerate(data["vulnerabilities"]):
476
+ if not isinstance(vuln, dict):
477
+ warnings.append(f" Warning: vulnerabilities[{i}] is not an object, skipping.")
478
+ continue
479
+ missing = required_fields - vuln.keys()
480
+ if missing:
481
+ warnings.append(
482
+ f" Warning: vulnerabilities[{i}] is missing field(s): {', '.join(sorted(missing))}"
483
+ )
484
+ return warnings
485
+
486
+
487
+ def count_severities(vulnerabilities: list) -> dict:
488
+ """Count vulnerabilities by severity level."""
489
+ counts = {"critical": 0, "high": 0, "medium": 0, "low": 0, "unknown": 0}
490
+ for vuln in vulnerabilities:
491
+ severity = vuln.get("severity", "unknown").lower()
492
+ if severity in counts:
493
+ counts[severity] += 1
494
+ else:
495
+ counts["unknown"] += 1
496
+ return counts
497
+
498
+
499
+ def _js_string(value: object) -> str:
500
+ """Encode a value as a JS string literal that is safe inside <script>.
501
+
502
+ json.dumps escapes for JavaScript but not for HTML. Scanner-supplied text
503
+ (CVE titles come from NVD and vendor advisory feeds) containing a literal
504
+ "</script>" would terminate the script element early, letting anything
505
+ after it parse as markup. Escaping the HTML-significant characters as
506
+ \\uXXXX leaves the decoded string identical while making breakout
507
+ impossible.
508
+
509
+ json.dumps defaults to ensure_ascii=True, so U+2028/U+2029 — legal in JSON
510
+ but historically line terminators in JS — are already escaped.
511
+ """
512
+ return (
513
+ json.dumps("" if value is None else str(value))
514
+ .replace("<", "\\u003c")
515
+ .replace(">", "\\u003e")
516
+ .replace("&", "\\u0026")
517
+ )
518
+
519
+
520
+ @overload
521
+ def _as_float(value: Any, default: float) -> float: ...
522
+ @overload
523
+ def _as_float(value: Any, default: None = None) -> float | None: ...
524
+
525
+
526
+ def _as_float(value: Any, default: float | None = None) -> float | None:
527
+ """Coerce untyped scanner data to a float, or `default` if it will not.
528
+
529
+ Scores and counts arrive from third-party JSON with no type guarantee, so
530
+ every numeric read goes through here rather than trusting the input. The
531
+ overloads let callers that supply a default treat the result as a float.
532
+ """
533
+ try:
534
+ return float(value)
535
+ except (TypeError, ValueError):
536
+ return default
537
+
538
+
539
+ def _js_number(value: Any) -> str:
540
+ """Encode a value as a JS number literal, or null if it is not numeric.
541
+
542
+ Interpolating an unchecked value straight into the array would emit
543
+ arbitrary source rather than a number.
544
+ """
545
+ number = _as_float(value)
546
+ return "null" if number is None else repr(number)
547
+
548
+
549
+ def _as_int(value: Any, default: int = 0) -> int:
550
+ """Coerce untyped scanner data to an int, falling back to `default`."""
551
+ number = _as_float(value)
552
+ return default if number is None else int(number)
553
+
554
+
555
+ def _js_int(value: object, default: int = 0) -> str:
556
+ """Encode a value as a JS integer literal, falling back to `default`."""
557
+ return str(_as_int(value, default))
558
+
559
+
560
+ # Advisory identifiers permitted in a link target. Both patterns allow only
561
+ # [A-Za-z0-9-], so a matching identifier cannot introduce path segments.
562
+ _CVE_RE = re.compile(r"^CVE-\d{4}-\d{4,}$", re.IGNORECASE)
563
+ _GHSA_RE = re.compile(r"^GHSA-[a-z0-9]{4}-[a-z0-9]{4}-[a-z0-9]{4}$", re.IGNORECASE)
564
+
565
+
566
+ def _advisory_href(identifier: str) -> str:
567
+ """Return the advisory URL for a well-formed identifier, or '#'.
568
+
569
+ Identifiers are scanner-supplied. Interpolating one unvalidated let ".."
570
+ segments collapse the path — "GHSA-../../../evil" resolved to
571
+ https://github.com/evil — so a crafted identifier could aim a row's link
572
+ at an attacker-controlled page on a domain the reader trusts.
573
+
574
+ The scheme was never at risk: both prefixes are hardcoded https origins,
575
+ and anything not matching a known prefix already returned "#".
576
+ """
577
+ ident = str(identifier).strip()
578
+ if _CVE_RE.match(ident):
579
+ return f"https://nvd.nist.gov/vuln/detail/{ident}"
580
+ if _GHSA_RE.match(ident):
581
+ return f"https://github.com/advisories/{ident}"
582
+ return "#"
583
+
584
+
585
+ # Scanners signal "no fix" with a missing key, an empty string, or "N/A".
586
+ NO_FIX = (None, "", "N/A")
587
+
588
+
589
+ def has_fix(vuln: dict) -> bool:
590
+ """Return True if the finding carries a fixed version.
591
+
592
+ Shared by the fixable count and the table cell, so the stat card cannot
593
+ disagree with the rows it summarises.
594
+ """
595
+ return vuln.get("fixed_version") not in NO_FIX
596
+
597
+
598
+ def count_fixable(vulnerabilities: list) -> int:
599
+ return sum(1 for v in vulnerabilities if has_fix(v))
600
+
601
+
602
+ def _csv_filename(package_name: str, scan_date: str) -> str:
603
+ """Name for the CSV a reader downloads from the report.
604
+
605
+ Package names carry registry paths, tags and digests, so they are reduced
606
+ to filename-safe characters — the value also lands in a download
607
+ attribute, where a path separator would be a directory traversal.
608
+ """
609
+ stem = re.sub(r"[^\w.-]+", "-", f"{package_name}-{scan_date}").strip("-.")
610
+ return f"{(stem or 'vulnly-report')[:80]}-vulnerabilities.csv"
611
+
612
+
613
+ def build_vuln_data_js(vulnerabilities: list) -> str:
614
+ """Build the VULN_DATA JavaScript array for the template."""
615
+ def _order(vuln: dict) -> tuple:
616
+ """Worst first: severity, then CVSS descending, unscored last.
617
+
618
+ Matches the report's default sort, so the emitted array is already in
619
+ the order the table shows.
620
+ """
621
+ rank = SEVERITY_ORDER.get(str(vuln.get("severity", "unknown")).lower(), 4)
622
+ # Unscored findings sort last within their severity.
623
+ score = _as_float(vuln.get("cvss"), -1.0)
624
+ return (rank, -score)
625
+
626
+ sorted_vulns = sorted(vulnerabilities, key=_order)
627
+
628
+ entries = []
629
+ for vuln in sorted_vulns:
630
+ identifier = vuln.get("identifier", "N/A")
631
+ package = vuln.get("package", "N/A")
632
+ severity = vuln.get("severity", "unknown").capitalize()
633
+ affected_version = vuln.get("affected_version", "N/A")
634
+ fixed_version = vuln.get("fixed_version", None)
635
+ title = vuln.get("title", f"{severity} vulnerability in {package}")
636
+ cvss = vuln.get("cvss", None)
637
+
638
+ fixed_js = _js_string(fixed_version) if has_fix(vuln) else "null"
639
+
640
+ cvss_js = _js_number(cvss)
641
+
642
+ href = _advisory_href(identifier)
643
+
644
+ entry = (
645
+ f" {{\n"
646
+ f" id: {_js_string(identifier)},\n"
647
+ f" pkg: {_js_string(package)},\n"
648
+ f" title: {_js_string(title)},\n"
649
+ f" sev: {_js_string(severity)},\n"
650
+ f" cvss: {cvss_js},\n"
651
+ f" affected: {_js_string(affected_version)},\n"
652
+ f" fixed: {fixed_js},\n"
653
+ f" href: {_js_string(href)}\n"
654
+ f" }}"
655
+ )
656
+ entries.append(entry)
657
+
658
+ return "[\n" + ",\n".join(entries) + "\n ]"
659
+
660
+
661
+ def build_pkg_data_js(packages: list) -> str:
662
+ """Build the PKG_DATA JavaScript array for the repo summary template."""
663
+ # Sort: vulnerable first, then no_issues_found, then no_scan
664
+ status_order = {"vulnerable": 0, "no_issues_found": 1, "no_scan": 2}
665
+
666
+ def sort_key(pkg: dict) -> tuple:
667
+ # `total` is untyped scanner data; negating a string would raise.
668
+ try:
669
+ total = int(pkg.get("total", 0))
670
+ except (TypeError, ValueError):
671
+ total = 0
672
+ return (status_order.get(pkg.get("status", ""), 3), -total)
673
+
674
+ sorted_pkgs = sorted(packages, key=sort_key)
675
+
676
+ entries = []
677
+ for pkg in sorted_pkgs:
678
+ entry = (
679
+ f" {{\n"
680
+ f" package: {_js_string(pkg.get('package', 'Unknown'))},\n"
681
+ f" slug_perm: {_js_string(pkg.get('slug_perm', ''))},\n"
682
+ f" status: {_js_string(pkg.get('status', 'unknown'))},\n"
683
+ f" critical: {_js_int(pkg.get('critical'))},\n"
684
+ f" high: {_js_int(pkg.get('high'))},\n"
685
+ f" medium: {_js_int(pkg.get('medium'))},\n"
686
+ f" low: {_js_int(pkg.get('low'))},\n"
687
+ f" unknown: {_js_int(pkg.get('unknown'))},\n"
688
+ f" total: {_js_int(pkg.get('total'))}\n"
689
+ f" }}"
690
+ )
691
+ entries.append(entry)
692
+
693
+ return "[\n" + ",\n".join(entries) + "\n ]"
694
+
695
+
696
+ # Maximum custom logo dimensions (pixels) and file size (bytes)
697
+ MAX_LOGO_WIDTH = 512
698
+ MAX_LOGO_HEIGHT = 512
699
+ MAX_LOGO_FILE_SIZE = 2 * 1024 * 1024 # 2 MB
700
+
701
+
702
+ def _get_image_dimensions(path: Path) -> tuple[int, int] | None:
703
+ """Read width and height from a PNG or JPEG file header. Returns (w, h) or None."""
704
+ try:
705
+ data = path.read_bytes()
706
+ # PNG: 8-byte signature, then IHDR chunk with width (4 bytes) and height (4 bytes)
707
+ if data[:8] == b"\x89PNG\r\n\x1a\n":
708
+ w, h = struct.unpack(">II", data[16:24])
709
+ return (w, h)
710
+ # JPEG: scan for SOF0/SOF2 markers
711
+ if data[:2] == b"\xff\xd8":
712
+ i = 2
713
+ while i < len(data) - 9:
714
+ if data[i] != 0xFF:
715
+ break
716
+ marker = data[i + 1]
717
+ if marker in (0xC0, 0xC2): # SOF0, SOF2
718
+ h, w = struct.unpack(">HH", data[i + 5 : i + 9])
719
+ return (w, h)
720
+ length = struct.unpack(">H", data[i + 2 : i + 4])[0]
721
+ i += 2 + length
722
+ except Exception:
723
+ pass
724
+ return None
725
+
726
+
727
+ def _validate_custom_logo(logo_path: Path) -> bool:
728
+ """Validate a custom logo's file size and dimensions. Prints warnings and returns False on failure."""
729
+ size = logo_path.stat().st_size
730
+ if size > MAX_LOGO_FILE_SIZE:
731
+ size_mb = size / (1024 * 1024)
732
+ print(
733
+ f"WARNING: Custom logo '{logo_path.name}' is {size_mb:.1f} MB "
734
+ f"(max {MAX_LOGO_FILE_SIZE // (1024 * 1024)} MB). Using default logo instead.",
735
+ file=sys.stderr,
736
+ )
737
+ return False
738
+ dims = _get_image_dimensions(logo_path)
739
+ if dims:
740
+ w, h = dims
741
+ if w > MAX_LOGO_WIDTH or h > MAX_LOGO_HEIGHT:
742
+ print(
743
+ f"WARNING: Custom logo '{logo_path.name}' is {w}×{h}px "
744
+ f"(max {MAX_LOGO_WIDTH}×{MAX_LOGO_HEIGHT}px). Using default logo instead.",
745
+ file=sys.stderr,
746
+ )
747
+ return False
748
+ return True
749
+
750
+
751
+ def _encode_logo_data_uri(logo_path: Path) -> str:
752
+ """Return a data URI string for the given image file."""
753
+ mime_type = mimetypes.guess_type(str(logo_path))[0] or "image/png"
754
+ logo_data = base64.b64encode(logo_path.read_bytes()).decode("ascii")
755
+ return f"data:{mime_type};base64,{logo_data}"
756
+
757
+
758
+ def _resolve_logo_html(logo_path: Path | None) -> str:
759
+ """Return logo HTML markup, shared by both report generators."""
760
+ # The header is 68px tall, so the mark has to sit inside that.
761
+ if logo_path and logo_path.is_file():
762
+ if _validate_custom_logo(logo_path):
763
+ data_uri = _encode_logo_data_uri(logo_path)
764
+ return f'<img src="{data_uri}" alt="Logo" style="height:40px;width:auto;border-radius:9px;">'
765
+ # Default: the icon plus a wordmark set in the report's own font, so it
766
+ # recolours with the theme. The icon carries no text of its own.
767
+ if DEFAULT_LOGO.is_file():
768
+ data_uri = _encode_logo_data_uri(DEFAULT_LOGO)
769
+ return (
770
+ f'<img src="{data_uri}" alt="" style="height:36px;width:auto;">'
771
+ f'<span class="logo-text">Vulnly</span>'
772
+ )
773
+ return '<div class="logo-mark">V</div>'
774
+
775
+
776
+ def _resolve_footer_logo_html() -> str:
777
+ """Return footer logo HTML — always uses the Vulnly logo."""
778
+ if DEFAULT_LOGO.is_file():
779
+ data_uri = _encode_logo_data_uri(DEFAULT_LOGO)
780
+ return f'<img src="{data_uri}" alt="" style="height:22px;width:auto;vertical-align:middle;margin-right:6px;">'
781
+ return ''
782
+
783
+
784
+ def generate_repo_summary_html(data: dict, template_path: Path = REPO_SUMMARY_TEMPLATE, logo_path: Path | None = None, output_path: Path | None = None, theme: str = "dark") -> str:
785
+ """Generate an HTML repo summary report by populating the template."""
786
+ template = template_path.read_text(encoding="utf-8")
787
+
788
+ logo_html = _resolve_logo_html(logo_path)
789
+ footer_logo_html = _resolve_footer_logo_html()
790
+
791
+ owner = data.get("owner", "unknown")
792
+ repository = data.get("repository", "unknown")
793
+ total_packages = data.get("total_packages", 0)
794
+ status_counts = data.get("status_counts", {})
795
+ total_vulns = data.get("total_vulnerabilities", 0)
796
+
797
+ report_date = datetime.datetime.now().strftime("%-d %b %Y")
798
+
799
+ count_vulnerable = status_counts.get("vulnerable", 0)
800
+ count_no_issues = status_counts.get("no_issues_found", 0)
801
+ count_no_scan = status_counts.get("no_scan", 0)
802
+
803
+ # Findings only. The owner, repository and package total are each stated
804
+ # once elsewhere in the hero, and the status breakdown is in the cards
805
+ # below, so this states the headline result and nothing more.
806
+ if total_vulns and count_vulnerable:
807
+ hero_summary = (
808
+ f"<strong>{total_vulns} "
809
+ f"{'vulnerability was' if total_vulns == 1 else 'vulnerabilities were'}</strong> "
810
+ f"detected across <strong>{count_vulnerable}</strong> "
811
+ f"{'package' if count_vulnerable == 1 else 'packages'}."
812
+ )
813
+ elif count_no_scan and not count_vulnerable:
814
+ # Nothing found, but the result is only as complete as the coverage.
815
+ hero_summary = (
816
+ f"No vulnerabilities were detected. "
817
+ f"<strong>{count_no_scan}</strong> "
818
+ f"{'package has' if count_no_scan == 1 else 'packages have'} not been scanned."
819
+ )
820
+ else:
821
+ hero_summary = "No vulnerabilities were detected."
822
+
823
+ # Executive summary
824
+ exec_p1 = (
825
+ f"A security summary of the <strong>{html.escape(repository)}</strong> repository "
826
+ f"owned by <strong>{html.escape(owner)}</strong> covers "
827
+ f"<strong>{total_packages} packages</strong>. "
828
+ f"Of these, <strong>{count_vulnerable} {'package is' if count_vulnerable == 1 else 'packages are'} vulnerable</strong>, "
829
+ f"<strong>{count_no_issues} {'has' if count_no_issues == 1 else 'have'} no issues</strong>, "
830
+ f"and <strong>{count_no_scan} {'has' if count_no_scan == 1 else 'have'} not been scanned</strong>."
831
+ )
832
+ exec_p2 = (
833
+ f"Across all scanned packages, a total of <strong>{total_vulns} vulnerabilities</strong> "
834
+ f"were identified: <strong>{data.get('total_critical', 0)} Critical</strong>, "
835
+ f"<strong>{data.get('total_high', 0)} High</strong>, "
836
+ f"<strong>{data.get('total_medium', 0)} Medium</strong>, and "
837
+ f"<strong>{data.get('total_low', 0)} Low</strong> severity."
838
+ )
839
+
840
+ if count_vulnerable > 0:
841
+ alert_text = (
842
+ f"{count_vulnerable} vulnerable "
843
+ f"{'package requires' if count_vulnerable == 1 else 'packages require'} "
844
+ f"review. A total of {total_vulns} vulnerabilities were found across these packages."
845
+ )
846
+ alert_class = ""
847
+ alert_icon = "⚡"
848
+ alert_title = "Action Required:"
849
+ else:
850
+ alert_text = (
851
+ f"No vulnerable packages were found in this repository. "
852
+ f"{count_no_scan} {'package has' if count_no_scan == 1 else 'packages have'} not yet been scanned."
853
+ )
854
+ alert_class = "info"
855
+ alert_icon = "✓"
856
+ alert_title = "Repository Clean:"
857
+
858
+ replacements = {
859
+ "{{THEME_ATTR}}": ' data-theme="light"' if theme == "light" else "",
860
+ "{{OWNER}}": html.escape(owner),
861
+ "{{REPOSITORY}}": html.escape(repository),
862
+ "{{TOTAL_PACKAGES}}": str(total_packages),
863
+ "{{HERO_SUMMARY}}": hero_summary,
864
+ "{{COUNT_VULNERABLE}}": str(count_vulnerable),
865
+ "{{COUNT_NO_ISSUES}}": str(count_no_issues),
866
+ "{{COUNT_NO_SCAN}}": str(count_no_scan),
867
+ "{{TOTAL_VULNS}}": str(total_vulns),
868
+ "{{COUNT_CRITICAL}}": str(data.get("total_critical", 0)),
869
+ "{{COUNT_HIGH}}": str(data.get("total_high", 0)),
870
+ "{{COUNT_MEDIUM}}": str(data.get("total_medium", 0)),
871
+ "{{COUNT_LOW}}": str(data.get("total_low", 0)),
872
+ "{{COUNT_UNKNOWN}}": str(data.get("total_unknown", 0)),
873
+ "{{EXEC_SUMMARY_P1}}": exec_p1,
874
+ "{{EXEC_SUMMARY_P2}}": exec_p2,
875
+ "{{ALERT_TEXT}}": alert_text,
876
+ "{{ALERT_CLASS}}": alert_class,
877
+ "{{ALERT_ICON}}": alert_icon,
878
+ "{{ALERT_TITLE}}": alert_title,
879
+ "{{REPORT_DATE}}": html.escape(report_date),
880
+ "{{LOGO_HTML}}": logo_html,
881
+ "{{FOOTER_LOGO_HTML}}": footer_logo_html,
882
+ "{{CHART_JS}}": _read_chart_js(),
883
+ "{{FONT_WOFF2_B64}}": _read_font_b64(),
884
+ "{{FAVICON_B64}}": _read_favicon_b64(),
885
+ }
886
+
887
+ template = _apply_replacements(template, replacements)
888
+
889
+ # Replace PKG_DATA array
890
+ pkg_data_js = build_pkg_data_js(data.get("packages", []))
891
+ replacement = f"const PKG_DATA = {pkg_data_js};"
892
+ template = re.sub(
893
+ r"const PKG_DATA = \[.*?\];",
894
+ lambda _: replacement,
895
+ template,
896
+ count=1,
897
+ flags=re.DOTALL,
898
+ )
899
+
900
+ return template
901
+
902
+
903
+ def generate_html(data: dict, template_path: Path = REPORT_TEMPLATE, logo_path: Path | None = None, output_path: Path | None = None, theme: str = "dark") -> str:
904
+ """Generate the full HTML report by populating the template."""
905
+ template = template_path.read_text(encoding="utf-8")
906
+
907
+ logo_html = _resolve_logo_html(logo_path)
908
+ footer_logo_html = _resolve_footer_logo_html()
909
+
910
+ vulnerabilities = data.get("vulnerabilities", [])
911
+ counts = count_severities(vulnerabilities)
912
+ total = len(vulnerabilities)
913
+
914
+ scan_path = data.get("repository", "Unknown")
915
+ pkg_name = data.get("package_name", "Unknown")
916
+ pkg_version = data.get("package_version", "Unknown")
917
+ # Snyk and Grype carry no scan timestamp, so their normalizers emit an
918
+ # empty string. The key is present, so a plain .get() default never fires
919
+ # and the empty value reached the report — fall back on falsy, not missing.
920
+ scan_date = data.get("scan_date") or datetime.datetime.now().strftime("%Y-%m-%d")
921
+
922
+ # Format display date (e.g. "13 Mar 2026")
923
+ try:
924
+ dt = datetime.datetime.strptime(scan_date, "%Y-%m-%d")
925
+ scan_date_display = dt.strftime("%-d %b %Y")
926
+ except ValueError:
927
+ scan_date_display = scan_date
928
+
929
+ report_date = datetime.datetime.now().strftime("%-d %b %Y")
930
+
931
+ # Only image/registry scanners set repository to the package name itself;
932
+ # for those the path adds nothing, so it is dropped rather than repeated.
933
+ path_is_distinct = bool(scan_path) and scan_path != pkg_name
934
+
935
+ if path_is_distinct:
936
+ path_clause = f"in the <strong>{html.escape(scan_path)}</strong> repository "
937
+ path_footer_html = f"Path: <strong>{html.escape(scan_path)}</strong> &bull;\n "
938
+ else:
939
+ path_clause = ""
940
+ path_footer_html = ""
941
+
942
+ # The line beneath the report title: what was scanned, once. The separator
943
+ # is only emitted between two present values, so a scan with just a version
944
+ # does not render a leading bullet.
945
+ identity_parts = []
946
+ if path_is_distinct:
947
+ identity_parts.append(html.escape(scan_path))
948
+ if pkg_version and pkg_version != "Unknown":
949
+ identity_parts.append(html.escape(pkg_version))
950
+ hero_identity = '<span class="identity-sep">&middot;</span>'.join(identity_parts)
951
+
952
+ # Generate executive summary text
953
+ scan_scope = (
954
+ f"A security scan of the <strong>{html.escape(pkg_name)}</strong> package "
955
+ f"(version <strong>{html.escape(pkg_version)}</strong>) {path_clause}"
956
+ )
957
+
958
+ urgent = counts["critical"] + counts["high"]
959
+
960
+ if total > 0:
961
+ if urgent > 0:
962
+ exec_p1 = (
963
+ f"{scan_scope}"
964
+ f"identified a total of <strong>{total} "
965
+ f"{'vulnerability' if total == 1 else 'vulnerabilities'}</strong>. "
966
+ f"Of these, <strong>{counts['critical']} "
967
+ f"{'is' if counts['critical'] == 1 else 'are'} rated Critical</strong> and "
968
+ f"<strong>{counts['high']} "
969
+ f"{'is' if counts['high'] == 1 else 'are'} rated High</strong> severity."
970
+ )
971
+ exec_p2 = (
972
+ f"The scan also found <strong>{counts['medium']} Medium</strong> and "
973
+ f"<strong>{counts['low']} Low</strong> severity issues. "
974
+ f"Addressing Critical and High severity vulnerabilities should be prioritised "
975
+ f"to reduce the risk of exploitation."
976
+ )
977
+ # Name only the severities actually present, so a High-only scan
978
+ # does not read "0 Critical and 3 High…".
979
+ urgent_parts = []
980
+ if counts["critical"]:
981
+ urgent_parts.append(f"{counts['critical']} Critical")
982
+ if counts["high"]:
983
+ urgent_parts.append(f"{counts['high']} High")
984
+ alert_text = (
985
+ f"{' and '.join(urgent_parts)} severity "
986
+ f"{'vulnerability requires' if urgent == 1 else 'vulnerabilities require'} "
987
+ f"immediate review and remediation. Consult the detailed table below for "
988
+ f"affected packages and available fixes."
989
+ )
990
+ alert_class = ""
991
+ alert_icon = "⚡"
992
+ alert_title = "Immediate Action Required:"
993
+ else:
994
+ exec_p1 = (
995
+ f"{scan_scope}"
996
+ f"identified a total of <strong>{total} "
997
+ f"{'vulnerability' if total == 1 else 'vulnerabilities'}</strong>, "
998
+ f"<strong>none of which are rated Critical or High</strong> severity."
999
+ )
1000
+ exec_p2 = (
1001
+ f"The scan found <strong>{counts['medium']} Medium</strong> and "
1002
+ f"<strong>{counts['low']} Low</strong> severity issues. "
1003
+ f"These do not require immediate action, but should be reviewed and "
1004
+ f"scheduled for remediation as part of routine maintenance."
1005
+ )
1006
+ alert_text = (
1007
+ f"No Critical or High severity vulnerabilities were found. The "
1008
+ f"{total} lower-severity {'issue' if total == 1 else 'issues'} listed below "
1009
+ f"can be addressed as part of routine maintenance."
1010
+ )
1011
+ alert_class = "warn"
1012
+ alert_icon = "ℹ"
1013
+ alert_title = "Review Recommended:"
1014
+
1015
+ badge_class = ""
1016
+ badge_text = "⚠ Vulnerabilities Detected"
1017
+ # Findings only — the package, path, version and date are all stated
1018
+ # once elsewhere in the hero. Scoped by affected package rather than
1019
+ # by image layer, since scans cover Maven, npm and others too.
1020
+ affected = len({v.get("package") for v in vulnerabilities if v.get("package")})
1021
+ if affected == 1:
1022
+ scope = " in 1 package"
1023
+ elif affected > 1:
1024
+ scope = f" across {affected} packages"
1025
+ else:
1026
+ scope = ""
1027
+ hero_vuln_summary = (
1028
+ f"<strong>{total} "
1029
+ f"{'vulnerability was' if total == 1 else 'vulnerabilities were'}</strong> "
1030
+ f"detected{scope}."
1031
+ )
1032
+ else:
1033
+ exec_p1 = (
1034
+ f"{scan_scope}"
1035
+ f"identified <strong>no known vulnerabilities</strong>."
1036
+ )
1037
+ exec_p2 = (
1038
+ f"No Critical, High, Medium or Low severity issues were reported by "
1039
+ f"<strong>{html.escape(data.get('scanner_source', 'the scanner'))}</strong>. "
1040
+ f"Re-scan when the package is updated, as new advisories are published "
1041
+ f"against existing versions over time."
1042
+ )
1043
+ alert_text = (
1044
+ f"This package is clean as of {html.escape(scan_date_display)}. "
1045
+ f"No remediation is required — continue to monitor for newly published "
1046
+ f"advisories affecting this version."
1047
+ )
1048
+ alert_class = "clean"
1049
+ alert_icon = "✓"
1050
+ alert_title = "No Vulnerabilities Detected:"
1051
+ badge_class = "clean"
1052
+ badge_text = "✓ No Vulnerabilities Detected"
1053
+ hero_vuln_summary = "No vulnerabilities were detected."
1054
+
1055
+ # Replace placeholders
1056
+ replacements = {
1057
+ "{{THEME_ATTR}}": ' data-theme="light"' if theme == "light" else "",
1058
+ "{{SCAN_PATH}}": html.escape(scan_path),
1059
+ "{{HERO_IDENTITY}}": hero_identity,
1060
+ "{{PATH_FOOTER_HTML}}": path_footer_html,
1061
+ "{{PACKAGE_NAME}}": html.escape(pkg_name),
1062
+ "{{PACKAGE_VERSION}}": html.escape(pkg_version),
1063
+ "{{SCAN_DATE}}": html.escape(scan_date),
1064
+ "{{SCAN_ID}}": html.escape(data.get("scan_id", "N/A")),
1065
+ "{{SCANNER_SOURCE}}": html.escape(data.get("scanner_source", "Unknown")),
1066
+ "{{REPORT_DATE}}": html.escape(report_date),
1067
+ "{{TOTAL_VULNS}}": str(total),
1068
+ "{{COUNT_CRITICAL}}": str(counts["critical"]),
1069
+ "{{COUNT_HIGH}}": str(counts["high"]),
1070
+ "{{COUNT_MEDIUM}}": str(counts["medium"]),
1071
+ "{{COUNT_LOW}}": str(counts["low"]),
1072
+ "{{COUNT_UNKNOWN}}": str(counts["unknown"]),
1073
+ "{{COUNT_FIXABLE}}": str(count_fixable(vulnerabilities)),
1074
+ "{{CSV_FILENAME}}": _csv_filename(pkg_name, scan_date),
1075
+ "{{EXEC_SUMMARY_P1}}": exec_p1,
1076
+ "{{EXEC_SUMMARY_P2}}": exec_p2,
1077
+ "{{HERO_VULN_SUMMARY}}": hero_vuln_summary,
1078
+ "{{ALERT_TEXT}}": alert_text,
1079
+ "{{ALERT_CLASS}}": alert_class,
1080
+ "{{ALERT_ICON}}": alert_icon,
1081
+ "{{ALERT_TITLE}}": alert_title,
1082
+ "{{BADGE_CLASS}}": badge_class,
1083
+ "{{BADGE_TEXT}}": badge_text,
1084
+ "{{LOGO_HTML}}": logo_html,
1085
+ "{{FOOTER_LOGO_HTML}}": footer_logo_html,
1086
+ "{{CHART_JS}}": _read_chart_js(),
1087
+ "{{FONT_WOFF2_B64}}": _read_font_b64(),
1088
+ "{{FAVICON_B64}}": _read_favicon_b64(),
1089
+ }
1090
+
1091
+ template = _apply_replacements(template, replacements)
1092
+
1093
+ # Replace the example VULN_DATA array with actual data
1094
+ vuln_data_js = build_vuln_data_js(vulnerabilities)
1095
+ replacement = f"const VULN_DATA = {vuln_data_js};"
1096
+ template = re.sub(
1097
+ r"const VULN_DATA = \[.*?\];",
1098
+ lambda _: replacement,
1099
+ template,
1100
+ count=1,
1101
+ flags=re.DOTALL,
1102
+ )
1103
+
1104
+ return template
1105
+
1106
+
1107
+ def main() -> None:
1108
+ parser = argparse.ArgumentParser(
1109
+ description="Generate an HTML vulnerability report from a JSON input file."
1110
+ )
1111
+ parser.add_argument(
1112
+ "-v", "--version",
1113
+ action="version",
1114
+ version=f"%(prog)s {__version__}",
1115
+ )
1116
+ parser.add_argument(
1117
+ "input",
1118
+ help="Path to the JSON file containing vulnerability data (use '-' for stdin)",
1119
+ )
1120
+ parser.add_argument(
1121
+ "-o",
1122
+ "--output",
1123
+ default=None,
1124
+ help="Output HTML file path (default: auto-generated from package metadata)",
1125
+ )
1126
+ parser.add_argument(
1127
+ "--logo",
1128
+ default=None,
1129
+ help="Path to a custom logo image (embedded in the report as a data URI)",
1130
+ )
1131
+ parser.add_argument(
1132
+ "--theme",
1133
+ choices=["dark", "light"],
1134
+ default="dark",
1135
+ help="Report colour theme (default: dark)",
1136
+ )
1137
+ parser.add_argument(
1138
+ "--source",
1139
+ default=None,
1140
+ help=(
1141
+ "Scanner source format: "
1142
+ f"{', '.join(sorted(SUPPORTED_SOURCES))} (auto-detected if omitted)"
1143
+ ),
1144
+ )
1145
+ args = parser.parse_args()
1146
+
1147
+ if args.source and args.source.lower() not in SUPPORTED_SOURCES:
1148
+ print(
1149
+ f"Error: Unsupported source '{args.source}'. "
1150
+ f"Supported sources: {', '.join(sorted(SUPPORTED_SOURCES))}",
1151
+ file=sys.stderr,
1152
+ )
1153
+ sys.exit(1)
1154
+
1155
+ # --- Load ---------------------------------------------------------------
1156
+ if args.input != "-" and not Path(args.input).exists():
1157
+ print(f"Error: Input file '{args.input}' not found.", file=sys.stderr)
1158
+ sys.exit(1)
1159
+
1160
+ try:
1161
+ data = load_vulnerability_data(args.input)
1162
+ except json.JSONDecodeError as exc:
1163
+ print(f"Error: Invalid JSON – {exc}", file=sys.stderr)
1164
+ sys.exit(1)
1165
+ # --- Normalize based on source ------------------------------------------
1166
+ source = (args.source or "").lower()
1167
+ is_repo_summary = False
1168
+
1169
+ # Check for repo summary format first (both explicit and auto-detect)
1170
+ if is_cloudsmith_repo_summary_format(data):
1171
+ is_repo_summary = True
1172
+ source = "cloudsmith"
1173
+ data = normalize_cloudsmith_repo_summary(data)
1174
+ elif source == "cloudsmith":
1175
+ if not is_cloudsmith_cli_format(data):
1176
+ print(
1177
+ "Error: --source cloudsmith was specified but the input does not "
1178
+ "look like Cloudsmith CLI output (expected a top-level 'data' object "
1179
+ "containing 'scans').",
1180
+ file=sys.stderr,
1181
+ )
1182
+ sys.exit(1)
1183
+ data = normalize_cloudsmith_cli(data)
1184
+ elif source == "trivy":
1185
+ if not is_trivy_format(data):
1186
+ print(
1187
+ "Error: --source trivy was specified but the input does not look "
1188
+ "like Trivy JSON output (expected a top-level 'Results' array).",
1189
+ file=sys.stderr,
1190
+ )
1191
+ sys.exit(1)
1192
+ data = normalize_trivy(data)
1193
+ elif source == "grype":
1194
+ if not is_grype_format(data):
1195
+ print(
1196
+ "Error: --source grype was specified but the input does not look "
1197
+ "like Grype JSON output (expected top-level 'matches' array and "
1198
+ "'descriptor' object).",
1199
+ file=sys.stderr,
1200
+ )
1201
+ sys.exit(1)
1202
+ data = normalize_grype(data)
1203
+ elif source == "snyk":
1204
+ if not is_snyk_format(data):
1205
+ print(
1206
+ "Error: --source snyk was specified but the input does not look "
1207
+ "like Snyk CLI output (expected top-level 'vulnerabilities' array, "
1208
+ "'packageManager', and 'projectName').",
1209
+ file=sys.stderr,
1210
+ )
1211
+ sys.exit(1)
1212
+ data = normalize_snyk(data)
1213
+ elif not source:
1214
+ # Auto-detect
1215
+ if is_cloudsmith_cli_format(data):
1216
+ source = "cloudsmith"
1217
+ data = normalize_cloudsmith_cli(data)
1218
+ elif is_trivy_format(data):
1219
+ source = "trivy"
1220
+ data = normalize_trivy(data)
1221
+ elif is_grype_format(data):
1222
+ source = "grype"
1223
+ data = normalize_grype(data)
1224
+ elif is_snyk_format(data):
1225
+ source = "snyk"
1226
+ data = normalize_snyk(data)
1227
+
1228
+ data["scanner_source"] = source.capitalize() if source else "Unknown"
1229
+
1230
+ # --- Logo ---------------------------------------------------------------
1231
+ logo_path = None
1232
+ if args.logo:
1233
+ logo_path = Path(args.logo)
1234
+ if not logo_path.exists():
1235
+ print(f"Error: Logo file '{args.logo}' not found.", file=sys.stderr)
1236
+ sys.exit(1)
1237
+
1238
+ # --- Repo summary flow --------------------------------------------------
1239
+ if is_repo_summary:
1240
+ if args.output is None:
1241
+ owner = data.get("owner", "unknown")
1242
+ repository = data.get("repository", "unknown")
1243
+ safe = re.sub(r'[^\w.\-]', '_', f"{owner}_{repository}_cloudsmith")
1244
+ reports_dir = Path("reports")
1245
+ reports_dir.mkdir(exist_ok=True)
1246
+ args.output = str(reports_dir / f"{safe}_repo_summary_report.html")
1247
+
1248
+ output_path = Path(args.output)
1249
+ report_html = generate_repo_summary_html(
1250
+ data, template_path=REPO_SUMMARY_TEMPLATE, logo_path=logo_path,
1251
+ output_path=output_path, theme=args.theme,
1252
+ )
1253
+
1254
+ with open(args.output, "w") as f:
1255
+ f.write(report_html)
1256
+
1257
+ total_pkgs = data.get("total_packages", 0)
1258
+ status = data.get("status_counts", {})
1259
+ total_vulns = data.get("total_vulnerabilities", 0)
1260
+ print(f"Repo summary generated: {args.output} ({total_pkgs} packages, {total_vulns} vulnerabilities)")
1261
+ print(
1262
+ f" VULNERABLE: {status.get('vulnerable', 0)} "
1263
+ f"NO ISSUES: {status.get('no_issues_found', 0)} "
1264
+ f"NOT SCANNED: {status.get('no_scan', 0)}"
1265
+ )
1266
+ return
1267
+
1268
+ # --- Validate -----------------------------------------------------------
1269
+ warnings = validate_data(data)
1270
+ for w in warnings:
1271
+ print(w, file=sys.stderr)
1272
+
1273
+ # --- Output filename ----------------------------------------------------
1274
+ if args.output is None:
1275
+ repo = data.get("repository", "unknown").replace("/", "_")
1276
+ pkg = data.get("package_name", "unknown")
1277
+ ver = data.get("package_version", "unknown")
1278
+ # Truncate long versions (e.g. Docker digests) to 12 chars
1279
+ if len(ver) > 20:
1280
+ ver = ver[:12]
1281
+ # Sanitise for filesystem safety
1282
+ source_tag = source if source else "unknown"
1283
+ safe = re.sub(r'[^\w.\-]', '_', f"{repo}_{pkg}_{ver}_{source_tag}")
1284
+ reports_dir = Path("reports")
1285
+ reports_dir.mkdir(exist_ok=True)
1286
+ args.output = str(reports_dir / f"{safe}_vulnerability_report.html")
1287
+
1288
+ # --- Generate -----------------------------------------------------------
1289
+ output_path = Path(args.output)
1290
+ report_html = generate_html(
1291
+ data, template_path=REPORT_TEMPLATE, logo_path=logo_path,
1292
+ output_path=output_path, theme=args.theme,
1293
+ )
1294
+
1295
+ with open(args.output, "w") as f:
1296
+ f.write(report_html)
1297
+
1298
+ # --- Terminal summary ---------------------------------------------------
1299
+ vulnerabilities = data.get("vulnerabilities", [])
1300
+ counts = count_severities(vulnerabilities)
1301
+ total = len(vulnerabilities)
1302
+ # SEVERITY_ORDER is the canonical ordering and includes `unknown`; listing
1303
+ # the levels by hand here previously dropped it, so a scan of only
1304
+ # unknown-severity findings reported a total and then "No vulnerabilities".
1305
+ breakdown = " ".join(
1306
+ f"{sev.upper()}: {counts[sev]}"
1307
+ for sev in sorted(SEVERITY_ORDER, key=lambda s: SEVERITY_ORDER[s])
1308
+ if counts[sev] > 0
1309
+ )
1310
+ print(f"Report generated: {args.output} ({total} vulnerabilities)")
1311
+ if breakdown:
1312
+ print(f" {breakdown}")
1313
+ else:
1314
+ print(" No vulnerabilities found.")
1315
+
1316
+ check_for_update()
1317
+
1318
+
1319
+ _VERSION_RE = re.compile(r"^(\d+)\.(\d+)\.(\d+)(?:(a|b|rc)(\d+))?$", re.IGNORECASE)
1320
+ # Pre-releases order before the release they lead to: 1.0.0a1 < 1.0.0b1 < 1.0.0rc1 < 1.0.0
1321
+ _PRE_ORDER = {"a": 0, "b": 1, "rc": 2}
1322
+ _RELEASE_RANK = 3
1323
+
1324
+
1325
+ def _version_key(version: str) -> tuple | None:
1326
+ """Return a sortable key for a version string, or None if unrecognised."""
1327
+ match = _VERSION_RE.match(str(version).strip())
1328
+ if not match:
1329
+ return None
1330
+ major, minor, patch, pre, pre_num = match.groups()
1331
+ if pre is None:
1332
+ return (int(major), int(minor), int(patch), _RELEASE_RANK, 0)
1333
+ return (int(major), int(minor), int(patch), _PRE_ORDER[pre.lower()], int(pre_num))
1334
+
1335
+
1336
+ UPDATE_CHECK_ENV = "VULNLY_NO_UPDATE_CHECK"
1337
+
1338
+
1339
+ def _update_check_enabled() -> bool:
1340
+ """Return False when the update check should be skipped entirely.
1341
+
1342
+ The check costs a network round trip on every invocation, which is noise
1343
+ in CI and unwanted on hosts with restricted egress. It is skipped when
1344
+ opted out, and when stderr is not a terminal — the notice is written
1345
+ there, so nobody would read it anyway.
1346
+ """
1347
+ if os.environ.get(UPDATE_CHECK_ENV, "").strip():
1348
+ return False
1349
+ try:
1350
+ return sys.stderr.isatty()
1351
+ except Exception:
1352
+ return False
1353
+
1354
+
1355
+ def check_for_update() -> None:
1356
+ """Check PyPI for a newer version and print a notice if available."""
1357
+ if not _update_check_enabled():
1358
+ return
1359
+ try:
1360
+ url = "https://pypi.org/pypi/vulnly/json"
1361
+ req = urllib.request.Request(url, headers={"Accept": "application/json"})
1362
+ with urllib.request.urlopen(req, timeout=3) as resp:
1363
+ data = json.loads(resp.read().decode())
1364
+ latest = data["info"]["version"]
1365
+
1366
+ # Comparing as strings reported an update whenever the versions merely
1367
+ # differed, so running a build ahead of PyPI advised "upgrading" to an
1368
+ # older release. Stay silent when either version is unrecognised rather
1369
+ # than guess.
1370
+ current_key = _version_key(__version__)
1371
+ latest_key = _version_key(latest)
1372
+ if current_key is None or latest_key is None:
1373
+ return
1374
+ if latest_key > current_key:
1375
+ print(
1376
+ f"\nUpdate available: {__version__} → {latest} "
1377
+ f"Run `pip install --upgrade vulnly` to update.",
1378
+ file=sys.stderr,
1379
+ )
1380
+ except Exception:
1381
+ pass # never block the user on a failed update check
1382
+
1383
+
1384
+ if __name__ == "__main__":
1385
+ main()