flagrante 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- flagrante/__init__.py +26 -0
- flagrante/classify.py +273 -0
- flagrante/cli.py +121 -0
- flagrante/html.py +362 -0
- flagrante/report.py +189 -0
- flagrante/sbom.py +207 -0
- flagrante/scan.py +62 -0
- flagrante/sources.py +273 -0
- flagrante-0.0.1.dist-info/METADATA +153 -0
- flagrante-0.0.1.dist-info/RECORD +14 -0
- flagrante-0.0.1.dist-info/WHEEL +5 -0
- flagrante-0.0.1.dist-info/entry_points.txt +2 -0
- flagrante-0.0.1.dist-info/licenses/LICENSE +202 -0
- flagrante-0.0.1.dist-info/top_level.txt +1 -0
flagrante/sbom.py
ADDED
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
"""SBOM parsing: CycloneDX and SPDX JSON -> a normalised component list.
|
|
2
|
+
|
|
3
|
+
Deliberately dependency-free. Both formats are read leniently: a malformed or
|
|
4
|
+
partially-populated SBOM should still yield whatever components it does carry,
|
|
5
|
+
because a manufacturer under time pressure is exactly who produces one.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import re
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from typing import Any, Iterable
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class SBOMError(ValueError):
|
|
17
|
+
"""The input could not be read as a supported SBOM."""
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class Component:
|
|
22
|
+
"""One software component as named by the SBOM."""
|
|
23
|
+
|
|
24
|
+
name: str
|
|
25
|
+
version: str | None
|
|
26
|
+
purl: str | None
|
|
27
|
+
ecosystem: str | None = None
|
|
28
|
+
licenses: tuple[str, ...] = field(default=())
|
|
29
|
+
|
|
30
|
+
@property
|
|
31
|
+
def label(self) -> str:
|
|
32
|
+
return f"{self.name}@{self.version}" if self.version else self.name
|
|
33
|
+
|
|
34
|
+
@property
|
|
35
|
+
def identifiable(self) -> bool:
|
|
36
|
+
"""Whether this component can be looked up against a vulnerability feed.
|
|
37
|
+
|
|
38
|
+
Without a purl we cannot query OSV reliably, and guessing an ecosystem
|
|
39
|
+
from a bare name invites false negatives -- the one direction this tool
|
|
40
|
+
must never fail in. Such components are reported as unresolvable rather
|
|
41
|
+
than silently dropped.
|
|
42
|
+
"""
|
|
43
|
+
return bool(self.purl)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass
|
|
47
|
+
class SBOMDocument:
|
|
48
|
+
format: str
|
|
49
|
+
spec_version: str | None
|
|
50
|
+
components: list[Component]
|
|
51
|
+
subject: str | None = None
|
|
52
|
+
|
|
53
|
+
@property
|
|
54
|
+
def identifiable(self) -> list[Component]:
|
|
55
|
+
return [c for c in self.components if c.identifiable]
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def unresolvable(self) -> list[Component]:
|
|
59
|
+
return [c for c in self.components if not c.identifiable]
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
_PURL_ECOSYSTEM = re.compile(r"^pkg:([a-zA-Z0-9._-]+)/")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _ecosystem_from_purl(purl: str | None) -> str | None:
|
|
66
|
+
if not purl:
|
|
67
|
+
return None
|
|
68
|
+
match = _PURL_ECOSYSTEM.match(purl)
|
|
69
|
+
return match.group(1).lower() if match else None
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _clean(value: Any) -> str | None:
|
|
73
|
+
if value is None:
|
|
74
|
+
return None
|
|
75
|
+
text = str(value).strip()
|
|
76
|
+
return text or None
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _cyclonedx_licenses(entry: dict) -> tuple[str, ...]:
|
|
80
|
+
out: list[str] = []
|
|
81
|
+
for item in entry.get("licenses") or []:
|
|
82
|
+
if not isinstance(item, dict):
|
|
83
|
+
continue
|
|
84
|
+
lic = item.get("license")
|
|
85
|
+
if isinstance(lic, dict):
|
|
86
|
+
name = _clean(lic.get("id") or lic.get("name"))
|
|
87
|
+
if name:
|
|
88
|
+
out.append(name)
|
|
89
|
+
expression = _clean(item.get("expression"))
|
|
90
|
+
if expression:
|
|
91
|
+
out.append(expression)
|
|
92
|
+
return tuple(dict.fromkeys(out))
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _walk_cyclonedx(entries: Iterable[dict]) -> Iterable[dict]:
|
|
96
|
+
"""Yield components including nested ones, which CycloneDX permits."""
|
|
97
|
+
for entry in entries or []:
|
|
98
|
+
if not isinstance(entry, dict):
|
|
99
|
+
continue
|
|
100
|
+
yield entry
|
|
101
|
+
nested = entry.get("components")
|
|
102
|
+
if isinstance(nested, list):
|
|
103
|
+
yield from _walk_cyclonedx(nested)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _parse_cyclonedx(doc: dict) -> SBOMDocument:
|
|
107
|
+
components: list[Component] = []
|
|
108
|
+
for entry in _walk_cyclonedx(doc.get("components") or []):
|
|
109
|
+
name = _clean(entry.get("name"))
|
|
110
|
+
if not name:
|
|
111
|
+
continue
|
|
112
|
+
group = _clean(entry.get("group"))
|
|
113
|
+
if group and not name.startswith(group):
|
|
114
|
+
name = f"{group}/{name}"
|
|
115
|
+
purl = _clean(entry.get("purl"))
|
|
116
|
+
components.append(
|
|
117
|
+
Component(
|
|
118
|
+
name=name,
|
|
119
|
+
version=_clean(entry.get("version")),
|
|
120
|
+
purl=purl,
|
|
121
|
+
ecosystem=_ecosystem_from_purl(purl),
|
|
122
|
+
licenses=_cyclonedx_licenses(entry),
|
|
123
|
+
)
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
metadata = doc.get("metadata") or {}
|
|
127
|
+
target = metadata.get("component") or {}
|
|
128
|
+
subject = _clean(target.get("name")) if isinstance(target, dict) else None
|
|
129
|
+
|
|
130
|
+
return SBOMDocument(
|
|
131
|
+
format="CycloneDX",
|
|
132
|
+
spec_version=_clean(doc.get("specVersion")),
|
|
133
|
+
components=components,
|
|
134
|
+
subject=subject,
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _spdx_purl(entry: dict) -> str | None:
|
|
139
|
+
for ref in entry.get("externalRefs") or []:
|
|
140
|
+
if not isinstance(ref, dict):
|
|
141
|
+
continue
|
|
142
|
+
if str(ref.get("referenceType", "")).lower() == "purl":
|
|
143
|
+
return _clean(ref.get("referenceLocator"))
|
|
144
|
+
return None
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _parse_spdx(doc: dict) -> SBOMDocument:
|
|
148
|
+
components: list[Component] = []
|
|
149
|
+
for entry in doc.get("packages") or []:
|
|
150
|
+
if not isinstance(entry, dict):
|
|
151
|
+
continue
|
|
152
|
+
name = _clean(entry.get("name"))
|
|
153
|
+
if not name:
|
|
154
|
+
continue
|
|
155
|
+
version = _clean(entry.get("versionInfo"))
|
|
156
|
+
purl = _spdx_purl(entry)
|
|
157
|
+
declared = _clean(entry.get("licenseDeclared"))
|
|
158
|
+
licenses = (declared,) if declared and declared != "NOASSERTION" else ()
|
|
159
|
+
components.append(
|
|
160
|
+
Component(
|
|
161
|
+
name=name,
|
|
162
|
+
version=version,
|
|
163
|
+
purl=purl,
|
|
164
|
+
ecosystem=_ecosystem_from_purl(purl),
|
|
165
|
+
licenses=licenses, # type: ignore[arg-type]
|
|
166
|
+
)
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
return SBOMDocument(
|
|
170
|
+
format="SPDX",
|
|
171
|
+
spec_version=_clean(doc.get("spdxVersion")),
|
|
172
|
+
components=components,
|
|
173
|
+
subject=_clean(doc.get("name")),
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def parse(raw: str | bytes) -> SBOMDocument:
|
|
178
|
+
"""Read a CycloneDX or SPDX JSON SBOM.
|
|
179
|
+
|
|
180
|
+
Raises SBOMError when the input is not JSON or carries no recognisable
|
|
181
|
+
format marker, so the caller can tell the user what to hand us instead.
|
|
182
|
+
"""
|
|
183
|
+
if isinstance(raw, bytes):
|
|
184
|
+
raw = raw.decode("utf-8-sig", errors="replace")
|
|
185
|
+
|
|
186
|
+
try:
|
|
187
|
+
doc = json.loads(raw)
|
|
188
|
+
except json.JSONDecodeError as exc:
|
|
189
|
+
raise SBOMError(f"not valid JSON ({exc.msg} at line {exc.lineno})") from exc
|
|
190
|
+
|
|
191
|
+
if not isinstance(doc, dict):
|
|
192
|
+
raise SBOMError("expected a JSON object at the top level")
|
|
193
|
+
|
|
194
|
+
if doc.get("bomFormat") == "CycloneDX" or "specVersion" in doc:
|
|
195
|
+
return _parse_cyclonedx(doc)
|
|
196
|
+
if "spdxVersion" in doc or "SPDXID" in doc:
|
|
197
|
+
return _parse_spdx(doc)
|
|
198
|
+
|
|
199
|
+
raise SBOMError(
|
|
200
|
+
"unrecognised SBOM format -- expected CycloneDX JSON (bomFormat) "
|
|
201
|
+
"or SPDX JSON (spdxVersion). Generate one with syft or cdxgen."
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def parse_file(path: str) -> SBOMDocument:
|
|
206
|
+
with open(path, "rb") as handle:
|
|
207
|
+
return parse(handle.read())
|
flagrante/scan.py
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Orchestration: SBOM in, Assessment out.
|
|
2
|
+
|
|
3
|
+
Kept separate from the CLI so the web service and the GitHub Action can share
|
|
4
|
+
exactly the same path. Any divergence between what the CLI says and what the
|
|
5
|
+
hosted tool says would be a correctness bug, not a cosmetic one.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Callable
|
|
11
|
+
|
|
12
|
+
from .classify import Assessment, assess
|
|
13
|
+
from .sbom import SBOMDocument, parse, parse_file
|
|
14
|
+
from .sources import fetch_epss, fetch_kev, query_osv
|
|
15
|
+
|
|
16
|
+
Progress = Callable[[str], None]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _noop(_message: str) -> None:
|
|
20
|
+
pass
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def scan_document(
|
|
24
|
+
document: SBOMDocument,
|
|
25
|
+
progress: Progress = _noop,
|
|
26
|
+
refresh_feeds: bool = False,
|
|
27
|
+
) -> Assessment:
|
|
28
|
+
identifiable = document.identifiable
|
|
29
|
+
progress(
|
|
30
|
+
f"{len(document.components)} components read, "
|
|
31
|
+
f"{len(identifiable)} with a package URL"
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
if not identifiable:
|
|
35
|
+
# Nothing queryable. We still return an assessment so the caller can
|
|
36
|
+
# report the unresolvable components, which is the whole story here.
|
|
37
|
+
return assess(document, {}, {}, {}, {})
|
|
38
|
+
|
|
39
|
+
purls = [c.purl for c in identifiable if c.purl]
|
|
40
|
+
progress("querying OSV for known vulnerabilities")
|
|
41
|
+
osv_by_purl, vulnerabilities = query_osv(purls)
|
|
42
|
+
|
|
43
|
+
cves = sorted({cve for v in vulnerabilities.values() for cve in v.cves})
|
|
44
|
+
progress(f"{len(vulnerabilities)} advisories, {len(cves)} distinct CVEs")
|
|
45
|
+
|
|
46
|
+
progress("fetching CISA KEV (confirmed exploited)")
|
|
47
|
+
kev = fetch_kev(refresh=refresh_feeds)
|
|
48
|
+
|
|
49
|
+
epss: dict[str, float] = {}
|
|
50
|
+
if cves:
|
|
51
|
+
progress("fetching EPSS exploitation probabilities")
|
|
52
|
+
epss = fetch_epss(cves)
|
|
53
|
+
|
|
54
|
+
return assess(document, osv_by_purl, vulnerabilities, kev, epss)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def scan_file(path: str, progress: Progress = _noop, refresh_feeds: bool = False) -> Assessment:
|
|
58
|
+
return scan_document(parse_file(path), progress, refresh_feeds)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def scan_text(raw: str | bytes, progress: Progress = _noop, refresh_feeds: bool = False) -> Assessment:
|
|
62
|
+
return scan_document(parse(raw), progress, refresh_feeds)
|
flagrante/sources.py
ADDED
|
@@ -0,0 +1,273 @@
|
|
|
1
|
+
"""Vulnerability and exploitation feeds.
|
|
2
|
+
|
|
3
|
+
Three public sources, no API keys, no dependencies:
|
|
4
|
+
|
|
5
|
+
OSV (osv.dev) component -> known vulnerabilities
|
|
6
|
+
CISA KEV CVE -> confirmed exploited in the wild
|
|
7
|
+
FIRST EPSS CVE -> probability of exploitation in 30 days
|
|
8
|
+
|
|
9
|
+
A note on failure handling, which matters more here than in most clients:
|
|
10
|
+
when a feed cannot be reached we raise. We never degrade to "nothing found".
|
|
11
|
+
An empty result and an unreachable exploitation feed look identical to a user
|
|
12
|
+
and mean opposite things, and only one of them is safe to act on.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
import os
|
|
19
|
+
import time
|
|
20
|
+
import urllib.error
|
|
21
|
+
import urllib.request
|
|
22
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
23
|
+
from dataclasses import dataclass
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Any, Sequence
|
|
26
|
+
|
|
27
|
+
OSV_BATCH_URL = "https://api.osv.dev/v1/querybatch"
|
|
28
|
+
OSV_QUERY_URL = "https://api.osv.dev/v1/query"
|
|
29
|
+
KEV_URL = (
|
|
30
|
+
"https://www.cisa.gov/sites/default/files/feeds/"
|
|
31
|
+
"known_exploited_vulnerabilities.json"
|
|
32
|
+
)
|
|
33
|
+
EPSS_URL = "https://api.first.org/data/v1/epss"
|
|
34
|
+
|
|
35
|
+
# Derived rather than written out, so the string the feeds see can never
|
|
36
|
+
# drift from the version actually shipped.
|
|
37
|
+
from . import __version__ as _version
|
|
38
|
+
|
|
39
|
+
USER_AGENT = f"flagrante/{_version} (CRA Article 14 exposure checker; +https://flagrante.dev)"
|
|
40
|
+
CACHE_TTL_SECONDS = 6 * 3600
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class FeedError(RuntimeError):
|
|
44
|
+
"""A required feed could not be reached or returned something unusable."""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _cache_dir() -> Path:
|
|
48
|
+
root = os.environ.get("FLAGRANTE_CACHE") or (Path.home() / ".cache" / "flagrante")
|
|
49
|
+
path = Path(root)
|
|
50
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
51
|
+
return path
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _cached(name: str, ttl: int = CACHE_TTL_SECONDS) -> Any | None:
|
|
55
|
+
path = _cache_dir() / name
|
|
56
|
+
if not path.exists():
|
|
57
|
+
return None
|
|
58
|
+
if time.time() - path.stat().st_mtime > ttl:
|
|
59
|
+
return None
|
|
60
|
+
try:
|
|
61
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
62
|
+
except (json.JSONDecodeError, OSError):
|
|
63
|
+
return None
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _store(name: str, payload: Any) -> None:
|
|
67
|
+
try:
|
|
68
|
+
(_cache_dir() / name).write_text(json.dumps(payload), encoding="utf-8")
|
|
69
|
+
except OSError:
|
|
70
|
+
pass # a cache miss is never worth failing a scan over
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _request(url: str, body: bytes | None = None, timeout: int = 45) -> Any:
|
|
74
|
+
headers = {"User-Agent": USER_AGENT, "Accept": "application/json"}
|
|
75
|
+
if body is not None:
|
|
76
|
+
headers["Content-Type"] = "application/json"
|
|
77
|
+
request = urllib.request.Request(url, data=body, headers=headers)
|
|
78
|
+
try:
|
|
79
|
+
with urllib.request.urlopen(request, timeout=timeout) as response:
|
|
80
|
+
return json.loads(response.read().decode("utf-8"))
|
|
81
|
+
except urllib.error.HTTPError as exc:
|
|
82
|
+
raise FeedError(f"{url} returned HTTP {exc.code}") from exc
|
|
83
|
+
except urllib.error.URLError as exc:
|
|
84
|
+
raise FeedError(f"could not reach {url}: {exc.reason}") from exc
|
|
85
|
+
except json.JSONDecodeError as exc:
|
|
86
|
+
raise FeedError(f"{url} returned malformed JSON") from exc
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _retrying(url: str, body: bytes | None = None, attempts: int = 3) -> Any:
|
|
90
|
+
delay = 1.0
|
|
91
|
+
last: FeedError | None = None
|
|
92
|
+
for _ in range(attempts):
|
|
93
|
+
try:
|
|
94
|
+
return _request(url, body)
|
|
95
|
+
except FeedError as exc:
|
|
96
|
+
last = exc
|
|
97
|
+
time.sleep(delay)
|
|
98
|
+
delay *= 2
|
|
99
|
+
raise last # type: ignore[misc]
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
# --------------------------------------------------------------------------
|
|
103
|
+
# CISA KEV -- the authoritative "confirmed exploited" list
|
|
104
|
+
# --------------------------------------------------------------------------
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@dataclass(frozen=True)
|
|
108
|
+
class KevEntry:
|
|
109
|
+
cve: str
|
|
110
|
+
vendor: str
|
|
111
|
+
product: str
|
|
112
|
+
name: str
|
|
113
|
+
date_added: str
|
|
114
|
+
due_date: str
|
|
115
|
+
ransomware: bool
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def fetch_kev(refresh: bool = False) -> dict[str, KevEntry]:
|
|
119
|
+
payload = None if refresh else _cached("kev.json")
|
|
120
|
+
if payload is None:
|
|
121
|
+
payload = _retrying(KEV_URL)
|
|
122
|
+
_store("kev.json", payload)
|
|
123
|
+
|
|
124
|
+
entries: dict[str, KevEntry] = {}
|
|
125
|
+
for item in payload.get("vulnerabilities") or []:
|
|
126
|
+
cve = str(item.get("cveID", "")).upper().strip()
|
|
127
|
+
if not cve:
|
|
128
|
+
continue
|
|
129
|
+
entries[cve] = KevEntry(
|
|
130
|
+
cve=cve,
|
|
131
|
+
vendor=str(item.get("vendorProject", "")),
|
|
132
|
+
product=str(item.get("product", "")),
|
|
133
|
+
name=str(
|
|
134
|
+
item.get("shortDescription", "") or item.get("vulnerabilityName", "")
|
|
135
|
+
),
|
|
136
|
+
date_added=str(item.get("dateAdded", "")),
|
|
137
|
+
due_date=str(item.get("dueDate", "")),
|
|
138
|
+
ransomware=str(item.get("knownRansomwareCampaignUse", "")).lower()
|
|
139
|
+
== "known",
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
if not entries:
|
|
143
|
+
raise FeedError("CISA KEV feed parsed but contained no entries")
|
|
144
|
+
return entries
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
# --------------------------------------------------------------------------
|
|
148
|
+
# EPSS -- probability that a CVE is exploited in the next 30 days
|
|
149
|
+
# --------------------------------------------------------------------------
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def fetch_epss(cves: Sequence[str], batch_size: int = 100) -> dict[str, float]:
|
|
153
|
+
unique = sorted({c.upper() for c in cves if c.upper().startswith("CVE-")})
|
|
154
|
+
if not unique:
|
|
155
|
+
return {}
|
|
156
|
+
|
|
157
|
+
scores: dict[str, float] = {}
|
|
158
|
+
for start in range(0, len(unique), batch_size):
|
|
159
|
+
chunk = unique[start : start + batch_size]
|
|
160
|
+
url = f"{EPSS_URL}?cve={','.join(chunk)}"
|
|
161
|
+
payload = _retrying(url)
|
|
162
|
+
for row in payload.get("data") or []:
|
|
163
|
+
cve = str(row.get("cve", "")).upper()
|
|
164
|
+
try:
|
|
165
|
+
scores[cve] = float(row.get("epss", 0.0))
|
|
166
|
+
except (TypeError, ValueError):
|
|
167
|
+
continue
|
|
168
|
+
return scores
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
# --------------------------------------------------------------------------
|
|
172
|
+
# OSV -- component to vulnerability
|
|
173
|
+
# --------------------------------------------------------------------------
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
@dataclass(frozen=True)
|
|
177
|
+
class Vulnerability:
|
|
178
|
+
id: str
|
|
179
|
+
aliases: tuple[str, ...]
|
|
180
|
+
summary: str
|
|
181
|
+
severity: str | None
|
|
182
|
+
|
|
183
|
+
@property
|
|
184
|
+
def cves(self) -> tuple[str, ...]:
|
|
185
|
+
found = [a.upper() for a in self.aliases if a.upper().startswith("CVE-")]
|
|
186
|
+
if self.id.upper().startswith("CVE-"):
|
|
187
|
+
found.append(self.id.upper())
|
|
188
|
+
return tuple(dict.fromkeys(found))
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _osv_batch(purls: Sequence[str]) -> list[list[str]]:
|
|
192
|
+
"""Cheap first pass: which purls have anything at all against them."""
|
|
193
|
+
queries = [{"package": {"purl": purl}} for purl in purls]
|
|
194
|
+
body = json.dumps({"queries": queries}).encode("utf-8")
|
|
195
|
+
payload = _retrying(OSV_BATCH_URL, body)
|
|
196
|
+
|
|
197
|
+
results: list[list[str]] = []
|
|
198
|
+
for entry in payload.get("results") or []:
|
|
199
|
+
ids = [str(v.get("id")) for v in (entry.get("vulns") or []) if v.get("id")]
|
|
200
|
+
results.append(ids)
|
|
201
|
+
|
|
202
|
+
while len(results) < len(purls):
|
|
203
|
+
results.append([])
|
|
204
|
+
return results
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _osv_full(purl: str) -> tuple[str, list[dict]]:
|
|
208
|
+
"""Full advisory records for one package.
|
|
209
|
+
|
|
210
|
+
/v1/query returns complete records, where /v1/querybatch returns bare ids.
|
|
211
|
+
Fetching details id-by-id turns one slow package into a hundred requests --
|
|
212
|
+
a five-component SBOM measured 56 seconds that way. Querying per package
|
|
213
|
+
instead makes the cost scale with packages that have findings, which is a
|
|
214
|
+
small minority, rather than with the number of advisories.
|
|
215
|
+
"""
|
|
216
|
+
body = json.dumps({"package": {"purl": purl}}).encode("utf-8")
|
|
217
|
+
try:
|
|
218
|
+
payload = _retrying(OSV_QUERY_URL, body, attempts=2)
|
|
219
|
+
except FeedError:
|
|
220
|
+
return purl, []
|
|
221
|
+
return purl, [v for v in (payload.get("vulns") or []) if isinstance(v, dict)]
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _severity_of(record: dict) -> str | None:
|
|
225
|
+
for item in record.get("severity") or []:
|
|
226
|
+
if isinstance(item, dict) and item.get("score"):
|
|
227
|
+
return str(item["score"])
|
|
228
|
+
database = record.get("database_specific") or {}
|
|
229
|
+
if isinstance(database, dict) and database.get("severity"):
|
|
230
|
+
return str(database["severity"])
|
|
231
|
+
return None
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _to_vulnerability(record: dict) -> Vulnerability:
|
|
235
|
+
aliases = tuple(str(a) for a in (record.get("aliases") or []))
|
|
236
|
+
return Vulnerability(
|
|
237
|
+
id=str(record.get("id", "")),
|
|
238
|
+
aliases=aliases,
|
|
239
|
+
summary=str(record.get("summary") or "").strip(),
|
|
240
|
+
severity=_severity_of(record),
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def query_osv(
|
|
245
|
+
purls: Sequence[str], batch_size: int = 200, workers: int = 16
|
|
246
|
+
) -> tuple[dict[str, list[str]], dict[str, Vulnerability]]:
|
|
247
|
+
"""Map each purl to its OSV advisory ids, plus a detail lookup."""
|
|
248
|
+
# Pass one: find the packages worth asking about in detail.
|
|
249
|
+
flagged: list[str] = []
|
|
250
|
+
for start in range(0, len(purls), batch_size):
|
|
251
|
+
chunk = list(purls[start : start + batch_size])
|
|
252
|
+
for purl, ids in zip(chunk, _osv_batch(chunk)):
|
|
253
|
+
if ids:
|
|
254
|
+
flagged.append(purl)
|
|
255
|
+
|
|
256
|
+
by_purl: dict[str, list[str]] = {purl: [] for purl in purls}
|
|
257
|
+
details: dict[str, Vulnerability] = {}
|
|
258
|
+
if not flagged:
|
|
259
|
+
return by_purl, details
|
|
260
|
+
|
|
261
|
+
# Pass two: full records, only for those.
|
|
262
|
+
with ThreadPoolExecutor(max_workers=workers) as pool:
|
|
263
|
+
for purl, records in pool.map(_osv_full, flagged):
|
|
264
|
+
ids = []
|
|
265
|
+
for record in records:
|
|
266
|
+
vuln = _to_vulnerability(record)
|
|
267
|
+
if not vuln.id:
|
|
268
|
+
continue
|
|
269
|
+
details[vuln.id] = vuln
|
|
270
|
+
ids.append(vuln.id)
|
|
271
|
+
by_purl[purl] = ids
|
|
272
|
+
|
|
273
|
+
return by_purl, details
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: flagrante
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: Which of your components carry vulnerabilities confirmed exploited in the wild -- the ones that start a CRA Article 14 24-hour reporting clock
|
|
5
|
+
Author: Flagrante
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Project-URL: Homepage, https://flagrante.dev
|
|
8
|
+
Project-URL: Source, https://github.com/pacordelcw/flagrante
|
|
9
|
+
Project-URL: Issues, https://github.com/pacordelcw/flagrante/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/pacordelcw/flagrante/commits/main
|
|
11
|
+
Keywords: cra,cyber-resilience-act,article-14,sbom,cyclonedx,spdx,kev,epss,osv,vulnerability,compliance,eu
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Environment :: Console
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Intended Audience :: Information Technology
|
|
16
|
+
Classifier: Intended Audience :: Legal Industry
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Security
|
|
24
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
25
|
+
Classifier: Typing :: Typed
|
|
26
|
+
Requires-Python: >=3.10
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
License-File: LICENSE
|
|
29
|
+
Dynamic: license-file
|
|
30
|
+
|
|
31
|
+
# flagrante
|
|
32
|
+
|
|
33
|
+
**Which of your components start a CRA Article 14 24-hour clock.**
|
|
34
|
+
|
|
35
|
+
Since 11 September 2026, a manufacturer placing a product with digital elements
|
|
36
|
+
on the EU market must send an early warning within **24 hours** of becoming
|
|
37
|
+
aware that a vulnerability in that product is **actively exploited**.
|
|
38
|
+
|
|
39
|
+
Every SBOM scanner on the market tells you which components are *vulnerable*.
|
|
40
|
+
That is not the question Article 14 asks. Thousands of CVEs are vulnerabilities;
|
|
41
|
+
a few hundred are being exploited. Only the second group starts a clock.
|
|
42
|
+
|
|
43
|
+
`flagrante` makes that distinction the whole output.
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
24-HOUR CLOCK LIKELY RUNNING (2)
|
|
47
|
+
Confirmed exploited in the wild -- CISA KEV
|
|
48
|
+
|
|
49
|
+
CVE-2021-44228 org.apache.logging.log4j/log4j-core@2.14.1
|
|
50
|
+
listed in CISA KEV since 2021-12-10 as exploited in the wild, linked to ransomware campaigns
|
|
51
|
+
|
|
52
|
+
ASSESS TODAY (16)
|
|
53
|
+
Elevated exploitation probability, not yet confirmed
|
|
54
|
+
...
|
|
55
|
+
|
|
56
|
+
VERDICT
|
|
57
|
+
1 component carries 2 vulnerabilities confirmed exploited in the wild. If any
|
|
58
|
+
of these ship in a product you place on the EU market, assess Article 14
|
|
59
|
+
reporting now.
|
|
60
|
+
|
|
61
|
+
ARTICLE 14 CASCADE
|
|
62
|
+
Early warning 2026-09-04 17:10 UTC (24h)
|
|
63
|
+
Notification 2026-09-06 17:10 UTC (72h)
|
|
64
|
+
Final report 2026-09-17 17:10 UTC (14d)
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## Use it
|
|
68
|
+
|
|
69
|
+
You need an SBOM. If you do not have one, generate it free — `flagrante` does not
|
|
70
|
+
duplicate that job, because [syft](https://github.com/anchore/syft) and
|
|
71
|
+
[cdxgen](https://github.com/CycloneDX/cdxgen) already do it well.
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
pip install flagrante
|
|
75
|
+
syft dir:. -o cyclonedx-json | flagrante
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Or against a file, in CI, or as JSON:
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
flagrante sbom.json
|
|
82
|
+
flagrante sbom.json --json > result.json
|
|
83
|
+
flagrante sbom.json --all # include findings with no exploitation signal
|
|
84
|
+
flagrante sbom.json --fail-on urgent
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Exit codes: `0` nothing at the chosen level, `1` confirmed exploited,
|
|
88
|
+
`2` urgent (with `--fail-on urgent`), `3` the scan could not complete.
|
|
89
|
+
|
|
90
|
+
### GitHub Action
|
|
91
|
+
|
|
92
|
+
```yaml
|
|
93
|
+
- uses: actions/checkout@v4
|
|
94
|
+
- run: syft dir:. -o cyclonedx-json > sbom.json
|
|
95
|
+
- uses: pacordelcw/flagrante@v1
|
|
96
|
+
with:
|
|
97
|
+
sbom: sbom.json
|
|
98
|
+
fail-on: exploited
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
## How it decides
|
|
102
|
+
|
|
103
|
+
Three public feeds, no API keys:
|
|
104
|
+
|
|
105
|
+
| Feed | Answers |
|
|
106
|
+
|---|---|
|
|
107
|
+
| [OSV](https://osv.dev) | which components carry known vulnerabilities |
|
|
108
|
+
| [CISA KEV](https://www.cisa.gov/known-exploited-vulnerabilities-catalog) | which CVEs are **confirmed exploited in the wild** |
|
|
109
|
+
| [FIRST EPSS](https://www.first.org/epss/) | probability of exploitation within 30 days |
|
|
110
|
+
|
|
111
|
+
| Tier | Trigger |
|
|
112
|
+
|---|---|
|
|
113
|
+
| **24-hour clock likely running** | CVE listed in CISA KEV |
|
|
114
|
+
| **Assess today** | EPSS ≥ 10%, not in KEV |
|
|
115
|
+
| **Track** | known vulnerability, no exploitation signal |
|
|
116
|
+
| **Cannot be checked** | no package URL in the SBOM |
|
|
117
|
+
|
|
118
|
+
## What this is not
|
|
119
|
+
|
|
120
|
+
`flagrante` is an **exposure indicator**. It is not a conformity assessment, not
|
|
121
|
+
legal advice, and not a determination that no reporting obligation exists.
|
|
122
|
+
|
|
123
|
+
Three limits worth stating plainly, because a tool in this space that hides
|
|
124
|
+
them is not worth trusting:
|
|
125
|
+
|
|
126
|
+
1. **It cannot tell whether vulnerable code is reachable in your product.**
|
|
127
|
+
A KEV match means the CVE is exploited somewhere in the world, not that
|
|
128
|
+
*your* product is being attacked. Article 14 turns on the vulnerability
|
|
129
|
+
being in the product and actively exploited. That judgement is yours.
|
|
130
|
+
|
|
131
|
+
2. **"Actively exploited" is not perfectly determinable from public data.**
|
|
132
|
+
KEV lags real-world exploitation, and EPSS is a model, not an observation.
|
|
133
|
+
Absence of a signal here is not evidence of absence in the world.
|
|
134
|
+
|
|
135
|
+
3. **There is deliberately no "you are compliant" outcome.** Over-flagging
|
|
136
|
+
costs you an afternoon; under-flagging costs up to €15 million or 2.5% of
|
|
137
|
+
global turnover. Every threshold in this tool is set by that asymmetry, and
|
|
138
|
+
there is a test suite that fails the build if a future change ever softens
|
|
139
|
+
the wording into reassurance.
|
|
140
|
+
|
|
141
|
+
If a feed is unreachable, `flagrante` refuses to print a result rather than
|
|
142
|
+
printing an empty one — an unreachable exploitation feed and a clean scan look
|
|
143
|
+
identical and mean opposite things.
|
|
144
|
+
|
|
145
|
+
## Development
|
|
146
|
+
|
|
147
|
+
No dependencies, no network in the test suite:
|
|
148
|
+
|
|
149
|
+
```bash
|
|
150
|
+
python -m unittest discover -s tests
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
Apache-2.0.
|