coretrace-python-analyzer 0.4.0__py3-none-any.whl → 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- coretrace_python/__init__.py +1 -1
- coretrace_python/bundled/dependency/reachable_vulnerability/reachable_vulnerability.py +54 -18
- coretrace_python/bundled/dependency/vulnerable_dependency/plugin.toml +1 -1
- coretrace_python/bundled/dependency/vulnerable_dependency/vulnerable_dependency.py +32 -7
- coretrace_python/bundled/models/python_stdlib/python_stdlib.py +25 -1
- coretrace_python/cache.py +50 -11
- coretrace_python/cli.py +33 -1
- coretrace_python/dependency/__init__.py +2 -0
- coretrace_python/dependency/advisories.py +38 -17
- coretrace_python/dependency/correlation.py +69 -13
- coretrace_python/dependency/graph.py +83 -11
- coretrace_python/dependency/policy.py +6 -2
- coretrace_python/dependency/vex.py +182 -0
- coretrace_python/engine.py +26 -6
- coretrace_python/interprocedural/__init__.py +6 -0
- coretrace_python/interprocedural/callgraph.py +112 -11
- coretrace_python/interprocedural/summaries.py +8 -1
- coretrace_python/plugins/api.py +18 -4
- coretrace_python/taint/__init__.py +4 -0
- coretrace_python/taint/engine.py +91 -16
- coretrace_python/taint/models.py +24 -0
- {coretrace_python_analyzer-0.4.0.dist-info → coretrace_python_analyzer-0.6.0.dist-info}/METADATA +1 -1
- {coretrace_python_analyzer-0.4.0.dist-info → coretrace_python_analyzer-0.6.0.dist-info}/RECORD +27 -26
- {coretrace_python_analyzer-0.4.0.dist-info → coretrace_python_analyzer-0.6.0.dist-info}/WHEEL +0 -0
- {coretrace_python_analyzer-0.4.0.dist-info → coretrace_python_analyzer-0.6.0.dist-info}/entry_points.txt +0 -0
- {coretrace_python_analyzer-0.4.0.dist-info → coretrace_python_analyzer-0.6.0.dist-info}/licenses/LICENSE +0 -0
- {coretrace_python_analyzer-0.4.0.dist-info → coretrace_python_analyzer-0.6.0.dist-info}/licenses/NOTICE +0 -0
coretrace_python/__init__.py
CHANGED
|
@@ -1,16 +1,24 @@
|
|
|
1
|
-
"""Calls, anywhere in the project, to an API affected by a vulnerable requirement
|
|
1
|
+
"""Calls, anywhere in the project, to an API affected by a vulnerable requirement, and
|
|
2
|
+
reads of an affected attribute whose getter runs the vulnerable code."""
|
|
2
3
|
|
|
3
4
|
from __future__ import annotations
|
|
4
5
|
|
|
5
|
-
from collections.abc import Sequence
|
|
6
|
+
from collections.abc import Iterator, Sequence
|
|
6
7
|
from typing import ClassVar
|
|
7
8
|
|
|
8
9
|
from coretrace_python.analysis import AnyAnalysis
|
|
9
|
-
from coretrace_python.dependency import DependencyAnalysis
|
|
10
|
-
from coretrace_python.dependency.correlation import
|
|
10
|
+
from coretrace_python.dependency import Advisory, DependencyAnalysis
|
|
11
|
+
from coretrace_python.dependency.correlation import (
|
|
12
|
+
ConditionCheck,
|
|
13
|
+
affected_symbols,
|
|
14
|
+
check_conditions,
|
|
15
|
+
evidence,
|
|
16
|
+
)
|
|
11
17
|
from coretrace_python.findings import Confidence, Finding
|
|
12
18
|
from coretrace_python.interprocedural import CallGraphAnalysis, ExternalSymbol
|
|
13
19
|
from coretrace_python.plugins import ProjectContext, ProjectPlugin
|
|
20
|
+
from coretrace_python.semantic.symbols import SymbolId
|
|
21
|
+
from coretrace_python.source import SourceSpan
|
|
14
22
|
|
|
15
23
|
|
|
16
24
|
class ReachableVulnerabilityPlugin(ProjectPlugin):
|
|
@@ -29,18 +37,46 @@ class ReachableVulnerabilityPlugin(ProjectPlugin):
|
|
|
29
37
|
if not isinstance(site.target, ExternalSymbol):
|
|
30
38
|
continue
|
|
31
39
|
for advisory in affected.get(site.target.symbol, ()):
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
40
|
+
entry = advisory.entry_point(site.target.symbol)
|
|
41
|
+
if entry is not None and entry.read:
|
|
42
|
+
continue # a call reads its callee first: reported with the reads
|
|
43
|
+
check = check_conditions(entry, site.arguments)
|
|
44
|
+
if check.contradicted is not None:
|
|
45
|
+
continue
|
|
46
|
+
findings.append(_reached(advisory, site.target.symbol, site.location, function, check))
|
|
47
|
+
# Reading ``request.form.get`` reads ``request.form``: one finding per entry
|
|
48
|
+
# point and function, where the function first reads it.
|
|
49
|
+
reported: set[tuple[Advisory, SymbolId]] = set()
|
|
50
|
+
for read in graph.reads(function):
|
|
51
|
+
for symbol in _read_through(read.symbol):
|
|
52
|
+
for advisory in affected.get(symbol, ()):
|
|
53
|
+
entry = advisory.entry_point(symbol)
|
|
54
|
+
if entry is None or not entry.read or (advisory, symbol) in reported:
|
|
55
|
+
continue
|
|
56
|
+
reported.add((advisory, symbol))
|
|
57
|
+
check = check_conditions(entry, None)
|
|
58
|
+
findings.append(_reached(advisory, symbol, read.location, function, check))
|
|
46
59
|
return findings
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _read_through(symbol: SymbolId) -> Iterator[SymbolId]:
|
|
63
|
+
"""``symbol`` and every symbol it is an attribute of, as reading it reads them all."""
|
|
64
|
+
|
|
65
|
+
components = symbol.canonical_name.split(".")
|
|
66
|
+
for end in range(len(components), 1, -1):
|
|
67
|
+
yield SymbolId(".".join(components[:end]))
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _reached(advisory: Advisory, symbol: SymbolId, span: SourceSpan, function: str, check: ConditionCheck) -> Finding:
|
|
71
|
+
return Finding(
|
|
72
|
+
rule_id="reachable-vulnerability",
|
|
73
|
+
message=(
|
|
74
|
+
f"{advisory.id}: {symbol} is affected in the required "
|
|
75
|
+
f"{advisory.package} {advisory.vulnerable}: {advisory.summary}"
|
|
76
|
+
),
|
|
77
|
+
severity=advisory.severity,
|
|
78
|
+
confidence=Confidence.HIGH,
|
|
79
|
+
span=span,
|
|
80
|
+
function=function,
|
|
81
|
+
metadata=evidence(advisory, symbol, "reachable", check),
|
|
82
|
+
)
|
|
@@ -7,23 +7,34 @@ from typing import ClassVar
|
|
|
7
7
|
|
|
8
8
|
from coretrace_python.analysis import AnyAnalysis
|
|
9
9
|
from coretrace_python.dependency import DependencyAnalysis
|
|
10
|
+
from coretrace_python.dependency.correlation import affected_symbols, check_conditions, ruled_out
|
|
10
11
|
from coretrace_python.findings import Confidence, Finding
|
|
12
|
+
from coretrace_python.interprocedural import CallGraphAnalysis, ExternalSymbol
|
|
11
13
|
from coretrace_python.plugins import ProjectContext, ProjectPlugin
|
|
12
14
|
|
|
13
15
|
|
|
14
16
|
class VulnerableDependencyPlugin(ProjectPlugin):
|
|
15
17
|
name: ClassVar[str] = "vulnerable-dependency"
|
|
16
|
-
requires: ClassVar[frozenset[AnyAnalysis]] = frozenset({DependencyAnalysis})
|
|
18
|
+
requires: ClassVar[frozenset[AnyAnalysis]] = frozenset({DependencyAnalysis, CallGraphAnalysis})
|
|
17
19
|
|
|
18
20
|
def analyze_project(self, ctx: ProjectContext) -> Sequence[Finding]:
|
|
19
21
|
findings: list[Finding] = []
|
|
20
22
|
imported = [s for module in ctx.modules for s in ctx.imports(module).all_symbols()]
|
|
23
|
+
excluded = _ruled_out(ctx)
|
|
21
24
|
for requirement in ctx.dependencies.requirements:
|
|
22
25
|
for advisory in ctx.advisories:
|
|
23
26
|
if not advisory.affects(requirement):
|
|
24
27
|
continue
|
|
25
28
|
pinned = requirement.pinned is not None
|
|
26
29
|
level = "imported" if advisory.imported_by(imported) else "declared"
|
|
30
|
+
metadata = {
|
|
31
|
+
"advisory": advisory.id,
|
|
32
|
+
"package": advisory.package,
|
|
33
|
+
"specifier": requirement.specifier,
|
|
34
|
+
"level": level,
|
|
35
|
+
}
|
|
36
|
+
if excluded.get(advisory.id):
|
|
37
|
+
metadata["ruled_out"] = "; ".join(excluded[advisory.id])
|
|
27
38
|
findings.append(
|
|
28
39
|
Finding(
|
|
29
40
|
rule_id="vulnerable-dependency",
|
|
@@ -35,12 +46,26 @@ class VulnerableDependencyPlugin(ProjectPlugin):
|
|
|
35
46
|
severity=advisory.severity,
|
|
36
47
|
confidence=Confidence.HIGH if pinned else Confidence.MEDIUM,
|
|
37
48
|
span=requirement.span,
|
|
38
|
-
metadata=
|
|
39
|
-
"advisory": advisory.id,
|
|
40
|
-
"package": advisory.package,
|
|
41
|
-
"specifier": requirement.specifier,
|
|
42
|
-
"level": level,
|
|
43
|
-
},
|
|
49
|
+
metadata=metadata,
|
|
44
50
|
)
|
|
45
51
|
)
|
|
46
52
|
return findings
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _ruled_out(ctx: ProjectContext) -> dict[str, list[str]]:
|
|
56
|
+
"""By advisory, the calls to its entry points whose arguments contradict one of its
|
|
57
|
+
conditions: the evidence that the requirement, imported, is not reached there."""
|
|
58
|
+
|
|
59
|
+
affected = affected_symbols(ctx.dependencies, ctx.advisories)
|
|
60
|
+
excluded: dict[str, list[str]] = {}
|
|
61
|
+
for module in sorted(ctx.modules):
|
|
62
|
+
graph = ctx.call_graph(module)
|
|
63
|
+
for function in graph.functions:
|
|
64
|
+
for site in graph.sites(function):
|
|
65
|
+
if not isinstance(site.target, ExternalSymbol):
|
|
66
|
+
continue
|
|
67
|
+
for advisory in affected.get(site.target.symbol, ()):
|
|
68
|
+
check = check_conditions(advisory.entry_point(site.target.symbol), site.arguments)
|
|
69
|
+
if check.contradicted is not None:
|
|
70
|
+
excluded.setdefault(advisory.id, []).append(ruled_out(module, site, check))
|
|
71
|
+
return excluded
|
|
@@ -6,9 +6,28 @@ from typing import ClassVar
|
|
|
6
6
|
|
|
7
7
|
from coretrace_python.plugins import ModelPlugin
|
|
8
8
|
from coretrace_python.semantic.symbols import SymbolId
|
|
9
|
-
from coretrace_python.taint import
|
|
9
|
+
from coretrace_python.taint import (
|
|
10
|
+
Model,
|
|
11
|
+
SafeArgument,
|
|
12
|
+
Sanitizer,
|
|
13
|
+
Sink,
|
|
14
|
+
Source,
|
|
15
|
+
TaintKind,
|
|
16
|
+
Validator,
|
|
17
|
+
)
|
|
10
18
|
|
|
11
19
|
_ENVIRONMENT_KINDS = TaintKind.ALL & ~(TaintKind.COMMAND | TaintKind.PATH)
|
|
20
|
+
# SafeLoader and BaseLoader build plain data only, in Python and in C, under every spelling.
|
|
21
|
+
_SAFE_YAML_LOADERS = (
|
|
22
|
+
"python.yaml.SafeLoader",
|
|
23
|
+
"python.yaml.loader.SafeLoader",
|
|
24
|
+
"python.yaml.BaseLoader",
|
|
25
|
+
"python.yaml.loader.BaseLoader",
|
|
26
|
+
"python.yaml.CSafeLoader",
|
|
27
|
+
"python.yaml.cyaml.CSafeLoader",
|
|
28
|
+
"python.yaml.CBaseLoader",
|
|
29
|
+
"python.yaml.cyaml.CBaseLoader",
|
|
30
|
+
)
|
|
12
31
|
_PROCESS_OUTPUT_KINDS = TaintKind.ALL & ~TaintKind.PATH
|
|
13
32
|
|
|
14
33
|
|
|
@@ -66,6 +85,11 @@ class PythonStdlibModels(ModelPlugin):
|
|
|
66
85
|
Sink(_sym("dill.loads"), TaintKind.DESERIALIZATION),
|
|
67
86
|
Sink(_sym("jsonpickle.decode"), TaintKind.DESERIALIZATION),
|
|
68
87
|
Sink(_sym("yaml.load"), TaintKind.DESERIALIZATION),
|
|
88
|
+
Sink(_sym("yaml.load_all"), TaintKind.DESERIALIZATION),
|
|
89
|
+
Sink(_sym("yaml.full_load_all"), TaintKind.DESERIALIZATION),
|
|
90
|
+
Sink(_sym("yaml.unsafe_load_all"), TaintKind.DESERIALIZATION),
|
|
91
|
+
SafeArgument(_sym("yaml.load"), "Loader", _SAFE_YAML_LOADERS, position=1, kinds=TaintKind.DESERIALIZATION),
|
|
92
|
+
SafeArgument(_sym("yaml.load_all"), "Loader", _SAFE_YAML_LOADERS, position=1, kinds=TaintKind.DESERIALIZATION),
|
|
69
93
|
Sink(_sym("yaml.unsafe_load"), TaintKind.DESERIALIZATION),
|
|
70
94
|
Sink(_sym("yaml.full_load"), TaintKind.DESERIALIZATION),
|
|
71
95
|
Sanitizer(_sym("os.path.basename"), TaintKind.PATH),
|
coretrace_python/cache.py
CHANGED
|
@@ -4,9 +4,10 @@ A module's results are stored under a key derived from everything they depend on
|
|
|
4
4
|
source text and identity, the engine, schema and plugin API versions, the plugins and
|
|
5
5
|
their code, the security models, the advisories, the dependency graph, and the keys of
|
|
6
6
|
the project modules it imports transitively. A module whose key is unchanged on a later
|
|
7
|
-
run is served from the cache: its summaries seed the project index, its call sites
|
|
8
|
-
the project plugins and its findings are reported as they were.
|
|
9
|
-
tampered or foreign file can never execute anything; an
|
|
7
|
+
run is served from the cache: its summaries seed the project index, its call sites and
|
|
8
|
+
symbol reads serve the project plugins and its findings are reported as they were.
|
|
9
|
+
Entries are JSON, so a tampered or foreign file can never execute anything; an
|
|
10
|
+
unreadable entry is a miss.
|
|
10
11
|
"""
|
|
11
12
|
|
|
12
13
|
from __future__ import annotations
|
|
@@ -22,32 +23,36 @@ from typing import Any
|
|
|
22
23
|
|
|
23
24
|
from coretrace_python.findings import Confidence, Finding, Severity
|
|
24
25
|
from coretrace_python.interprocedural import (
|
|
26
|
+
Arguments,
|
|
25
27
|
CallSite,
|
|
26
28
|
ExternalCall,
|
|
27
29
|
ExternalSymbol,
|
|
28
30
|
FunctionSummary,
|
|
29
31
|
KnownFunction,
|
|
32
|
+
ModuleFunction,
|
|
30
33
|
ModuleGraph,
|
|
31
34
|
Mutation,
|
|
32
35
|
NonlocalWrite,
|
|
33
36
|
SummaryIndex,
|
|
37
|
+
SymbolRead,
|
|
34
38
|
Target,
|
|
35
39
|
UnknownTarget,
|
|
36
40
|
)
|
|
37
41
|
from coretrace_python.semantic.symbols import SymbolId
|
|
38
42
|
from coretrace_python.source import SourceId, SourceSpan
|
|
39
43
|
|
|
40
|
-
CACHE_FORMAT =
|
|
44
|
+
CACHE_FORMAT = 8
|
|
41
45
|
|
|
42
46
|
|
|
43
47
|
@dataclass(frozen=True)
|
|
44
48
|
class CachedModule:
|
|
45
49
|
"""Everything a later run needs from one module without lowering it again."""
|
|
46
50
|
|
|
47
|
-
functions: tuple[
|
|
51
|
+
functions: tuple[ModuleFunction, ...]
|
|
48
52
|
summaries: Mapping[str, FunctionSummary]
|
|
49
53
|
sites: tuple[CallSite, ...]
|
|
50
54
|
findings: tuple[Finding, ...]
|
|
55
|
+
reads: tuple[SymbolRead, ...] = ()
|
|
51
56
|
|
|
52
57
|
def __post_init__(self) -> None:
|
|
53
58
|
object.__setattr__(self, "summaries", MappingProxyType(dict(self.summaries)))
|
|
@@ -128,10 +133,11 @@ class ProjectCache:
|
|
|
128
133
|
def encode(module: CachedModule) -> dict[str, Any]:
|
|
129
134
|
return {
|
|
130
135
|
"format": CACHE_FORMAT,
|
|
131
|
-
"functions":
|
|
136
|
+
"functions": [[f.name, _encode_span(f.span), f.entry_point] for f in module.functions],
|
|
132
137
|
"summaries": {name: _encode_summary(s) for name, s in module.summaries.items()},
|
|
133
138
|
"sites": [_encode_site(site) for site in module.sites],
|
|
134
139
|
"findings": [_encode_finding(finding) for finding in module.findings],
|
|
140
|
+
"reads": [[r.function, _encode_span(r.location), str(r.symbol)] for r in module.reads],
|
|
135
141
|
}
|
|
136
142
|
|
|
137
143
|
|
|
@@ -139,10 +145,11 @@ def decode(data: Mapping[str, Any]) -> CachedModule:
|
|
|
139
145
|
if data["format"] != CACHE_FORMAT:
|
|
140
146
|
raise ValueError(f"unsupported cache format {data['format']!r}")
|
|
141
147
|
return CachedModule(
|
|
142
|
-
tuple(
|
|
148
|
+
tuple(_decode_function(function) for function in data["functions"]),
|
|
143
149
|
{_string(name): _decode_summary(s) for name, s in data["summaries"].items()},
|
|
144
150
|
tuple(_decode_site(site) for site in data["sites"]),
|
|
145
151
|
tuple(_decode_finding(finding) for finding in data["findings"]),
|
|
152
|
+
tuple(_decode_read(read) for read in data["reads"]),
|
|
146
153
|
)
|
|
147
154
|
|
|
148
155
|
|
|
@@ -187,6 +194,16 @@ def _decode_span(data: Any) -> SourceSpan:
|
|
|
187
194
|
)
|
|
188
195
|
|
|
189
196
|
|
|
197
|
+
def _decode_function(data: Any) -> ModuleFunction:
|
|
198
|
+
name, span, entry_point = data
|
|
199
|
+
return ModuleFunction(_string(name), _decode_span(span), None if entry_point is None else _string(entry_point))
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _decode_read(data: Any) -> SymbolRead:
|
|
203
|
+
function, span, symbol = data
|
|
204
|
+
return SymbolRead(_string(function), _decode_span(span), SymbolId(_string(symbol)))
|
|
205
|
+
|
|
206
|
+
|
|
190
207
|
def _encode_finding(finding: Finding) -> dict[str, Any]:
|
|
191
208
|
return {
|
|
192
209
|
"rule": finding.rule_id,
|
|
@@ -219,6 +236,7 @@ def _encode_call(call: ExternalCall) -> dict[str, Any]:
|
|
|
219
236
|
"keywords": sorted(call.keyword_dependencies),
|
|
220
237
|
"location": _encode_span(call.location),
|
|
221
238
|
"call_site": None if call.call_site is None else _encode_span(call.call_site),
|
|
239
|
+
"given": _encode_arguments(call.arguments),
|
|
222
240
|
}
|
|
223
241
|
|
|
224
242
|
|
|
@@ -230,6 +248,7 @@ def _decode_call(data: Mapping[str, Any]) -> ExternalCall:
|
|
|
230
248
|
_indices(data["keywords"]),
|
|
231
249
|
_decode_span(data["location"]),
|
|
232
250
|
None if site is None else _decode_span(site),
|
|
251
|
+
_decode_arguments(data["given"]),
|
|
233
252
|
)
|
|
234
253
|
|
|
235
254
|
|
|
@@ -311,8 +330,7 @@ def _encode_site(site: CallSite) -> dict[str, Any]:
|
|
|
311
330
|
"caller": site.caller,
|
|
312
331
|
"location": _encode_span(site.location),
|
|
313
332
|
"target": _encode_target(site.target),
|
|
314
|
-
"arguments": site.arguments,
|
|
315
|
-
"keywords": site.keywords,
|
|
333
|
+
"arguments": _encode_arguments(site.arguments),
|
|
316
334
|
}
|
|
317
335
|
|
|
318
336
|
|
|
@@ -321,6 +339,27 @@ def _decode_site(data: Mapping[str, Any]) -> CallSite:
|
|
|
321
339
|
_string(data["caller"]),
|
|
322
340
|
_decode_span(data["location"]),
|
|
323
341
|
_decode_target(data["target"]),
|
|
324
|
-
|
|
325
|
-
|
|
342
|
+
_decode_arguments(data["arguments"]),
|
|
343
|
+
)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _encode_arguments(arguments: Arguments) -> dict[str, Any]:
|
|
347
|
+
return {
|
|
348
|
+
"positional": list(arguments.positional),
|
|
349
|
+
"keywords": [[name, value] for name, value in arguments.keywords],
|
|
350
|
+
"unpacked": arguments.unpacked,
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def _decode_arguments(data: Mapping[str, Any]) -> Arguments:
|
|
355
|
+
def denoted(value: Any) -> str | None:
|
|
356
|
+
return None if value is None else _string(value)
|
|
357
|
+
|
|
358
|
+
unpacked = data["unpacked"]
|
|
359
|
+
if not isinstance(unpacked, bool):
|
|
360
|
+
raise TypeError(f"expected a boolean, got {unpacked!r}")
|
|
361
|
+
return Arguments(
|
|
362
|
+
tuple(denoted(value) for value in data["positional"]),
|
|
363
|
+
tuple((_string(name), denoted(value)) for name, value in data["keywords"]),
|
|
364
|
+
unpacked,
|
|
326
365
|
)
|
coretrace_python/cli.py
CHANGED
|
@@ -3,13 +3,20 @@ from __future__ import annotations
|
|
|
3
3
|
import argparse
|
|
4
4
|
import sys
|
|
5
5
|
import zipfile
|
|
6
|
+
from datetime import UTC, datetime
|
|
6
7
|
from pathlib import Path
|
|
7
8
|
|
|
8
9
|
from coretrace_python import __version__, engine
|
|
9
10
|
from coretrace_python.analysis import AnalysisError
|
|
10
11
|
from coretrace_python.cache import ProjectCache
|
|
11
12
|
from coretrace_python.cfg import CFGError
|
|
12
|
-
from coretrace_python.dependency import
|
|
13
|
+
from coretrace_python.dependency import (
|
|
14
|
+
dump_advisories,
|
|
15
|
+
import_osv,
|
|
16
|
+
read_osv,
|
|
17
|
+
render_sbom,
|
|
18
|
+
render_vex,
|
|
19
|
+
)
|
|
13
20
|
from coretrace_python.findings import Severity
|
|
14
21
|
from coretrace_python.findings.baseline import Baseline, BaselineError
|
|
15
22
|
from coretrace_python.frontend import HIRBuildError, ParseError, build_hir
|
|
@@ -106,6 +113,14 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
106
113
|
metavar="PATH",
|
|
107
114
|
help="with --check on a directory, write a CycloneDX bill of materials to PATH",
|
|
108
115
|
)
|
|
116
|
+
parser.add_argument(
|
|
117
|
+
"--vex",
|
|
118
|
+
type=Path,
|
|
119
|
+
default=None,
|
|
120
|
+
metavar="PATH",
|
|
121
|
+
help="with --check on a directory, write to PATH an OpenVEX document saying whether "
|
|
122
|
+
"each advisory affecting a requirement affects the project",
|
|
123
|
+
)
|
|
109
124
|
parser.add_argument(
|
|
110
125
|
"--advisories",
|
|
111
126
|
action="append",
|
|
@@ -182,6 +197,9 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
182
197
|
if args.sbom is not None and not (args.check and args.path.is_dir()):
|
|
183
198
|
print("error: --sbom only applies to --check on a directory", file=sys.stderr)
|
|
184
199
|
return EXIT_ERROR
|
|
200
|
+
if args.vex is not None and not (args.check and args.path.is_dir()):
|
|
201
|
+
print("error: --vex only applies to --check on a directory", file=sys.stderr)
|
|
202
|
+
return EXIT_ERROR
|
|
185
203
|
if args.advisories and not (args.check and args.path.is_dir()):
|
|
186
204
|
print("error: --advisories only applies to --check on a directory", file=sys.stderr)
|
|
187
205
|
return EXIT_ERROR
|
|
@@ -227,6 +245,20 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
227
245
|
render_sbom(analysis.dependencies, analysis.advisories, engine.TOOL_NAME, __version__),
|
|
228
246
|
encoding="utf-8",
|
|
229
247
|
)
|
|
248
|
+
if args.vex is not None:
|
|
249
|
+
args.vex.write_text(
|
|
250
|
+
render_vex(
|
|
251
|
+
analysis.dependencies,
|
|
252
|
+
analysis.advisories,
|
|
253
|
+
(*analysis.findings, *analysis.suppressed, *analysis.accepted),
|
|
254
|
+
analysis.coverage,
|
|
255
|
+
args.path,
|
|
256
|
+
engine.TOOL_NAME,
|
|
257
|
+
__version__,
|
|
258
|
+
datetime.now(UTC),
|
|
259
|
+
),
|
|
260
|
+
encoding="utf-8",
|
|
261
|
+
)
|
|
230
262
|
else:
|
|
231
263
|
file_analysis = engine.analyze_file(SourceManager().load_file(args.path), plugin_roots)
|
|
232
264
|
findings, coverage = file_analysis.findings, file_analysis.coverage
|
|
@@ -22,6 +22,7 @@ from coretrace_python.dependency.graph import (
|
|
|
22
22
|
)
|
|
23
23
|
from coretrace_python.dependency.policy import POLICY_FILE, Policy, apply_policy, load_policy
|
|
24
24
|
from coretrace_python.dependency.sbom import render_sbom
|
|
25
|
+
from coretrace_python.dependency.vex import render_vex
|
|
25
26
|
|
|
26
27
|
__all__ = [
|
|
27
28
|
"ADVISORY_FILE",
|
|
@@ -45,4 +46,5 @@ __all__ = [
|
|
|
45
46
|
"parse_dependencies",
|
|
46
47
|
"read_osv",
|
|
47
48
|
"render_sbom",
|
|
49
|
+
"render_vex",
|
|
48
50
|
]
|
|
@@ -138,14 +138,7 @@ def dump_advisories(advisories: Iterable[Advisory]) -> str:
|
|
|
138
138
|
"severity": a.severity.value,
|
|
139
139
|
"affected_symbols": [str(s) for s in a.affected_symbols],
|
|
140
140
|
"aliases": list(a.aliases),
|
|
141
|
-
"entry_points": [
|
|
142
|
-
{
|
|
143
|
-
"symbol": str(e.symbol),
|
|
144
|
-
"justification": e.justification,
|
|
145
|
-
"conditions": [_condition_entry(c) for c in e.conditions],
|
|
146
|
-
}
|
|
147
|
-
for e in a.entry_points
|
|
148
|
-
],
|
|
141
|
+
"entry_points": [_entry_point_entry(e) for e in a.entry_points],
|
|
149
142
|
"modules": list(a.modules),
|
|
150
143
|
}
|
|
151
144
|
for a in advisories
|
|
@@ -154,12 +147,27 @@ def dump_advisories(advisories: Iterable[Advisory]) -> str:
|
|
|
154
147
|
return json.dumps(document, indent=2) + "\n"
|
|
155
148
|
|
|
156
149
|
|
|
150
|
+
def _entry_point_entry(entry_point: AdvisoryEntryPoint) -> dict[str, Any]:
|
|
151
|
+
entry: dict[str, Any] = {
|
|
152
|
+
"symbol": str(entry_point.symbol),
|
|
153
|
+
"justification": entry_point.justification,
|
|
154
|
+
"conditions": [_condition_entry(c) for c in entry_point.conditions],
|
|
155
|
+
}
|
|
156
|
+
if entry_point.read:
|
|
157
|
+
entry["read"] = True
|
|
158
|
+
return entry
|
|
159
|
+
|
|
160
|
+
|
|
157
161
|
def _condition_entry(condition: Condition) -> dict[str, Any]:
|
|
158
162
|
entry: dict[str, Any] = {"kind": condition.kind, "text": condition.text}
|
|
159
163
|
if condition.argument is not None:
|
|
160
164
|
entry["argument"] = condition.argument
|
|
161
165
|
if condition.values:
|
|
162
166
|
entry["values"] = list(condition.values)
|
|
167
|
+
if condition.position is not None:
|
|
168
|
+
entry["position"] = condition.position
|
|
169
|
+
if condition.default:
|
|
170
|
+
entry["default"] = True
|
|
163
171
|
return entry
|
|
164
172
|
|
|
165
173
|
|
|
@@ -190,16 +198,29 @@ def _advisory(entry: Mapping[str, Any]) -> Advisory:
|
|
|
190
198
|
|
|
191
199
|
|
|
192
200
|
def _entry_point(entry: Mapping[str, Any]) -> AdvisoryEntryPoint:
|
|
201
|
+
read = entry.get("read", False)
|
|
202
|
+
if not isinstance(read, bool):
|
|
203
|
+
raise TypeError(f"entry point read must be true or false, got {read!r}")
|
|
193
204
|
return AdvisoryEntryPoint(
|
|
194
205
|
SymbolId(str(entry["symbol"])),
|
|
195
206
|
str(entry["justification"]),
|
|
196
|
-
tuple(
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
)
|
|
207
|
+
tuple(_condition(c) for c in entry.get("conditions") or []),
|
|
208
|
+
read,
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _condition(entry: Mapping[str, Any]) -> Condition:
|
|
213
|
+
position = entry.get("position")
|
|
214
|
+
if position is not None and (not isinstance(position, int) or isinstance(position, bool) or position < 0):
|
|
215
|
+
raise TypeError(f"condition position must be a non-negative integer, got {position!r}")
|
|
216
|
+
default = entry.get("default", False)
|
|
217
|
+
if not isinstance(default, bool):
|
|
218
|
+
raise TypeError(f"condition default must be true or false, got {default!r}")
|
|
219
|
+
return Condition(
|
|
220
|
+
str(entry["kind"]),
|
|
221
|
+
str(entry["text"]),
|
|
222
|
+
None if entry.get("argument") is None else str(entry["argument"]),
|
|
223
|
+
tuple(str(v) for v in entry.get("values") or []),
|
|
224
|
+
position,
|
|
225
|
+
default,
|
|
205
226
|
)
|
|
@@ -10,10 +10,18 @@ are correlated here into one high-confidence ``exploitable-vulnerability`` findi
|
|
|
10
10
|
from __future__ import annotations
|
|
11
11
|
|
|
12
12
|
from collections.abc import Iterable, Mapping
|
|
13
|
-
|
|
14
|
-
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
|
|
15
|
+
from coretrace_python.dependency.graph import (
|
|
16
|
+
DIRECT,
|
|
17
|
+
Advisory,
|
|
18
|
+
AdvisoryEntryPoint,
|
|
19
|
+
Condition,
|
|
20
|
+
DependencyGraph,
|
|
21
|
+
)
|
|
15
22
|
from coretrace_python.findings import Confidence, Finding, Severity
|
|
16
23
|
from coretrace_python.findings.refutation import Status, Verdict, Verdicts
|
|
24
|
+
from coretrace_python.interprocedural import Arguments, CallSite, ExternalSymbol
|
|
17
25
|
from coretrace_python.semantic.symbols import SymbolId
|
|
18
26
|
from coretrace_python.taint import Sink, TaintFlow, TaintKind
|
|
19
27
|
|
|
@@ -40,10 +48,56 @@ def advisory_sinks(affected: Affected) -> tuple[Sink, ...]:
|
|
|
40
48
|
return tuple(Sink(symbol, TaintKind.ADVISORY) for symbol in affected)
|
|
41
49
|
|
|
42
50
|
|
|
43
|
-
|
|
51
|
+
@dataclass(frozen=True)
|
|
52
|
+
class ConditionCheck:
|
|
53
|
+
"""What one call tells of an entry point's conditions: those it meets, those left to
|
|
54
|
+
review (every semantic one, and every argument one the call does not decide), and
|
|
55
|
+
the argument condition it contradicts with what it passes instead (None: absent),
|
|
56
|
+
in which case the call does not reach the vulnerability."""
|
|
57
|
+
|
|
58
|
+
met: tuple[Condition, ...] = ()
|
|
59
|
+
pending: tuple[Condition, ...] = ()
|
|
60
|
+
contradicted: Condition | None = None
|
|
61
|
+
passed: str | None = None
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def check_conditions(entry: AdvisoryEntryPoint | None, arguments: Arguments | None) -> ConditionCheck:
|
|
65
|
+
"""Decide ``entry``'s conditions against what the call's ``arguments`` denote."""
|
|
66
|
+
|
|
67
|
+
if entry is None:
|
|
68
|
+
return ConditionCheck()
|
|
69
|
+
met: list[Condition] = []
|
|
70
|
+
pending: list[Condition] = []
|
|
71
|
+
for condition in entry.conditions:
|
|
72
|
+
given = arguments.given(condition.argument, condition.position) if condition.checkable and arguments is not None else None
|
|
73
|
+
if given is None:
|
|
74
|
+
pending.append(condition)
|
|
75
|
+
continue
|
|
76
|
+
explicit, value = given
|
|
77
|
+
if explicit and value is None:
|
|
78
|
+
pending.append(condition)
|
|
79
|
+
continue
|
|
80
|
+
affected = value in condition.values if explicit else condition.default
|
|
81
|
+
if not affected:
|
|
82
|
+
return ConditionCheck(tuple(met), tuple(pending), condition, value)
|
|
83
|
+
met.append(condition)
|
|
84
|
+
return ConditionCheck(tuple(met), tuple(pending))
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def ruled_out(module: str, site: CallSite, check: ConditionCheck) -> str:
|
|
88
|
+
"""Which call a contradicted condition rules out, and why: ``app:12
|
|
89
|
+
python.yaml.load(Loader=python.yaml.SafeLoader)``."""
|
|
90
|
+
|
|
91
|
+
assert check.contradicted is not None and isinstance(site.target, ExternalSymbol)
|
|
92
|
+
argument = check.contradicted.argument or f"#{check.contradicted.position}"
|
|
93
|
+
passed = f"{argument} absent" if check.passed is None else f"{argument}={check.passed}"
|
|
94
|
+
return f"{module}:{site.location.start_line} {site.target.symbol}({passed})"
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def evidence(advisory: Advisory, symbol: SymbolId, level: str, check: ConditionCheck) -> dict[str, str]:
|
|
44
98
|
"""What a finding keeps of the advisory for ``symbol``: the level of evidence
|
|
45
|
-
established, how the symbol relates to the vulnerability and
|
|
46
|
-
|
|
99
|
+
established, how the symbol relates to the vulnerability, and its conditions — those
|
|
100
|
+
the call meets and those left to review."""
|
|
47
101
|
|
|
48
102
|
metadata = {"advisory": advisory.id, "package": advisory.package, "symbol": str(symbol), "level": level}
|
|
49
103
|
entry = advisory.entry_point(symbol)
|
|
@@ -54,9 +108,10 @@ def evidence(advisory: Advisory, symbol: SymbolId, level: str) -> dict[str, str]
|
|
|
54
108
|
metadata["justification"] = entry.justification
|
|
55
109
|
if entry.conditions:
|
|
56
110
|
metadata["conditions"] = "; ".join(c.text for c in entry.conditions)
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
111
|
+
if check.met:
|
|
112
|
+
metadata["conditions_met"] = "; ".join(c.text for c in check.met)
|
|
113
|
+
if check.pending:
|
|
114
|
+
metadata["conditions_pending_review"] = "; ".join(c.text for c in check.pending)
|
|
60
115
|
return metadata
|
|
61
116
|
|
|
62
117
|
|
|
@@ -77,21 +132,22 @@ def correlate(
|
|
|
77
132
|
if verdict is not None and verdict.status is Status.REFUTED:
|
|
78
133
|
continue
|
|
79
134
|
hotspot = verdict is not None and verdict.status is Status.HOTSPOT
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
135
|
+
for advisory in affected.get(flow.sink.symbol, ()):
|
|
136
|
+
check = check_conditions(advisory.entry_point(flow.sink.symbol), flow.sink_arguments)
|
|
137
|
+
if check.contradicted is None:
|
|
138
|
+
findings.append(_exploitable(function, flow, advisory, verdict, hotspot, check))
|
|
83
139
|
return tuple(findings)
|
|
84
140
|
|
|
85
141
|
|
|
86
142
|
def _exploitable(
|
|
87
|
-
function: str, flow: TaintFlow, advisory: Advisory, verdict: Verdict | None, hotspot: bool
|
|
143
|
+
function: str, flow: TaintFlow, advisory: Advisory, verdict: Verdict | None, hotspot: bool, check: ConditionCheck
|
|
88
144
|
) -> Finding:
|
|
89
145
|
message = (
|
|
90
146
|
f"{advisory.id}: {flow.source.label} input reaches {flow.sink.symbol}, affected in "
|
|
91
147
|
f"the required {advisory.package} {advisory.vulnerable}: {advisory.summary}"
|
|
92
148
|
)
|
|
93
149
|
metadata = {
|
|
94
|
-
**evidence(advisory, flow.sink.symbol, "exploitable"),
|
|
150
|
+
**evidence(advisory, flow.sink.symbol, "exploitable", check),
|
|
95
151
|
"source": str(flow.source.symbol),
|
|
96
152
|
"source_label": flow.source.label,
|
|
97
153
|
"verdict": "hotspot" if hotspot else "vulnerability",
|