coretrace-python-analyzer 0.3.0__py3-none-any.whl → 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. coretrace_python/__init__.py +1 -1
  2. coretrace_python/bundled/dependency/reachable_vulnerability/reachable_vulnerability.py +19 -26
  3. coretrace_python/bundled/dependency/sample_advisories/sample_advisories.py +2 -0
  4. coretrace_python/bundled/dependency/vulnerable_dependency/plugin.toml +1 -1
  5. coretrace_python/bundled/dependency/vulnerable_dependency/vulnerable_dependency.py +34 -6
  6. coretrace_python/cache.py +37 -8
  7. coretrace_python/dependency/__init__.py +4 -0
  8. coretrace_python/dependency/advisories.py +55 -5
  9. coretrace_python/dependency/correlation.py +139 -45
  10. coretrace_python/dependency/graph.py +79 -5
  11. coretrace_python/engine.py +26 -11
  12. coretrace_python/interprocedural/__init__.py +4 -0
  13. coretrace_python/interprocedural/callgraph.py +71 -11
  14. coretrace_python/interprocedural/modulegraph.py +4 -1
  15. coretrace_python/interprocedural/summaries.py +8 -1
  16. coretrace_python/ir/lowering.py +57 -3
  17. coretrace_python/plugins/__init__.py +4 -0
  18. coretrace_python/plugins/api.py +65 -5
  19. coretrace_python/plugins/loader.py +4 -4
  20. coretrace_python/semantic/scopes.py +3 -1
  21. coretrace_python/taint/__init__.py +2 -0
  22. coretrace_python/taint/engine.py +85 -16
  23. {coretrace_python_analyzer-0.3.0.dist-info → coretrace_python_analyzer-0.5.0.dist-info}/METADATA +1 -1
  24. {coretrace_python_analyzer-0.3.0.dist-info → coretrace_python_analyzer-0.5.0.dist-info}/RECORD +28 -28
  25. {coretrace_python_analyzer-0.3.0.dist-info → coretrace_python_analyzer-0.5.0.dist-info}/WHEEL +0 -0
  26. {coretrace_python_analyzer-0.3.0.dist-info → coretrace_python_analyzer-0.5.0.dist-info}/entry_points.txt +0 -0
  27. {coretrace_python_analyzer-0.3.0.dist-info → coretrace_python_analyzer-0.5.0.dist-info}/licenses/LICENSE +0 -0
  28. {coretrace_python_analyzer-0.3.0.dist-info → coretrace_python_analyzer-0.5.0.dist-info}/licenses/NOTICE +0 -0
@@ -1,4 +1,4 @@
1
1
  """CoreTrace's Python static analysis frontend."""
2
2
 
3
- __version__ = "0.3.0"
3
+ __version__ = "0.5.0"
4
4
 
@@ -7,6 +7,7 @@ from typing import ClassVar
7
7
 
8
8
  from coretrace_python.analysis import AnyAnalysis
9
9
  from coretrace_python.dependency import DependencyAnalysis
10
+ from coretrace_python.dependency.correlation import affected_symbols, check_conditions, evidence
10
11
  from coretrace_python.findings import Confidence, Finding
11
12
  from coretrace_python.interprocedural import CallGraphAnalysis, ExternalSymbol
12
13
  from coretrace_python.plugins import ProjectContext, ProjectPlugin
@@ -17,12 +18,7 @@ class ReachableVulnerabilityPlugin(ProjectPlugin):
17
18
  requires: ClassVar[frozenset[AnyAnalysis]] = frozenset({DependencyAnalysis, CallGraphAnalysis})
18
19
 
19
20
  def analyze_project(self, ctx: ProjectContext) -> Sequence[Finding]:
20
- affected = {}
21
- for requirement in ctx.dependencies.requirements:
22
- for advisory in ctx.advisories:
23
- if advisory.affects(requirement):
24
- for symbol in advisory.affected_symbols:
25
- affected.setdefault(symbol, advisory)
21
+ affected = affected_symbols(ctx.dependencies, ctx.advisories)
26
22
  if not affected:
27
23
  return ()
28
24
  findings: list[Finding] = []
@@ -32,25 +28,22 @@ class ReachableVulnerabilityPlugin(ProjectPlugin):
32
28
  for site in graph.sites(function):
33
29
  if not isinstance(site.target, ExternalSymbol):
34
30
  continue
35
- advisory = affected.get(site.target.symbol)
36
- if advisory is None:
37
- continue
38
- findings.append(
39
- Finding(
40
- rule_id="reachable-vulnerability",
41
- message=(
42
- f"{advisory.id}: {site.target.symbol} is affected in the required "
43
- f"{advisory.package} {advisory.vulnerable}: {advisory.summary}"
44
- ),
45
- severity=advisory.severity,
46
- confidence=Confidence.HIGH,
47
- span=site.location,
48
- function=function,
49
- metadata={
50
- "advisory": advisory.id,
51
- "package": advisory.package,
52
- "symbol": str(site.target.symbol),
53
- },
31
+ for advisory in affected.get(site.target.symbol, ()):
32
+ check = check_conditions(advisory.entry_point(site.target.symbol), site.arguments)
33
+ if check.contradicted is not None:
34
+ continue
35
+ findings.append(
36
+ Finding(
37
+ rule_id="reachable-vulnerability",
38
+ message=(
39
+ f"{advisory.id}: {site.target.symbol} is affected in the required "
40
+ f"{advisory.package} {advisory.vulnerable}: {advisory.summary}"
41
+ ),
42
+ severity=advisory.severity,
43
+ confidence=Confidence.HIGH,
44
+ span=site.location,
45
+ function=function,
46
+ metadata=evidence(advisory, site.target.symbol, "reachable", check),
47
+ )
54
48
  )
55
- )
56
49
  return findings
@@ -28,6 +28,7 @@ class SampleAdvisories(ModelPlugin):
28
28
  "yaml.load and full_load can execute arbitrary code from untrusted documents",
29
29
  Severity.CRITICAL,
30
30
  (_sym("yaml.load"), _sym("yaml.full_load"), _sym("yaml.unsafe_load")),
31
+ modules=("yaml",),
31
32
  ),
32
33
  Advisory(
33
34
  "CVE-2018-18074",
@@ -105,5 +106,6 @@ class SampleAdvisories(ModelPlugin):
105
106
  Severity.HIGH,
106
107
  (_sym("PIL.Image.open"),),
107
108
  ("GHSA-cfh3-3jmp-rvhc",),
109
+ modules=("PIL",),
108
110
  ),
109
111
  )
@@ -1,7 +1,7 @@
1
1
  name = "vulnerable-dependency"
2
2
  version = "1.0.0"
3
3
  plugin_api = ">=1,<2"
4
- requires = ["dependency.graph"]
4
+ requires = ["dependency.graph", "interprocedural.callgraph"]
5
5
  provides = ["vulnerability.vulnerable-dependency"]
6
6
 
7
7
  [entrypoint]
@@ -7,21 +7,34 @@ from typing import ClassVar
7
7
 
8
8
  from coretrace_python.analysis import AnyAnalysis
9
9
  from coretrace_python.dependency import DependencyAnalysis
10
+ from coretrace_python.dependency.correlation import affected_symbols, check_conditions, ruled_out
10
11
  from coretrace_python.findings import Confidence, Finding
12
+ from coretrace_python.interprocedural import CallGraphAnalysis, ExternalSymbol
11
13
  from coretrace_python.plugins import ProjectContext, ProjectPlugin
12
14
 
13
15
 
14
16
  class VulnerableDependencyPlugin(ProjectPlugin):
15
17
  name: ClassVar[str] = "vulnerable-dependency"
16
- requires: ClassVar[frozenset[AnyAnalysis]] = frozenset({DependencyAnalysis})
18
+ requires: ClassVar[frozenset[AnyAnalysis]] = frozenset({DependencyAnalysis, CallGraphAnalysis})
17
19
 
18
20
  def analyze_project(self, ctx: ProjectContext) -> Sequence[Finding]:
19
21
  findings: list[Finding] = []
22
+ imported = [s for module in ctx.modules for s in ctx.imports(module).all_symbols()]
23
+ excluded = _ruled_out(ctx)
20
24
  for requirement in ctx.dependencies.requirements:
21
25
  for advisory in ctx.advisories:
22
26
  if not advisory.affects(requirement):
23
27
  continue
24
28
  pinned = requirement.pinned is not None
29
+ level = "imported" if advisory.imported_by(imported) else "declared"
30
+ metadata = {
31
+ "advisory": advisory.id,
32
+ "package": advisory.package,
33
+ "specifier": requirement.specifier,
34
+ "level": level,
35
+ }
36
+ if excluded.get(advisory.id):
37
+ metadata["ruled_out"] = "; ".join(excluded[advisory.id])
25
38
  findings.append(
26
39
  Finding(
27
40
  rule_id="vulnerable-dependency",
@@ -33,11 +46,26 @@ class VulnerableDependencyPlugin(ProjectPlugin):
33
46
  severity=advisory.severity,
34
47
  confidence=Confidence.HIGH if pinned else Confidence.MEDIUM,
35
48
  span=requirement.span,
36
- metadata={
37
- "advisory": advisory.id,
38
- "package": advisory.package,
39
- "specifier": requirement.specifier,
40
- },
49
+ metadata=metadata,
41
50
  )
42
51
  )
43
52
  return findings
53
+
54
+
55
+ def _ruled_out(ctx: ProjectContext) -> dict[str, list[str]]:
56
+ """By advisory, the calls to its entry points whose arguments contradict one of its
57
+ conditions: the evidence that the requirement, imported, is not reached there."""
58
+
59
+ affected = affected_symbols(ctx.dependencies, ctx.advisories)
60
+ excluded: dict[str, list[str]] = {}
61
+ for module in sorted(ctx.modules):
62
+ graph = ctx.call_graph(module)
63
+ for function in graph.functions:
64
+ for site in graph.sites(function):
65
+ if not isinstance(site.target, ExternalSymbol):
66
+ continue
67
+ for advisory in affected.get(site.target.symbol, ()):
68
+ check = check_conditions(advisory.entry_point(site.target.symbol), site.arguments)
69
+ if check.contradicted is not None:
70
+ excluded.setdefault(advisory.id, []).append(ruled_out(module, site, check))
71
+ return excluded
coretrace_python/cache.py CHANGED
@@ -22,11 +22,13 @@ from typing import Any
22
22
 
23
23
  from coretrace_python.findings import Confidence, Finding, Severity
24
24
  from coretrace_python.interprocedural import (
25
+ Arguments,
25
26
  CallSite,
26
27
  ExternalCall,
27
28
  ExternalSymbol,
28
29
  FunctionSummary,
29
30
  KnownFunction,
31
+ ModuleFunction,
30
32
  ModuleGraph,
31
33
  Mutation,
32
34
  NonlocalWrite,
@@ -37,14 +39,14 @@ from coretrace_python.interprocedural import (
37
39
  from coretrace_python.semantic.symbols import SymbolId
38
40
  from coretrace_python.source import SourceId, SourceSpan
39
41
 
40
- CACHE_FORMAT = 4
42
+ CACHE_FORMAT = 7
41
43
 
42
44
 
43
45
  @dataclass(frozen=True)
44
46
  class CachedModule:
45
47
  """Everything a later run needs from one module without lowering it again."""
46
48
 
47
- functions: tuple[str, ...]
49
+ functions: tuple[ModuleFunction, ...]
48
50
  summaries: Mapping[str, FunctionSummary]
49
51
  sites: tuple[CallSite, ...]
50
52
  findings: tuple[Finding, ...]
@@ -128,7 +130,7 @@ class ProjectCache:
128
130
  def encode(module: CachedModule) -> dict[str, Any]:
129
131
  return {
130
132
  "format": CACHE_FORMAT,
131
- "functions": list(module.functions),
133
+ "functions": [[f.name, _encode_span(f.span), f.entry_point] for f in module.functions],
132
134
  "summaries": {name: _encode_summary(s) for name, s in module.summaries.items()},
133
135
  "sites": [_encode_site(site) for site in module.sites],
134
136
  "findings": [_encode_finding(finding) for finding in module.findings],
@@ -139,7 +141,7 @@ def decode(data: Mapping[str, Any]) -> CachedModule:
139
141
  if data["format"] != CACHE_FORMAT:
140
142
  raise ValueError(f"unsupported cache format {data['format']!r}")
141
143
  return CachedModule(
142
- tuple(_string(name) for name in data["functions"]),
144
+ tuple(_decode_function(function) for function in data["functions"]),
143
145
  {_string(name): _decode_summary(s) for name, s in data["summaries"].items()},
144
146
  tuple(_decode_site(site) for site in data["sites"]),
145
147
  tuple(_decode_finding(finding) for finding in data["findings"]),
@@ -187,6 +189,11 @@ def _decode_span(data: Any) -> SourceSpan:
187
189
  )
188
190
 
189
191
 
192
+ def _decode_function(data: Any) -> ModuleFunction:
193
+ name, span, entry_point = data
194
+ return ModuleFunction(_string(name), _decode_span(span), None if entry_point is None else _string(entry_point))
195
+
196
+
190
197
  def _encode_finding(finding: Finding) -> dict[str, Any]:
191
198
  return {
192
199
  "rule": finding.rule_id,
@@ -219,6 +226,7 @@ def _encode_call(call: ExternalCall) -> dict[str, Any]:
219
226
  "keywords": sorted(call.keyword_dependencies),
220
227
  "location": _encode_span(call.location),
221
228
  "call_site": None if call.call_site is None else _encode_span(call.call_site),
229
+ "given": _encode_arguments(call.arguments),
222
230
  }
223
231
 
224
232
 
@@ -230,6 +238,7 @@ def _decode_call(data: Mapping[str, Any]) -> ExternalCall:
230
238
  _indices(data["keywords"]),
231
239
  _decode_span(data["location"]),
232
240
  None if site is None else _decode_span(site),
241
+ _decode_arguments(data["given"]),
233
242
  )
234
243
 
235
244
 
@@ -311,8 +320,7 @@ def _encode_site(site: CallSite) -> dict[str, Any]:
311
320
  "caller": site.caller,
312
321
  "location": _encode_span(site.location),
313
322
  "target": _encode_target(site.target),
314
- "arguments": site.arguments,
315
- "keywords": site.keywords,
323
+ "arguments": _encode_arguments(site.arguments),
316
324
  }
317
325
 
318
326
 
@@ -321,6 +329,27 @@ def _decode_site(data: Mapping[str, Any]) -> CallSite:
321
329
  _string(data["caller"]),
322
330
  _decode_span(data["location"]),
323
331
  _decode_target(data["target"]),
324
- _integer(data["arguments"]),
325
- _integer(data["keywords"]),
332
+ _decode_arguments(data["arguments"]),
333
+ )
334
+
335
+
336
+ def _encode_arguments(arguments: Arguments) -> dict[str, Any]:
337
+ return {
338
+ "positional": list(arguments.positional),
339
+ "keywords": [[name, value] for name, value in arguments.keywords],
340
+ "unpacked": arguments.unpacked,
341
+ }
342
+
343
+
344
+ def _decode_arguments(data: Mapping[str, Any]) -> Arguments:
345
+ def denoted(value: Any) -> str | None:
346
+ return None if value is None else _string(value)
347
+
348
+ unpacked = data["unpacked"]
349
+ if not isinstance(unpacked, bool):
350
+ raise TypeError(f"expected a boolean, got {unpacked!r}")
351
+ return Arguments(
352
+ tuple(denoted(value) for value in data["positional"]),
353
+ tuple((_string(name), denoted(value)) for name, value in data["keywords"]),
354
+ unpacked,
326
355
  )
@@ -11,6 +11,8 @@ from coretrace_python.dependency.advisories import (
11
11
  from coretrace_python.dependency.graph import (
12
12
  DEPENDENCY_FILES,
13
13
  Advisory,
14
+ AdvisoryEntryPoint,
15
+ Condition,
14
16
  DependencyAnalysis,
15
17
  DependencyGraph,
16
18
  Requirement,
@@ -26,7 +28,9 @@ __all__ = [
26
28
  "DEPENDENCY_FILES",
27
29
  "POLICY_FILE",
28
30
  "Advisory",
31
+ "AdvisoryEntryPoint",
29
32
  "AdvisoryFileError",
33
+ "Condition",
30
34
  "DependencyAnalysis",
31
35
  "DependencyGraph",
32
36
  "Policy",
@@ -5,8 +5,8 @@ OSV dump into ``Advisory`` values, keeping the PyPI ecosystem and turning each r
5
5
  events into a version specifier; ``dump_advisories`` writes them as a small JSON file
6
6
  that a project keeps at its root as ``advisories.json`` or passes with ``--advisories``.
7
7
  OSV records name no affected APIs, so imported advisories feed the requirement checks
8
- and the SBOM; a file completed by hand with ``affected_symbols`` also feeds the
9
- reachability and correlation checks.
8
+ and the SBOM; a file completed by hand with ``affected_symbols``, ``entry_points`` and
9
+ their ``conditions`` also feeds the reachability and correlation checks.
10
10
  """
11
11
 
12
12
  from __future__ import annotations
@@ -17,7 +17,7 @@ from collections.abc import Iterable, Iterator, Mapping
17
17
  from pathlib import Path
18
18
  from typing import Any
19
19
 
20
- from coretrace_python.dependency.graph import Advisory, normalize
20
+ from coretrace_python.dependency.graph import Advisory, AdvisoryEntryPoint, Condition, normalize
21
21
  from coretrace_python.findings import Severity
22
22
  from coretrace_python.semantic.symbols import SymbolId
23
23
 
@@ -84,8 +84,9 @@ def import_osv(records: Iterable[Mapping[str, Any]]) -> tuple[Advisory, ...]:
84
84
  if str(package.get("ecosystem", "")).lower() != "pypi" or not package.get("name"):
85
85
  continue
86
86
  name = normalize(str(package["name"]))
87
- for specifier in _specifiers(affected.get("ranges") or []):
88
- advisories.append(Advisory(identifier, name, specifier, summary, severity, (), aliases))
87
+ specifiers = _specifiers(affected.get("ranges") or [])
88
+ if specifiers:
89
+ advisories.append(Advisory(identifier, name, " || ".join(specifiers), summary, severity, (), aliases))
89
90
  return tuple(advisories)
90
91
 
91
92
 
@@ -137,6 +138,15 @@ def dump_advisories(advisories: Iterable[Advisory]) -> str:
137
138
  "severity": a.severity.value,
138
139
  "affected_symbols": [str(s) for s in a.affected_symbols],
139
140
  "aliases": list(a.aliases),
141
+ "entry_points": [
142
+ {
143
+ "symbol": str(e.symbol),
144
+ "justification": e.justification,
145
+ "conditions": [_condition_entry(c) for c in e.conditions],
146
+ }
147
+ for e in a.entry_points
148
+ ],
149
+ "modules": list(a.modules),
140
150
  }
141
151
  for a in advisories
142
152
  ],
@@ -144,6 +154,19 @@ def dump_advisories(advisories: Iterable[Advisory]) -> str:
144
154
  return json.dumps(document, indent=2) + "\n"
145
155
 
146
156
 
157
+ def _condition_entry(condition: Condition) -> dict[str, Any]:
158
+ entry: dict[str, Any] = {"kind": condition.kind, "text": condition.text}
159
+ if condition.argument is not None:
160
+ entry["argument"] = condition.argument
161
+ if condition.values:
162
+ entry["values"] = list(condition.values)
163
+ if condition.position is not None:
164
+ entry["position"] = condition.position
165
+ if condition.default:
166
+ entry["default"] = True
167
+ return entry
168
+
169
+
147
170
  def load_advisories(path: Path) -> tuple[Advisory, ...]:
148
171
  try:
149
172
  document = json.loads(path.read_text(encoding="utf-8"))
@@ -165,4 +188,31 @@ def _advisory(entry: Mapping[str, Any]) -> Advisory:
165
188
  Severity(entry["severity"]),
166
189
  tuple(SymbolId(str(s)) for s in entry.get("affected_symbols") or []),
167
190
  tuple(str(a) for a in entry.get("aliases") or []),
191
+ tuple(_entry_point(e) for e in entry.get("entry_points") or []),
192
+ tuple(str(m) for m in entry.get("modules") or []),
193
+ )
194
+
195
+
196
+ def _entry_point(entry: Mapping[str, Any]) -> AdvisoryEntryPoint:
197
+ return AdvisoryEntryPoint(
198
+ SymbolId(str(entry["symbol"])),
199
+ str(entry["justification"]),
200
+ tuple(_condition(c) for c in entry.get("conditions") or []),
201
+ )
202
+
203
+
204
+ def _condition(entry: Mapping[str, Any]) -> Condition:
205
+ position = entry.get("position")
206
+ if position is not None and (not isinstance(position, int) or isinstance(position, bool) or position < 0):
207
+ raise TypeError(f"condition position must be a non-negative integer, got {position!r}")
208
+ default = entry.get("default", False)
209
+ if not isinstance(default, bool):
210
+ raise TypeError(f"condition default must be true or false, got {default!r}")
211
+ return Condition(
212
+ str(entry["kind"]),
213
+ str(entry["text"]),
214
+ None if entry.get("argument") is None else str(entry["argument"]),
215
+ tuple(str(v) for v in entry.get("values") or []),
216
+ position,
217
+ default,
168
218
  )
@@ -10,78 +10,172 @@ are correlated here into one high-confidence ``exploitable-vulnerability`` findi
10
10
  from __future__ import annotations
11
11
 
12
12
  from collections.abc import Iterable, Mapping
13
+ from dataclasses import dataclass
13
14
 
14
- from coretrace_python.dependency.graph import Advisory, DependencyGraph
15
+ from coretrace_python.dependency.graph import (
16
+ DIRECT,
17
+ Advisory,
18
+ AdvisoryEntryPoint,
19
+ Condition,
20
+ DependencyGraph,
21
+ )
15
22
  from coretrace_python.findings import Confidence, Finding, Severity
16
- from coretrace_python.findings.refutation import Status, Verdicts
23
+ from coretrace_python.findings.refutation import Status, Verdict, Verdicts
24
+ from coretrace_python.interprocedural import Arguments, CallSite, ExternalSymbol
17
25
  from coretrace_python.semantic.symbols import SymbolId
18
26
  from coretrace_python.taint import Sink, TaintFlow, TaintKind
19
27
 
28
+ Affected = Mapping[SymbolId, tuple[Advisory, ...]]
20
29
 
21
- def affected_symbols(
22
- dependencies: DependencyGraph, advisories: Iterable[Advisory]
23
- ) -> Mapping[SymbolId, Advisory]:
24
- """The APIs affected by advisories whose package is required in a vulnerable version."""
25
30
 
26
- affected: dict[SymbolId, Advisory] = {}
31
+ def affected_symbols(dependencies: DependencyGraph, advisories: Iterable[Advisory]) -> Affected:
32
+ """The APIs affected by advisories whose package is required in a vulnerable version,
33
+ each with every such advisory: a call to ``yaml.load`` reaches every CVE of the
34
+ pinned release, not the first one listed."""
35
+
36
+ affected: dict[SymbolId, list[Advisory]] = {}
27
37
  for requirement in dependencies.requirements:
28
38
  for advisory in advisories:
29
39
  if advisory.affects(requirement):
30
- for symbol in advisory.affected_symbols:
31
- affected.setdefault(symbol, advisory)
32
- return affected
40
+ for symbol in advisory.reachable_symbols:
41
+ found = affected.setdefault(symbol, [])
42
+ if advisory not in found:
43
+ found.append(advisory)
44
+ return {symbol: tuple(found) for symbol, found in affected.items()}
33
45
 
34
46
 
35
- def advisory_sinks(affected: Mapping[SymbolId, Advisory]) -> tuple[Sink, ...]:
47
+ def advisory_sinks(affected: Affected) -> tuple[Sink, ...]:
36
48
  return tuple(Sink(symbol, TaintKind.ADVISORY) for symbol in affected)
37
49
 
38
50
 
51
+ @dataclass(frozen=True)
52
+ class ConditionCheck:
53
+ """What one call tells of an entry point's conditions: those it meets, those left to
54
+ review (every semantic one, and every argument one the call does not decide), and
55
+ the argument condition it contradicts with what it passes instead (None: absent),
56
+ in which case the call does not reach the vulnerability."""
57
+
58
+ met: tuple[Condition, ...] = ()
59
+ pending: tuple[Condition, ...] = ()
60
+ contradicted: Condition | None = None
61
+ passed: str | None = None
62
+
63
+
64
+ def check_conditions(entry: AdvisoryEntryPoint | None, arguments: Arguments | None) -> ConditionCheck:
65
+ """Decide ``entry``'s conditions against what the call's ``arguments`` denote."""
66
+
67
+ if entry is None:
68
+ return ConditionCheck()
69
+ met: list[Condition] = []
70
+ pending: list[Condition] = []
71
+ for condition in entry.conditions:
72
+ given = _given(condition, arguments) if condition.checkable and arguments is not None else None
73
+ if given is None:
74
+ pending.append(condition)
75
+ continue
76
+ explicit, value = given
77
+ if explicit and value is None:
78
+ pending.append(condition)
79
+ continue
80
+ affected = value in condition.values if explicit else condition.default
81
+ if not affected:
82
+ return ConditionCheck(tuple(met), tuple(pending), condition, value)
83
+ met.append(condition)
84
+ return ConditionCheck(tuple(met), tuple(pending))
85
+
86
+
87
+ def _given(condition: Condition, arguments: Arguments) -> tuple[bool, str | None] | None:
88
+ """Whether the call gives the condition's argument, and what it denotes: ``(True,
89
+ value)`` when given, ``(False, None)`` when surely absent, None when it cannot tell."""
90
+
91
+ for name, value in arguments.keywords:
92
+ if name == condition.argument:
93
+ return True, value
94
+ if condition.position is not None and condition.position < len(arguments.positional):
95
+ return True, arguments.positional[condition.position]
96
+ return None if arguments.unpacked else (False, None)
97
+
98
+
99
+ def ruled_out(module: str, site: CallSite, check: ConditionCheck) -> str:
100
+ """Which call a contradicted condition rules out, and why: ``app:12
101
+ python.yaml.load(Loader=python.yaml.SafeLoader)``."""
102
+
103
+ assert check.contradicted is not None and isinstance(site.target, ExternalSymbol)
104
+ argument = check.contradicted.argument or f"#{check.contradicted.position}"
105
+ passed = f"{argument} absent" if check.passed is None else f"{argument}={check.passed}"
106
+ return f"{module}:{site.location.start_line} {site.target.symbol}({passed})"
107
+
108
+
109
+ def evidence(advisory: Advisory, symbol: SymbolId, level: str, check: ConditionCheck) -> dict[str, str]:
110
+ """What a finding keeps of the advisory for ``symbol``: the level of evidence
111
+ established, how the symbol relates to the vulnerability, and its conditions — those
112
+ the call meets and those left to review."""
113
+
114
+ metadata = {"advisory": advisory.id, "package": advisory.package, "symbol": str(symbol), "level": level}
115
+ entry = advisory.entry_point(symbol)
116
+ if entry is None:
117
+ metadata["justification"] = DIRECT
118
+ return metadata
119
+ metadata["entry_point"] = str(entry.symbol)
120
+ metadata["justification"] = entry.justification
121
+ if entry.conditions:
122
+ metadata["conditions"] = "; ".join(c.text for c in entry.conditions)
123
+ if check.met:
124
+ metadata["conditions_met"] = "; ".join(c.text for c in check.met)
125
+ if check.pending:
126
+ metadata["conditions_pending_review"] = "; ".join(c.text for c in check.pending)
127
+ return metadata
128
+
129
+
39
130
  def correlate(
40
131
  function: str,
41
132
  flows: Iterable[TaintFlow],
42
133
  verdicts: Verdicts | None,
43
- affected: Mapping[SymbolId, Advisory],
134
+ affected: Affected,
44
135
  ) -> tuple[Finding, ...]:
45
- """Exploitable-vulnerability findings for the non-refuted ADVISORY flows of a function."""
136
+ """Exploitable-vulnerability findings for the non-refuted ADVISORY flows of a function,
137
+ one per advisory the sink is affected by."""
46
138
 
47
139
  findings: list[Finding] = []
48
140
  for flow in flows:
49
141
  if not flow.kinds & TaintKind.ADVISORY:
50
142
  continue
51
- advisory = affected.get(flow.sink.symbol)
52
- if advisory is None:
53
- continue
54
143
  verdict = verdicts.verdict(flow) if verdicts is not None else None
55
144
  if verdict is not None and verdict.status is Status.REFUTED:
56
145
  continue
57
146
  hotspot = verdict is not None and verdict.status is Status.HOTSPOT
58
- message = (
59
- f"{advisory.id}: {flow.source.label} input reaches {flow.sink.symbol}, affected in "
60
- f"the required {advisory.package} {advisory.vulnerable}: {advisory.summary}"
61
- )
62
- metadata = {
63
- "advisory": advisory.id,
64
- "package": advisory.package,
65
- "symbol": str(flow.sink.symbol),
66
- "source": str(flow.source.symbol),
67
- "source_label": flow.source.label,
68
- "verdict": "hotspot" if hotspot else "vulnerability",
69
- }
70
- if verdict is not None:
71
- metadata["evidence"] = verdict.evidence
72
- if flow.through is not None and flow.sink_location is not None:
73
- message += f" through {flow.through}"
74
- metadata["through"] = flow.through
75
- metadata["sink_line"] = str(flow.sink_location.start_line)
76
- findings.append(
77
- Finding(
78
- rule_id="exploitable-vulnerability",
79
- message=message,
80
- severity=Severity.CRITICAL,
81
- confidence=Confidence.MEDIUM if hotspot else Confidence.HIGH,
82
- span=flow.location,
83
- function=function,
84
- metadata=metadata,
85
- )
86
- )
147
+ for advisory in affected.get(flow.sink.symbol, ()):
148
+ check = check_conditions(advisory.entry_point(flow.sink.symbol), flow.sink_arguments)
149
+ if check.contradicted is None:
150
+ findings.append(_exploitable(function, flow, advisory, verdict, hotspot, check))
87
151
  return tuple(findings)
152
+
153
+
154
+ def _exploitable(
155
+ function: str, flow: TaintFlow, advisory: Advisory, verdict: Verdict | None, hotspot: bool, check: ConditionCheck
156
+ ) -> Finding:
157
+ message = (
158
+ f"{advisory.id}: {flow.source.label} input reaches {flow.sink.symbol}, affected in "
159
+ f"the required {advisory.package} {advisory.vulnerable}: {advisory.summary}"
160
+ )
161
+ metadata = {
162
+ **evidence(advisory, flow.sink.symbol, "exploitable", check),
163
+ "source": str(flow.source.symbol),
164
+ "source_label": flow.source.label,
165
+ "verdict": "hotspot" if hotspot else "vulnerability",
166
+ }
167
+ if verdict is not None:
168
+ metadata["evidence"] = verdict.evidence
169
+ if flow.through is not None and flow.sink_location is not None:
170
+ message += f" through {flow.through}"
171
+ metadata["through"] = flow.through
172
+ metadata["sink_line"] = str(flow.sink_location.start_line)
173
+ return Finding(
174
+ rule_id="exploitable-vulnerability",
175
+ message=message,
176
+ severity=Severity.CRITICAL,
177
+ confidence=Confidence.MEDIUM if hotspot else Confidence.HIGH,
178
+ span=flow.location,
179
+ function=function,
180
+ metadata=metadata,
181
+ )