coretrace-python-analyzer 0.4.0__py3-none-any.whl → 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. coretrace_python/__init__.py +1 -1
  2. coretrace_python/bundled/dependency/reachable_vulnerability/reachable_vulnerability.py +54 -18
  3. coretrace_python/bundled/dependency/vulnerable_dependency/plugin.toml +1 -1
  4. coretrace_python/bundled/dependency/vulnerable_dependency/vulnerable_dependency.py +32 -7
  5. coretrace_python/bundled/models/python_stdlib/python_stdlib.py +25 -1
  6. coretrace_python/cache.py +50 -11
  7. coretrace_python/cli.py +33 -1
  8. coretrace_python/dependency/__init__.py +2 -0
  9. coretrace_python/dependency/advisories.py +38 -17
  10. coretrace_python/dependency/correlation.py +69 -13
  11. coretrace_python/dependency/graph.py +83 -11
  12. coretrace_python/dependency/policy.py +6 -2
  13. coretrace_python/dependency/vex.py +182 -0
  14. coretrace_python/engine.py +26 -6
  15. coretrace_python/interprocedural/__init__.py +6 -0
  16. coretrace_python/interprocedural/callgraph.py +112 -11
  17. coretrace_python/interprocedural/summaries.py +8 -1
  18. coretrace_python/plugins/api.py +18 -4
  19. coretrace_python/taint/__init__.py +4 -0
  20. coretrace_python/taint/engine.py +91 -16
  21. coretrace_python/taint/models.py +24 -0
  22. {coretrace_python_analyzer-0.4.0.dist-info → coretrace_python_analyzer-0.6.0.dist-info}/METADATA +1 -1
  23. {coretrace_python_analyzer-0.4.0.dist-info → coretrace_python_analyzer-0.6.0.dist-info}/RECORD +27 -26
  24. {coretrace_python_analyzer-0.4.0.dist-info → coretrace_python_analyzer-0.6.0.dist-info}/WHEEL +0 -0
  25. {coretrace_python_analyzer-0.4.0.dist-info → coretrace_python_analyzer-0.6.0.dist-info}/entry_points.txt +0 -0
  26. {coretrace_python_analyzer-0.4.0.dist-info → coretrace_python_analyzer-0.6.0.dist-info}/licenses/LICENSE +0 -0
  27. {coretrace_python_analyzer-0.4.0.dist-info → coretrace_python_analyzer-0.6.0.dist-info}/licenses/NOTICE +0 -0
@@ -1,4 +1,4 @@
1
1
  """CoreTrace's Python static analysis frontend."""
2
2
 
3
- __version__ = "0.4.0"
3
+ __version__ = "0.6.0"
4
4
 
@@ -1,16 +1,24 @@
1
- """Calls, anywhere in the project, to an API affected by a vulnerable requirement."""
1
+ """Calls, anywhere in the project, to an API affected by a vulnerable requirement, and
2
+ reads of an affected attribute whose getter runs the vulnerable code."""
2
3
 
3
4
  from __future__ import annotations
4
5
 
5
- from collections.abc import Sequence
6
+ from collections.abc import Iterator, Sequence
6
7
  from typing import ClassVar
7
8
 
8
9
  from coretrace_python.analysis import AnyAnalysis
9
- from coretrace_python.dependency import DependencyAnalysis
10
- from coretrace_python.dependency.correlation import affected_symbols, evidence
10
+ from coretrace_python.dependency import Advisory, DependencyAnalysis
11
+ from coretrace_python.dependency.correlation import (
12
+ ConditionCheck,
13
+ affected_symbols,
14
+ check_conditions,
15
+ evidence,
16
+ )
11
17
  from coretrace_python.findings import Confidence, Finding
12
18
  from coretrace_python.interprocedural import CallGraphAnalysis, ExternalSymbol
13
19
  from coretrace_python.plugins import ProjectContext, ProjectPlugin
20
+ from coretrace_python.semantic.symbols import SymbolId
21
+ from coretrace_python.source import SourceSpan
14
22
 
15
23
 
16
24
  class ReachableVulnerabilityPlugin(ProjectPlugin):
@@ -29,18 +37,46 @@ class ReachableVulnerabilityPlugin(ProjectPlugin):
29
37
  if not isinstance(site.target, ExternalSymbol):
30
38
  continue
31
39
  for advisory in affected.get(site.target.symbol, ()):
32
- findings.append(
33
- Finding(
34
- rule_id="reachable-vulnerability",
35
- message=(
36
- f"{advisory.id}: {site.target.symbol} is affected in the required "
37
- f"{advisory.package} {advisory.vulnerable}: {advisory.summary}"
38
- ),
39
- severity=advisory.severity,
40
- confidence=Confidence.HIGH,
41
- span=site.location,
42
- function=function,
43
- metadata=evidence(advisory, site.target.symbol, "reachable"),
44
- )
45
- )
40
+ entry = advisory.entry_point(site.target.symbol)
41
+ if entry is not None and entry.read:
42
+ continue # a call reads its callee first: reported with the reads
43
+ check = check_conditions(entry, site.arguments)
44
+ if check.contradicted is not None:
45
+ continue
46
+ findings.append(_reached(advisory, site.target.symbol, site.location, function, check))
47
+ # Reading ``request.form.get`` reads ``request.form``: one finding per entry
48
+ # point and function, where the function first reads it.
49
+ reported: set[tuple[Advisory, SymbolId]] = set()
50
+ for read in graph.reads(function):
51
+ for symbol in _read_through(read.symbol):
52
+ for advisory in affected.get(symbol, ()):
53
+ entry = advisory.entry_point(symbol)
54
+ if entry is None or not entry.read or (advisory, symbol) in reported:
55
+ continue
56
+ reported.add((advisory, symbol))
57
+ check = check_conditions(entry, None)
58
+ findings.append(_reached(advisory, symbol, read.location, function, check))
46
59
  return findings
60
+
61
+
62
+ def _read_through(symbol: SymbolId) -> Iterator[SymbolId]:
63
+ """``symbol`` and every symbol it is an attribute of, as reading it reads them all."""
64
+
65
+ components = symbol.canonical_name.split(".")
66
+ for end in range(len(components), 1, -1):
67
+ yield SymbolId(".".join(components[:end]))
68
+
69
+
70
+ def _reached(advisory: Advisory, symbol: SymbolId, span: SourceSpan, function: str, check: ConditionCheck) -> Finding:
71
+ return Finding(
72
+ rule_id="reachable-vulnerability",
73
+ message=(
74
+ f"{advisory.id}: {symbol} is affected in the required "
75
+ f"{advisory.package} {advisory.vulnerable}: {advisory.summary}"
76
+ ),
77
+ severity=advisory.severity,
78
+ confidence=Confidence.HIGH,
79
+ span=span,
80
+ function=function,
81
+ metadata=evidence(advisory, symbol, "reachable", check),
82
+ )
@@ -1,7 +1,7 @@
1
1
  name = "vulnerable-dependency"
2
2
  version = "1.0.0"
3
3
  plugin_api = ">=1,<2"
4
- requires = ["dependency.graph"]
4
+ requires = ["dependency.graph", "interprocedural.callgraph"]
5
5
  provides = ["vulnerability.vulnerable-dependency"]
6
6
 
7
7
  [entrypoint]
@@ -7,23 +7,34 @@ from typing import ClassVar
7
7
 
8
8
  from coretrace_python.analysis import AnyAnalysis
9
9
  from coretrace_python.dependency import DependencyAnalysis
10
+ from coretrace_python.dependency.correlation import affected_symbols, check_conditions, ruled_out
10
11
  from coretrace_python.findings import Confidence, Finding
12
+ from coretrace_python.interprocedural import CallGraphAnalysis, ExternalSymbol
11
13
  from coretrace_python.plugins import ProjectContext, ProjectPlugin
12
14
 
13
15
 
14
16
  class VulnerableDependencyPlugin(ProjectPlugin):
15
17
  name: ClassVar[str] = "vulnerable-dependency"
16
- requires: ClassVar[frozenset[AnyAnalysis]] = frozenset({DependencyAnalysis})
18
+ requires: ClassVar[frozenset[AnyAnalysis]] = frozenset({DependencyAnalysis, CallGraphAnalysis})
17
19
 
18
20
  def analyze_project(self, ctx: ProjectContext) -> Sequence[Finding]:
19
21
  findings: list[Finding] = []
20
22
  imported = [s for module in ctx.modules for s in ctx.imports(module).all_symbols()]
23
+ excluded = _ruled_out(ctx)
21
24
  for requirement in ctx.dependencies.requirements:
22
25
  for advisory in ctx.advisories:
23
26
  if not advisory.affects(requirement):
24
27
  continue
25
28
  pinned = requirement.pinned is not None
26
29
  level = "imported" if advisory.imported_by(imported) else "declared"
30
+ metadata = {
31
+ "advisory": advisory.id,
32
+ "package": advisory.package,
33
+ "specifier": requirement.specifier,
34
+ "level": level,
35
+ }
36
+ if excluded.get(advisory.id):
37
+ metadata["ruled_out"] = "; ".join(excluded[advisory.id])
27
38
  findings.append(
28
39
  Finding(
29
40
  rule_id="vulnerable-dependency",
@@ -35,12 +46,26 @@ class VulnerableDependencyPlugin(ProjectPlugin):
35
46
  severity=advisory.severity,
36
47
  confidence=Confidence.HIGH if pinned else Confidence.MEDIUM,
37
48
  span=requirement.span,
38
- metadata={
39
- "advisory": advisory.id,
40
- "package": advisory.package,
41
- "specifier": requirement.specifier,
42
- "level": level,
43
- },
49
+ metadata=metadata,
44
50
  )
45
51
  )
46
52
  return findings
53
+
54
+
55
+ def _ruled_out(ctx: ProjectContext) -> dict[str, list[str]]:
56
+ """By advisory, the calls to its entry points whose arguments contradict one of its
57
+ conditions: the evidence that the requirement, imported, is not reached there."""
58
+
59
+ affected = affected_symbols(ctx.dependencies, ctx.advisories)
60
+ excluded: dict[str, list[str]] = {}
61
+ for module in sorted(ctx.modules):
62
+ graph = ctx.call_graph(module)
63
+ for function in graph.functions:
64
+ for site in graph.sites(function):
65
+ if not isinstance(site.target, ExternalSymbol):
66
+ continue
67
+ for advisory in affected.get(site.target.symbol, ()):
68
+ check = check_conditions(advisory.entry_point(site.target.symbol), site.arguments)
69
+ if check.contradicted is not None:
70
+ excluded.setdefault(advisory.id, []).append(ruled_out(module, site, check))
71
+ return excluded
@@ -6,9 +6,28 @@ from typing import ClassVar
6
6
 
7
7
  from coretrace_python.plugins import ModelPlugin
8
8
  from coretrace_python.semantic.symbols import SymbolId
9
- from coretrace_python.taint import Model, Sanitizer, Sink, Source, TaintKind, Validator
9
+ from coretrace_python.taint import (
10
+ Model,
11
+ SafeArgument,
12
+ Sanitizer,
13
+ Sink,
14
+ Source,
15
+ TaintKind,
16
+ Validator,
17
+ )
10
18
 
11
19
  _ENVIRONMENT_KINDS = TaintKind.ALL & ~(TaintKind.COMMAND | TaintKind.PATH)
20
+ # SafeLoader and BaseLoader build plain data only, in Python and in C, under every spelling.
21
+ _SAFE_YAML_LOADERS = (
22
+ "python.yaml.SafeLoader",
23
+ "python.yaml.loader.SafeLoader",
24
+ "python.yaml.BaseLoader",
25
+ "python.yaml.loader.BaseLoader",
26
+ "python.yaml.CSafeLoader",
27
+ "python.yaml.cyaml.CSafeLoader",
28
+ "python.yaml.CBaseLoader",
29
+ "python.yaml.cyaml.CBaseLoader",
30
+ )
12
31
  _PROCESS_OUTPUT_KINDS = TaintKind.ALL & ~TaintKind.PATH
13
32
 
14
33
 
@@ -66,6 +85,11 @@ class PythonStdlibModels(ModelPlugin):
66
85
  Sink(_sym("dill.loads"), TaintKind.DESERIALIZATION),
67
86
  Sink(_sym("jsonpickle.decode"), TaintKind.DESERIALIZATION),
68
87
  Sink(_sym("yaml.load"), TaintKind.DESERIALIZATION),
88
+ Sink(_sym("yaml.load_all"), TaintKind.DESERIALIZATION),
89
+ Sink(_sym("yaml.full_load_all"), TaintKind.DESERIALIZATION),
90
+ Sink(_sym("yaml.unsafe_load_all"), TaintKind.DESERIALIZATION),
91
+ SafeArgument(_sym("yaml.load"), "Loader", _SAFE_YAML_LOADERS, position=1, kinds=TaintKind.DESERIALIZATION),
92
+ SafeArgument(_sym("yaml.load_all"), "Loader", _SAFE_YAML_LOADERS, position=1, kinds=TaintKind.DESERIALIZATION),
69
93
  Sink(_sym("yaml.unsafe_load"), TaintKind.DESERIALIZATION),
70
94
  Sink(_sym("yaml.full_load"), TaintKind.DESERIALIZATION),
71
95
  Sanitizer(_sym("os.path.basename"), TaintKind.PATH),
coretrace_python/cache.py CHANGED
@@ -4,9 +4,10 @@ A module's results are stored under a key derived from everything they depend on
4
4
  source text and identity, the engine, schema and plugin API versions, the plugins and
5
5
  their code, the security models, the advisories, the dependency graph, and the keys of
6
6
  the project modules it imports transitively. A module whose key is unchanged on a later
7
- run is served from the cache: its summaries seed the project index, its call sites serve
8
- the project plugins and its findings are reported as they were. Entries are JSON, so a
9
- tampered or foreign file can never execute anything; an unreadable entry is a miss.
7
+ run is served from the cache: its summaries seed the project index, its call sites and
8
+ symbol reads serve the project plugins and its findings are reported as they were.
9
+ Entries are JSON, so a tampered or foreign file can never execute anything; an
10
+ unreadable entry is a miss.
10
11
  """
11
12
 
12
13
  from __future__ import annotations
@@ -22,32 +23,36 @@ from typing import Any
22
23
 
23
24
  from coretrace_python.findings import Confidence, Finding, Severity
24
25
  from coretrace_python.interprocedural import (
26
+ Arguments,
25
27
  CallSite,
26
28
  ExternalCall,
27
29
  ExternalSymbol,
28
30
  FunctionSummary,
29
31
  KnownFunction,
32
+ ModuleFunction,
30
33
  ModuleGraph,
31
34
  Mutation,
32
35
  NonlocalWrite,
33
36
  SummaryIndex,
37
+ SymbolRead,
34
38
  Target,
35
39
  UnknownTarget,
36
40
  )
37
41
  from coretrace_python.semantic.symbols import SymbolId
38
42
  from coretrace_python.source import SourceId, SourceSpan
39
43
 
40
- CACHE_FORMAT = 5
44
+ CACHE_FORMAT = 8
41
45
 
42
46
 
43
47
  @dataclass(frozen=True)
44
48
  class CachedModule:
45
49
  """Everything a later run needs from one module without lowering it again."""
46
50
 
47
- functions: tuple[str, ...]
51
+ functions: tuple[ModuleFunction, ...]
48
52
  summaries: Mapping[str, FunctionSummary]
49
53
  sites: tuple[CallSite, ...]
50
54
  findings: tuple[Finding, ...]
55
+ reads: tuple[SymbolRead, ...] = ()
51
56
 
52
57
  def __post_init__(self) -> None:
53
58
  object.__setattr__(self, "summaries", MappingProxyType(dict(self.summaries)))
@@ -128,10 +133,11 @@ class ProjectCache:
128
133
  def encode(module: CachedModule) -> dict[str, Any]:
129
134
  return {
130
135
  "format": CACHE_FORMAT,
131
- "functions": list(module.functions),
136
+ "functions": [[f.name, _encode_span(f.span), f.entry_point] for f in module.functions],
132
137
  "summaries": {name: _encode_summary(s) for name, s in module.summaries.items()},
133
138
  "sites": [_encode_site(site) for site in module.sites],
134
139
  "findings": [_encode_finding(finding) for finding in module.findings],
140
+ "reads": [[r.function, _encode_span(r.location), str(r.symbol)] for r in module.reads],
135
141
  }
136
142
 
137
143
 
@@ -139,10 +145,11 @@ def decode(data: Mapping[str, Any]) -> CachedModule:
139
145
  if data["format"] != CACHE_FORMAT:
140
146
  raise ValueError(f"unsupported cache format {data['format']!r}")
141
147
  return CachedModule(
142
- tuple(_string(name) for name in data["functions"]),
148
+ tuple(_decode_function(function) for function in data["functions"]),
143
149
  {_string(name): _decode_summary(s) for name, s in data["summaries"].items()},
144
150
  tuple(_decode_site(site) for site in data["sites"]),
145
151
  tuple(_decode_finding(finding) for finding in data["findings"]),
152
+ tuple(_decode_read(read) for read in data["reads"]),
146
153
  )
147
154
 
148
155
 
@@ -187,6 +194,16 @@ def _decode_span(data: Any) -> SourceSpan:
187
194
  )
188
195
 
189
196
 
197
+ def _decode_function(data: Any) -> ModuleFunction:
198
+ name, span, entry_point = data
199
+ return ModuleFunction(_string(name), _decode_span(span), None if entry_point is None else _string(entry_point))
200
+
201
+
202
+ def _decode_read(data: Any) -> SymbolRead:
203
+ function, span, symbol = data
204
+ return SymbolRead(_string(function), _decode_span(span), SymbolId(_string(symbol)))
205
+
206
+
190
207
  def _encode_finding(finding: Finding) -> dict[str, Any]:
191
208
  return {
192
209
  "rule": finding.rule_id,
@@ -219,6 +236,7 @@ def _encode_call(call: ExternalCall) -> dict[str, Any]:
219
236
  "keywords": sorted(call.keyword_dependencies),
220
237
  "location": _encode_span(call.location),
221
238
  "call_site": None if call.call_site is None else _encode_span(call.call_site),
239
+ "given": _encode_arguments(call.arguments),
222
240
  }
223
241
 
224
242
 
@@ -230,6 +248,7 @@ def _decode_call(data: Mapping[str, Any]) -> ExternalCall:
230
248
  _indices(data["keywords"]),
231
249
  _decode_span(data["location"]),
232
250
  None if site is None else _decode_span(site),
251
+ _decode_arguments(data["given"]),
233
252
  )
234
253
 
235
254
 
@@ -311,8 +330,7 @@ def _encode_site(site: CallSite) -> dict[str, Any]:
311
330
  "caller": site.caller,
312
331
  "location": _encode_span(site.location),
313
332
  "target": _encode_target(site.target),
314
- "arguments": site.arguments,
315
- "keywords": site.keywords,
333
+ "arguments": _encode_arguments(site.arguments),
316
334
  }
317
335
 
318
336
 
@@ -321,6 +339,27 @@ def _decode_site(data: Mapping[str, Any]) -> CallSite:
321
339
  _string(data["caller"]),
322
340
  _decode_span(data["location"]),
323
341
  _decode_target(data["target"]),
324
- _integer(data["arguments"]),
325
- _integer(data["keywords"]),
342
+ _decode_arguments(data["arguments"]),
343
+ )
344
+
345
+
346
+ def _encode_arguments(arguments: Arguments) -> dict[str, Any]:
347
+ return {
348
+ "positional": list(arguments.positional),
349
+ "keywords": [[name, value] for name, value in arguments.keywords],
350
+ "unpacked": arguments.unpacked,
351
+ }
352
+
353
+
354
+ def _decode_arguments(data: Mapping[str, Any]) -> Arguments:
355
+ def denoted(value: Any) -> str | None:
356
+ return None if value is None else _string(value)
357
+
358
+ unpacked = data["unpacked"]
359
+ if not isinstance(unpacked, bool):
360
+ raise TypeError(f"expected a boolean, got {unpacked!r}")
361
+ return Arguments(
362
+ tuple(denoted(value) for value in data["positional"]),
363
+ tuple((_string(name), denoted(value)) for name, value in data["keywords"]),
364
+ unpacked,
326
365
  )
coretrace_python/cli.py CHANGED
@@ -3,13 +3,20 @@ from __future__ import annotations
3
3
  import argparse
4
4
  import sys
5
5
  import zipfile
6
+ from datetime import UTC, datetime
6
7
  from pathlib import Path
7
8
 
8
9
  from coretrace_python import __version__, engine
9
10
  from coretrace_python.analysis import AnalysisError
10
11
  from coretrace_python.cache import ProjectCache
11
12
  from coretrace_python.cfg import CFGError
12
- from coretrace_python.dependency import dump_advisories, import_osv, read_osv, render_sbom
13
+ from coretrace_python.dependency import (
14
+ dump_advisories,
15
+ import_osv,
16
+ read_osv,
17
+ render_sbom,
18
+ render_vex,
19
+ )
13
20
  from coretrace_python.findings import Severity
14
21
  from coretrace_python.findings.baseline import Baseline, BaselineError
15
22
  from coretrace_python.frontend import HIRBuildError, ParseError, build_hir
@@ -106,6 +113,14 @@ def build_parser() -> argparse.ArgumentParser:
106
113
  metavar="PATH",
107
114
  help="with --check on a directory, write a CycloneDX bill of materials to PATH",
108
115
  )
116
+ parser.add_argument(
117
+ "--vex",
118
+ type=Path,
119
+ default=None,
120
+ metavar="PATH",
121
+ help="with --check on a directory, write to PATH an OpenVEX document saying whether "
122
+ "each advisory affecting a requirement affects the project",
123
+ )
109
124
  parser.add_argument(
110
125
  "--advisories",
111
126
  action="append",
@@ -182,6 +197,9 @@ def main(argv: list[str] | None = None) -> int:
182
197
  if args.sbom is not None and not (args.check and args.path.is_dir()):
183
198
  print("error: --sbom only applies to --check on a directory", file=sys.stderr)
184
199
  return EXIT_ERROR
200
+ if args.vex is not None and not (args.check and args.path.is_dir()):
201
+ print("error: --vex only applies to --check on a directory", file=sys.stderr)
202
+ return EXIT_ERROR
185
203
  if args.advisories and not (args.check and args.path.is_dir()):
186
204
  print("error: --advisories only applies to --check on a directory", file=sys.stderr)
187
205
  return EXIT_ERROR
@@ -227,6 +245,20 @@ def main(argv: list[str] | None = None) -> int:
227
245
  render_sbom(analysis.dependencies, analysis.advisories, engine.TOOL_NAME, __version__),
228
246
  encoding="utf-8",
229
247
  )
248
+ if args.vex is not None:
249
+ args.vex.write_text(
250
+ render_vex(
251
+ analysis.dependencies,
252
+ analysis.advisories,
253
+ (*analysis.findings, *analysis.suppressed, *analysis.accepted),
254
+ analysis.coverage,
255
+ args.path,
256
+ engine.TOOL_NAME,
257
+ __version__,
258
+ datetime.now(UTC),
259
+ ),
260
+ encoding="utf-8",
261
+ )
230
262
  else:
231
263
  file_analysis = engine.analyze_file(SourceManager().load_file(args.path), plugin_roots)
232
264
  findings, coverage = file_analysis.findings, file_analysis.coverage
@@ -22,6 +22,7 @@ from coretrace_python.dependency.graph import (
22
22
  )
23
23
  from coretrace_python.dependency.policy import POLICY_FILE, Policy, apply_policy, load_policy
24
24
  from coretrace_python.dependency.sbom import render_sbom
25
+ from coretrace_python.dependency.vex import render_vex
25
26
 
26
27
  __all__ = [
27
28
  "ADVISORY_FILE",
@@ -45,4 +46,5 @@ __all__ = [
45
46
  "parse_dependencies",
46
47
  "read_osv",
47
48
  "render_sbom",
49
+ "render_vex",
48
50
  ]
@@ -138,14 +138,7 @@ def dump_advisories(advisories: Iterable[Advisory]) -> str:
138
138
  "severity": a.severity.value,
139
139
  "affected_symbols": [str(s) for s in a.affected_symbols],
140
140
  "aliases": list(a.aliases),
141
- "entry_points": [
142
- {
143
- "symbol": str(e.symbol),
144
- "justification": e.justification,
145
- "conditions": [_condition_entry(c) for c in e.conditions],
146
- }
147
- for e in a.entry_points
148
- ],
141
+ "entry_points": [_entry_point_entry(e) for e in a.entry_points],
149
142
  "modules": list(a.modules),
150
143
  }
151
144
  for a in advisories
@@ -154,12 +147,27 @@ def dump_advisories(advisories: Iterable[Advisory]) -> str:
154
147
  return json.dumps(document, indent=2) + "\n"
155
148
 
156
149
 
150
+ def _entry_point_entry(entry_point: AdvisoryEntryPoint) -> dict[str, Any]:
151
+ entry: dict[str, Any] = {
152
+ "symbol": str(entry_point.symbol),
153
+ "justification": entry_point.justification,
154
+ "conditions": [_condition_entry(c) for c in entry_point.conditions],
155
+ }
156
+ if entry_point.read:
157
+ entry["read"] = True
158
+ return entry
159
+
160
+
157
161
  def _condition_entry(condition: Condition) -> dict[str, Any]:
158
162
  entry: dict[str, Any] = {"kind": condition.kind, "text": condition.text}
159
163
  if condition.argument is not None:
160
164
  entry["argument"] = condition.argument
161
165
  if condition.values:
162
166
  entry["values"] = list(condition.values)
167
+ if condition.position is not None:
168
+ entry["position"] = condition.position
169
+ if condition.default:
170
+ entry["default"] = True
163
171
  return entry
164
172
 
165
173
 
@@ -190,16 +198,29 @@ def _advisory(entry: Mapping[str, Any]) -> Advisory:
190
198
 
191
199
 
192
200
  def _entry_point(entry: Mapping[str, Any]) -> AdvisoryEntryPoint:
201
+ read = entry.get("read", False)
202
+ if not isinstance(read, bool):
203
+ raise TypeError(f"entry point read must be true or false, got {read!r}")
193
204
  return AdvisoryEntryPoint(
194
205
  SymbolId(str(entry["symbol"])),
195
206
  str(entry["justification"]),
196
- tuple(
197
- Condition(
198
- str(c["kind"]),
199
- str(c["text"]),
200
- None if c.get("argument") is None else str(c["argument"]),
201
- tuple(str(v) for v in c.get("values") or []),
202
- )
203
- for c in entry.get("conditions") or []
204
- ),
207
+ tuple(_condition(c) for c in entry.get("conditions") or []),
208
+ read,
209
+ )
210
+
211
+
212
+ def _condition(entry: Mapping[str, Any]) -> Condition:
213
+ position = entry.get("position")
214
+ if position is not None and (not isinstance(position, int) or isinstance(position, bool) or position < 0):
215
+ raise TypeError(f"condition position must be a non-negative integer, got {position!r}")
216
+ default = entry.get("default", False)
217
+ if not isinstance(default, bool):
218
+ raise TypeError(f"condition default must be true or false, got {default!r}")
219
+ return Condition(
220
+ str(entry["kind"]),
221
+ str(entry["text"]),
222
+ None if entry.get("argument") is None else str(entry["argument"]),
223
+ tuple(str(v) for v in entry.get("values") or []),
224
+ position,
225
+ default,
205
226
  )
@@ -10,10 +10,18 @@ are correlated here into one high-confidence ``exploitable-vulnerability`` findi
10
10
  from __future__ import annotations
11
11
 
12
12
  from collections.abc import Iterable, Mapping
13
-
14
- from coretrace_python.dependency.graph import DIRECT, Advisory, DependencyGraph
13
+ from dataclasses import dataclass
14
+
15
+ from coretrace_python.dependency.graph import (
16
+ DIRECT,
17
+ Advisory,
18
+ AdvisoryEntryPoint,
19
+ Condition,
20
+ DependencyGraph,
21
+ )
15
22
  from coretrace_python.findings import Confidence, Finding, Severity
16
23
  from coretrace_python.findings.refutation import Status, Verdict, Verdicts
24
+ from coretrace_python.interprocedural import Arguments, CallSite, ExternalSymbol
17
25
  from coretrace_python.semantic.symbols import SymbolId
18
26
  from coretrace_python.taint import Sink, TaintFlow, TaintKind
19
27
 
@@ -40,10 +48,56 @@ def advisory_sinks(affected: Affected) -> tuple[Sink, ...]:
40
48
  return tuple(Sink(symbol, TaintKind.ADVISORY) for symbol in affected)
41
49
 
42
50
 
43
- def evidence(advisory: Advisory, symbol: SymbolId, level: str) -> dict[str, str]:
51
+ @dataclass(frozen=True)
52
+ class ConditionCheck:
53
+ """What one call tells of an entry point's conditions: those it meets, those left to
54
+ review (every semantic one, and every argument one the call does not decide), and
55
+ the argument condition it contradicts with what it passes instead (None: absent),
56
+ in which case the call does not reach the vulnerability."""
57
+
58
+ met: tuple[Condition, ...] = ()
59
+ pending: tuple[Condition, ...] = ()
60
+ contradicted: Condition | None = None
61
+ passed: str | None = None
62
+
63
+
64
+ def check_conditions(entry: AdvisoryEntryPoint | None, arguments: Arguments | None) -> ConditionCheck:
65
+ """Decide ``entry``'s conditions against what the call's ``arguments`` denote."""
66
+
67
+ if entry is None:
68
+ return ConditionCheck()
69
+ met: list[Condition] = []
70
+ pending: list[Condition] = []
71
+ for condition in entry.conditions:
72
+ given = arguments.given(condition.argument, condition.position) if condition.checkable and arguments is not None else None
73
+ if given is None:
74
+ pending.append(condition)
75
+ continue
76
+ explicit, value = given
77
+ if explicit and value is None:
78
+ pending.append(condition)
79
+ continue
80
+ affected = value in condition.values if explicit else condition.default
81
+ if not affected:
82
+ return ConditionCheck(tuple(met), tuple(pending), condition, value)
83
+ met.append(condition)
84
+ return ConditionCheck(tuple(met), tuple(pending))
85
+
86
+
87
+ def ruled_out(module: str, site: CallSite, check: ConditionCheck) -> str:
88
+ """Which call a contradicted condition rules out, and why: ``app:12
89
+ python.yaml.load(Loader=python.yaml.SafeLoader)``."""
90
+
91
+ assert check.contradicted is not None and isinstance(site.target, ExternalSymbol)
92
+ argument = check.contradicted.argument or f"#{check.contradicted.position}"
93
+ passed = f"{argument} absent" if check.passed is None else f"{argument}={check.passed}"
94
+ return f"{module}:{site.location.start_line} {site.target.symbol}({passed})"
95
+
96
+
97
+ def evidence(advisory: Advisory, symbol: SymbolId, level: str, check: ConditionCheck) -> dict[str, str]:
44
98
  """What a finding keeps of the advisory for ``symbol``: the level of evidence
45
- established, how the symbol relates to the vulnerability and under which
46
- conditions, the ones the engine could not check listed as pending review."""
99
+ established, how the symbol relates to the vulnerability, and its conditions — those
100
+ the call meets and those left to review."""
47
101
 
48
102
  metadata = {"advisory": advisory.id, "package": advisory.package, "symbol": str(symbol), "level": level}
49
103
  entry = advisory.entry_point(symbol)
@@ -54,9 +108,10 @@ def evidence(advisory: Advisory, symbol: SymbolId, level: str) -> dict[str, str]
54
108
  metadata["justification"] = entry.justification
55
109
  if entry.conditions:
56
110
  metadata["conditions"] = "; ".join(c.text for c in entry.conditions)
57
- # ponytail: no condition is checked yet, so every one awaits review; argument
58
- # conditions get checked at the call site once call sites carry argument symbols.
59
- metadata["conditions_pending_review"] = "; ".join(c.text for c in entry.conditions)
111
+ if check.met:
112
+ metadata["conditions_met"] = "; ".join(c.text for c in check.met)
113
+ if check.pending:
114
+ metadata["conditions_pending_review"] = "; ".join(c.text for c in check.pending)
60
115
  return metadata
61
116
 
62
117
 
@@ -77,21 +132,22 @@ def correlate(
77
132
  if verdict is not None and verdict.status is Status.REFUTED:
78
133
  continue
79
134
  hotspot = verdict is not None and verdict.status is Status.HOTSPOT
80
- findings.extend(
81
- _exploitable(function, flow, advisory, verdict, hotspot) for advisory in affected.get(flow.sink.symbol, ())
82
- )
135
+ for advisory in affected.get(flow.sink.symbol, ()):
136
+ check = check_conditions(advisory.entry_point(flow.sink.symbol), flow.sink_arguments)
137
+ if check.contradicted is None:
138
+ findings.append(_exploitable(function, flow, advisory, verdict, hotspot, check))
83
139
  return tuple(findings)
84
140
 
85
141
 
86
142
  def _exploitable(
87
- function: str, flow: TaintFlow, advisory: Advisory, verdict: Verdict | None, hotspot: bool
143
+ function: str, flow: TaintFlow, advisory: Advisory, verdict: Verdict | None, hotspot: bool, check: ConditionCheck
88
144
  ) -> Finding:
89
145
  message = (
90
146
  f"{advisory.id}: {flow.source.label} input reaches {flow.sink.symbol}, affected in "
91
147
  f"the required {advisory.package} {advisory.vulnerable}: {advisory.summary}"
92
148
  )
93
149
  metadata = {
94
- **evidence(advisory, flow.sink.symbol, "exploitable"),
150
+ **evidence(advisory, flow.sink.symbol, "exploitable", check),
95
151
  "source": str(flow.source.symbol),
96
152
  "source_label": flow.source.label,
97
153
  "verdict": "hotspot" if hotspot else "vulnerability",