mcp-capdiff 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_audit/__init__.py +6 -0
- mcp_audit/__main__.py +6 -0
- mcp_audit/analysis/__init__.py +1 -0
- mcp_audit/analysis/diff.py +235 -0
- mcp_audit/analysis/scan.py +29 -0
- mcp_audit/cli/__init__.py +1 -0
- mcp_audit/cli/main.py +112 -0
- mcp_audit/models/__init__.py +12 -0
- mcp_audit/models/capability.py +73 -0
- mcp_audit/models/finding.py +87 -0
- mcp_audit/models/report.py +120 -0
- mcp_audit/policy/__init__.py +1 -0
- mcp_audit/policy/evaluator.py +29 -0
- mcp_audit/policy/loader.py +138 -0
- mcp_audit/reporters/__init__.py +1 -0
- mcp_audit/reporters/json.py +9 -0
- mcp_audit/reporters/manifest.py +30 -0
- mcp_audit/reporters/sarif.py +94 -0
- mcp_audit/reporters/terminal.py +97 -0
- mcp_audit/rules/__init__.py +1 -0
- mcp_audit/rules/engine.py +177 -0
- mcp_audit/scanners/__init__.py +1 -0
- mcp_audit/scanners/source/__init__.py +1 -0
- mcp_audit/scanners/source/python.py +441 -0
- mcp_capdiff-0.1.1.dist-info/METADATA +173 -0
- mcp_capdiff-0.1.1.dist-info/RECORD +30 -0
- mcp_capdiff-0.1.1.dist-info/WHEEL +5 -0
- mcp_capdiff-0.1.1.dist-info/entry_points.txt +2 -0
- mcp_capdiff-0.1.1.dist-info/licenses/LICENSE +21 -0
- mcp_capdiff-0.1.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from mcp_audit.models.capability import Tool
|
|
6
|
+
from mcp_audit.models.finding import Evidence, Finding, Location, Severity
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def evaluate_tools(tools: list[Tool]) -> list[Finding]:
|
|
10
|
+
findings: list[Finding] = []
|
|
11
|
+
for tool in tools:
|
|
12
|
+
findings.extend(_tool_findings(tool))
|
|
13
|
+
findings.extend(_cross_tool_findings(tools))
|
|
14
|
+
return findings
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _tool_findings(tool: Tool) -> list[Finding]:
|
|
18
|
+
capability = tool.capability
|
|
19
|
+
findings: list[Finding] = []
|
|
20
|
+
|
|
21
|
+
if capability.execution in {"shell", "arbitrary_code"}:
|
|
22
|
+
evidence = _evidence(tool, {"execution"})
|
|
23
|
+
findings.append(
|
|
24
|
+
Finding(
|
|
25
|
+
rule_id="MCP001",
|
|
26
|
+
title="Arbitrary shell execution",
|
|
27
|
+
severity=Severity.CRITICAL,
|
|
28
|
+
tool=tool.name,
|
|
29
|
+
message="A tool parameter can reach shell or dynamic code execution.",
|
|
30
|
+
impact="An MCP client could execute commands with the server process's privileges.",
|
|
31
|
+
recommendation="Replace arbitrary command input with an allowlisted operation and avoid shell=True.",
|
|
32
|
+
location=_evidence_location(evidence, tool),
|
|
33
|
+
evidence=evidence,
|
|
34
|
+
)
|
|
35
|
+
)
|
|
36
|
+
if capability.filesystem == "unrestricted":
|
|
37
|
+
evidence = _evidence(tool, {"filesystem"})
|
|
38
|
+
findings.append(
|
|
39
|
+
Finding(
|
|
40
|
+
rule_id="MCP002",
|
|
41
|
+
title="Unrestricted filesystem access",
|
|
42
|
+
severity=Severity.HIGH,
|
|
43
|
+
tool=tool.name,
|
|
44
|
+
message="A tool-controlled path reaches a filesystem operation without an approved-root check.",
|
|
45
|
+
impact="The MCP client may read or overwrite files available to the server process.",
|
|
46
|
+
recommendation="Resolve paths under an allowlisted root and reject paths that escape it.",
|
|
47
|
+
location=_evidence_location(evidence, tool),
|
|
48
|
+
evidence=evidence,
|
|
49
|
+
)
|
|
50
|
+
)
|
|
51
|
+
if capability.network == "unrestricted_outbound":
|
|
52
|
+
evidence = _evidence(tool, {"network"})
|
|
53
|
+
findings.append(
|
|
54
|
+
Finding(
|
|
55
|
+
rule_id="MCP003",
|
|
56
|
+
title="Arbitrary URL or SSRF surface",
|
|
57
|
+
severity=Severity.HIGH,
|
|
58
|
+
tool=tool.name,
|
|
59
|
+
message="A tool-controlled destination reaches an outbound HTTP request without a hostname allowlist.",
|
|
60
|
+
impact="The MCP client may reach arbitrary internet, internal, localhost, or metadata endpoints.",
|
|
61
|
+
recommendation="Parse the URL, require HTTPS, block private ranges, and enforce an explicit hostname allowlist.",
|
|
62
|
+
location=_evidence_location(evidence, tool),
|
|
63
|
+
evidence=evidence,
|
|
64
|
+
destination=capability.network_destination,
|
|
65
|
+
)
|
|
66
|
+
)
|
|
67
|
+
if capability.side_effect in {"external_write", "destructive_action"} and not capability.requires_approval:
|
|
68
|
+
evidence = _evidence(tool, {"network", "approval", "side_effect", "destructive"}) or tool.evidence
|
|
69
|
+
findings.append(
|
|
70
|
+
Finding(
|
|
71
|
+
rule_id="MCP004",
|
|
72
|
+
title="High-impact side effect without approval",
|
|
73
|
+
severity=Severity.HIGH if capability.side_effect == "destructive_action" else Severity.MEDIUM,
|
|
74
|
+
tool=tool.name,
|
|
75
|
+
message="The tool performs an external or high-impact action without an approval boundary.",
|
|
76
|
+
impact="A model or compromised client could trigger consequential actions without human confirmation.",
|
|
77
|
+
recommendation="Require explicit approval for external writes, destructive actions, and financial actions.",
|
|
78
|
+
location=_evidence_location(evidence, tool),
|
|
79
|
+
evidence=evidence,
|
|
80
|
+
)
|
|
81
|
+
)
|
|
82
|
+
for parameter in tool.parameters:
|
|
83
|
+
if not parameter.bounded and any(token in parameter.name.lower() for token in {"path", "url", "host", "query", "command"}):
|
|
84
|
+
findings.append(
|
|
85
|
+
Finding(
|
|
86
|
+
rule_id="MCP007",
|
|
87
|
+
title="Unbounded input",
|
|
88
|
+
severity=Severity.MEDIUM,
|
|
89
|
+
tool=tool.name,
|
|
90
|
+
message=f"Security-sensitive parameter '{parameter.name}' has no detected bounds or allowlist.",
|
|
91
|
+
impact="Oversized or unconstrained input increases injection, traversal, and resource-exhaustion risk.",
|
|
92
|
+
recommendation="Add length, scheme, enum, path-root, or hostname allowlist validation.",
|
|
93
|
+
location=_location(tool),
|
|
94
|
+
evidence=[Evidence(message=f"Parameter type: {parameter.annotation or 'untyped'}", kind="parameter")],
|
|
95
|
+
)
|
|
96
|
+
)
|
|
97
|
+
if capability.destructive and not capability.requires_approval:
|
|
98
|
+
evidence = _evidence(tool, {"destructive"})
|
|
99
|
+
findings.append(
|
|
100
|
+
Finding(
|
|
101
|
+
rule_id="MCP010",
|
|
102
|
+
title="Destructive tool exposed",
|
|
103
|
+
severity=Severity.HIGH,
|
|
104
|
+
tool=tool.name,
|
|
105
|
+
message="The tool name or description indicates a destructive action without explicit approval.",
|
|
106
|
+
impact="A mistaken or malicious invocation could irreversibly delete or revoke data or access.",
|
|
107
|
+
recommendation="Require human approval and scoped authorization for destructive tools.",
|
|
108
|
+
location=_location(tool),
|
|
109
|
+
evidence=evidence or [Evidence(message="Tool semantics indicate a destructive action.", kind="capability")],
|
|
110
|
+
)
|
|
111
|
+
)
|
|
112
|
+
return findings
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _cross_tool_findings(tools: list[Tool]) -> list[Finding]:
|
|
116
|
+
findings: list[Finding] = []
|
|
117
|
+
sensitive_sources = [tool for tool in tools if tool.capability.data_access and tool.capability.sensitivity != "public"]
|
|
118
|
+
external_sinks = [
|
|
119
|
+
tool
|
|
120
|
+
for tool in tools
|
|
121
|
+
if tool.capability.network == "unrestricted_outbound" or tool.capability.side_effect in {"external_write", "public_action"}
|
|
122
|
+
]
|
|
123
|
+
for source in sensitive_sources:
|
|
124
|
+
for sink in external_sinks:
|
|
125
|
+
if source.name == sink.name or source.context != sink.context:
|
|
126
|
+
continue
|
|
127
|
+
unrestricted = sink.capability.network == "unrestricted_outbound"
|
|
128
|
+
data_classes = sorted(source.capability.data_access)
|
|
129
|
+
sink_evidence = _evidence(sink, {"network", "side_effect"})
|
|
130
|
+
findings.append(
|
|
131
|
+
Finding(
|
|
132
|
+
rule_id="MCP005",
|
|
133
|
+
title="Potential data exfiltration path",
|
|
134
|
+
severity=Severity.CRITICAL if unrestricted else Severity.HIGH,
|
|
135
|
+
tool=f"{source.name} -> {sink.name}",
|
|
136
|
+
message=f"{', '.join(data_classes)} data returned by '{source.name}' can enter agent context and reach '{sink.name}'.",
|
|
137
|
+
impact="Sensitive data exposed to agent context may be sent to an external destination in a later tool call.",
|
|
138
|
+
recommendation="Constrain the sink destination, require approval, and separate sensitive read tools from external write tools.",
|
|
139
|
+
location=_evidence_location(sink_evidence, sink),
|
|
140
|
+
evidence=[
|
|
141
|
+
Evidence(
|
|
142
|
+
message=f"Source '{source.name}' returns classified data: {', '.join(data_classes)}.",
|
|
143
|
+
kind="capability_source",
|
|
144
|
+
location=_location(source),
|
|
145
|
+
),
|
|
146
|
+
Evidence(
|
|
147
|
+
message=f"Sink '{sink.name}' permits {sink.capability.network or sink.capability.side_effect}.",
|
|
148
|
+
kind="capability_sink",
|
|
149
|
+
location=_evidence_location(sink_evidence, sink),
|
|
150
|
+
snippet=sink_evidence[0].snippet if sink_evidence else None,
|
|
151
|
+
),
|
|
152
|
+
],
|
|
153
|
+
path=[source.name, "agent_context", sink.name],
|
|
154
|
+
data_classification=data_classes,
|
|
155
|
+
destination=sink.capability.network_destination or sink.capability.network,
|
|
156
|
+
)
|
|
157
|
+
)
|
|
158
|
+
return findings
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _evidence(tool: Tool, kinds: set[str]) -> list[Evidence]:
|
|
162
|
+
return [item for item in tool.evidence if item.kind in kinds]
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _evidence_location(evidence: list[Evidence], tool: Tool) -> Location:
|
|
166
|
+
return next((item.location for item in evidence if item.location is not None), _location(tool))
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _location(tool: Tool) -> Location:
|
|
170
|
+
source = tool.source
|
|
171
|
+
if ":" not in source:
|
|
172
|
+
return Location(path=source)
|
|
173
|
+
path, line = source.rsplit(":", 1)
|
|
174
|
+
try:
|
|
175
|
+
return Location(path=str(Path(path)), line=int(line))
|
|
176
|
+
except ValueError:
|
|
177
|
+
return Location(path=source)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Source and runtime scanners."""
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Source scanners."""
|
|
@@ -0,0 +1,441 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import ast
|
|
4
|
+
import os
|
|
5
|
+
import re
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from mcp_audit.models.capability import Capability, Parameter, Tool
|
|
9
|
+
from mcp_audit.models.finding import Evidence, Location
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
SENSITIVE_TOKENS = {
|
|
13
|
+
"confidential": {"confidential", "private"},
|
|
14
|
+
"financial": {"invoice", "payment", "payroll", "refund", "charge", "bank"},
|
|
15
|
+
"pii": {"customer", "email", "employee", "person", "user", "pii"},
|
|
16
|
+
"secret": {"secret", "token", "password", "credential", "api_key", "apikey"},
|
|
17
|
+
}
|
|
18
|
+
NETWORK_CALLS = {
|
|
19
|
+
"requests.get", "requests.post", "requests.put", "requests.patch", "requests.delete",
|
|
20
|
+
"requests.request", "httpx.get", "httpx.post", "httpx.put", "httpx.patch",
|
|
21
|
+
"httpx.delete", "urllib.request.urlopen",
|
|
22
|
+
}
|
|
23
|
+
NETWORK_WRITES = {
|
|
24
|
+
"requests.post", "requests.put", "requests.patch", "requests.delete",
|
|
25
|
+
"httpx.post", "httpx.put", "httpx.patch", "httpx.delete",
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def scan_python_sources(root: Path) -> list[Tool]:
|
|
30
|
+
files = _python_files(root)
|
|
31
|
+
tools: list[Tool] = []
|
|
32
|
+
for file_path in files:
|
|
33
|
+
try:
|
|
34
|
+
source_text = file_path.read_text(encoding="utf-8")
|
|
35
|
+
tree = ast.parse(source_text, filename=str(file_path))
|
|
36
|
+
except SyntaxError as exc:
|
|
37
|
+
raise ValueError(f"cannot parse Python source {file_path}:{exc.lineno}: {exc.msg}") from exc
|
|
38
|
+
except UnicodeDecodeError as exc:
|
|
39
|
+
raise ValueError(f"Python source is not UTF-8: {file_path}") from exc
|
|
40
|
+
except OSError as exc:
|
|
41
|
+
raise OSError(exc.errno, f"cannot read Python source {file_path}: {exc.strerror}", str(file_path)) from exc
|
|
42
|
+
context_path = file_path.name if root.is_file() else str(file_path.relative_to(root))
|
|
43
|
+
visitor = FastMCPVisitor(file_path, source_text, context_path)
|
|
44
|
+
visitor.visit(tree)
|
|
45
|
+
tools.extend(visitor.tools)
|
|
46
|
+
return tools
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _python_files(root: Path) -> list[Path]:
|
|
50
|
+
try:
|
|
51
|
+
root.stat()
|
|
52
|
+
except FileNotFoundError as exc:
|
|
53
|
+
raise FileNotFoundError(f"scan target does not exist: {root}") from exc
|
|
54
|
+
except OSError as exc:
|
|
55
|
+
raise OSError(exc.errno, f"cannot access scan target {root}: {exc.strerror}", str(root)) from exc
|
|
56
|
+
|
|
57
|
+
if root.is_file():
|
|
58
|
+
if root.suffix != ".py":
|
|
59
|
+
raise ValueError(f"scan target is not a Python file: {root}")
|
|
60
|
+
return [root]
|
|
61
|
+
if not root.is_dir():
|
|
62
|
+
raise ValueError(f"scan target is not a file or directory: {root}")
|
|
63
|
+
|
|
64
|
+
files: list[Path] = []
|
|
65
|
+
|
|
66
|
+
def raise_walk_error(error: OSError) -> None:
|
|
67
|
+
raise OSError(error.errno, f"cannot traverse scan target: {error.strerror}", error.filename) from error
|
|
68
|
+
|
|
69
|
+
for directory, dirnames, filenames in os.walk(root, onerror=raise_walk_error):
|
|
70
|
+
dirnames[:] = [name for name in dirnames if name not in {".git", ".venv", "venv", "__pycache__"}]
|
|
71
|
+
base = Path(directory)
|
|
72
|
+
files.extend(base / name for name in filenames if name.endswith(".py"))
|
|
73
|
+
return sorted(files)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class FastMCPVisitor(ast.NodeVisitor):
|
|
77
|
+
def __init__(self, file_path: Path, source_text: str, context_path: str) -> None:
|
|
78
|
+
self.file_path = file_path
|
|
79
|
+
self.source_text = source_text
|
|
80
|
+
self.context_path = context_path
|
|
81
|
+
self.tools: list[Tool] = []
|
|
82
|
+
|
|
83
|
+
def visit_FunctionDef(self, node: ast.FunctionDef) -> None:
|
|
84
|
+
if _is_tool_function(node):
|
|
85
|
+
self.tools.append(_tool_from_function(self.file_path, self.source_text, self.context_path, node))
|
|
86
|
+
self.generic_visit(node)
|
|
87
|
+
|
|
88
|
+
visit_AsyncFunctionDef = visit_FunctionDef
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _is_tool_function(node: ast.FunctionDef | ast.AsyncFunctionDef) -> bool:
|
|
92
|
+
for decorator in node.decorator_list:
|
|
93
|
+
target = decorator.func if isinstance(decorator, ast.Call) else decorator
|
|
94
|
+
if isinstance(target, ast.Attribute) and target.attr == "tool":
|
|
95
|
+
return True
|
|
96
|
+
if isinstance(target, ast.Name) and target.id in {"tool", "mcp_tool"}:
|
|
97
|
+
return True
|
|
98
|
+
return False
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _tool_from_function(
|
|
102
|
+
file_path: Path,
|
|
103
|
+
source_text: str,
|
|
104
|
+
context_path: str,
|
|
105
|
+
node: ast.FunctionDef | ast.AsyncFunctionDef,
|
|
106
|
+
) -> Tool:
|
|
107
|
+
name = _tool_name(node)
|
|
108
|
+
description = ast.get_docstring(node) or ""
|
|
109
|
+
parameters = [_parameter(arg, node) for arg in node.args.args if arg.arg not in {"self", "cls"}]
|
|
110
|
+
analyzer = FunctionAnalyzer(file_path, source_text, parameters)
|
|
111
|
+
analyzer.visit(node)
|
|
112
|
+
capability = Capability()
|
|
113
|
+
lower_text = f"{name.lower()} {description.lower()}"
|
|
114
|
+
data_classes = _data_classes(lower_text)
|
|
115
|
+
|
|
116
|
+
if analyzer.shell_execution:
|
|
117
|
+
capability.execution = "shell"
|
|
118
|
+
if analyzer.arbitrary_code:
|
|
119
|
+
capability.execution = "arbitrary_code"
|
|
120
|
+
if analyzer.filesystem_read:
|
|
121
|
+
capability.filesystem = "allowlisted_files" if analyzer.path_guard else "unrestricted"
|
|
122
|
+
capability.data_access.update(data_classes or {"internal"})
|
|
123
|
+
if analyzer.database_read:
|
|
124
|
+
capability.data_access.update(data_classes or {"internal"})
|
|
125
|
+
if capability.data_access:
|
|
126
|
+
capability.sensitivity = _highest_sensitivity(capability.data_access)
|
|
127
|
+
if analyzer.filesystem_write:
|
|
128
|
+
capability.filesystem = "allowlisted_files" if analyzer.path_guard else "unrestricted"
|
|
129
|
+
capability.side_effect = "local_write"
|
|
130
|
+
if analyzer.network_call:
|
|
131
|
+
capability.network = "allowlisted_hosts" if analyzer.host_guard or analyzer.fixed_network_destination else "unrestricted_outbound"
|
|
132
|
+
capability.network_destination = analyzer.network_destination
|
|
133
|
+
if analyzer.network_write:
|
|
134
|
+
capability.side_effect = "external_write"
|
|
135
|
+
if _looks_external_write(lower_text):
|
|
136
|
+
capability.side_effect = "external_write"
|
|
137
|
+
analyzer.evidence.append(
|
|
138
|
+
Evidence(
|
|
139
|
+
message="Tool name or description indicates an external write.",
|
|
140
|
+
kind="side_effect",
|
|
141
|
+
location=_function_location(file_path, node),
|
|
142
|
+
)
|
|
143
|
+
)
|
|
144
|
+
if _looks_destructive(lower_text):
|
|
145
|
+
capability.destructive = True
|
|
146
|
+
capability.side_effect = "destructive_action"
|
|
147
|
+
analyzer.evidence.append(
|
|
148
|
+
Evidence(
|
|
149
|
+
message="Tool name or description indicates a destructive action.",
|
|
150
|
+
kind="destructive",
|
|
151
|
+
location=_function_location(file_path, node),
|
|
152
|
+
)
|
|
153
|
+
)
|
|
154
|
+
if "financial" in data_classes:
|
|
155
|
+
capability.financial = True
|
|
156
|
+
capability.data_access.add("financial")
|
|
157
|
+
capability.requires_approval = _requires_approval(node) or _has_approval_guard(node)
|
|
158
|
+
capability.audit_metadata = any(token in lower_text for token in {"audit", "trace", "request_id"})
|
|
159
|
+
|
|
160
|
+
for parameter in parameters:
|
|
161
|
+
if (
|
|
162
|
+
analyzer.host_guard and parameter.name in analyzer.host_guarded_parameters
|
|
163
|
+
) or (
|
|
164
|
+
analyzer.path_guard and parameter.name in analyzer.path_guarded_parameters
|
|
165
|
+
):
|
|
166
|
+
parameter.bounded = True
|
|
167
|
+
|
|
168
|
+
return Tool(
|
|
169
|
+
name=name,
|
|
170
|
+
source=f"{file_path}:{node.lineno}",
|
|
171
|
+
context=_tool_context(context_path, node),
|
|
172
|
+
description=description,
|
|
173
|
+
parameters=parameters,
|
|
174
|
+
capability=capability,
|
|
175
|
+
evidence=analyzer.evidence,
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
class FunctionAnalyzer(ast.NodeVisitor):
|
|
180
|
+
def __init__(self, file_path: Path, source_text: str, parameters: list[Parameter]) -> None:
|
|
181
|
+
self.file_path = file_path
|
|
182
|
+
self.source_text = source_text
|
|
183
|
+
self.parameter_names = {parameter.name for parameter in parameters}
|
|
184
|
+
self.aliases: dict[str, set[str]] = {}
|
|
185
|
+
self.host_guarded_parameters: set[str] = set()
|
|
186
|
+
self.path_guarded_parameters: set[str] = set()
|
|
187
|
+
self.shell_execution = False
|
|
188
|
+
self.arbitrary_code = False
|
|
189
|
+
self.filesystem_read = False
|
|
190
|
+
self.filesystem_write = False
|
|
191
|
+
self.database_read = False
|
|
192
|
+
self.network_call = False
|
|
193
|
+
self.network_write = False
|
|
194
|
+
self.path_guard_lines: list[int] = []
|
|
195
|
+
self.host_guard_lines: list[int] = []
|
|
196
|
+
self.filesystem_operation_lines: list[int] = []
|
|
197
|
+
self.network_call_lines: list[int] = []
|
|
198
|
+
self.fixed_network_destination = False
|
|
199
|
+
self.network_destination: str | None = None
|
|
200
|
+
self.evidence: list[Evidence] = []
|
|
201
|
+
|
|
202
|
+
def visit_Assign(self, node: ast.Assign) -> None:
|
|
203
|
+
roots = self._parameter_roots(node.value)
|
|
204
|
+
for target in node.targets:
|
|
205
|
+
if isinstance(target, ast.Name) and roots:
|
|
206
|
+
self.aliases[target.id] = roots
|
|
207
|
+
self.generic_visit(node)
|
|
208
|
+
|
|
209
|
+
def visit_Call(self, node: ast.Call) -> None:
|
|
210
|
+
call_name = _call_name(node.func)
|
|
211
|
+
if call_name in {"eval", "exec", "compile"}:
|
|
212
|
+
self.arbitrary_code = True
|
|
213
|
+
self._record(node, f"Dynamic {call_name}(...) executes model-controlled code.", "execution")
|
|
214
|
+
if call_name in {"subprocess.run", "subprocess.call", "subprocess.Popen", "os.system"}:
|
|
215
|
+
if call_name == "os.system" or _keyword_is_true(node, "shell"):
|
|
216
|
+
self.shell_execution = True
|
|
217
|
+
self._record(node, f"{call_name}(...) invokes a command through a shell.", "execution")
|
|
218
|
+
elif call_name.startswith("subprocess.") and _has_model_controlled_executable(node, self._parameter_roots):
|
|
219
|
+
self.shell_execution = True
|
|
220
|
+
self._record(node, f"{call_name}(...) uses a model-controlled executable or interpreter payload.", "execution")
|
|
221
|
+
if call_name in {"open", "read_text", "read_bytes", "Path.open", "Path.read_text", "Path.read_bytes"} or call_name.endswith((".read_text", ".read_bytes")):
|
|
222
|
+
self.filesystem_read = True
|
|
223
|
+
self.filesystem_operation_lines.append(node.lineno)
|
|
224
|
+
self._record(node, "A filesystem read is reachable from this MCP tool.", "filesystem")
|
|
225
|
+
if call_name in {"write_text", "write_bytes", "Path.write_text", "Path.write_bytes"} or call_name.endswith((".write_text", ".write_bytes")):
|
|
226
|
+
self.filesystem_write = True
|
|
227
|
+
self.filesystem_operation_lines.append(node.lineno)
|
|
228
|
+
self._record(node, "A filesystem write is reachable from this MCP tool.", "filesystem")
|
|
229
|
+
if call_name.endswith((".fetchone", ".fetchall", ".fetchmany")) or _is_select_query(node, call_name):
|
|
230
|
+
self.database_read = True
|
|
231
|
+
self._record(node, "A database read returns data through this MCP tool.", "data_access")
|
|
232
|
+
if call_name in NETWORK_CALLS:
|
|
233
|
+
self.network_call = True
|
|
234
|
+
self.network_call_lines.append(node.lineno)
|
|
235
|
+
self.network_write = self.network_write or call_name in NETWORK_WRITES
|
|
236
|
+
argument = node.args[0] if node.args else _keyword_value(node, "url")
|
|
237
|
+
roots = self._parameter_roots(argument)
|
|
238
|
+
if isinstance(argument, ast.Constant) and isinstance(argument.value, str):
|
|
239
|
+
self.fixed_network_destination = True
|
|
240
|
+
self.network_destination = argument.value
|
|
241
|
+
elif roots:
|
|
242
|
+
self.network_destination = "unrestricted external URL"
|
|
243
|
+
self._record(node, f"Tool input {', '.join(sorted(roots)) or 'data'} flows into {call_name}(...).", "network")
|
|
244
|
+
self.generic_visit(node)
|
|
245
|
+
|
|
246
|
+
def visit_If(self, node: ast.If) -> None:
|
|
247
|
+
if _rejects_branch(node.body):
|
|
248
|
+
roots = self._parameter_roots(node.test)
|
|
249
|
+
if _is_host_allowlist_guard(node.test) and roots:
|
|
250
|
+
self.host_guard_lines.append(node.lineno)
|
|
251
|
+
self.host_guarded_parameters.update(roots)
|
|
252
|
+
self._record(node.test, "Destination hostname is checked against an explicit allowlist.", "network_guard")
|
|
253
|
+
if _is_path_root_guard(node.test) and roots:
|
|
254
|
+
self.path_guard_lines.append(node.lineno)
|
|
255
|
+
self.path_guarded_parameters.update(roots)
|
|
256
|
+
self._record(node.test, "Resolved path is checked against an approved root.", "filesystem_guard")
|
|
257
|
+
if _is_approval_guard(node.test):
|
|
258
|
+
self._record(node.test, "The operation is rejected unless approval is present.", "approval")
|
|
259
|
+
self.generic_visit(node)
|
|
260
|
+
|
|
261
|
+
@property
|
|
262
|
+
def host_guard(self) -> bool:
|
|
263
|
+
return _guard_precedes_operations(self.host_guard_lines, self.network_call_lines)
|
|
264
|
+
|
|
265
|
+
@property
|
|
266
|
+
def path_guard(self) -> bool:
|
|
267
|
+
return _guard_precedes_operations(self.path_guard_lines, self.filesystem_operation_lines)
|
|
268
|
+
|
|
269
|
+
def _parameter_roots(self, node: ast.AST | None) -> set[str]:
|
|
270
|
+
if node is None:
|
|
271
|
+
return set()
|
|
272
|
+
roots: set[str] = set()
|
|
273
|
+
for child in ast.walk(node):
|
|
274
|
+
if isinstance(child, ast.Name):
|
|
275
|
+
if child.id in self.parameter_names:
|
|
276
|
+
roots.add(child.id)
|
|
277
|
+
roots.update(self.aliases.get(child.id, set()))
|
|
278
|
+
return roots
|
|
279
|
+
|
|
280
|
+
def _record(self, node: ast.AST, message: str, kind: str) -> None:
|
|
281
|
+
snippet = ast.get_source_segment(self.source_text, node)
|
|
282
|
+
self.evidence.append(
|
|
283
|
+
Evidence(
|
|
284
|
+
message=message,
|
|
285
|
+
kind=kind,
|
|
286
|
+
snippet=snippet.strip() if snippet else None,
|
|
287
|
+
location=Location(
|
|
288
|
+
path=str(self.file_path),
|
|
289
|
+
line=getattr(node, "lineno", 1),
|
|
290
|
+
column=getattr(node, "col_offset", 0) + 1,
|
|
291
|
+
end_line=getattr(node, "end_lineno", None),
|
|
292
|
+
end_column=(getattr(node, "end_col_offset", 0) + 1) if hasattr(node, "end_col_offset") else None,
|
|
293
|
+
),
|
|
294
|
+
)
|
|
295
|
+
)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _tool_name(node: ast.FunctionDef | ast.AsyncFunctionDef) -> str:
|
|
299
|
+
for decorator in node.decorator_list:
|
|
300
|
+
if isinstance(decorator, ast.Call):
|
|
301
|
+
for keyword in decorator.keywords:
|
|
302
|
+
if keyword.arg == "name" and isinstance(keyword.value, ast.Constant):
|
|
303
|
+
return str(keyword.value.value)
|
|
304
|
+
return node.name
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _tool_context(context_path: str, node: ast.FunctionDef | ast.AsyncFunctionDef) -> str:
|
|
308
|
+
for decorator in node.decorator_list:
|
|
309
|
+
target = decorator.func if isinstance(decorator, ast.Call) else decorator
|
|
310
|
+
if isinstance(target, ast.Attribute) and target.attr == "tool":
|
|
311
|
+
owner = _call_name(target.value) or "mcp"
|
|
312
|
+
return f"{context_path}:{owner}"
|
|
313
|
+
return context_path
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def _parameter(arg: ast.arg, node: ast.FunctionDef | ast.AsyncFunctionDef) -> Parameter:
|
|
317
|
+
annotation = ast.unparse(arg.annotation) if arg.annotation else None
|
|
318
|
+
source = ast.unparse(node)
|
|
319
|
+
bounded = any(token in source for token in {f"{arg.arg}.strip", f"len({arg.arg})", "max_length", "constr("})
|
|
320
|
+
return Parameter(name=arg.arg, annotation=annotation, bounded=bounded)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _call_name(node: ast.AST) -> str:
|
|
324
|
+
if isinstance(node, ast.Name):
|
|
325
|
+
return node.id
|
|
326
|
+
if isinstance(node, ast.Attribute):
|
|
327
|
+
parent = _call_name(node.value)
|
|
328
|
+
return f"{parent}.{node.attr}" if parent else node.attr
|
|
329
|
+
if isinstance(node, ast.Call):
|
|
330
|
+
return _call_name(node.func)
|
|
331
|
+
return ""
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _keyword_is_true(node: ast.Call, name: str) -> bool:
|
|
335
|
+
return any(keyword.arg == name and isinstance(keyword.value, ast.Constant) and keyword.value.value is True for keyword in node.keywords)
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
def _keyword_value(node: ast.Call, name: str) -> ast.AST | None:
|
|
339
|
+
return next((keyword.value for keyword in node.keywords if keyword.arg == name), None)
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def _has_model_controlled_executable(node: ast.Call, roots_for) -> bool:
|
|
343
|
+
if not node.args:
|
|
344
|
+
return False
|
|
345
|
+
command = node.args[0]
|
|
346
|
+
roots = roots_for(command)
|
|
347
|
+
if not roots:
|
|
348
|
+
return False
|
|
349
|
+
if isinstance(command, (ast.List, ast.Tuple)) and command.elts:
|
|
350
|
+
executable = command.elts[0]
|
|
351
|
+
if isinstance(executable, ast.Constant) and isinstance(executable.value, str):
|
|
352
|
+
return executable.value in {"bash", "sh", "zsh", "cmd", "powershell", "pwsh", "python", "python3"}
|
|
353
|
+
return bool(roots_for(executable))
|
|
354
|
+
return True
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def _requires_approval(node: ast.FunctionDef | ast.AsyncFunctionDef) -> bool:
|
|
358
|
+
for decorator in node.decorator_list:
|
|
359
|
+
if isinstance(decorator, ast.Call):
|
|
360
|
+
for keyword in decorator.keywords:
|
|
361
|
+
if keyword.arg in {"requires_approval", "approval_required"} and isinstance(keyword.value, ast.Constant):
|
|
362
|
+
return bool(keyword.value.value)
|
|
363
|
+
return False
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def _has_approval_guard(node: ast.FunctionDef | ast.AsyncFunctionDef) -> bool:
|
|
367
|
+
return any(isinstance(child, ast.If) and _is_approval_guard(child.test) and _rejects_branch(child.body) for child in ast.walk(node))
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def _is_approval_guard(node: ast.AST) -> bool:
|
|
371
|
+
text = ast.unparse(node).lower()
|
|
372
|
+
return any(token in text for token in {"approved", "approval", "confirmed", "confirm"})
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def _rejects_branch(body: list[ast.stmt]) -> bool:
|
|
376
|
+
return any(isinstance(child, (ast.Raise, ast.Return)) for statement in body for child in ast.walk(statement))
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def _guard_precedes_operations(guard_lines: list[int], operation_lines: list[int]) -> bool:
|
|
380
|
+
return bool(guard_lines and operation_lines) and all(any(guard < operation for guard in guard_lines) for operation in operation_lines)
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def _is_host_allowlist_guard(node: ast.AST) -> bool:
|
|
384
|
+
for comparison in (child for child in ast.walk(node) if isinstance(child, ast.Compare)):
|
|
385
|
+
text = ast.unparse(comparison).lower()
|
|
386
|
+
has_host = ".hostname" in text or ".netloc" in text
|
|
387
|
+
has_membership = any(isinstance(operator, (ast.In, ast.NotIn)) for operator in comparison.ops)
|
|
388
|
+
has_allowlist = any(isinstance(child, (ast.Set, ast.List, ast.Tuple)) for child in ast.walk(comparison)) or any(
|
|
389
|
+
isinstance(child, ast.Name) and any(token in child.id.lower() for token in {"allowed", "allowlist", "trusted"})
|
|
390
|
+
for child in ast.walk(comparison)
|
|
391
|
+
)
|
|
392
|
+
if has_host and has_membership and has_allowlist:
|
|
393
|
+
return True
|
|
394
|
+
return False
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def _is_path_root_guard(node: ast.AST) -> bool:
|
|
398
|
+
text = ast.unparse(node).lower()
|
|
399
|
+
return (".parents" in text or "relative_to(" in text or "is_relative_to(" in text) and any(
|
|
400
|
+
token in text for token in {"root", "base", "directory", "workspace"}
|
|
401
|
+
)
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
def _is_select_query(node: ast.Call, call_name: str) -> bool:
|
|
405
|
+
if not call_name.endswith(".execute") or not node.args:
|
|
406
|
+
return False
|
|
407
|
+
query = node.args[0]
|
|
408
|
+
return isinstance(query, ast.Constant) and isinstance(query.value, str) and query.value.lstrip().lower().startswith("select")
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
def _data_classes(text: str) -> set[str]:
|
|
412
|
+
return {classification for classification, tokens in SENSITIVE_TOKENS.items() if any(token in text for token in tokens)}
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def _function_location(file_path: Path, node: ast.FunctionDef | ast.AsyncFunctionDef) -> Location:
|
|
416
|
+
return Location(
|
|
417
|
+
path=str(file_path),
|
|
418
|
+
line=node.lineno,
|
|
419
|
+
column=node.col_offset + 1,
|
|
420
|
+
end_line=node.end_lineno,
|
|
421
|
+
end_column=(node.end_col_offset + 1) if node.end_col_offset is not None else None,
|
|
422
|
+
)
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def _semantic_tokens(text: str) -> set[str]:
|
|
426
|
+
return set(re.findall(r"[a-z0-9]+", text.lower()))
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def _highest_sensitivity(data_classes: set[str]) -> str:
|
|
430
|
+
for value in ("secret", "pii", "financial", "confidential", "internal"):
|
|
431
|
+
if value in data_classes:
|
|
432
|
+
return value
|
|
433
|
+
return "public"
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def _looks_external_write(text: str) -> bool:
|
|
437
|
+
return bool(_semantic_tokens(text) & {"send", "email", "slack", "webhook", "publish", "post", "message"})
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def _looks_destructive(text: str) -> bool:
|
|
441
|
+
return bool(_semantic_tokens(text) & {"delete", "drop", "remove", "revoke", "terminate", "destroy"})
|