sniffmcp-cli 0.4.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sniffmcp/__init__.py +2 -0
- sniffmcp/__main__.py +3 -0
- sniffmcp/checks.py +425 -0
- sniffmcp/cli.py +242 -0
- sniffmcp/client.py +160 -0
- sniffmcp/crawldb.py +121 -0
- sniffmcp/crawler.py +467 -0
- sniffmcp/engine.py +56 -0
- sniffmcp/fleet.py +182 -0
- sniffmcp/injection.py +183 -0
- sniffmcp/models.py +53 -0
- sniffmcp/osv.py +132 -0
- sniffmcp/report.py +194 -0
- sniffmcp/scoring.py +37 -0
- sniffmcp/server.py +150 -0
- sniffmcp/state.py +107 -0
- sniffmcp/watcher.py +90 -0
- sniffmcp_cli-0.4.1.dist-info/METADATA +265 -0
- sniffmcp_cli-0.4.1.dist-info/RECORD +23 -0
- sniffmcp_cli-0.4.1.dist-info/WHEEL +5 -0
- sniffmcp_cli-0.4.1.dist-info/entry_points.txt +3 -0
- sniffmcp_cli-0.4.1.dist-info/licenses/LICENSE +202 -0
- sniffmcp_cli-0.4.1.dist-info/top_level.txt +1 -0
sniffmcp/__init__.py
ADDED
sniffmcp/__main__.py
ADDED
sniffmcp/checks.py
ADDED
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
"""Static checks over a server's config and manifest.
|
|
2
|
+
|
|
3
|
+
Check IDs (SM-01..SM-10) are sniffmcp's own and do not follow OWASP's numbering;
|
|
4
|
+
report.OWASP_FOR_CHECK maps each to the OWASP MCP Top 10 categories it covers.
|
|
5
|
+
|
|
6
|
+
Each check is a pure function: (context) -> list[Finding]. The context is a
|
|
7
|
+
ManifestContext built by the client after connecting to the target server.
|
|
8
|
+
Pure functions = unit-testable without any MCP transport.
|
|
9
|
+
|
|
10
|
+
Calibration rule: a popular, legitimate server must land at A/B. Findings
|
|
11
|
+
describe evidence that something is wrong; what a server is *able* to do
|
|
12
|
+
(write files, run commands) is reported at low/info so the user can decide,
|
|
13
|
+
not punished as if it were a defect. Tool *descriptions* are analysed in
|
|
14
|
+
injection.py (INJ-HEUR); this module covers config, transport, capabilities,
|
|
15
|
+
server instructions, prompts and drift.
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
import difflib, os, re
|
|
19
|
+
from .models import Finding
|
|
20
|
+
from .injection import scan_text, tool_texts
|
|
21
|
+
|
|
22
|
+
# ---------- context ----------
|
|
23
|
+
class ManifestContext:
|
|
24
|
+
def __init__(self, target, transport, config, tools, resources=None, prompts=None,
|
|
25
|
+
server_info=None, instructions=None, prior_manifest=None):
|
|
26
|
+
self.target = target
|
|
27
|
+
self.transport = transport
|
|
28
|
+
self.config = config or {} # raw server entry from client config
|
|
29
|
+
self.tools = tools or []
|
|
30
|
+
self.resources = resources or []
|
|
31
|
+
self.prompts = prompts or []
|
|
32
|
+
self.server_info = server_info or {}
|
|
33
|
+
self.instructions = instructions or "" # server-level instructions field
|
|
34
|
+
self.prior_manifest = prior_manifest # dict from state store, or None
|
|
35
|
+
|
|
36
|
+
# ---------- helpers ----------
|
|
37
|
+
SECRET_PATTERNS = [
|
|
38
|
+
("OpenAI/Anthropic-style key", re.compile(r"\bsk-(ant-)?[A-Za-z0-9_-]{20,}")),
|
|
39
|
+
("GitHub token", re.compile(r"\b(ghp|gho|ghu|ghs|ghr)_[A-Za-z0-9]{30,}|\bgithub_pat_[A-Za-z0-9_]{40,}")),
|
|
40
|
+
("AWS access key", re.compile(r"\b(AKIA|ASIA)[0-9A-Z]{16}\b")),
|
|
41
|
+
("Slack token", re.compile(r"\bxox[baprs]-[A-Za-z0-9-]{10,}")),
|
|
42
|
+
("Stripe key", re.compile(r"\b[rs]k_(live|test)_[A-Za-z0-9]{16,}")),
|
|
43
|
+
("Google API key", re.compile(r"\bAIza[0-9A-Za-z_-]{35}\b")),
|
|
44
|
+
("Bearer token", re.compile(r"(?i)\bbearer\s+[A-Za-z0-9._~+/-]{24,}")),
|
|
45
|
+
("JWT", re.compile(r"\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}")),
|
|
46
|
+
("Private key PEM", re.compile(r"-----BEGIN ([A-Z]+ )?PRIVATE KEY-----")),
|
|
47
|
+
]
|
|
48
|
+
SENSITIVE_NAME = re.compile(r"(?i)(token|secret|password|passwd|api_?key|private_?key|credential|auth)")
|
|
49
|
+
ENV_REF = re.compile(r"^\$\{[^}]+\}$|^\$[A-Za-z_][A-Za-z0-9_]*$")
|
|
50
|
+
SENSITIVE_PATH = re.compile( # concrete secret locations only; .env.example etc. are templates
|
|
51
|
+
r"(?i)(\.ssh/|\bid_(rsa|ed25519|ecdsa)\b|\.aws/credentials|\.kube/config|\.netrc|"
|
|
52
|
+
r"\.env(\.(local|production|prod|development|dev|staging))?(?=$|[?#\s\"'])|"
|
|
53
|
+
r"\.git-credentials|\.docker/config\.json|/etc/shadow|\.pem$|\.p12$|keychain)")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def redact(value: str) -> str:
|
|
57
|
+
"""Never echo a secret into a report (reports end up in CI logs and SARIF uploads)."""
|
|
58
|
+
value = str(value)
|
|
59
|
+
return f"{value[:4]}…({len(value)} chars)" if len(value) > 4 else "****"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def find_secrets(text):
|
|
63
|
+
hits = []
|
|
64
|
+
for label, rx in SECRET_PATTERNS:
|
|
65
|
+
for m in rx.finditer(text or ""):
|
|
66
|
+
hits.append((label, redact(m.group(0))))
|
|
67
|
+
return hits
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def name_tokens(name: str) -> set[str]:
|
|
71
|
+
"""'deleteFile' / 'delete_file' / 'delete-file' -> {'delete', 'file'}."""
|
|
72
|
+
spaced = re.sub(r"([a-z0-9])([A-Z])", r"\1 \2", name or "")
|
|
73
|
+
return {t.lower() for t in re.split(r"[^A-Za-z0-9]+", spaced) if t}
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
DESTRUCTIVE = {"delete", "remove", "rm", "rmdir", "drop", "truncate", "destroy", "wipe", "purge",
|
|
77
|
+
"erase", "kill", "terminate", "unlink", "revoke", "reset", "overwrite"}
|
|
78
|
+
WRITE = {"write", "create", "update", "edit", "modify", "set", "put", "patch", "insert", "upsert",
|
|
79
|
+
"move", "rename", "push", "merge", "send", "post", "publish", "upload", "deploy",
|
|
80
|
+
"commit", "close", "archive", "add", "fork", "transfer", "pay", "charge", "refund"}
|
|
81
|
+
EXEC = {"exec", "shell", "eval", "bash", "terminal", "spawn", "subprocess", "powershell"}
|
|
82
|
+
SENSITIVE_TOOL = {"env", "environ", "environment", "credential", "credentials", "secret", "secrets",
|
|
83
|
+
"password", "passwords", "passwd", "keychain", "keyring", "vault", "ssh",
|
|
84
|
+
"apikey", "privatekey"}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def tool_capability(tool: dict) -> set[str]:
|
|
88
|
+
toks = name_tokens(tool.get("name", ""))
|
|
89
|
+
caps = set()
|
|
90
|
+
if toks & DESTRUCTIVE:
|
|
91
|
+
caps.add("destructive")
|
|
92
|
+
if toks & WRITE:
|
|
93
|
+
caps.add("write")
|
|
94
|
+
# "execute"/"run" alone is often a namespace (execute_read_portfolio, seen in the wild)
|
|
95
|
+
if toks & EXEC or (toks & {"run", "invoke", "execute"}
|
|
96
|
+
and toks & {"code", "command", "commands", "cmd", "script", "python", "js", "process", "sql"}):
|
|
97
|
+
caps.add("exec")
|
|
98
|
+
if toks & SENSITIVE_TOOL:
|
|
99
|
+
caps.add("sensitive")
|
|
100
|
+
return caps
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _short_list(names, n=6):
|
|
104
|
+
names = list(names)
|
|
105
|
+
return ", ".join(names[:n]) + (f" (+{len(names) - n} more)" if len(names) > n else "")
|
|
106
|
+
|
|
107
|
+
# ---------- checks (IDs are frozen; see report.RULES_META and report.OWASP_FOR_CHECK) ----------
|
|
108
|
+
|
|
109
|
+
def check_tool_poisoning(ctx: ManifestContext) -> list[Finding]:
|
|
110
|
+
"""SM-01: server-level instructions carrying hidden directives.
|
|
111
|
+
(Tool descriptions and parameter descriptions: injection.analyze_descriptions.)"""
|
|
112
|
+
out = []
|
|
113
|
+
for label, sev, snippet in scan_text(ctx.instructions):
|
|
114
|
+
out.append(Finding("SM-01", sev, f"Server instructions: {label}", f"…{snippet}…",
|
|
115
|
+
"Server instructions are injected into every session. Review them; never let them override user intent."))
|
|
116
|
+
return out
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def check_token_mismanagement(ctx: ManifestContext) -> list[Finding]:
|
|
120
|
+
"""SM-02: credentials stored in plaintext in the server entry."""
|
|
121
|
+
out = []
|
|
122
|
+
args = " ".join(str(a) for a in ctx.config.get("args") or [])
|
|
123
|
+
for label, snip in find_secrets(args):
|
|
124
|
+
out.append(Finding("SM-02", "high", f"{label} passed as a command-line argument", snip,
|
|
125
|
+
"Arguments are visible to every local process (ps, /proc). Pass it via env or an OAuth flow."))
|
|
126
|
+
url = str(ctx.config.get("url", ""))
|
|
127
|
+
if re.search(r"(?i)[?&](api_?key|token|key|secret|access_token)=[^&]{8,}", url):
|
|
128
|
+
out.append(Finding("SM-02", "high", "Credential in remote server URL query string",
|
|
129
|
+
re.sub(r"=([^&]{4})[^&]*", r"=\1…", url),
|
|
130
|
+
"URLs end up in proxy and server logs. Send credentials in a header or use OAuth."))
|
|
131
|
+
for k, v in (ctx.config.get("env") or {}).items():
|
|
132
|
+
v = str(v)
|
|
133
|
+
if ENV_REF.match(v.strip()):
|
|
134
|
+
continue # ${VAR} reference: the secret lives outside the config file
|
|
135
|
+
if SENSITIVE_NAME.search(k) and len(v) >= 8 or find_secrets(v):
|
|
136
|
+
out.append(Finding("SM-02", "medium", f"Plaintext credential in config env '{k}'", redact(v),
|
|
137
|
+
"Any process (or any other MCP server with file access) that can read this config can read the key. "
|
|
138
|
+
"Reference it as ${VAR} from your shell environment, or use OAuth.", ))
|
|
139
|
+
for k, v in (ctx.config.get("headers") or {}).items():
|
|
140
|
+
v = str(v)
|
|
141
|
+
if not ENV_REF.search(v.split()[-1] if v.split() else v) and (SENSITIVE_NAME.search(k) or find_secrets(v)):
|
|
142
|
+
out.append(Finding("SM-02", "medium", f"Plaintext credential in config header '{k}'", redact(v),
|
|
143
|
+
"Reference it as ${VAR} instead of storing it in the config file."))
|
|
144
|
+
return out
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _drift_text_added(old: str, new: str) -> str:
|
|
148
|
+
sm = difflib.SequenceMatcher(None, old, new)
|
|
149
|
+
return " ".join(new[j1:j2] for op, _, _, j1, j2 in sm.get_opcodes() if op in ("insert", "replace"))
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _required(tool: dict) -> set[str]:
|
|
153
|
+
return set((tool.get("inputSchema") or {}).get("required") or [])
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def check_rug_pull(ctx: ManifestContext) -> list[Finding]:
|
|
157
|
+
"""SM-03: drift vs a stored baseline, classified so benign updates don't page anyone.
|
|
158
|
+
risky -> new injection indicators, new destructive/exec/sensitive tools, annotation downgrades
|
|
159
|
+
breaking -> tool removed, required parameter added (breaks skills/workflows)
|
|
160
|
+
benign -> wording changes, new harmless tools"""
|
|
161
|
+
out = []
|
|
162
|
+
if not ctx.prior_manifest:
|
|
163
|
+
return [Finding("SM-03", "info", "No baseline stored — first scan", "",
|
|
164
|
+
"Run `sniffmcp watch` to baseline this server and detect post-install changes.")]
|
|
165
|
+
prior = {t.get("name"): t for t in ctx.prior_manifest.get("tools", [])}
|
|
166
|
+
now = {t.get("name"): t for t in ctx.tools}
|
|
167
|
+
|
|
168
|
+
for name in sorted(now.keys() - prior.keys()):
|
|
169
|
+
t = now[name]
|
|
170
|
+
hits = [h for _, text in tool_texts(t) for h in scan_text(text)]
|
|
171
|
+
caps = tool_capability(t) & {"destructive", "exec", "sensitive"}
|
|
172
|
+
if hits:
|
|
173
|
+
out.append(Finding("SM-03", "critical", f"New tool '{name}' arrived with injection indicators",
|
|
174
|
+
"; ".join(f"{l}: …{s}…" for l, _, s in hits[:3]),
|
|
175
|
+
"This is the rug-pull pattern. Remove the server until reviewed.", tool_name=name, kind="risky"))
|
|
176
|
+
elif caps:
|
|
177
|
+
out.append(Finding("SM-03", "high", f"New {'/'.join(sorted(caps))} tool '{name}' added after baseline",
|
|
178
|
+
(t.get("description") or "")[:160],
|
|
179
|
+
"A server gaining powerful tools after install deserves a review.", tool_name=name, kind="risky"))
|
|
180
|
+
else:
|
|
181
|
+
out.append(Finding("SM-03", "info", f"New tool '{name}' added", (t.get("description") or "")[:160],
|
|
182
|
+
"Benign-looking addition.", tool_name=name, kind="benign"))
|
|
183
|
+
|
|
184
|
+
for name in sorted(prior.keys() - now.keys()):
|
|
185
|
+
out.append(Finding("SM-03", "low", f"Tool '{name}' removed", "Skills or workflows calling it will break.",
|
|
186
|
+
"Check the server changelog.", tool_name=name, kind="breaking"))
|
|
187
|
+
|
|
188
|
+
for name in sorted(prior.keys() & now.keys()):
|
|
189
|
+
old, new = prior[name], now[name]
|
|
190
|
+
old_texts, new_texts = dict(tool_texts(old)), dict(tool_texts(new))
|
|
191
|
+
added = " ".join(_drift_text_added(old_texts.get(k, ""), v)
|
|
192
|
+
for k, v in new_texts.items() if v != old_texts.get(k, ""))
|
|
193
|
+
new_hits = [h for h in scan_text(added) if h[0] not in {x[0] for t in old_texts.values() for x in scan_text(t)}]
|
|
194
|
+
if new_hits:
|
|
195
|
+
out.append(Finding("SM-03", "critical", f"Tool '{name}' metadata changed and gained injection indicators",
|
|
196
|
+
"; ".join(f"{l}: …{s}…" for l, _, s in new_hits[:3]),
|
|
197
|
+
"Description mutation after install is the Deadbugz pattern. Stop using this server and review.",
|
|
198
|
+
tool_name=name, kind="risky"))
|
|
199
|
+
elif added.strip():
|
|
200
|
+
urls = re.findall(r"https?://[^\s)\"']+", added)
|
|
201
|
+
sev, why = ("medium", f"new URL(s): {_short_list(urls, 3)}") if urls else ("info", "wording change")
|
|
202
|
+
out.append(Finding("SM-03", sev, f"Tool '{name}' description changed",
|
|
203
|
+
f"{why}; added text: {added.strip()[:160]}", "Review the diff.", tool_name=name,
|
|
204
|
+
kind="risky" if urls else "benign"))
|
|
205
|
+
newly_required = _required(new) - _required(old)
|
|
206
|
+
if newly_required:
|
|
207
|
+
out.append(Finding("SM-03", "low", f"Tool '{name}' has new required parameter(s)",
|
|
208
|
+
_short_list(sorted(newly_required)), "Existing callers will fail until updated.",
|
|
209
|
+
tool_name=name, kind="breaking"))
|
|
210
|
+
oa, na = old.get("annotations") or {}, new.get("annotations") or {}
|
|
211
|
+
if oa.get("readOnlyHint") and not na.get("readOnlyHint") or (na.get("destructiveHint") and not oa.get("destructiveHint")):
|
|
212
|
+
out.append(Finding("SM-03", "medium", f"Tool '{name}' is no longer read-only / became destructive",
|
|
213
|
+
f"annotations {oa} -> {na}",
|
|
214
|
+
"Clients may have auto-approved this tool as read-only. Re-check your approval settings.",
|
|
215
|
+
tool_name=name, kind="risky"))
|
|
216
|
+
return out
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def check_excessive_privileges(ctx: ManifestContext) -> list[Finding]:
|
|
220
|
+
"""SM-04: what the server can do, plus annotations that lie about it.
|
|
221
|
+
Clients use readOnlyHint to auto-approve calls, so a destructive tool that
|
|
222
|
+
claims to be read-only is a real defect; merely having write tools is not."""
|
|
223
|
+
out = []
|
|
224
|
+
by_cap: dict[str, list[str]] = {"exec": [], "destructive": [], "write": []}
|
|
225
|
+
for t in ctx.tools:
|
|
226
|
+
name, caps = t.get("name", ""), tool_capability(t)
|
|
227
|
+
for c in by_cap:
|
|
228
|
+
if c in caps:
|
|
229
|
+
by_cap[c].append(name)
|
|
230
|
+
ann = t.get("annotations") or {}
|
|
231
|
+
# Names are weak evidence: real read-only tools are called diagnose_sales_drop,
|
|
232
|
+
# list_wipe_jobs, execute_sql_readonly. Only a name that *starts* with a strong
|
|
233
|
+
# destructive verb contradicts readOnlyHint, and even then it's "check", not "lie".
|
|
234
|
+
first = re.split(r"[^a-z0-9]+", re.sub(r"([a-z0-9])([A-Z])", r"\1_\2", name).lower())[0]
|
|
235
|
+
if ann.get("readOnlyHint") and first in {"delete", "drop", "destroy", "wipe", "truncate", "purge", "erase", "rm"}:
|
|
236
|
+
out.append(Finding("SM-04", "medium", f"Tool '{name}' claims readOnlyHint but its name starts with '{first}'",
|
|
237
|
+
f"annotations: {ann}",
|
|
238
|
+
"Clients may auto-approve read-only tools. Check what it does before auto-approving.",
|
|
239
|
+
tool_name=name))
|
|
240
|
+
elif "destructive" in caps and ann.get("destructiveHint") is False:
|
|
241
|
+
out.append(Finding("SM-04", "medium", f"Tool '{name}' declares destructiveHint=false",
|
|
242
|
+
f"annotations: {ann}", "Require manual approval for this tool regardless of its hint.", tool_name=name))
|
|
243
|
+
if by_cap["exec"]:
|
|
244
|
+
out.append(Finding("SM-04", "medium", "Can execute commands or code on the host",
|
|
245
|
+
_short_list(by_cap["exec"]),
|
|
246
|
+
"Keep manual approval on for these tools; run the server in a container if you can."))
|
|
247
|
+
if by_cap["destructive"]:
|
|
248
|
+
out.append(Finding("SM-04", "low", "Has destructive tools", _short_list(by_cap["destructive"]),
|
|
249
|
+
"Keep manual approval on for these tools."))
|
|
250
|
+
if by_cap["write"]:
|
|
251
|
+
out.append(Finding("SM-04", "info", f"Has {len(by_cap['write'])} write-capable tool(s)",
|
|
252
|
+
_short_list(by_cap["write"]), "Scope the server's credentials to what you need."))
|
|
253
|
+
return out
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def check_sensitive_data_scope(ctx: ManifestContext) -> list[Finding]:
|
|
257
|
+
"""SM-05: tools and resources aimed at secrets, env, keychains."""
|
|
258
|
+
out = []
|
|
259
|
+
names = [t.get("name") for t in ctx.tools if "sensitive" in tool_capability(t)]
|
|
260
|
+
if names:
|
|
261
|
+
out.append(Finding("SM-05", "medium", "Tools that read secrets, credentials or environment",
|
|
262
|
+
_short_list(names), "Make sure their output can't be forwarded by another tool (e.g. a web fetch or message send)."))
|
|
263
|
+
for r in ctx.resources:
|
|
264
|
+
uri = str(r.get("uri", ""))
|
|
265
|
+
if SENSITIVE_PATH.search(uri):
|
|
266
|
+
out.append(Finding("SM-05", "high", "Resource URI points at a sensitive path", uri,
|
|
267
|
+
"Do not expose credential-bearing resources to the agent; remove or gate them."))
|
|
268
|
+
return out
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
LOCAL_HOSTS = re.compile(r"^http://(localhost|127\.\d+\.\d+\.\d+|\[::1\]|0\.0\.0\.0)(:\d+)?(/|$)")
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def check_transport_security(ctx: ManifestContext) -> list[Finding]:
|
|
275
|
+
"""SM-06: cleartext remote transport."""
|
|
276
|
+
out = []
|
|
277
|
+
url = str(ctx.config.get("url", ""))
|
|
278
|
+
if url.startswith("http://") and not LOCAL_HOSTS.match(url):
|
|
279
|
+
has_creds = bool(ctx.config.get("headers"))
|
|
280
|
+
out.append(Finding("SM-06", "critical" if has_creds else "high",
|
|
281
|
+
"Remote MCP server over plain HTTP" + (" with credentials in headers" if has_creds else ""), url,
|
|
282
|
+
"Tool schemas, results and any credentials travel in cleartext. Use https."))
|
|
283
|
+
elif url and ctx.transport != "stdio" and not ctx.config.get("headers"):
|
|
284
|
+
out.append(Finding("SM-06", "info", "Remote server with no client credentials configured", url,
|
|
285
|
+
"Fine for public read-only servers; if the client does OAuth, this is expected."))
|
|
286
|
+
return out
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
NPM_RUNNERS = {"npx", "bunx", "pnpx"}
|
|
290
|
+
PY_RUNNERS = {"uvx"}
|
|
291
|
+
DOCKER_VALUE_FLAGS = {"-e", "--env", "-v", "--volume", "-p", "--publish", "--name", "--network", "-w",
|
|
292
|
+
"--workdir", "--env-file", "--mount", "-u", "--user", "--entrypoint", "--platform",
|
|
293
|
+
"-l", "--label", "--add-host", "--cpus", "-m", "--memory"}
|
|
294
|
+
NPM_PINNED = re.compile(r"^(@[^/@]+/)?[^@/]+@v?\d+\.\d+\.\d+([-+][\w.]+)?$")
|
|
295
|
+
PY_PINNED = re.compile(r"(==|@)v?\d+(\.\d+)+([-+.\w]*)$|@[0-9a-f]{7,40}$")
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def launch_spec(config: dict) -> tuple[str, str] | None:
|
|
299
|
+
"""(ecosystem, package spec) for servers launched through a package runner."""
|
|
300
|
+
cmd = os.path.basename(str(config.get("command") or "")).removesuffix(".cmd").removesuffix(".exe")
|
|
301
|
+
args = [str(a) for a in config.get("args") or []]
|
|
302
|
+
if cmd in ("npm", "pnpm", "yarn") and args[:1] in (["exec"], ["dlx"]):
|
|
303
|
+
cmd, args = "npx", args[1:]
|
|
304
|
+
if cmd == "pipx" and args[:1] == ["run"]:
|
|
305
|
+
cmd, args = "uvx", args[1:]
|
|
306
|
+
if cmd == "uv" and args[:2] == ["tool", "run"]:
|
|
307
|
+
cmd, args = "uvx", args[2:]
|
|
308
|
+
if cmd in NPM_RUNNERS:
|
|
309
|
+
for i, a in enumerate(args):
|
|
310
|
+
if a in ("-p", "--package") and i + 1 < len(args):
|
|
311
|
+
return "npm", args[i + 1]
|
|
312
|
+
if a.startswith("--package="):
|
|
313
|
+
return "npm", a.split("=", 1)[1]
|
|
314
|
+
pos = [a for a in args if not a.startswith("-")]
|
|
315
|
+
return ("npm", pos[0]) if pos else None
|
|
316
|
+
if cmd in PY_RUNNERS:
|
|
317
|
+
for i, a in enumerate(args):
|
|
318
|
+
if a == "--from" and i + 1 < len(args):
|
|
319
|
+
return "pypi", args[i + 1]
|
|
320
|
+
pos, skip = [], False
|
|
321
|
+
for a in args:
|
|
322
|
+
if skip:
|
|
323
|
+
skip = False
|
|
324
|
+
elif a in ("--python", "--with", "--index-url", "-p", "-w"):
|
|
325
|
+
skip = True
|
|
326
|
+
elif not a.startswith("-"):
|
|
327
|
+
pos.append(a)
|
|
328
|
+
return ("pypi", pos[0]) if pos else None
|
|
329
|
+
if cmd in ("docker", "podman") and "run" in args:
|
|
330
|
+
rest, skip = args[args.index("run") + 1:], False
|
|
331
|
+
for a in rest:
|
|
332
|
+
if skip:
|
|
333
|
+
skip = False
|
|
334
|
+
elif a in DOCKER_VALUE_FLAGS:
|
|
335
|
+
skip = True
|
|
336
|
+
elif not a.startswith("-"):
|
|
337
|
+
return "docker", a
|
|
338
|
+
return None
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def check_unpinned_dependencies(ctx: ManifestContext) -> list[Finding]:
|
|
342
|
+
"""SM-07: server fetched from a registry at launch without an exact version."""
|
|
343
|
+
spec = launch_spec(ctx.config)
|
|
344
|
+
if not spec:
|
|
345
|
+
return []
|
|
346
|
+
eco, pkg = spec
|
|
347
|
+
fix = "Pin an exact version so a malicious or broken publish can't reach you silently."
|
|
348
|
+
if eco == "npm" and not NPM_PINNED.match(pkg):
|
|
349
|
+
return [Finding("SM-07", "medium", "npm package launched without an exact version", pkg,
|
|
350
|
+
fix + " e.g. npx -y pkg@1.2.3")]
|
|
351
|
+
if eco == "pypi" and not PY_PINNED.search(pkg) and not pkg.startswith(("/", ".")):
|
|
352
|
+
return [Finding("SM-07", "medium", "Python package launched without an exact version", pkg,
|
|
353
|
+
fix + " e.g. uvx pkg==1.2.3")]
|
|
354
|
+
if eco == "docker":
|
|
355
|
+
if "@sha256:" in pkg:
|
|
356
|
+
return []
|
|
357
|
+
tag = pkg.rsplit("/", 1)[-1].partition(":")[2]
|
|
358
|
+
if not tag or tag == "latest":
|
|
359
|
+
return [Finding("SM-07", "medium", "Container image without a tag or digest", pkg,
|
|
360
|
+
fix + " Use image@sha256:<digest>.")]
|
|
361
|
+
return [Finding("SM-07", "info", "Container image pinned by tag (tags are mutable)", pkg,
|
|
362
|
+
"Pin by digest for full reproducibility.")]
|
|
363
|
+
return []
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def check_attack_surface(ctx: ManifestContext) -> list[Finding]:
|
|
367
|
+
"""SM-08: very large tool sets, duplicate names."""
|
|
368
|
+
out = []
|
|
369
|
+
n = len(ctx.tools)
|
|
370
|
+
if n > 40:
|
|
371
|
+
chars = sum(len(str(t.get("description", ""))) + len(str(t.get("inputSchema", ""))) for t in ctx.tools)
|
|
372
|
+
out.append(Finding("SM-08", "low", f"Large tool surface ({n} tools)",
|
|
373
|
+
f"~{chars // 4:,} tokens of tool definitions",
|
|
374
|
+
"Disable the tools you don't use; each one is context cost and another injection surface."))
|
|
375
|
+
names = [t.get("name", "") for t in ctx.tools]
|
|
376
|
+
dups = sorted({x for x in names if names.count(x) > 1})
|
|
377
|
+
if dups:
|
|
378
|
+
out.append(Finding("SM-08", "high", "Duplicate tool names", ", ".join(dups),
|
|
379
|
+
"Duplicate names can shadow a legitimate tool with a malicious twin."))
|
|
380
|
+
return out
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def check_resource_exposure(ctx: ManifestContext) -> list[Finding]:
|
|
384
|
+
"""SM-09: prompts carrying hidden directives (prompts are instructions surfaced to the model)."""
|
|
385
|
+
out = []
|
|
386
|
+
for p in ctx.prompts:
|
|
387
|
+
texts = [p.get("description") or ""] + [a.get("description") or "" for a in p.get("arguments") or []]
|
|
388
|
+
for label, sev, snippet in (h for t in texts for h in scan_text(t)):
|
|
389
|
+
out.append(Finding("SM-09", sev, f"Prompt '{p.get('name')}': {label}", f"…{snippet}…",
|
|
390
|
+
"Prompts are executable instructions surfaced to the model; treat like tool descriptions."))
|
|
391
|
+
return out
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
PIPE_TO_SHELL = re.compile(r"(?i)(curl|wget|iwr|invoke-webrequest)\b[^|]*\|\s*(sh|bash|zsh|python3?|iex|node)\b|\biex\s*\(")
|
|
395
|
+
TLS_OFF = {"NODE_TLS_REJECT_UNAUTHORIZED": "0", "PYTHONHTTPSVERIFY": "0"}
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def check_config_hygiene(ctx: ManifestContext) -> list[Finding]:
|
|
399
|
+
"""SM-10: dangerous launch configuration."""
|
|
400
|
+
out = []
|
|
401
|
+
cmdline = " ".join([str(ctx.config.get("command") or "")] + [str(a) for a in ctx.config.get("args") or []])
|
|
402
|
+
if PIPE_TO_SHELL.search(cmdline):
|
|
403
|
+
out.append(Finding("SM-10", "high", "Launch command downloads and executes a script", cmdline[:160],
|
|
404
|
+
"Every launch runs whatever that URL serves today. Install a pinned package instead."))
|
|
405
|
+
for k, bad in TLS_OFF.items():
|
|
406
|
+
if str((ctx.config.get("env") or {}).get(k, "")) == bad:
|
|
407
|
+
out.append(Finding("SM-10", "medium", f"TLS verification disabled ({k}={bad})", k,
|
|
408
|
+
"Remove it; the server's outbound traffic can be intercepted."))
|
|
409
|
+
return out
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
ALL_CHECKS = [check_tool_poisoning, check_token_mismanagement, check_rug_pull,
|
|
413
|
+
check_excessive_privileges, check_sensitive_data_scope,
|
|
414
|
+
check_transport_security, check_unpinned_dependencies,
|
|
415
|
+
check_attack_surface, check_resource_exposure, check_config_hygiene]
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def run_all_checks(ctx: ManifestContext) -> list[Finding]:
|
|
419
|
+
findings = []
|
|
420
|
+
for c in ALL_CHECKS:
|
|
421
|
+
try:
|
|
422
|
+
findings.extend(c(ctx))
|
|
423
|
+
except Exception as e: # a broken check must never kill the scan
|
|
424
|
+
findings.append(Finding(c.__name__, "info", f"check errored: {e}", "", ""))
|
|
425
|
+
return findings
|