sniffmcp-cli 0.4.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sniffmcp/crawler.py ADDED
@@ -0,0 +1,467 @@
1
+ """Ecosystem crawler: the official MCP registry -> remote manifest snapshots and
2
+ npm/PyPI release history, stored over time in crawldb.
3
+
4
+ Rules this crawler keeps:
5
+ - Probes only public addresses. Registry entries are untrusted; one pointing at
6
+ localhost, a LAN range or cloud metadata must not turn this into an SSRF tool.
7
+ - Only list calls (tools/resources/prompts). Never calls a tool.
8
+ - At most one probe per endpoint per --min-age hours; at most 2 concurrent per host.
9
+ - Endpoints whose registry entry declares a required auth header are recorded,
10
+ not probed (a guaranteed 401 every day helps nobody).
11
+ - Packages are never downloaded or executed: npm/PyPI metadata only.
12
+ - Stores facts (what changed, when). Verdicts stay in local reports.
13
+ """
14
+ from __future__ import annotations
15
+ import asyncio, ipaddress, json, socket, statistics, time
16
+ from collections import Counter
17
+ from urllib.parse import urlparse, quote
18
+
19
+ import httpx2
20
+
21
+ from . import __version__, osv
22
+ from .checks import ManifestContext, check_rug_pull
23
+ from .client import ConnectError, fetch_manifest
24
+ from .crawldb import CrawlDB
25
+ from .engine import report_from_manifest
26
+ from .injection import scan_text
27
+ from .models import canonical_manifest_hash
28
+
29
+ REGISTRY = "https://registry.modelcontextprotocol.io/v0/servers"
30
+ USER_AGENT = f"sniffmcp-crawler/{__version__} (MCP ecosystem research; list calls only, ~1 probe/day)"
31
+ CLIENT_NAME = "sniffmcp-crawler"
32
+ DAY = 86400
33
+
34
+
35
+ # ---------------------------------------------------------------- registry
36
+ async def sync_registry(db: CrawlDB, page_size: int = 100, max_pages: int = 1000) -> dict:
37
+ now, cursor, n = time.time(), None, 0
38
+ async with httpx2.AsyncClient(timeout=60, headers={"User-Agent": USER_AGENT}) as http:
39
+ for _ in range(max_pages):
40
+ params = {"limit": page_size, "version": "latest"}
41
+ if cursor:
42
+ params["cursor"] = cursor
43
+ data = await _get_json(http, REGISTRY, params)
44
+ for entry in data.get("servers", []):
45
+ _ingest(db, entry, now)
46
+ n += 1
47
+ db.commit()
48
+ cursor = (data.get("metadata") or {}).get("nextCursor")
49
+ if not cursor:
50
+ break
51
+ return {"servers": n}
52
+
53
+
54
+ async def _get_json(http, url, params, attempts: int = 4) -> dict:
55
+ """The registry is slow (10-20s pages) and returns occasional 5xx; back off and retry."""
56
+ for i in range(attempts):
57
+ try:
58
+ r = await http.get(url, params=params)
59
+ if r.status_code < 500:
60
+ r.raise_for_status()
61
+ return r.json()
62
+ err = f"HTTP {r.status_code}"
63
+ except (httpx2.TransportError, httpx2.TimeoutException) as e:
64
+ err = f"{type(e).__name__}: {e}"
65
+ if i < attempts - 1:
66
+ await asyncio.sleep(5 * 3 ** i)
67
+ raise RuntimeError(f"registry unavailable after {attempts} attempts ({err}); progress so far is saved")
68
+
69
+
70
+ def _ingest(db: CrawlDB, entry: dict, now: float) -> None:
71
+ s = entry["server"]
72
+ db.upsert_server(entry, now)
73
+ for rem in s.get("remotes") or []:
74
+ url = rem.get("url") or ""
75
+ transport = "sse" if rem.get("type") == "sse" else "streamable_http"
76
+ auth = any(h.get("isRequired") for h in rem.get("headers") or [])
77
+ skip = ("templated-url" if "{" in url else "auth-declared" if auth
78
+ else None if url.startswith(("https://", "http://")) else "bad-url")
79
+ db.upsert_endpoint(url, transport, s["name"], auth, skip, now)
80
+ for pkg in s.get("packages") or []:
81
+ eco = {"npm": "npm", "pypi": "pypi"}.get(pkg.get("registryType"))
82
+ if eco and pkg.get("identifier"):
83
+ db.upsert_package(eco, pkg["identifier"], s["name"], now)
84
+
85
+
86
+ # ---------------------------------------------------------------- remotes
87
+ async def public_address_check(url: str) -> str | None:
88
+ """None if the URL resolves only to globally routable addresses, else the reason."""
89
+ u = urlparse(url)
90
+ if u.scheme not in ("https", "http") or not u.hostname:
91
+ return "bad-url"
92
+ try:
93
+ infos = await asyncio.get_running_loop().getaddrinfo(u.hostname, u.port or 443,
94
+ type=socket.SOCK_STREAM)
95
+ except OSError:
96
+ return "dns-failure"
97
+ for info in infos:
98
+ ip = ipaddress.ip_address(info[4][0].split("%")[0])
99
+ if not ip.is_global:
100
+ return "non-public-address"
101
+ return None
102
+
103
+
104
+ async def probe_remotes(db: CrawlDB, concurrency: int = 8, per_host: int = 2,
105
+ min_age_h: float = 20, limit: int | None = None, timeout: float = 20,
106
+ per_host_cap: int | None = 100, progress=None) -> dict:
107
+ due = db.due_endpoints(min_age_h * 3600, limit, per_host_cap)
108
+ total_sem = asyncio.Semaphore(concurrency)
109
+ host_sems: dict[str, asyncio.Semaphore] = {}
110
+ stats: Counter = Counter()
111
+
112
+ async def one(ep):
113
+ host = urlparse(ep["url"]).hostname or ""
114
+ hs = host_sems.setdefault(host, asyncio.Semaphore(per_host))
115
+ async with hs, total_sem: # host slot first, so one busy host can't park every global slot
116
+ status, err, mhash, t0 = await _probe(db, ep, timeout)
117
+ db.conn.execute("INSERT INTO probes(endpoint_id,taken_at,status,error,manifest_hash,duration_ms)"
118
+ " VALUES(?,?,?,?,?,?)", (ep["id"], time.time(), status, err, mhash,
119
+ int((time.time() - t0) * 1000)))
120
+ db.commit()
121
+ stats[status] += 1
122
+ if progress:
123
+ progress(sum(stats.values()), len(due), ep["url"], status)
124
+
125
+ await asyncio.gather(*(one(ep) for ep in due))
126
+ return {"probed": len(due), **stats}
127
+
128
+
129
+ async def _probe(db: CrawlDB, ep: dict, timeout: float):
130
+ t0 = time.time()
131
+ blocked = await public_address_check(ep["url"])
132
+ if blocked == "dns-failure":
133
+ return "error", "DNS lookup failed", None, t0
134
+ if blocked:
135
+ return "skipped", blocked, None, t0
136
+ config = {"url": ep["url"], "headers": {"User-Agent": USER_AGENT}, "timeout": timeout}
137
+ if ep["transport"] == "sse":
138
+ config["type"] = "sse"
139
+ try:
140
+ manifest = await fetch_manifest(config, client_name=CLIENT_NAME)
141
+ except ConnectError as e:
142
+ msg = str(e)
143
+ status = ("auth" if msg.startswith("requires authentication") else
144
+ "timeout" if msg.startswith("timed out") else "error")
145
+ return status, msg[:300], None, t0
146
+ except Exception as e: # never let one endpoint kill the crawl
147
+ return "error", f"{type(e).__name__}: {e}"[:300], None, t0
148
+ # server instructions are model-visible too, so they're part of the change hash
149
+ mhash = canonical_manifest_hash([{"name": "__instructions__", "description": manifest.get("instructions") or ""}]
150
+ + manifest["tools"], manifest.get("resources"), manifest.get("prompts"))
151
+ prev = db.last_snapshot(ep["id"])
152
+ if not prev or prev["manifest_hash"] != mhash:
153
+ _record_snapshot(db, ep, manifest, mhash, prev)
154
+ return "ok", None, mhash, t0
155
+
156
+
157
+ def _record_snapshot(db: CrawlDB, ep: dict, manifest: dict, mhash: str, prev: dict | None) -> None:
158
+ now = time.time()
159
+ rep = report_from_manifest({"url": ep["url"]}, manifest)
160
+ findings = [f.__dict__ for f in rep.findings if f.severity != "info"]
161
+ stored = {k: manifest.get(k) for k in ("tools", "resources", "prompts", "instructions",
162
+ "server_info", "protocol_version")}
163
+ db.conn.execute(
164
+ "INSERT INTO snapshots(endpoint_id,taken_at,manifest_hash,manifest,protocol_version,tool_count,score,grade,findings)"
165
+ " VALUES(?,?,?,?,?,?,?,?,?)",
166
+ (ep["id"], now, mhash, json.dumps(stored), manifest.get("protocol_version"),
167
+ len(manifest["tools"]), rep.score, rep.grade, json.dumps(findings)))
168
+ if not prev:
169
+ return
170
+ old = json.loads(prev["manifest"])
171
+ ctx = ManifestContext(ep["url"], ep["transport"], {"url": ep["url"]}, manifest["tools"],
172
+ manifest.get("resources"), manifest.get("prompts"),
173
+ prior_manifest={"tools": old.get("tools") or []})
174
+ changes = [(f.kind or "benign", f.severity, f.title, f.tool_name, f.evidence) for f in check_rug_pull(ctx)]
175
+ if (old.get("instructions") or "") != (manifest.get("instructions") or ""):
176
+ old_labels = {h[0] for h in scan_text(old.get("instructions") or "")}
177
+ new_hits = [h for h in scan_text(manifest.get("instructions") or "") if h[0] not in old_labels]
178
+ changes.append(("risky" if new_hits else "benign", "critical" if new_hits else "info",
179
+ "Server instructions changed" + (f": {new_hits[0][0]}" if new_hits else ""), None,
180
+ (manifest.get("instructions") or "")[:200]))
181
+ if not changes:
182
+ changes.append(("benign", "info", "Resources or prompts changed", None, ""))
183
+ db.conn.executemany(
184
+ "INSERT INTO changes(endpoint_id,detected_at,from_hash,to_hash,kind,severity,title,tool_name,evidence)"
185
+ " VALUES(?,?,?,?,?,?,?,?,?)",
186
+ [(ep["id"], now, prev["manifest_hash"], mhash, *c) for c in changes])
187
+
188
+
189
+ def rescore(db: CrawlDB) -> int:
190
+ """Re-run the current rules over every stored snapshot (after a rule fix)."""
191
+ rows = db.q("SELECT s.id, s.manifest, e.url FROM snapshots s JOIN endpoints e ON e.id=s.endpoint_id")
192
+ for r in rows:
193
+ rep = report_from_manifest({"url": r["url"]}, json.loads(r["manifest"]))
194
+ db.conn.execute("UPDATE snapshots SET score=?, grade=?, findings=? WHERE id=?",
195
+ (rep.score, rep.grade, json.dumps([f.__dict__ for f in rep.findings if f.severity != "info"]),
196
+ r["id"]))
197
+ db.commit()
198
+ return len(rows)
199
+
200
+
201
+ # ---------------------------------------------------------------- packages
202
+ INSTALL_HOOKS = ("preinstall", "install", "postinstall")
203
+
204
+
205
+ def npm_versions(doc: dict) -> list[dict]:
206
+ times = doc.get("time") or {}
207
+ out = []
208
+ for ver, v in (doc.get("versions") or {}).items():
209
+ scripts = {k: v["scripts"][k] for k in INSTALL_HOOKS if k in (v.get("scripts") or {})}
210
+ user = v.get("_npmUser") or {}
211
+ dist = v.get("dist") or {}
212
+ out.append({"version": ver, "published_at": times.get(ver),
213
+ "install_scripts": scripts, "publisher": user.get("name"),
214
+ "provenance": bool(dist.get("attestations") or user.get("trustedPublisher")),
215
+ "sdist_only": False, "integrity": dist.get("integrity")})
216
+ return sorted(out, key=lambda x: x["published_at"] or "")
217
+
218
+
219
+ def pypi_versions(doc: dict) -> list[dict]:
220
+ out = []
221
+ for ver, files in (doc.get("releases") or {}).items():
222
+ if not files:
223
+ continue
224
+ out.append({"version": ver,
225
+ "published_at": min(f.get("upload_time_iso_8601") or "" for f in files),
226
+ "install_scripts": {}, "publisher": None, "provenance": False,
227
+ "sdist_only": all(f.get("packagetype") == "sdist" for f in files),
228
+ "integrity": (files[0].get("digests") or {}).get("sha256")})
229
+ return sorted(out, key=lambda x: x["published_at"] or "")
230
+
231
+
232
+ def release_events(versions: list[dict]) -> list[tuple[str, str, str, str, str]]:
233
+ """(version, published_at, kind, severity, detail) over the full release history.
234
+ Signals, not verdicts: each is a known precursor in package takeovers."""
235
+ events, publishers, prev = [], set(), None
236
+ for v in versions:
237
+ if v["install_scripts"] and (prev is None or not prev["install_scripts"]):
238
+ events.append((v["version"], v["published_at"],
239
+ "install-script-added" if prev else "install-script", "high" if prev else "medium",
240
+ json.dumps(v["install_scripts"])[:300]))
241
+ if prev and prev["provenance"] and not v["provenance"]:
242
+ events.append((v["version"], v["published_at"], "provenance-dropped", "high",
243
+ f"{prev['version']} had provenance/trusted publishing; {v['version']} does not"))
244
+ # A move *to* trusted publishing (provenance on) is an improvement, not a takeover.
245
+ if v["publisher"] and publishers and v["publisher"] not in publishers and not v["provenance"]:
246
+ events.append((v["version"], v["published_at"], "new-publisher", "medium",
247
+ f"first release by '{v['publisher']}' (previous: {', '.join(sorted(publishers))[:200]})"))
248
+ if prev and v["sdist_only"] and not prev["sdist_only"]:
249
+ events.append((v["version"], v["published_at"], "wheel-dropped", "medium",
250
+ "source-only release: installing runs the package's build code"))
251
+ if v["publisher"]:
252
+ publishers.add(v["publisher"])
253
+ prev = v
254
+ return events
255
+
256
+
257
+ async def crawl_packages(db: CrawlDB, concurrency: int = 8, min_age_h: float = 20,
258
+ limit: int | None = None, progress=None) -> dict:
259
+ cutoff = time.time() - min_age_h * 3600
260
+ due = db.q("SELECT * FROM packages WHERE COALESCE(last_checked,0) < ? ORDER BY ecosystem, name", cutoff)
261
+ due = due[:limit] if limit else due
262
+ sem, stats = asyncio.Semaphore(concurrency), Counter()
263
+ async with httpx2.AsyncClient(timeout=30, headers={"User-Agent": USER_AGENT}) as http:
264
+ async def one(p):
265
+ async with sem:
266
+ status = await _crawl_package(db, http, p)
267
+ stats[status] += 1
268
+ if progress:
269
+ progress(sum(stats.values()), len(due), f"{p['ecosystem']}:{p['name']}", status)
270
+ await asyncio.gather(*(one(p) for p in due))
271
+ return {"checked": len(due), **stats}
272
+
273
+
274
+ async def _crawl_package(db: CrawlDB, http, p: dict) -> str:
275
+ eco, name, now = p["ecosystem"], p["name"], time.time()
276
+ url = (f"https://registry.npmjs.org/{quote(name, safe='@')}" if eco == "npm"
277
+ else f"https://pypi.org/pypi/{quote(name)}/json")
278
+ try:
279
+ r = await http.get(url)
280
+ if r.status_code == 404:
281
+ raise LookupError("not found on registry")
282
+ r.raise_for_status()
283
+ versions = npm_versions(r.json()) if eco == "npm" else pypi_versions(r.json())
284
+ except Exception as e:
285
+ db.conn.execute("UPDATE packages SET last_checked=?, error=? WHERE ecosystem=? AND name=?",
286
+ (now, f"{type(e).__name__}: {e}"[:300], eco, name))
287
+ db.commit()
288
+ return "error"
289
+ db.conn.executemany(
290
+ "INSERT OR IGNORE INTO package_versions(ecosystem,name,version,published_at,install_scripts,"
291
+ "publisher,provenance,sdist_only,integrity,first_seen) VALUES(?,?,?,?,?,?,?,?,?,?)",
292
+ [(eco, name, v["version"], v["published_at"], json.dumps(v["install_scripts"]), v["publisher"],
293
+ int(v["provenance"]), int(v["sdist_only"]), v["integrity"], now) for v in versions])
294
+ db.conn.executemany(
295
+ "INSERT OR IGNORE INTO package_events(ecosystem,name,version,published_at,detected_at,kind,severity,detail)"
296
+ " VALUES(?,?,?,?,?,?,?,?)", [(eco, name, *e[:2], now, *e[2:]) for e in release_events(versions)])
297
+ db.conn.execute("UPDATE packages SET last_checked=?, error=NULL WHERE ecosystem=? AND name=?", (now, eco, name))
298
+ db.commit()
299
+ return "ok"
300
+
301
+
302
+ # ---------------------------------------------------------------- advisories
303
+ async def crawl_advisories(db: CrawlDB) -> dict:
304
+ """OSV for every tracked package: malware (MAL-) in any version, plus advisories
305
+ affecting the current latest version. Old CVEs on superseded versions are skipped."""
306
+ rows = db.q("SELECT ecosystem, name, version, published_at FROM package_versions")
307
+ latest: dict[tuple[str, str], tuple[str, str]] = {}
308
+ for r in rows:
309
+ key = (r["ecosystem"], r["name"])
310
+ if key not in latest or (r["published_at"] or "") > latest[key][1]:
311
+ latest[key] = (r["version"], r["published_at"] or "")
312
+ keys = sorted(latest)
313
+ now = time.time()
314
+ async with httpx2.AsyncClient(timeout=60, headers={"User-Agent": USER_AGENT}) as http:
315
+ at_latest = await osv.query_batch(http, [(e, n, latest[(e, n)][0]) for e, n in keys])
316
+ any_ver = await osv.query_batch(http, [(e, n, None) for e, n in keys])
317
+ wanted: dict[tuple[str, str], dict[str, bool]] = {}
318
+ for k, lat, anyv in zip(keys, at_latest, any_ver):
319
+ ids = {i: True for i in lat}
320
+ ids.update({i: i in lat for i in anyv if i.startswith("MAL-")})
321
+ if ids:
322
+ wanted[k] = ids
323
+ info = await osv.details(http, {i for ids in wanted.values() for i in ids})
324
+ for (eco, name), ids in wanted.items():
325
+ for vid, affects in ids.items():
326
+ v = info.get(vid, {"id": vid})
327
+ db.conn.execute(
328
+ "INSERT INTO package_advisories(ecosystem,name,vuln_id,kind,severity,summary,aliases,affects_latest,"
329
+ "latest_version,first_seen,last_seen) VALUES(?,?,?,?,?,?,?,?,?,?,?) ON CONFLICT(ecosystem,name,vuln_id)"
330
+ " DO UPDATE SET affects_latest=excluded.affects_latest, latest_version=excluded.latest_version,"
331
+ " severity=excluded.severity, summary=excluded.summary, last_seen=excluded.last_seen",
332
+ (eco, name, vid, "malware" if vid.startswith("MAL-") else "vulnerability", osv.severity(v),
333
+ (v.get("summary") or "")[:300], json.dumps(v.get("aliases") or []), int(affects),
334
+ latest[(eco, name)][0], now, now))
335
+ db.commit()
336
+ mal = sum(1 for ids in wanted.values() if any(i.startswith("MAL-") for i in ids))
337
+ return {"packages": len(keys), "with_advisories": len(wanted), "with_malware_history": mal}
338
+
339
+
340
+ # ---------------------------------------------------------------- report
341
+ def _cell(s, n=110) -> str:
342
+ s = " ".join(str(s or "").split()).replace("`", "'").replace("|", "/")
343
+ return f"`{s[:n]}{'…' if len(s) > n else ''}`" if s else ""
344
+
345
+
346
+ def _pct(a, b) -> str:
347
+ return f"{100 * a / b:.0f}%" if b else "–"
348
+
349
+
350
+ def report(db: CrawlDB, days: int = 7) -> str:
351
+ since = time.time() - days * DAY
352
+ since_iso = time.strftime("%Y-%m-%dT%H:%M:%S", time.gmtime(since))
353
+ L = [f"# MCP ecosystem crawl — {time.strftime('%Y-%m-%d')}", ""]
354
+
355
+ servers = db.one("SELECT COUNT(*) FROM servers")
356
+ active = db.one("SELECT COUNT(*) FROM servers WHERE status='active'")
357
+ with_remote = db.one("SELECT COUNT(DISTINCT server_name) FROM endpoints")
358
+ pk = Counter({r["ecosystem"]: r["n"] for r in db.q("SELECT ecosystem, COUNT(*) n FROM packages GROUP BY ecosystem")})
359
+ pubs = Counter(r["name"].split("/")[0] for r in db.q("SELECT name FROM servers"))
360
+ bulk = [(p, n) for p, n in pubs.most_common(10) if n >= 100]
361
+ hosts = Counter(urlparse(r["url"]).hostname for r in db.q("SELECT url FROM endpoints"))
362
+ L += ["## Registry", f"- {servers} servers in the official registry ({active} active), "
363
+ f"from {len(pubs)} publisher namespaces",
364
+ f"- bulk publishers (≥100 entries): " + (", ".join(f"{p} {n}" for p, n in bulk) or "none"),
365
+ f"- busiest endpoint hosts: " + ", ".join(f"{h} {n}" for h, n in hosts.most_common(5)),
366
+ f"- {with_remote} with remote endpoints; packages tracked: "
367
+ + ", ".join(f"{k} {v}" for k, v in sorted(pk.items())), ""]
368
+
369
+ skip = Counter({r["skip_reason"] or "probed": r["n"] for r in
370
+ db.q("SELECT skip_reason, COUNT(*) n FROM endpoints GROUP BY skip_reason")})
371
+ latest = db.q("SELECT p.status, p.error, e.url, e.server_name, e.id FROM probes p JOIN endpoints e ON e.id=p.endpoint_id"
372
+ " WHERE p.id=(SELECT MAX(id) FROM probes WHERE endpoint_id=p.endpoint_id)")
373
+ st = Counter(r["status"] for r in latest)
374
+ n_ep, n_probed = sum(skip.values()), len(latest)
375
+ # rates over endpoints we have evidence for: declared-auth ones plus those actually probed
376
+ known = skip.get("auth-declared", 0) + n_probed
377
+ auth_total = skip.get("auth-declared", 0) + st.get("auth", 0)
378
+ L += ["## Remote endpoints", f"- {n_ep} endpoints: " + ", ".join(f"{k} {v}" for k, v in skip.most_common()),
379
+ f"- probed so far: {n_probed} of {skip.get('probed', 0)}; latest result: "
380
+ + ", ".join(f"{k} {v}" for k, v in st.most_common()),
381
+ f"- **require auth: {auth_total} of {known} known ({_pct(auth_total, known)})** "
382
+ f"(declared in registry {skip.get('auth-declared', 0)}, 401/403 on probe {st.get('auth', 0)})",
383
+ f"- **openly scannable: {st.get('ok', 0)} of {n_probed} probed ({_pct(st.get('ok', 0), n_probed)})**", ""]
384
+
385
+ ok_ids = [r["id"] for r in latest if r["status"] == "ok"]
386
+ snaps = [db.last_snapshot(i) for i in ok_ids]
387
+ snaps = [s for s in snaps if s]
388
+ if snaps:
389
+ grades = Counter(s["grade"] for s in snaps)
390
+ protos = Counter(s["protocol_version"] for s in snaps)
391
+ tools = [s["tool_count"] for s in snaps]
392
+ toks = [len(json.dumps(json.loads(s["manifest"]).get("tools") or [])) // 4 for s in snaps]
393
+ q90 = lambda xs: sorted(xs)[int(0.9 * (len(xs) - 1))]
394
+ L += ["### Reachable servers", f"- grades: " + ", ".join(f"{g} {grades.get(g, 0)}" for g in "ABCDF"),
395
+ f"- protocol versions: " + ", ".join(f"{k} {v}" for k, v in protos.most_common()),
396
+ f"- tools per server: median {statistics.median(tools):.0f}, p90 {q90(tools)}, max {max(tools)}",
397
+ f"- tool-definition tokens per server (≈chars/4): median {statistics.median(toks):,.0f}, "
398
+ f"p90 {q90(toks):,}, max {max(toks):,}", ""]
399
+ by_check: Counter = Counter()
400
+ serious = []
401
+ name_of = {r["id"]: r["server_name"] for r in latest}
402
+ for s in snaps:
403
+ fs = json.loads(s["findings"])
404
+ for cid in {f["check_id"] for f in fs if f["severity"] in ("medium", "high", "critical")}:
405
+ by_check[cid] += 1
406
+ serious += [(name_of.get(s["endpoint_id"]), f) for f in fs if f["severity"] in ("high", "critical")]
407
+ L += ["### Findings (servers with ≥1 medium+ finding, by check)",
408
+ ", ".join(f"{k} {v}" for k, v in by_check.most_common()) or "none", ""]
409
+ if serious:
410
+ L += ["### High / critical findings — verify by hand before citing", "",
411
+ "| server | severity | finding | evidence |", "|---|---|---|---|"]
412
+ L += [f"| {_cell(n, 50)} | {f['severity']} | {_cell(f['title'], 70)} | {_cell(f['evidence'])} |"
413
+ for n, f in serious[:60]]
414
+ L.append("")
415
+
416
+ ch = db.q("SELECT c.*, e.server_name FROM changes c JOIN endpoints e ON e.id=c.endpoint_id"
417
+ " WHERE c.detected_at>=? ORDER BY c.id", since)
418
+ L += [f"## Manifest changes in the last {days} days",
419
+ f"- {len(ch)} change records on {len({c['endpoint_id'] for c in ch})} endpoints: "
420
+ + ", ".join(f"{k} {v}" for k, v in Counter(c["kind"] for c in ch).most_common()), ""]
421
+ notable = [c for c in ch if c["kind"] != "benign"]
422
+ if notable:
423
+ L += ["| server | kind | severity | change | evidence |", "|---|---|---|---|---|"]
424
+ L += [f"| {_cell(c['server_name'], 50)} | {c['kind']} | {c['severity']} | {_cell(c['title'], 70)} | {_cell(c['evidence'])} |"
425
+ for c in notable[:60]]
426
+ L.append("")
427
+
428
+ recent = db.one("SELECT COUNT(*) FROM package_versions WHERE published_at>=?", since_iso)
429
+ ev_all = Counter({(r["kind"]): r["n"] for r in db.q("SELECT kind, COUNT(*) n FROM package_events GROUP BY kind")})
430
+ ev_recent = db.q("SELECT * FROM package_events WHERE published_at>=? AND severity IN ('high','medium')"
431
+ " ORDER BY published_at DESC", since_iso)
432
+ errs = db.one("SELECT COUNT(*) FROM packages WHERE error IS NOT NULL")
433
+ L += ["## Packages (npm / PyPI metadata; nothing executed)",
434
+ f"- {recent} versions published in the last {days} days; {errs} packages not found or failed",
435
+ f"- release-history signals, all time: " + (", ".join(f"{k} {v}" for k, v in ev_all.most_common()) or "none"),
436
+ ""]
437
+ if ev_recent:
438
+ L += [f"### Signals on releases from the last {days} days", "",
439
+ "| package | version | signal | detail |", "|---|---|---|---|"]
440
+ L += [f"| {_cell(e['ecosystem'] + ':' + e['name'], 60)} | {_cell(e['version'], 20)} | {e['kind']} | {_cell(e['detail'])} |"
441
+ for e in ev_recent[:60]]
442
+ L.append("")
443
+ adv = db.q("SELECT a.*, p.server_name FROM package_advisories a LEFT JOIN packages p"
444
+ " ON p.ecosystem=a.ecosystem AND p.name=a.name")
445
+ if adv:
446
+ mal = [a for a in adv if a["kind"] == "malware"]
447
+ vul = [a for a in adv if a["kind"] == "vulnerability" and a["affects_latest"]]
448
+ sev = Counter(a["severity"] for a in vul)
449
+ L += ["## Known advisories (OSV.dev)",
450
+ f"- **malware advisories: {len({(a['ecosystem'], a['name']) for a in mal})} packages** "
451
+ f"({len({(a['ecosystem'], a['name']) for a in mal if a['affects_latest']})} with the current latest version affected)",
452
+ f"- latest version has known vulnerabilities: {len({(a['ecosystem'], a['name']) for a in vul})} packages "
453
+ f"(advisories: " + ", ".join(f"{k} {v}" for k, v in sev.most_common()) + ")", ""]
454
+ if mal:
455
+ L += ["| package | registry entry | advisory | affects latest |", "|---|---|---|---|"]
456
+ L += [f"| {_cell(a['ecosystem'] + ':' + a['name'], 50)} | {_cell(a['server_name'], 50)} | "
457
+ f"{_cell(a['vuln_id'] + ' ' + a['summary'], 80)} | {'yes' if a['affects_latest'] else 'no'} |" for a in mal[:60]]
458
+ L.append("")
459
+ top = [a for a in vul if a["severity"] == "high"]
460
+ if top:
461
+ L += ["### High-severity advisories on current latest versions", "",
462
+ "| package | latest | advisory |", "|---|---|---|"]
463
+ L += [f"| {_cell(a['ecosystem'] + ':' + a['name'], 50)} | {_cell(a['latest_version'], 20)} | "
464
+ f"{_cell(a['vuln_id'] + ' ' + a['summary'], 90)} |" for a in top[:40]]
465
+ L.append("")
466
+ L += ["---", "Facts from automated list calls, registry metadata and OSV.dev. A finding is a lead to check, not a verdict."]
467
+ return "\n".join(L)
sniffmcp/engine.py ADDED
@@ -0,0 +1,56 @@
1
+ """One scan pipeline shared by the CLI, the MCP server, the fleet scanner and tests."""
2
+ from __future__ import annotations
3
+ from . import osv
4
+ from .client import fetch_manifest, transport_of
5
+ from .checks import ManifestContext, launch_spec, run_all_checks
6
+ from .injection import analyze_descriptions
7
+ from .models import Finding, ScanReport, canonical_manifest_hash
8
+ from .scoring import score_findings
9
+
10
+
11
+ def target_of(config: dict) -> str:
12
+ if config.get("url"):
13
+ return str(config["url"])
14
+ return " ".join([str(config.get("command", "?"))] + [str(a) for a in config.get("args") or []])[:200]
15
+
16
+
17
+ def report_from_manifest(config: dict, manifest: dict, prior: dict | None = None,
18
+ llm_hook=None, extra: list[Finding] | None = None) -> ScanReport:
19
+ tools = manifest.get("tools", [])
20
+ transport = transport_of(config)
21
+ ctx = ManifestContext(target_of(config), transport, config, tools,
22
+ manifest.get("resources"), manifest.get("prompts"),
23
+ manifest.get("server_info"), manifest.get("instructions"),
24
+ prior_manifest=prior)
25
+ findings = run_all_checks(ctx) + analyze_descriptions(tools, llm_hook=llm_hook) + list(extra or [])
26
+ for err in manifest.get("errors") or []:
27
+ findings.append(Finding("SM-10", "info", "Partial manifest", err,
28
+ "Some listings failed; the scan may be incomplete."))
29
+ score, grade = score_findings(findings)
30
+ return ScanReport(target=ctx.target, transport=transport, score=score, grade=grade,
31
+ findings=findings,
32
+ manifest_hash=canonical_manifest_hash(tools, manifest.get("resources"),
33
+ manifest.get("prompts")),
34
+ tool_count=len(tools))
35
+
36
+
37
+ async def scan(config: dict, prior: dict | None = None, launch: bool = True) -> tuple[ScanReport, dict]:
38
+ """Check the package against OSV, then connect and audit.
39
+ Known malware is reported without ever being launched. With launch=False, stdio
40
+ servers are never started (config + OSV checks only), which is what CI wants for a
41
+ config file that came from a pull request.
42
+ Raises client.ConnectError if the server can't be reached."""
43
+ pre: list[Finding] = []
44
+ spec = launch_spec(config) if transport_of(config) == "stdio" else None
45
+ if spec and spec[0] in osv.ECOSYSTEM and not osv.offline():
46
+ pre = await osv.advisories_for_spec(*spec)
47
+ if osv.blocks_launch(pre):
48
+ manifest = {"tools": [], "errors": ["not launched: the package is known malware (OSV)"]}
49
+ return report_from_manifest(config, manifest, prior, extra=pre), manifest
50
+ if not launch and transport_of(config) == "stdio":
51
+ manifest = {"tools": []}
52
+ pre.append(Finding("SM-10", "info", "Not launched (--no-launch): config and package checks only", "",
53
+ "Run without --no-launch on a trusted machine to inspect the server's tools."))
54
+ return report_from_manifest(config, manifest, prior, extra=pre), manifest
55
+ manifest = await fetch_manifest(config)
56
+ return report_from_manifest(config, manifest, prior, extra=pre), manifest