sniffmcp-cli 0.4.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sniffmcp/__init__.py +2 -0
- sniffmcp/__main__.py +3 -0
- sniffmcp/checks.py +425 -0
- sniffmcp/cli.py +242 -0
- sniffmcp/client.py +160 -0
- sniffmcp/crawldb.py +121 -0
- sniffmcp/crawler.py +467 -0
- sniffmcp/engine.py +56 -0
- sniffmcp/fleet.py +182 -0
- sniffmcp/injection.py +183 -0
- sniffmcp/models.py +53 -0
- sniffmcp/osv.py +132 -0
- sniffmcp/report.py +194 -0
- sniffmcp/scoring.py +37 -0
- sniffmcp/server.py +150 -0
- sniffmcp/state.py +107 -0
- sniffmcp/watcher.py +90 -0
- sniffmcp_cli-0.4.1.dist-info/METADATA +265 -0
- sniffmcp_cli-0.4.1.dist-info/RECORD +23 -0
- sniffmcp_cli-0.4.1.dist-info/WHEEL +5 -0
- sniffmcp_cli-0.4.1.dist-info/entry_points.txt +3 -0
- sniffmcp_cli-0.4.1.dist-info/licenses/LICENSE +202 -0
- sniffmcp_cli-0.4.1.dist-info/top_level.txt +1 -0
sniffmcp/cli.py
ADDED
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
"""sniffmcp CLI. Same engine as the MCP server.
|
|
2
|
+
|
|
3
|
+
Exit contract (stable):
|
|
4
|
+
0 scan completed, no active finding at/above --fail-on
|
|
5
|
+
1 scan completed, at least one active finding at/above threshold
|
|
6
|
+
2 target or configuration error
|
|
7
|
+
3 analysis incomplete (server unreachable / timed out)
|
|
8
|
+
|
|
9
|
+
Usage:
|
|
10
|
+
sniffmcp scan '{"command":"npx","args":["-y","pkg@1.2.3"]}'
|
|
11
|
+
sniffmcp scan https://mcp.example.com/mcp
|
|
12
|
+
sniffmcp scan cfg.json --format sarif --output results.sarif
|
|
13
|
+
sniffmcp fleet # every server in your Claude/Cursor/Windsurf configs
|
|
14
|
+
sniffmcp watch cfg.json --name myserver --webhook https://hooks.slack.com/...
|
|
15
|
+
sniffmcp watch --name myserver --accept # accept the current manifest as the new baseline
|
|
16
|
+
sniffmcp watch --run # check all watches on their intervals, forever
|
|
17
|
+
sniffmcp crawl run --output crawl-report.md # daily: registry sync, probe public servers, package history
|
|
18
|
+
"""
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
import argparse, asyncio, json, os, sys
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
from .client import ConnectError, transport_of
|
|
24
|
+
from .engine import scan
|
|
25
|
+
from .scoring import summarize
|
|
26
|
+
from .report import to_sarif, fleet_to_sarif, to_json_report, diff_baseline, apply_suppressions, gate, SEVERITY_RANK
|
|
27
|
+
from .state import Store
|
|
28
|
+
from . import watcher
|
|
29
|
+
from .fleet import scan_fleet, fleet_console
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def load_config(arg: str) -> dict:
|
|
33
|
+
"""Inline JSON, a path to a JSON file, or an http(s) URL."""
|
|
34
|
+
arg = arg.strip()
|
|
35
|
+
if arg.startswith(("https://", "http://")):
|
|
36
|
+
return {"url": arg}
|
|
37
|
+
if arg.startswith("{"):
|
|
38
|
+
return json.loads(arg)
|
|
39
|
+
return json.loads(Path(arg).read_text())
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def console_print(report) -> None:
|
|
43
|
+
print(f"\n {report.target}\n score {report.score}/100 grade {report.grade} ({report.tool_count} tools)")
|
|
44
|
+
s = summarize(report.findings)
|
|
45
|
+
print(f" critical:{s['critical']} high:{s['high']} medium:{s['medium']} low:{s['low']} info:{s['info']}")
|
|
46
|
+
for f in sorted(report.findings, key=lambda f: -SEVERITY_RANK.get(f.severity, 0)):
|
|
47
|
+
mark = " (suppressed)" if f.suppressed else ""
|
|
48
|
+
print(f" [{f.severity:8}] {f.check_id:8} {f.title}{mark}")
|
|
49
|
+
if f.evidence and f.severity != "info":
|
|
50
|
+
print(f" {f.evidence[:150]}")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
async def cmd_scan(args) -> int:
|
|
54
|
+
try:
|
|
55
|
+
config = load_config(args.config)
|
|
56
|
+
except Exception as e:
|
|
57
|
+
print(f"error: config must be JSON, a JSON file path, or a URL ({e})", file=sys.stderr)
|
|
58
|
+
return 2
|
|
59
|
+
try:
|
|
60
|
+
report, _ = await scan(config, launch=not args.no_launch)
|
|
61
|
+
except ConnectError as e:
|
|
62
|
+
print(f"error: {e}", file=sys.stderr)
|
|
63
|
+
return 3
|
|
64
|
+
findings = report.findings
|
|
65
|
+
apply_suppressions(findings, Path(args.suppress).read_text() if args.suppress else "", args.suppress)
|
|
66
|
+
if args.baseline:
|
|
67
|
+
try:
|
|
68
|
+
diff_baseline(findings, json.loads(Path(args.baseline).read_text()))
|
|
69
|
+
except Exception as e:
|
|
70
|
+
print(f"warning: baseline ignored ({e})", file=sys.stderr)
|
|
71
|
+
|
|
72
|
+
if args.format == "sarif":
|
|
73
|
+
out = json.dumps(to_sarif(findings, report.target), indent=2)
|
|
74
|
+
elif args.format == "json":
|
|
75
|
+
out = json.dumps(to_json_report(report.target, report.score, report.grade, findings,
|
|
76
|
+
report.manifest_hash), indent=2)
|
|
77
|
+
else:
|
|
78
|
+
console_print(report)
|
|
79
|
+
out = None
|
|
80
|
+
if out and args.output:
|
|
81
|
+
Path(args.output).write_text(out)
|
|
82
|
+
print(f"wrote {args.output}")
|
|
83
|
+
elif out:
|
|
84
|
+
print(out)
|
|
85
|
+
return gate(findings, args.fail_on)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
async def cmd_watch(args) -> int:
|
|
89
|
+
store = Store()
|
|
90
|
+
if args.run:
|
|
91
|
+
print("watching all servers; Ctrl+C to stop")
|
|
92
|
+
await watcher.run_forever(store)
|
|
93
|
+
return 0
|
|
94
|
+
if not args.name:
|
|
95
|
+
print("error: --name is required", file=sys.stderr)
|
|
96
|
+
return 2
|
|
97
|
+
if args.accept:
|
|
98
|
+
try:
|
|
99
|
+
ok = await watcher.accept(store, args.name)
|
|
100
|
+
except ConnectError as e:
|
|
101
|
+
print(f"error: {e}", file=sys.stderr)
|
|
102
|
+
return 3
|
|
103
|
+
print(f"baseline for '{args.name}' updated" if ok else f"no watch named '{args.name}'")
|
|
104
|
+
return 0 if ok else 2
|
|
105
|
+
if args.config:
|
|
106
|
+
try:
|
|
107
|
+
config = load_config(args.config)
|
|
108
|
+
except Exception as e:
|
|
109
|
+
print(f"error: {e}", file=sys.stderr)
|
|
110
|
+
return 2
|
|
111
|
+
store.add_watch(args.name, config, transport_of(config), args.interval, args.webhook)
|
|
112
|
+
watch = store.get_watch(args.name)
|
|
113
|
+
if not watch:
|
|
114
|
+
print(f"error: no watch named '{args.name}'; pass a config to create it", file=sys.stderr)
|
|
115
|
+
return 2
|
|
116
|
+
had_baseline = bool(watch.get("baseline_hash"))
|
|
117
|
+
findings = await watcher.check_watch(store, watch)
|
|
118
|
+
if not had_baseline and store.get_watch(args.name).get("baseline_hash"):
|
|
119
|
+
print(f"baseline set for '{args.name}'")
|
|
120
|
+
elif not findings:
|
|
121
|
+
print(f"'{args.name}': no changes since baseline")
|
|
122
|
+
for f in findings:
|
|
123
|
+
print(f" [{f.severity:8}] [{f.kind or '-':8}] {f.title}")
|
|
124
|
+
if f.evidence:
|
|
125
|
+
print(f" {f.evidence[:150]}")
|
|
126
|
+
return gate(findings, "high")
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
async def cmd_crawl(args) -> int:
|
|
130
|
+
from . import crawler
|
|
131
|
+
from .crawldb import CrawlDB
|
|
132
|
+
db = CrawlDB(args.db)
|
|
133
|
+
|
|
134
|
+
def progress(done, total, what, status):
|
|
135
|
+
if status != "ok" or done % 50 == 0 or done == total:
|
|
136
|
+
print(f" [{done}/{total}] {status:8} {what[:90]}", file=sys.stderr)
|
|
137
|
+
|
|
138
|
+
if args.action in ("sync", "run"):
|
|
139
|
+
print(f"sync: {await crawler.sync_registry(db)}", file=sys.stderr)
|
|
140
|
+
if args.action in ("probe", "run"):
|
|
141
|
+
print(f"probe: {await crawler.probe_remotes(db, concurrency=args.concurrency, min_age_h=args.min_age, limit=args.limit, per_host_cap=args.per_host_cap, progress=progress)}",
|
|
142
|
+
file=sys.stderr)
|
|
143
|
+
if args.action in ("packages", "run"):
|
|
144
|
+
print(f"packages: {await crawler.crawl_packages(db, concurrency=args.concurrency, min_age_h=args.min_age, limit=args.limit, progress=progress)}",
|
|
145
|
+
file=sys.stderr)
|
|
146
|
+
if args.action in ("advisories", "run"):
|
|
147
|
+
print(f"advisories: {await crawler.crawl_advisories(db)}", file=sys.stderr)
|
|
148
|
+
if args.action == "rescore":
|
|
149
|
+
print(f"rescored {crawler.rescore(db)} snapshots with the current rules", file=sys.stderr)
|
|
150
|
+
if args.action in ("report", "run", "rescore"):
|
|
151
|
+
md = crawler.report(db, days=args.days)
|
|
152
|
+
if args.output:
|
|
153
|
+
Path(args.output).write_text(md)
|
|
154
|
+
print(f"wrote {args.output}", file=sys.stderr)
|
|
155
|
+
else:
|
|
156
|
+
print(md)
|
|
157
|
+
return 0
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def main(argv=None) -> int:
|
|
161
|
+
ap = argparse.ArgumentParser(prog="sniffmcp")
|
|
162
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
163
|
+
|
|
164
|
+
s = sub.add_parser("scan", help="scan one MCP server")
|
|
165
|
+
s.add_argument("config", help="inline JSON, path to a JSON file, or an MCP URL")
|
|
166
|
+
s.add_argument("--format", choices=["console", "json", "sarif"], default="console")
|
|
167
|
+
s.add_argument("--output", default=None)
|
|
168
|
+
s.add_argument("--fail-on", default="high", choices=list(SEVERITY_RANK.keys()))
|
|
169
|
+
s.add_argument("--baseline", default=None, help="previous JSON report; matched findings don't gate")
|
|
170
|
+
s.add_argument("--suppress", default=None, help="suppressions file (directives or JSON rules)")
|
|
171
|
+
s.add_argument("--offline", action="store_true", help="skip OSV malware/vulnerability lookups")
|
|
172
|
+
s.add_argument("--no-launch", action="store_true", help="never start a stdio server: config + OSV checks only")
|
|
173
|
+
|
|
174
|
+
wch = sub.add_parser("watch", help="baseline a server and report post-install changes")
|
|
175
|
+
wch.add_argument("config", nargs="?", help="config to create/update the watch")
|
|
176
|
+
wch.add_argument("--name")
|
|
177
|
+
wch.add_argument("--interval", type=int, default=3600)
|
|
178
|
+
wch.add_argument("--webhook", default=None)
|
|
179
|
+
wch.add_argument("--accept", action="store_true", help="accept the current manifest as the baseline")
|
|
180
|
+
wch.add_argument("--run", action="store_true", help="check every watch on its interval, forever")
|
|
181
|
+
|
|
182
|
+
fl = sub.add_parser("fleet", help="scan every installed MCP server from client configs")
|
|
183
|
+
fl.add_argument("paths", nargs="*", help="config file paths; omit to auto-discover")
|
|
184
|
+
fl.add_argument("--format", choices=["console", "json", "sarif"], default="console")
|
|
185
|
+
fl.add_argument("--output", default=None)
|
|
186
|
+
fl.add_argument("--fail-on", default="high", choices=list(SEVERITY_RANK.keys()))
|
|
187
|
+
fl.add_argument("--offline", action="store_true", help="skip OSV malware/vulnerability lookups")
|
|
188
|
+
fl.add_argument("--no-launch", action="store_true",
|
|
189
|
+
help="never start stdio servers: config + OSV checks only (use in CI on untrusted configs)")
|
|
190
|
+
|
|
191
|
+
cr = sub.add_parser("crawl", help="crawl the official MCP registry and snapshot public servers over time")
|
|
192
|
+
cr.add_argument("action", choices=["sync", "probe", "packages", "advisories", "report", "run", "rescore"],
|
|
193
|
+
help="run = sync + probe + packages + advisories + report (what a daily cron should call)")
|
|
194
|
+
cr.add_argument("--db", default=None, help="crawl database (default ~/.sniffmcp/crawl.db)")
|
|
195
|
+
cr.add_argument("--limit", type=int, default=None, help="probe/check at most N items")
|
|
196
|
+
cr.add_argument("--concurrency", type=int, default=8)
|
|
197
|
+
cr.add_argument("--min-age", type=float, default=20, help="hours before an item is re-probed")
|
|
198
|
+
cr.add_argument("--per-host-cap", type=int, default=100, help="max endpoints probed per host per run (rotates)")
|
|
199
|
+
cr.add_argument("--days", type=int, default=7, help="report window")
|
|
200
|
+
cr.add_argument("--output", default=None, help="write the report here instead of stdout")
|
|
201
|
+
|
|
202
|
+
sv = sub.add_parser("serve", help="run as an MCP server")
|
|
203
|
+
sv.add_argument("--http", action="store_true")
|
|
204
|
+
sv.add_argument("--port", type=int, default=8930)
|
|
205
|
+
|
|
206
|
+
args = ap.parse_args(argv)
|
|
207
|
+
if getattr(args, "offline", False):
|
|
208
|
+
os.environ["SNIFFMCP_OFFLINE"] = "1"
|
|
209
|
+
if args.cmd == "fleet":
|
|
210
|
+
try:
|
|
211
|
+
fleet = asyncio.run(scan_fleet(args.paths or None, args.fail_on, launch=not args.no_launch))
|
|
212
|
+
except ValueError as e:
|
|
213
|
+
print(f"error: {e}", file=sys.stderr)
|
|
214
|
+
return 2
|
|
215
|
+
if args.format in ("json", "sarif"):
|
|
216
|
+
out = json.dumps(fleet if args.format == "json" else fleet_to_sarif(fleet), indent=2)
|
|
217
|
+
if args.output:
|
|
218
|
+
Path(args.output).write_text(out)
|
|
219
|
+
print(f"wrote {args.output}")
|
|
220
|
+
else:
|
|
221
|
+
print(out)
|
|
222
|
+
else:
|
|
223
|
+
fleet_console(fleet)
|
|
224
|
+
return fleet["summary"]["gate"]
|
|
225
|
+
if args.cmd == "scan":
|
|
226
|
+
return asyncio.run(cmd_scan(args))
|
|
227
|
+
if args.cmd == "watch":
|
|
228
|
+
return asyncio.run(cmd_watch(args))
|
|
229
|
+
if args.cmd == "crawl":
|
|
230
|
+
return asyncio.run(cmd_crawl(args))
|
|
231
|
+
if args.cmd == "serve":
|
|
232
|
+
from .server import mcp # mcp.run() owns its event loop; don't wrap it in asyncio.run
|
|
233
|
+
if args.http:
|
|
234
|
+
mcp.run(transport="streamable-http", host="127.0.0.1", port=args.port)
|
|
235
|
+
else:
|
|
236
|
+
mcp.run(transport="stdio")
|
|
237
|
+
return 0
|
|
238
|
+
return 2
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
if __name__ == "__main__":
|
|
242
|
+
sys.exit(main())
|
sniffmcp/client.py
ADDED
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""Connect to a target MCP server and pull the manifest the agent would see.
|
|
2
|
+
|
|
3
|
+
Uses the official SDK (mcp>=2). `Client(mode="auto")` probes the 2026-07-28
|
|
4
|
+
`server/discover` and falls back to the legacy initialize handshake, so both
|
|
5
|
+
protocol generations scan the same way.
|
|
6
|
+
|
|
7
|
+
Config shapes accepted (same as Claude / Cursor / Claude Code configs):
|
|
8
|
+
{"command": "npx", "args": ["-y", "pkg@1.2.3"], "env": {...}, "cwd": "..."}
|
|
9
|
+
{"url": "https://host/mcp", "headers": {...}} # Streamable HTTP
|
|
10
|
+
{"type": "sse", "url": "https://host/sse", "headers": {...}} # legacy SSE
|
|
11
|
+
Optional: "timeout" (seconds).
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
import os, re, tempfile
|
|
15
|
+
|
|
16
|
+
import anyio
|
|
17
|
+
from mcp import Client
|
|
18
|
+
from mcp.client.stdio import StdioServerParameters, stdio_client
|
|
19
|
+
from mcp.client.streamable_http import streamable_http_client
|
|
20
|
+
from mcp.client.sse import sse_client
|
|
21
|
+
from mcp.shared._httpx_utils import create_mcp_http_client
|
|
22
|
+
from mcp.types import Implementation
|
|
23
|
+
|
|
24
|
+
from . import __version__
|
|
25
|
+
|
|
26
|
+
DEFAULT_TIMEOUT_STDIO = 90.0 # first `npx -y` / `uvx` run downloads the package
|
|
27
|
+
DEFAULT_TIMEOUT_HTTP = 30.0
|
|
28
|
+
MAX_PAGES = 50
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class ConnectError(Exception):
|
|
32
|
+
"""Target could not be reached, spawned, or did not complete the handshake."""
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
_VAR = re.compile(r"\$\{([A-Za-z_][A-Za-z0-9_]*)(?::-([^}]*))?\}")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _expand(value: str) -> str:
|
|
39
|
+
"""Expand ${VAR} and ${VAR:-default} the way Claude Code / Cursor configs do."""
|
|
40
|
+
return _VAR.sub(lambda m: os.environ.get(m.group(1), m.group(2) or ""), value)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def transport_of(config: dict) -> str:
|
|
44
|
+
if "url" not in config:
|
|
45
|
+
return "stdio"
|
|
46
|
+
if config.get("type") == "sse" or config.get("transport") == "sse":
|
|
47
|
+
return "sse"
|
|
48
|
+
return "streamable_http"
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _innermost(e: BaseException) -> BaseException:
|
|
52
|
+
while isinstance(e, BaseExceptionGroup) and e.exceptions:
|
|
53
|
+
e = e.exceptions[0]
|
|
54
|
+
return e
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
async def _list_all(fetch, attr: str) -> list[dict]:
|
|
58
|
+
out, cursor = [], None
|
|
59
|
+
for _ in range(MAX_PAGES):
|
|
60
|
+
page = await fetch(cursor=cursor)
|
|
61
|
+
out.extend(x.model_dump(mode="json", by_alias=True, exclude_none=True)
|
|
62
|
+
for x in getattr(page, attr))
|
|
63
|
+
cursor = page.next_cursor
|
|
64
|
+
if not cursor:
|
|
65
|
+
break
|
|
66
|
+
return out
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
async def _collect(client: Client) -> dict:
|
|
70
|
+
caps = client.server_capabilities
|
|
71
|
+
manifest = {"tools": [], "resources": [], "prompts": [],
|
|
72
|
+
"server_info": {}, "instructions": client.instructions or "",
|
|
73
|
+
"protocol_version": client.protocol_version, "errors": []}
|
|
74
|
+
if client.server_info:
|
|
75
|
+
manifest["server_info"] = client.server_info.model_dump(mode="json", exclude_none=True)
|
|
76
|
+
# Only ask for what the server advertises; some servers error on unknown methods.
|
|
77
|
+
for attr, cap, fetch in (("tools", caps.tools, client.list_tools),
|
|
78
|
+
("resources", caps.resources, client.list_resources),
|
|
79
|
+
("prompts", caps.prompts, client.list_prompts)):
|
|
80
|
+
if cap is None:
|
|
81
|
+
continue
|
|
82
|
+
try:
|
|
83
|
+
manifest[attr] = await _list_all(fetch, attr)
|
|
84
|
+
except Exception as e: # a failed list is reported, not fatal
|
|
85
|
+
manifest["errors"].append(f"{attr}/list failed: {_innermost(e)}")
|
|
86
|
+
return manifest
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _transport_for(config: dict, errlog):
|
|
90
|
+
kind = transport_of(config)
|
|
91
|
+
if kind == "stdio":
|
|
92
|
+
cmd = config.get("command")
|
|
93
|
+
if not cmd or not isinstance(cmd, str):
|
|
94
|
+
raise ConnectError("config has neither 'command' nor 'url'")
|
|
95
|
+
env = {k: _expand(str(v)) for k, v in (config.get("env") or {}).items()}
|
|
96
|
+
params = StdioServerParameters(command=_expand(cmd),
|
|
97
|
+
args=[_expand(str(a)) for a in config.get("args") or []],
|
|
98
|
+
env=env or None, cwd=config.get("cwd"))
|
|
99
|
+
return stdio_client(params, errlog=errlog)
|
|
100
|
+
url = _expand(config["url"])
|
|
101
|
+
headers = {k: _expand(str(v)) for k, v in (config.get("headers") or {}).items()}
|
|
102
|
+
if kind == "sse":
|
|
103
|
+
return sse_client(url, headers=headers or None)
|
|
104
|
+
return streamable_http_client(url, http_client=create_mcp_http_client(headers=headers or None))
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
async def fetch_manifest(config: dict, client_name: str = "sniffmcp") -> dict:
|
|
108
|
+
"""Connect, negotiate, and return tools/resources/prompts/server_info/instructions."""
|
|
109
|
+
kind = transport_of(config)
|
|
110
|
+
timeout = float(config.get("timeout") or
|
|
111
|
+
(DEFAULT_TIMEOUT_STDIO if kind == "stdio" else DEFAULT_TIMEOUT_HTTP))
|
|
112
|
+
with tempfile.TemporaryFile(mode="w+", encoding="utf-8", errors="replace") as errlog:
|
|
113
|
+
try:
|
|
114
|
+
transport = _transport_for(config, errlog)
|
|
115
|
+
with anyio.fail_after(timeout):
|
|
116
|
+
async with Client(transport, mode="auto", cache=None,
|
|
117
|
+
client_info=Implementation(name=client_name, version=__version__)) as client:
|
|
118
|
+
return await _collect(client)
|
|
119
|
+
except ConnectError:
|
|
120
|
+
raise
|
|
121
|
+
except TimeoutError:
|
|
122
|
+
raise ConnectError(f"timed out after {timeout:.0f}s{_stderr_tail(errlog)}") from None
|
|
123
|
+
except Exception as e:
|
|
124
|
+
inner = _innermost(e)
|
|
125
|
+
if kind != "stdio":
|
|
126
|
+
hint = await _http_status_hint(config)
|
|
127
|
+
if hint:
|
|
128
|
+
raise ConnectError(hint) from None
|
|
129
|
+
raise ConnectError(f"{type(inner).__name__}: {inner}{_stderr_tail(errlog)}") from None
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
async def _http_status_hint(config: dict) -> str | None:
|
|
133
|
+
"""After a failed remote connect, say *why* in terms a user can act on."""
|
|
134
|
+
import httpx2
|
|
135
|
+
headers = {"Content-Type": "application/json", "Accept": "application/json, text/event-stream",
|
|
136
|
+
**{k: _expand(str(v)) for k, v in (config.get("headers") or {}).items()}}
|
|
137
|
+
body = {"jsonrpc": "2.0", "id": 1, "method": "initialize",
|
|
138
|
+
"params": {"protocolVersion": "2025-06-18", "capabilities": {},
|
|
139
|
+
"clientInfo": {"name": "sniffmcp", "version": "0"}}}
|
|
140
|
+
try:
|
|
141
|
+
async with httpx2.AsyncClient(timeout=10) as http:
|
|
142
|
+
r = await http.post(_expand(config["url"]), json=body, headers=headers)
|
|
143
|
+
except Exception:
|
|
144
|
+
return None
|
|
145
|
+
if r.status_code in (401, 403):
|
|
146
|
+
oauth = "resource_metadata" in r.headers.get("www-authenticate", "")
|
|
147
|
+
return (f"requires authentication (HTTP {r.status_code}"
|
|
148
|
+
f"{', OAuth' if oauth else ''}); pass credentials in \"headers\" to scan it")
|
|
149
|
+
if r.status_code >= 400:
|
|
150
|
+
return f"HTTP {r.status_code} from server"
|
|
151
|
+
return None
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _stderr_tail(errlog) -> str:
|
|
155
|
+
try:
|
|
156
|
+
errlog.seek(0)
|
|
157
|
+
lines = [l.strip() for l in errlog.read().splitlines() if l.strip()]
|
|
158
|
+
except Exception:
|
|
159
|
+
return ""
|
|
160
|
+
return (" | stderr: " + " / ".join(lines[-3:])[:300]) if lines else ""
|
sniffmcp/crawldb.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
"""SQLite store for the ecosystem crawl: registry entries, remote endpoint
|
|
2
|
+
snapshots, manifest changes, and npm/PyPI release history.
|
|
3
|
+
|
|
4
|
+
Snapshots are stored only when a manifest changes; every probe is logged
|
|
5
|
+
separately so availability and auth stats stay honest.
|
|
6
|
+
Default path ~/.sniffmcp/crawl.db (override with SNIFFMCP_CRAWL_DB or --db).
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
import json, os, sqlite3, time
|
|
10
|
+
|
|
11
|
+
SCHEMA = """
|
|
12
|
+
CREATE TABLE IF NOT EXISTS servers(
|
|
13
|
+
name TEXT PRIMARY KEY, version TEXT, title TEXT, description TEXT, repo_url TEXT,
|
|
14
|
+
status TEXT, published_at TEXT, updated_at TEXT, raw TEXT, first_seen REAL, last_seen REAL);
|
|
15
|
+
CREATE TABLE IF NOT EXISTS server_versions(
|
|
16
|
+
name TEXT, version TEXT, published_at TEXT, first_seen REAL, PRIMARY KEY(name, version));
|
|
17
|
+
CREATE TABLE IF NOT EXISTS endpoints(
|
|
18
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT, url TEXT UNIQUE, transport TEXT, server_name TEXT,
|
|
19
|
+
auth_declared INTEGER, skip_reason TEXT, first_seen REAL, last_seen REAL);
|
|
20
|
+
CREATE TABLE IF NOT EXISTS probes(
|
|
21
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT, endpoint_id INTEGER, taken_at REAL, status TEXT,
|
|
22
|
+
error TEXT, manifest_hash TEXT, duration_ms INTEGER);
|
|
23
|
+
CREATE TABLE IF NOT EXISTS snapshots(
|
|
24
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT, endpoint_id INTEGER, taken_at REAL, manifest_hash TEXT,
|
|
25
|
+
manifest TEXT, protocol_version TEXT, tool_count INTEGER, score INTEGER, grade TEXT, findings TEXT);
|
|
26
|
+
CREATE TABLE IF NOT EXISTS changes(
|
|
27
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT, endpoint_id INTEGER, detected_at REAL, from_hash TEXT,
|
|
28
|
+
to_hash TEXT, kind TEXT, severity TEXT, title TEXT, tool_name TEXT, evidence TEXT);
|
|
29
|
+
CREATE TABLE IF NOT EXISTS packages(
|
|
30
|
+
ecosystem TEXT, name TEXT, server_name TEXT, first_seen REAL, last_checked REAL, error TEXT,
|
|
31
|
+
PRIMARY KEY(ecosystem, name));
|
|
32
|
+
CREATE TABLE IF NOT EXISTS package_versions(
|
|
33
|
+
ecosystem TEXT, name TEXT, version TEXT, published_at TEXT, install_scripts TEXT,
|
|
34
|
+
publisher TEXT, provenance INTEGER, sdist_only INTEGER, integrity TEXT, first_seen REAL,
|
|
35
|
+
PRIMARY KEY(ecosystem, name, version));
|
|
36
|
+
CREATE TABLE IF NOT EXISTS package_events(
|
|
37
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT, ecosystem TEXT, name TEXT, version TEXT,
|
|
38
|
+
published_at TEXT, detected_at REAL, kind TEXT, severity TEXT, detail TEXT,
|
|
39
|
+
UNIQUE(ecosystem, name, version, kind));
|
|
40
|
+
CREATE TABLE IF NOT EXISTS package_advisories(
|
|
41
|
+
ecosystem TEXT, name TEXT, vuln_id TEXT, kind TEXT, severity TEXT, summary TEXT, aliases TEXT,
|
|
42
|
+
affects_latest INTEGER, latest_version TEXT, first_seen REAL, last_seen REAL,
|
|
43
|
+
PRIMARY KEY(ecosystem, name, vuln_id));
|
|
44
|
+
CREATE INDEX IF NOT EXISTS probes_ep ON probes(endpoint_id, taken_at);
|
|
45
|
+
CREATE INDEX IF NOT EXISTS snaps_ep ON snapshots(endpoint_id, id);
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def default_path() -> str:
|
|
50
|
+
return os.environ.get("SNIFFMCP_CRAWL_DB") or os.path.join(
|
|
51
|
+
os.path.expanduser("~"), ".sniffmcp", "crawl.db")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class CrawlDB:
|
|
55
|
+
def __init__(self, path: str | None = None):
|
|
56
|
+
path = path or default_path()
|
|
57
|
+
if path != ":memory:":
|
|
58
|
+
os.makedirs(os.path.dirname(os.path.abspath(path)), exist_ok=True)
|
|
59
|
+
self.conn = sqlite3.connect(path, timeout=60) # wait for other writers (e.g. a concurrent rescore) instead of failing
|
|
60
|
+
self.conn.row_factory = sqlite3.Row
|
|
61
|
+
self.conn.execute("PRAGMA journal_mode=WAL")
|
|
62
|
+
self.conn.executescript(SCHEMA)
|
|
63
|
+
self.conn.commit()
|
|
64
|
+
|
|
65
|
+
def q(self, sql, *args) -> list[dict]:
|
|
66
|
+
return [dict(r) for r in self.conn.execute(sql, args)]
|
|
67
|
+
|
|
68
|
+
def one(self, sql, *args):
|
|
69
|
+
row = self.conn.execute(sql, args).fetchone()
|
|
70
|
+
return row[0] if row else None
|
|
71
|
+
|
|
72
|
+
# ---- registry ----
|
|
73
|
+
def upsert_server(self, entry: dict, now: float) -> None:
|
|
74
|
+
s, meta = entry["server"], entry.get("_meta", {}).get("io.modelcontextprotocol.registry/official", {})
|
|
75
|
+
self.conn.execute(
|
|
76
|
+
"INSERT INTO servers(name,version,title,description,repo_url,status,published_at,updated_at,raw,first_seen,last_seen)"
|
|
77
|
+
" VALUES(?,?,?,?,?,?,?,?,?,?,?) ON CONFLICT(name) DO UPDATE SET version=excluded.version,"
|
|
78
|
+
" title=excluded.title, description=excluded.description, repo_url=excluded.repo_url,"
|
|
79
|
+
" status=excluded.status, published_at=excluded.published_at, updated_at=excluded.updated_at,"
|
|
80
|
+
" raw=excluded.raw, last_seen=excluded.last_seen",
|
|
81
|
+
(s["name"], s.get("version"), s.get("title"), s.get("description"),
|
|
82
|
+
(s.get("repository") or {}).get("url"), meta.get("status"), meta.get("publishedAt"),
|
|
83
|
+
meta.get("updatedAt"), json.dumps(s), now, now))
|
|
84
|
+
self.conn.execute("INSERT OR IGNORE INTO server_versions(name,version,published_at,first_seen) VALUES(?,?,?,?)",
|
|
85
|
+
(s["name"], s.get("version"), meta.get("publishedAt"), now))
|
|
86
|
+
|
|
87
|
+
def upsert_endpoint(self, url, transport, server_name, auth_declared, skip_reason, now) -> None:
|
|
88
|
+
self.conn.execute(
|
|
89
|
+
"INSERT INTO endpoints(url,transport,server_name,auth_declared,skip_reason,first_seen,last_seen)"
|
|
90
|
+
" VALUES(?,?,?,?,?,?,?) ON CONFLICT(url) DO UPDATE SET transport=excluded.transport,"
|
|
91
|
+
" server_name=excluded.server_name, auth_declared=excluded.auth_declared,"
|
|
92
|
+
" skip_reason=excluded.skip_reason, last_seen=excluded.last_seen",
|
|
93
|
+
(url, transport, server_name, int(auth_declared), skip_reason, now, now))
|
|
94
|
+
|
|
95
|
+
def upsert_package(self, ecosystem, name, server_name, now) -> None:
|
|
96
|
+
self.conn.execute("INSERT OR IGNORE INTO packages(ecosystem,name,server_name,first_seen) VALUES(?,?,?,?)",
|
|
97
|
+
(ecosystem, name, server_name, now))
|
|
98
|
+
|
|
99
|
+
# ---- probes / snapshots ----
|
|
100
|
+
def due_endpoints(self, min_age_s: float, limit: int | None, per_host_cap: int | None = None) -> list[dict]:
|
|
101
|
+
"""Stalest first (never-probed, then oldest), at most per_host_cap per host per run.
|
|
102
|
+
Bulk publishers register thousands of entries on one host; the cap samples them
|
|
103
|
+
on rotation instead of hitting one operator thousands of times a day."""
|
|
104
|
+
from urllib.parse import urlparse
|
|
105
|
+
sql = ("SELECT e.*, COALESCE((SELECT MAX(taken_at) FROM probes p WHERE p.endpoint_id=e.id), 0) AS last_probe"
|
|
106
|
+
" FROM endpoints e WHERE e.skip_reason IS NULL AND last_probe < ? ORDER BY last_probe, e.id")
|
|
107
|
+
rows, per_host = [], {}
|
|
108
|
+
for r in self.q(sql, time.time() - min_age_s):
|
|
109
|
+
host = urlparse(r["url"]).hostname or ""
|
|
110
|
+
if per_host_cap and per_host.get(host, 0) >= per_host_cap:
|
|
111
|
+
continue
|
|
112
|
+
per_host[host] = per_host.get(host, 0) + 1
|
|
113
|
+
rows.append(r)
|
|
114
|
+
return rows[:limit] if limit else rows
|
|
115
|
+
|
|
116
|
+
def last_snapshot(self, endpoint_id) -> dict | None:
|
|
117
|
+
rows = self.q("SELECT * FROM snapshots WHERE endpoint_id=? ORDER BY id DESC LIMIT 1", endpoint_id)
|
|
118
|
+
return rows[0] if rows else None
|
|
119
|
+
|
|
120
|
+
def commit(self):
|
|
121
|
+
self.conn.commit()
|