grim-mcp 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
grim/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """GRIM — security audit orchestrator for code, deps, exposure, and active compromise."""
2
+
3
+ __version__ = "0.2.0"
grim/__main__.py ADDED
@@ -0,0 +1,211 @@
1
+ """GRIM CLI entry point.
2
+
3
+ Usage:
4
+ grim version
5
+ grim mcp # run MCP server (stdio)
6
+ grim list # list tools
7
+ grim scan PATH [--format md|json] [--out FILE] [--tools ...] [--no-network] [--deep]
8
+ grim tool NAME --path P [--path-b P2] [--format md|json]
9
+ grim diff A B [--format md|json] [--deep]
10
+ grim plan PATH [--no-network] [--deep]
11
+ grim sbom PATH [--format cyclonedx|spdx] [--out FILE]
12
+ grim ledger PATH [--ledger FILE] [--tools ...]
13
+ grim iocs PATH [--deep]
14
+ grim update-feeds [--url URL]
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import argparse
20
+ import json
21
+ import sys
22
+
23
+ from . import __version__
24
+ from .core.report import render_markdown, render_summary_line
25
+ from .tools import TOOLS, call_tool
26
+
27
+
28
+ def main(argv: list[str] | None = None) -> int:
29
+ parser = argparse.ArgumentParser(prog="grim", description="GRIM — security audit for code, deps, exposure, drift")
30
+ sub = parser.add_subparsers(dest="command")
31
+
32
+ sub.add_parser("version", help="print version")
33
+ sub.add_parser("mcp", help="run MCP server over stdio")
34
+ sub.add_parser("list", help="list available tools")
35
+
36
+ p_scan = sub.add_parser("scan", help="one-shot audit of a path or archive")
37
+ p_scan.add_argument("path")
38
+ p_scan.add_argument("--format", choices=["md", "json"], default="md")
39
+ p_scan.add_argument("--out", default=None)
40
+ p_scan.add_argument("--tools", default=None, help="subset: exposure,secrets,code,deps,iocs")
41
+ p_scan.add_argument("--no-network", action="store_true")
42
+ p_scan.add_argument("--deep", action="store_true", help="nested archives + IoC hash matching")
43
+
44
+ p_tool = sub.add_parser("tool", help="run a single tool")
45
+ p_tool.add_argument("name")
46
+ p_tool.add_argument("--path", default=None)
47
+ p_tool.add_argument("--path-a", default=None)
48
+ p_tool.add_argument("--path-b", default=None)
49
+ p_tool.add_argument("--format", choices=["md", "json"], default="md")
50
+ p_tool.add_argument("--raw", action="store_true", help="print raw JSON result payload")
51
+
52
+ p_diff = sub.add_parser("diff", help="diff two artifacts (baseline vs current)")
53
+ p_diff.add_argument("path_a")
54
+ p_diff.add_argument("path_b")
55
+ p_diff.add_argument("--format", choices=["md", "json"], default="md")
56
+ p_diff.add_argument("--deep", action="store_true")
57
+
58
+ p_plan = sub.add_parser("plan", help="show the audit plan for a target")
59
+ p_plan.add_argument("path")
60
+ p_plan.add_argument("--no-network", action="store_true")
61
+ p_plan.add_argument("--deep", action="store_true")
62
+
63
+ p_sbom = sub.add_parser("sbom", help="emit CycloneDX/SPDX SBOM")
64
+ p_sbom.add_argument("path")
65
+ p_sbom.add_argument("--format", choices=["cyclonedx", "spdx"], default="cyclonedx")
66
+ p_sbom.add_argument("--out", default=None)
67
+
68
+ p_ledger = sub.add_parser("ledger", help="merge a scan into the persistent findings ledger")
69
+ p_ledger.add_argument("path")
70
+ p_ledger.add_argument("--ledger", default=None)
71
+ p_ledger.add_argument("--tools", default=None)
72
+ p_ledger.add_argument("--deep", action="store_true")
73
+
74
+ p_iocs = sub.add_parser("iocs", help="match file hashes against the IoC store")
75
+ p_iocs.add_argument("path")
76
+ p_iocs.add_argument("--deep", action="store_true")
77
+
78
+ p_feeds = sub.add_parser("update-feeds", help="sync the IoC store from a feed URL")
79
+ p_feeds.add_argument("--url", default=None)
80
+ p_feeds.add_argument("--ioc-path", default=None)
81
+
82
+ args = parser.parse_args(argv)
83
+
84
+ if args.command in (None, "version"):
85
+ print(f"grim {__version__}")
86
+ return 0
87
+ if args.command == "mcp":
88
+ from .mcp.server import run_server
89
+
90
+ run_server()
91
+ return 0
92
+ if args.command == "list":
93
+ for name, meta in TOOLS.items():
94
+ print(f"{name:16s} {meta['description']}")
95
+ return 0
96
+ if args.command == "scan":
97
+ tool_args = {"path": args.path}
98
+ if args.tools:
99
+ tool_args["tools"] = [t.strip() for t in args.tools.split(",") if t.strip()]
100
+ if args.no_network:
101
+ tool_args["network"] = False
102
+ if args.deep:
103
+ tool_args["deep"] = True
104
+ payload = call_tool("scan", tool_args)
105
+ return _emit(payload, args.format, args.out)
106
+ if args.command == "tool":
107
+ if args.name not in TOOLS:
108
+ print(f"unknown tool: {args.name}", file=sys.stderr)
109
+ return 2
110
+ tool_args = _build_tool_args(args.name, args)
111
+ payload = call_tool(args.name, tool_args)
112
+ if args.raw:
113
+ print(json.dumps(payload, indent=2, default=str))
114
+ return 0 if payload.get("ok") else 1
115
+ return _emit(payload, args.format, None)
116
+ if args.command == "diff":
117
+ payload = call_tool("diff_artifacts", {"path_a": args.path_a, "path_b": args.path_b,
118
+ "deep": bool(args.deep)})
119
+ return _emit(payload, args.format, None)
120
+ if args.command == "plan":
121
+ return _print_json(call_tool("plan", {"path": args.path,
122
+ "network": not args.no_network,
123
+ "deep": args.deep}))
124
+ if args.command == "sbom":
125
+ payload = call_tool("sbom", {"path": args.path, "format": args.format})
126
+ if not payload.get("ok"):
127
+ print(f"error: {payload.get('error')}", file=sys.stderr)
128
+ return 1
129
+ text = json.dumps(payload["bom"], indent=2, default=str)
130
+ if args.out:
131
+ with open(args.out, "w", encoding="utf-8") as fh:
132
+ fh.write(text)
133
+ print(f"wrote {args.out} ({payload['component_count']} components)")
134
+ else:
135
+ print(text)
136
+ return 0
137
+ if args.command == "ledger":
138
+ tool_args: dict = {"path": args.path, "deep": bool(args.deep)}
139
+ if args.ledger:
140
+ tool_args["ledger_path"] = args.ledger
141
+ if args.tools:
142
+ tool_args["tools"] = [t.strip() for t in args.tools.split(",") if t.strip()]
143
+ return _print_json(call_tool("ledger", tool_args))
144
+ if args.command == "iocs":
145
+ payload = call_tool("scan_iocs", {"path": args.path, "deep": bool(args.deep)})
146
+ return _emit(payload, "md", None)
147
+ if args.command == "update-feeds":
148
+ tool_args = {"url": args.url, "ioc_path": args.ioc_path}
149
+ return _print_json(call_tool("update_feeds", {k: v for k, v in tool_args.items() if v}))
150
+ parser.print_help()
151
+ return 1
152
+
153
+
154
+ def _print_json(payload: dict) -> int:
155
+ print(json.dumps(payload, indent=2, default=str))
156
+ return 0 if payload.get("ok") else 1
157
+
158
+
159
+ def _build_tool_args(name: str, args: argparse.Namespace) -> dict:
160
+ schema_props = TOOLS[name]["schema"].get("properties", {})
161
+ provided = {
162
+ "path": args.path,
163
+ "path_a": args.path_a,
164
+ "path_b": args.path_b,
165
+ }
166
+ return {k: v for k, v in provided.items() if k in schema_props and v is not None}
167
+
168
+
169
+ def _emit(payload: dict, fmt: str, out: str | None) -> int:
170
+ if not payload.get("ok"):
171
+ print(f"error: {payload.get('error')}", file=sys.stderr)
172
+ return 1
173
+ if fmt == "json":
174
+ text = json.dumps(payload, indent=2, default=str)
175
+ else:
176
+ findings = [_to_finding(d) for d in payload.get("findings", [])]
177
+ text = render_markdown(findings, payload.get("meta", {}))
178
+ text += "\n" + render_summary_line(findings) + "\n"
179
+ if out:
180
+ with open(out, "w", encoding="utf-8") as fh:
181
+ fh.write(text)
182
+ print(f"wrote {out} ({render_summary_line([_to_finding(d) for d in payload.get('findings', [])])})")
183
+ else:
184
+ print(text)
185
+ return 0
186
+
187
+
188
+ def _to_finding(d: dict):
189
+ from .core.findings import Finding
190
+
191
+ return Finding(
192
+ id=d.get("id", ""),
193
+ severity=d.get("severity", "info"),
194
+ confidence=float(d.get("confidence", 0.5)),
195
+ category=d.get("category", ""),
196
+ owasp=d.get("owasp", ""),
197
+ title=d.get("title", ""),
198
+ description=d.get("description", ""),
199
+ location=d.get("location", {}),
200
+ evidence=d.get("evidence", ""),
201
+ remediation=d.get("remediation", ""),
202
+ references=d.get("references", []),
203
+ engine=d.get("engine", "grim"),
204
+ first_seen=d.get("first_seen", ""),
205
+ tags=d.get("tags", []),
206
+ mitre=d.get("mitre", []),
207
+ )
208
+
209
+
210
+ if __name__ == "__main__":
211
+ raise SystemExit(main())
grim/core/__init__.py ADDED
File without changes
grim/core/attack.py ADDED
@@ -0,0 +1,78 @@
1
+ """MITRE ATT&CK tagging for GRIM findings.
2
+
3
+ Maps finding signals (tags, category, title) to ATT&CK technique IDs so reports can
4
+ be correlated with threat-intel tooling. Pure lookup, no network, no deps.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from typing import Iterable
10
+
11
+ from .findings import Finding
12
+
13
+ # (signal substrings matched against tags/category/title, technique id, name)
14
+ RULES: list[tuple[tuple[str, ...], str, str]] = [
15
+ (("webshell", "web shell", "webshell indicator"), "T1505.003", "Server Software Component: Web Shell"),
16
+ (("polyglot",), "T1505.003", "Server Software Component: Web Shell"),
17
+ (("upload", "cwe-434", "arbitrary file"), "T1190", "Exploit Public-Facing Application"),
18
+ (("elf", "native executable", "pe binary"), "T1204.002", "User Execution: Malicious File"),
19
+ (("stager", "downloader", "payload host", "c2"), "T1105", "Ingress Tool Transfer"),
20
+ (("piped shell", "curl | sh"), "T1059.004", "Command and Scripting Interpreter: Unix Shell"),
21
+ (("shell", "command execution", "exec.command", "child_process", "process.start"), "T1059", "Command and Scripting Interpreter"),
22
+ (("eval", "dynamic code execution", "deserialization"), "T1059", "Command and Scripting Interpreter"),
23
+ (("sql", "injection"), "T1190", "Exploit Public-Facing Application"),
24
+ (("xxe", "xml parsing"), "T1190", "Exploit Public-Facing Application"),
25
+ (("secrets", "cwe-798", "credential", "environment file", ".env"), "T1552.001", "Unsecured Credentials: Credentials In Files"),
26
+ (("dependencies", "supply chain", "cwe-1395"), "T1195.002", "Supply Chain Compromise: Compromise Software Supply Chain"),
27
+ (("drift", "changed"), "T1565.001", "Data Manipulation: Stored Data Manipulation"),
28
+ (("removed", "cleanup"), "T1070.004", "Indicator Removal: File Deletion"),
29
+ (("encoded", "obfuscat", "gzinflate", "str_rot13"), "T1027", "Obfuscated Files or Information"),
30
+ (("miner", "resource hijack"), "T1496", "Resource Hijacking"),
31
+ (("log", "cwe-532"), "T1552.001", "Unsecured Credentials: Credentials In Files"),
32
+ (("backup", "cwe-530", "dump"), "T1552.001", "Unsecured Credentials: Credentials In Files"),
33
+ (("exposed .git", "vcs", "cwe-538"), "T1213", "Data from Information Repositories"),
34
+ (("ioc", "known malware"), "T1204.002", "User Execution: Malicious File"),
35
+ (("hardcoded-ip", "suspicious"), "T1071.001", "Application Layer Protocol: Web Protocols"),
36
+ ]
37
+
38
+
39
+ def techniques_for(finding: Finding) -> list[tuple[str, str]]:
40
+ """Return ordered, de-duplicated (id, name) techniques for a finding."""
41
+ haystack = " ".join(
42
+ [finding.category.lower(), finding.title.lower(), " ".join(finding.tags).lower()]
43
+ )
44
+ out: list[tuple[str, str]] = []
45
+ seen: set[str] = set()
46
+ for signals, tid, name in RULES:
47
+ if tid in seen:
48
+ continue
49
+ if any(sig in haystack for sig in signals):
50
+ seen.add(tid)
51
+ out.append((tid, name))
52
+ return out
53
+
54
+
55
+ def enrich_one(finding: Finding) -> Finding:
56
+ for tid, _name in techniques_for(finding):
57
+ if tid not in finding.mitre:
58
+ finding.mitre.append(tid)
59
+ tag = f"attack:{tid}"
60
+ if tag not in finding.tags:
61
+ finding.tags.append(tag)
62
+ return finding
63
+
64
+
65
+ def enrich(findings: Iterable[Finding]) -> list[Finding]:
66
+ """In-place MITRE tagging; returns the list for convenience."""
67
+ out = list(findings)
68
+ for f in out:
69
+ enrich_one(f)
70
+ return out
71
+
72
+
73
+ def catalog() -> list[dict[str, str]]:
74
+ """All techniques GRIM can emit, for documentation/discovery."""
75
+ seen: dict[str, str] = {}
76
+ for _signals, tid, name in RULES:
77
+ seen.setdefault(tid, name)
78
+ return [{"id": k, "name": v} for k, v in sorted(seen.items())]
grim/core/detector.py ADDED
@@ -0,0 +1,159 @@
1
+ """Stack detection: identify project type(s) from manifests and file patterns."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import os
7
+ from pathlib import Path
8
+
9
+ MANIFESTS = {
10
+ "package.json": "node",
11
+ "composer.json": "php",
12
+ "requirements.txt": "python",
13
+ "pyproject.toml": "python",
14
+ "Pipfile": "python",
15
+ "go.mod": "go",
16
+ "Gemfile": "ruby",
17
+ "pom.xml": "java",
18
+ "build.gradle": "java",
19
+ }
20
+
21
+ JS_FRAMEWORKS = {
22
+ "next": "nextjs",
23
+ "nuxt": "nuxt",
24
+ "astro": "astro",
25
+ "react": "react",
26
+ "vue": "vue",
27
+ "svelte": "svelte",
28
+ "@angular/core": "angular",
29
+ "express": "express",
30
+ "fastify": "fastify",
31
+ }
32
+
33
+ PHP_FRAMEWORKS = {
34
+ "laravel/framework": "laravel",
35
+ "symfony/symfony": "symfony",
36
+ "cakephp/cakephp": "cakephp",
37
+ }
38
+
39
+ CONFIG_MARKERS = {
40
+ "next.config.js": "nextjs",
41
+ "next.config.mjs": "nextjs",
42
+ "nuxt.config.ts": "nuxt",
43
+ "astro.config.mjs": "astro",
44
+ "angular.json": "angular",
45
+ }
46
+
47
+ MAX_DEPTH = 4
48
+ SKIP_DIRS = {".git", "node_modules", "vendor", ".venv", "venv", "__pycache__", "dist", "build"}
49
+
50
+
51
+ def detect_stack(path: str) -> dict:
52
+ """Return {stacks: [{language, framework, manifest, path}], web_dirs: [...]}."""
53
+ p = Path(path)
54
+ if not p.exists():
55
+ raise FileNotFoundError(f"path not found: {path}")
56
+
57
+ stacks: list[dict] = []
58
+ manifests_found: list[str] = []
59
+ web_dirs: list[str] = []
60
+ lockfiles: list[str] = []
61
+
62
+ if p.is_file():
63
+ manifests_found.append(p.name)
64
+ else:
65
+ for root, dirs, files in os.walk(p):
66
+ rel = os.path.relpath(root, p)
67
+ depth = 0 if rel == "." else rel.count(os.sep) + 1
68
+ dirs[:] = [d for d in dirs if d not in SKIP_DIRS and depth < MAX_DEPTH]
69
+ base = os.path.basename(root).lower()
70
+ if base in {"public", "public_html", "www", "htdocs", "web"}:
71
+ web_dirs.append(str(Path(root)))
72
+ for fn in files:
73
+ if fn in MANIFESTS:
74
+ manifests_found.append(str(Path(root) / fn))
75
+ if fn in CONFIG_MARKERS:
76
+ manifests_found.append(str(Path(root) / fn))
77
+ if fn.endswith(".lock") and fn in {"composer.lock", "package-lock.json", "yarn.lock", "pnpm-lock.yaml"}:
78
+ lockfiles.append(str(Path(root) / fn))
79
+
80
+ for m in manifests_found:
81
+ mp = Path(m)
82
+ lang = MANIFESTS.get(mp.name)
83
+ framework = None
84
+ if lang is None and mp.name in CONFIG_MARKERS:
85
+ framework = CONFIG_MARKERS[mp.name]
86
+ lang = "node"
87
+ if mp.name == "package.json":
88
+ framework = framework or _js_framework(mp)
89
+ if mp.name == "composer.json":
90
+ framework = _php_framework(mp)
91
+
92
+ stacks.append(
93
+ {
94
+ "language": lang or "unknown",
95
+ "framework": framework,
96
+ "manifest": m,
97
+ }
98
+ )
99
+
100
+ # Static site detection
101
+ if not stacks:
102
+ for candidate in ["index.html", "index.htm"]:
103
+ if (p / candidate).is_file() if p.is_dir() else False:
104
+ stacks.append({"language": "static", "framework": "html", "manifest": candidate})
105
+
106
+ return {
107
+ "path": str(p),
108
+ "is_archive": p.is_file() and _looks_archive(p.name),
109
+ "stacks": _dedupe_stacks(stacks),
110
+ "web_dirs": web_dirs[:20],
111
+ "lockfiles": lockfiles[:20],
112
+ }
113
+
114
+
115
+ def _js_framework(package_json: Path) -> str | None:
116
+ try:
117
+ data = json.loads(package_json.read_text(encoding="utf-8", errors="ignore"))
118
+ except Exception:
119
+ return None
120
+ deps = {}
121
+ deps.update(data.get("dependencies") or {})
122
+ deps.update(data.get("devDependencies") or {})
123
+ for key, fw in JS_FRAMEWORKS.items():
124
+ if key in deps:
125
+ return fw
126
+ return None
127
+
128
+
129
+ def _php_framework(composer_json: Path) -> str | None:
130
+ try:
131
+ data = json.loads(composer_json.read_text(encoding="utf-8", errors="ignore"))
132
+ except Exception:
133
+ return None
134
+ deps = {}
135
+ deps.update(data.get("require") or {})
136
+ deps.update(data.get("require-dev") or {})
137
+ for key, fw in PHP_FRAMEWORKS.items():
138
+ if key in deps:
139
+ return fw
140
+ return None
141
+
142
+
143
+ def _dedupe_stacks(stacks: list[dict]) -> list[dict]:
144
+ seen = set()
145
+ out = []
146
+ for s in stacks:
147
+ key = (s["language"], s["framework"])
148
+ if key in seen:
149
+ continue
150
+ seen.add(key)
151
+ out.append(s)
152
+ return out
153
+
154
+
155
+ def _looks_archive(name: str) -> bool:
156
+ lowered = name.lower()
157
+ return lowered.endswith(
158
+ (".tar", ".tar.gz", ".tgz", ".tar.bz2", ".tar.xz", ".zip", ".7z", ".rar", ".gz", ".bz2", ".xz")
159
+ )
grim/core/findings.py ADDED
@@ -0,0 +1,111 @@
1
+ """Finding model, ranking, and dedupe for GRIM."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ from dataclasses import dataclass, field, asdict
8
+ from datetime import datetime, timezone
9
+ from typing import Any, Iterable
10
+
11
+ SEVERITY_ORDER = {"critical": 4, "high": 3, "medium": 2, "low": 1, "info": 0}
12
+ SEVERITY_EMOJI = {"critical": "!!", "high": "!", "medium": "+", "low": "-", "info": "."}
13
+
14
+
15
+ def _now() -> str:
16
+ return datetime.now(timezone.utc).isoformat(timespec="seconds")
17
+
18
+
19
+ @dataclass
20
+ class Finding:
21
+ id: str
22
+ severity: str
23
+ confidence: float
24
+ category: str
25
+ owasp: str
26
+ title: str
27
+ description: str = ""
28
+ location: dict[str, Any] = field(default_factory=dict)
29
+ evidence: str = ""
30
+ remediation: str = ""
31
+ references: list[str] = field(default_factory=list)
32
+ engine: str = "grim"
33
+ first_seen: str = field(default_factory=_now)
34
+ tags: list[str] = field(default_factory=list)
35
+ mitre: list[str] = field(default_factory=list)
36
+
37
+ def to_dict(self) -> dict[str, Any]:
38
+ d = asdict(self)
39
+ d["severity_rank"] = SEVERITY_ORDER.get(self.severity, 0)
40
+ return d
41
+
42
+ def fingerprint(self) -> str:
43
+ key = json.dumps(
44
+ [
45
+ self.severity,
46
+ self.category,
47
+ self.location.get("file", self.location.get("url", "")),
48
+ self.location.get("line", 0),
49
+ self.title,
50
+ ],
51
+ sort_keys=True,
52
+ )
53
+ return hashlib.sha256(key.encode()).hexdigest()[:16]
54
+
55
+
56
+ SEVERITY_CVSS_BANDS = [
57
+ (9.0, "critical"),
58
+ (7.0, "high"),
59
+ (4.0, "medium"),
60
+ (0.1, "low"),
61
+ (0.0, "info"),
62
+ ]
63
+
64
+
65
+ def cvss_to_severity(score: float) -> str:
66
+ for threshold, sev in SEVERITY_CVSS_BANDS:
67
+ if score >= threshold:
68
+ return sev
69
+ return "info"
70
+
71
+
72
+ def rank(findings: Iterable[Finding]) -> list[Finding]:
73
+ """Dedupe and sort by severity x confidence, then by location."""
74
+ seen: dict[str, Finding] = {}
75
+ for f in findings:
76
+ fp = f.fingerprint()
77
+ if fp in seen:
78
+ existing = seen[fp]
79
+ # keep the higher-confidence version, merge tags
80
+ if f.confidence > existing.confidence:
81
+ existing.confidence = f.confidence
82
+ existing.evidence = f.evidence or existing.evidence
83
+ for t in f.tags:
84
+ if t not in existing.tags:
85
+ existing.tags.append(t)
86
+ continue
87
+ seen[fp] = f
88
+ ordered = sorted(
89
+ seen.values(),
90
+ key=lambda f: (
91
+ -SEVERITY_ORDER.get(f.severity, 0),
92
+ -f.confidence,
93
+ str(f.location.get("file", f.location.get("url", ""))),
94
+ int(f.location.get("line", 0) or 0),
95
+ ),
96
+ )
97
+ return ordered
98
+
99
+
100
+ def summarize(findings: Iterable[Finding]) -> dict[str, Any]:
101
+ counts: dict[str, int] = {k: 0 for k in SEVERITY_ORDER}
102
+ total = 0
103
+ for f in findings:
104
+ counts[f.severity] = counts.get(f.severity, 0) + 1
105
+ total += 1
106
+ return {"total": total, "by_severity": counts}
107
+
108
+
109
+ def make_id(prefix: str, title: str, location: str = "") -> str:
110
+ digest = hashlib.sha1(f"{prefix}|{title}|{location}".encode()).hexdigest()[:6].upper()
111
+ return f"GRIM-{prefix}-{digest}"