grim-mcp 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- grim/__init__.py +3 -0
- grim/__main__.py +211 -0
- grim/core/__init__.py +0 -0
- grim/core/attack.py +78 -0
- grim/core/detector.py +159 -0
- grim/core/findings.py +111 -0
- grim/core/ledger.py +161 -0
- grim/core/planner.py +79 -0
- grim/core/report.py +93 -0
- grim/engines/__init__.py +0 -0
- grim/engines/codepatterns.py +612 -0
- grim/engines/deps.py +358 -0
- grim/engines/diffscan.py +235 -0
- grim/engines/exposure.py +496 -0
- grim/engines/flow.py +262 -0
- grim/engines/secrets.py +393 -0
- grim/feeds/__init__.py +1 -0
- grim/feeds/iocs.py +213 -0
- grim/mcp/__init__.py +0 -0
- grim/mcp/server.py +99 -0
- grim/sbom.py +144 -0
- grim/tools.py +384 -0
- grim_mcp-0.2.0.dist-info/LICENSE +21 -0
- grim_mcp-0.2.0.dist-info/METADATA +555 -0
- grim_mcp-0.2.0.dist-info/RECORD +28 -0
- grim_mcp-0.2.0.dist-info/WHEEL +5 -0
- grim_mcp-0.2.0.dist-info/entry_points.txt +2 -0
- grim_mcp-0.2.0.dist-info/top_level.txt +1 -0
grim/__init__.py
ADDED
grim/__main__.py
ADDED
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
"""GRIM CLI entry point.
|
|
2
|
+
|
|
3
|
+
Usage:
|
|
4
|
+
grim version
|
|
5
|
+
grim mcp # run MCP server (stdio)
|
|
6
|
+
grim list # list tools
|
|
7
|
+
grim scan PATH [--format md|json] [--out FILE] [--tools ...] [--no-network] [--deep]
|
|
8
|
+
grim tool NAME --path P [--path-b P2] [--format md|json]
|
|
9
|
+
grim diff A B [--format md|json] [--deep]
|
|
10
|
+
grim plan PATH [--no-network] [--deep]
|
|
11
|
+
grim sbom PATH [--format cyclonedx|spdx] [--out FILE]
|
|
12
|
+
grim ledger PATH [--ledger FILE] [--tools ...]
|
|
13
|
+
grim iocs PATH [--deep]
|
|
14
|
+
grim update-feeds [--url URL]
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import argparse
|
|
20
|
+
import json
|
|
21
|
+
import sys
|
|
22
|
+
|
|
23
|
+
from . import __version__
|
|
24
|
+
from .core.report import render_markdown, render_summary_line
|
|
25
|
+
from .tools import TOOLS, call_tool
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def main(argv: list[str] | None = None) -> int:
|
|
29
|
+
parser = argparse.ArgumentParser(prog="grim", description="GRIM — security audit for code, deps, exposure, drift")
|
|
30
|
+
sub = parser.add_subparsers(dest="command")
|
|
31
|
+
|
|
32
|
+
sub.add_parser("version", help="print version")
|
|
33
|
+
sub.add_parser("mcp", help="run MCP server over stdio")
|
|
34
|
+
sub.add_parser("list", help="list available tools")
|
|
35
|
+
|
|
36
|
+
p_scan = sub.add_parser("scan", help="one-shot audit of a path or archive")
|
|
37
|
+
p_scan.add_argument("path")
|
|
38
|
+
p_scan.add_argument("--format", choices=["md", "json"], default="md")
|
|
39
|
+
p_scan.add_argument("--out", default=None)
|
|
40
|
+
p_scan.add_argument("--tools", default=None, help="subset: exposure,secrets,code,deps,iocs")
|
|
41
|
+
p_scan.add_argument("--no-network", action="store_true")
|
|
42
|
+
p_scan.add_argument("--deep", action="store_true", help="nested archives + IoC hash matching")
|
|
43
|
+
|
|
44
|
+
p_tool = sub.add_parser("tool", help="run a single tool")
|
|
45
|
+
p_tool.add_argument("name")
|
|
46
|
+
p_tool.add_argument("--path", default=None)
|
|
47
|
+
p_tool.add_argument("--path-a", default=None)
|
|
48
|
+
p_tool.add_argument("--path-b", default=None)
|
|
49
|
+
p_tool.add_argument("--format", choices=["md", "json"], default="md")
|
|
50
|
+
p_tool.add_argument("--raw", action="store_true", help="print raw JSON result payload")
|
|
51
|
+
|
|
52
|
+
p_diff = sub.add_parser("diff", help="diff two artifacts (baseline vs current)")
|
|
53
|
+
p_diff.add_argument("path_a")
|
|
54
|
+
p_diff.add_argument("path_b")
|
|
55
|
+
p_diff.add_argument("--format", choices=["md", "json"], default="md")
|
|
56
|
+
p_diff.add_argument("--deep", action="store_true")
|
|
57
|
+
|
|
58
|
+
p_plan = sub.add_parser("plan", help="show the audit plan for a target")
|
|
59
|
+
p_plan.add_argument("path")
|
|
60
|
+
p_plan.add_argument("--no-network", action="store_true")
|
|
61
|
+
p_plan.add_argument("--deep", action="store_true")
|
|
62
|
+
|
|
63
|
+
p_sbom = sub.add_parser("sbom", help="emit CycloneDX/SPDX SBOM")
|
|
64
|
+
p_sbom.add_argument("path")
|
|
65
|
+
p_sbom.add_argument("--format", choices=["cyclonedx", "spdx"], default="cyclonedx")
|
|
66
|
+
p_sbom.add_argument("--out", default=None)
|
|
67
|
+
|
|
68
|
+
p_ledger = sub.add_parser("ledger", help="merge a scan into the persistent findings ledger")
|
|
69
|
+
p_ledger.add_argument("path")
|
|
70
|
+
p_ledger.add_argument("--ledger", default=None)
|
|
71
|
+
p_ledger.add_argument("--tools", default=None)
|
|
72
|
+
p_ledger.add_argument("--deep", action="store_true")
|
|
73
|
+
|
|
74
|
+
p_iocs = sub.add_parser("iocs", help="match file hashes against the IoC store")
|
|
75
|
+
p_iocs.add_argument("path")
|
|
76
|
+
p_iocs.add_argument("--deep", action="store_true")
|
|
77
|
+
|
|
78
|
+
p_feeds = sub.add_parser("update-feeds", help="sync the IoC store from a feed URL")
|
|
79
|
+
p_feeds.add_argument("--url", default=None)
|
|
80
|
+
p_feeds.add_argument("--ioc-path", default=None)
|
|
81
|
+
|
|
82
|
+
args = parser.parse_args(argv)
|
|
83
|
+
|
|
84
|
+
if args.command in (None, "version"):
|
|
85
|
+
print(f"grim {__version__}")
|
|
86
|
+
return 0
|
|
87
|
+
if args.command == "mcp":
|
|
88
|
+
from .mcp.server import run_server
|
|
89
|
+
|
|
90
|
+
run_server()
|
|
91
|
+
return 0
|
|
92
|
+
if args.command == "list":
|
|
93
|
+
for name, meta in TOOLS.items():
|
|
94
|
+
print(f"{name:16s} {meta['description']}")
|
|
95
|
+
return 0
|
|
96
|
+
if args.command == "scan":
|
|
97
|
+
tool_args = {"path": args.path}
|
|
98
|
+
if args.tools:
|
|
99
|
+
tool_args["tools"] = [t.strip() for t in args.tools.split(",") if t.strip()]
|
|
100
|
+
if args.no_network:
|
|
101
|
+
tool_args["network"] = False
|
|
102
|
+
if args.deep:
|
|
103
|
+
tool_args["deep"] = True
|
|
104
|
+
payload = call_tool("scan", tool_args)
|
|
105
|
+
return _emit(payload, args.format, args.out)
|
|
106
|
+
if args.command == "tool":
|
|
107
|
+
if args.name not in TOOLS:
|
|
108
|
+
print(f"unknown tool: {args.name}", file=sys.stderr)
|
|
109
|
+
return 2
|
|
110
|
+
tool_args = _build_tool_args(args.name, args)
|
|
111
|
+
payload = call_tool(args.name, tool_args)
|
|
112
|
+
if args.raw:
|
|
113
|
+
print(json.dumps(payload, indent=2, default=str))
|
|
114
|
+
return 0 if payload.get("ok") else 1
|
|
115
|
+
return _emit(payload, args.format, None)
|
|
116
|
+
if args.command == "diff":
|
|
117
|
+
payload = call_tool("diff_artifacts", {"path_a": args.path_a, "path_b": args.path_b,
|
|
118
|
+
"deep": bool(args.deep)})
|
|
119
|
+
return _emit(payload, args.format, None)
|
|
120
|
+
if args.command == "plan":
|
|
121
|
+
return _print_json(call_tool("plan", {"path": args.path,
|
|
122
|
+
"network": not args.no_network,
|
|
123
|
+
"deep": args.deep}))
|
|
124
|
+
if args.command == "sbom":
|
|
125
|
+
payload = call_tool("sbom", {"path": args.path, "format": args.format})
|
|
126
|
+
if not payload.get("ok"):
|
|
127
|
+
print(f"error: {payload.get('error')}", file=sys.stderr)
|
|
128
|
+
return 1
|
|
129
|
+
text = json.dumps(payload["bom"], indent=2, default=str)
|
|
130
|
+
if args.out:
|
|
131
|
+
with open(args.out, "w", encoding="utf-8") as fh:
|
|
132
|
+
fh.write(text)
|
|
133
|
+
print(f"wrote {args.out} ({payload['component_count']} components)")
|
|
134
|
+
else:
|
|
135
|
+
print(text)
|
|
136
|
+
return 0
|
|
137
|
+
if args.command == "ledger":
|
|
138
|
+
tool_args: dict = {"path": args.path, "deep": bool(args.deep)}
|
|
139
|
+
if args.ledger:
|
|
140
|
+
tool_args["ledger_path"] = args.ledger
|
|
141
|
+
if args.tools:
|
|
142
|
+
tool_args["tools"] = [t.strip() for t in args.tools.split(",") if t.strip()]
|
|
143
|
+
return _print_json(call_tool("ledger", tool_args))
|
|
144
|
+
if args.command == "iocs":
|
|
145
|
+
payload = call_tool("scan_iocs", {"path": args.path, "deep": bool(args.deep)})
|
|
146
|
+
return _emit(payload, "md", None)
|
|
147
|
+
if args.command == "update-feeds":
|
|
148
|
+
tool_args = {"url": args.url, "ioc_path": args.ioc_path}
|
|
149
|
+
return _print_json(call_tool("update_feeds", {k: v for k, v in tool_args.items() if v}))
|
|
150
|
+
parser.print_help()
|
|
151
|
+
return 1
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _print_json(payload: dict) -> int:
|
|
155
|
+
print(json.dumps(payload, indent=2, default=str))
|
|
156
|
+
return 0 if payload.get("ok") else 1
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _build_tool_args(name: str, args: argparse.Namespace) -> dict:
|
|
160
|
+
schema_props = TOOLS[name]["schema"].get("properties", {})
|
|
161
|
+
provided = {
|
|
162
|
+
"path": args.path,
|
|
163
|
+
"path_a": args.path_a,
|
|
164
|
+
"path_b": args.path_b,
|
|
165
|
+
}
|
|
166
|
+
return {k: v for k, v in provided.items() if k in schema_props and v is not None}
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _emit(payload: dict, fmt: str, out: str | None) -> int:
|
|
170
|
+
if not payload.get("ok"):
|
|
171
|
+
print(f"error: {payload.get('error')}", file=sys.stderr)
|
|
172
|
+
return 1
|
|
173
|
+
if fmt == "json":
|
|
174
|
+
text = json.dumps(payload, indent=2, default=str)
|
|
175
|
+
else:
|
|
176
|
+
findings = [_to_finding(d) for d in payload.get("findings", [])]
|
|
177
|
+
text = render_markdown(findings, payload.get("meta", {}))
|
|
178
|
+
text += "\n" + render_summary_line(findings) + "\n"
|
|
179
|
+
if out:
|
|
180
|
+
with open(out, "w", encoding="utf-8") as fh:
|
|
181
|
+
fh.write(text)
|
|
182
|
+
print(f"wrote {out} ({render_summary_line([_to_finding(d) for d in payload.get('findings', [])])})")
|
|
183
|
+
else:
|
|
184
|
+
print(text)
|
|
185
|
+
return 0
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _to_finding(d: dict):
|
|
189
|
+
from .core.findings import Finding
|
|
190
|
+
|
|
191
|
+
return Finding(
|
|
192
|
+
id=d.get("id", ""),
|
|
193
|
+
severity=d.get("severity", "info"),
|
|
194
|
+
confidence=float(d.get("confidence", 0.5)),
|
|
195
|
+
category=d.get("category", ""),
|
|
196
|
+
owasp=d.get("owasp", ""),
|
|
197
|
+
title=d.get("title", ""),
|
|
198
|
+
description=d.get("description", ""),
|
|
199
|
+
location=d.get("location", {}),
|
|
200
|
+
evidence=d.get("evidence", ""),
|
|
201
|
+
remediation=d.get("remediation", ""),
|
|
202
|
+
references=d.get("references", []),
|
|
203
|
+
engine=d.get("engine", "grim"),
|
|
204
|
+
first_seen=d.get("first_seen", ""),
|
|
205
|
+
tags=d.get("tags", []),
|
|
206
|
+
mitre=d.get("mitre", []),
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
if __name__ == "__main__":
|
|
211
|
+
raise SystemExit(main())
|
grim/core/__init__.py
ADDED
|
File without changes
|
grim/core/attack.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""MITRE ATT&CK tagging for GRIM findings.
|
|
2
|
+
|
|
3
|
+
Maps finding signals (tags, category, title) to ATT&CK technique IDs so reports can
|
|
4
|
+
be correlated with threat-intel tooling. Pure lookup, no network, no deps.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from typing import Iterable
|
|
10
|
+
|
|
11
|
+
from .findings import Finding
|
|
12
|
+
|
|
13
|
+
# (signal substrings matched against tags/category/title, technique id, name)
|
|
14
|
+
RULES: list[tuple[tuple[str, ...], str, str]] = [
|
|
15
|
+
(("webshell", "web shell", "webshell indicator"), "T1505.003", "Server Software Component: Web Shell"),
|
|
16
|
+
(("polyglot",), "T1505.003", "Server Software Component: Web Shell"),
|
|
17
|
+
(("upload", "cwe-434", "arbitrary file"), "T1190", "Exploit Public-Facing Application"),
|
|
18
|
+
(("elf", "native executable", "pe binary"), "T1204.002", "User Execution: Malicious File"),
|
|
19
|
+
(("stager", "downloader", "payload host", "c2"), "T1105", "Ingress Tool Transfer"),
|
|
20
|
+
(("piped shell", "curl | sh"), "T1059.004", "Command and Scripting Interpreter: Unix Shell"),
|
|
21
|
+
(("shell", "command execution", "exec.command", "child_process", "process.start"), "T1059", "Command and Scripting Interpreter"),
|
|
22
|
+
(("eval", "dynamic code execution", "deserialization"), "T1059", "Command and Scripting Interpreter"),
|
|
23
|
+
(("sql", "injection"), "T1190", "Exploit Public-Facing Application"),
|
|
24
|
+
(("xxe", "xml parsing"), "T1190", "Exploit Public-Facing Application"),
|
|
25
|
+
(("secrets", "cwe-798", "credential", "environment file", ".env"), "T1552.001", "Unsecured Credentials: Credentials In Files"),
|
|
26
|
+
(("dependencies", "supply chain", "cwe-1395"), "T1195.002", "Supply Chain Compromise: Compromise Software Supply Chain"),
|
|
27
|
+
(("drift", "changed"), "T1565.001", "Data Manipulation: Stored Data Manipulation"),
|
|
28
|
+
(("removed", "cleanup"), "T1070.004", "Indicator Removal: File Deletion"),
|
|
29
|
+
(("encoded", "obfuscat", "gzinflate", "str_rot13"), "T1027", "Obfuscated Files or Information"),
|
|
30
|
+
(("miner", "resource hijack"), "T1496", "Resource Hijacking"),
|
|
31
|
+
(("log", "cwe-532"), "T1552.001", "Unsecured Credentials: Credentials In Files"),
|
|
32
|
+
(("backup", "cwe-530", "dump"), "T1552.001", "Unsecured Credentials: Credentials In Files"),
|
|
33
|
+
(("exposed .git", "vcs", "cwe-538"), "T1213", "Data from Information Repositories"),
|
|
34
|
+
(("ioc", "known malware"), "T1204.002", "User Execution: Malicious File"),
|
|
35
|
+
(("hardcoded-ip", "suspicious"), "T1071.001", "Application Layer Protocol: Web Protocols"),
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def techniques_for(finding: Finding) -> list[tuple[str, str]]:
|
|
40
|
+
"""Return ordered, de-duplicated (id, name) techniques for a finding."""
|
|
41
|
+
haystack = " ".join(
|
|
42
|
+
[finding.category.lower(), finding.title.lower(), " ".join(finding.tags).lower()]
|
|
43
|
+
)
|
|
44
|
+
out: list[tuple[str, str]] = []
|
|
45
|
+
seen: set[str] = set()
|
|
46
|
+
for signals, tid, name in RULES:
|
|
47
|
+
if tid in seen:
|
|
48
|
+
continue
|
|
49
|
+
if any(sig in haystack for sig in signals):
|
|
50
|
+
seen.add(tid)
|
|
51
|
+
out.append((tid, name))
|
|
52
|
+
return out
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def enrich_one(finding: Finding) -> Finding:
|
|
56
|
+
for tid, _name in techniques_for(finding):
|
|
57
|
+
if tid not in finding.mitre:
|
|
58
|
+
finding.mitre.append(tid)
|
|
59
|
+
tag = f"attack:{tid}"
|
|
60
|
+
if tag not in finding.tags:
|
|
61
|
+
finding.tags.append(tag)
|
|
62
|
+
return finding
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def enrich(findings: Iterable[Finding]) -> list[Finding]:
|
|
66
|
+
"""In-place MITRE tagging; returns the list for convenience."""
|
|
67
|
+
out = list(findings)
|
|
68
|
+
for f in out:
|
|
69
|
+
enrich_one(f)
|
|
70
|
+
return out
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def catalog() -> list[dict[str, str]]:
|
|
74
|
+
"""All techniques GRIM can emit, for documentation/discovery."""
|
|
75
|
+
seen: dict[str, str] = {}
|
|
76
|
+
for _signals, tid, name in RULES:
|
|
77
|
+
seen.setdefault(tid, name)
|
|
78
|
+
return [{"id": k, "name": v} for k, v in sorted(seen.items())]
|
grim/core/detector.py
ADDED
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""Stack detection: identify project type(s) from manifests and file patterns."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
MANIFESTS = {
|
|
10
|
+
"package.json": "node",
|
|
11
|
+
"composer.json": "php",
|
|
12
|
+
"requirements.txt": "python",
|
|
13
|
+
"pyproject.toml": "python",
|
|
14
|
+
"Pipfile": "python",
|
|
15
|
+
"go.mod": "go",
|
|
16
|
+
"Gemfile": "ruby",
|
|
17
|
+
"pom.xml": "java",
|
|
18
|
+
"build.gradle": "java",
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
JS_FRAMEWORKS = {
|
|
22
|
+
"next": "nextjs",
|
|
23
|
+
"nuxt": "nuxt",
|
|
24
|
+
"astro": "astro",
|
|
25
|
+
"react": "react",
|
|
26
|
+
"vue": "vue",
|
|
27
|
+
"svelte": "svelte",
|
|
28
|
+
"@angular/core": "angular",
|
|
29
|
+
"express": "express",
|
|
30
|
+
"fastify": "fastify",
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
PHP_FRAMEWORKS = {
|
|
34
|
+
"laravel/framework": "laravel",
|
|
35
|
+
"symfony/symfony": "symfony",
|
|
36
|
+
"cakephp/cakephp": "cakephp",
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
CONFIG_MARKERS = {
|
|
40
|
+
"next.config.js": "nextjs",
|
|
41
|
+
"next.config.mjs": "nextjs",
|
|
42
|
+
"nuxt.config.ts": "nuxt",
|
|
43
|
+
"astro.config.mjs": "astro",
|
|
44
|
+
"angular.json": "angular",
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
MAX_DEPTH = 4
|
|
48
|
+
SKIP_DIRS = {".git", "node_modules", "vendor", ".venv", "venv", "__pycache__", "dist", "build"}
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def detect_stack(path: str) -> dict:
|
|
52
|
+
"""Return {stacks: [{language, framework, manifest, path}], web_dirs: [...]}."""
|
|
53
|
+
p = Path(path)
|
|
54
|
+
if not p.exists():
|
|
55
|
+
raise FileNotFoundError(f"path not found: {path}")
|
|
56
|
+
|
|
57
|
+
stacks: list[dict] = []
|
|
58
|
+
manifests_found: list[str] = []
|
|
59
|
+
web_dirs: list[str] = []
|
|
60
|
+
lockfiles: list[str] = []
|
|
61
|
+
|
|
62
|
+
if p.is_file():
|
|
63
|
+
manifests_found.append(p.name)
|
|
64
|
+
else:
|
|
65
|
+
for root, dirs, files in os.walk(p):
|
|
66
|
+
rel = os.path.relpath(root, p)
|
|
67
|
+
depth = 0 if rel == "." else rel.count(os.sep) + 1
|
|
68
|
+
dirs[:] = [d for d in dirs if d not in SKIP_DIRS and depth < MAX_DEPTH]
|
|
69
|
+
base = os.path.basename(root).lower()
|
|
70
|
+
if base in {"public", "public_html", "www", "htdocs", "web"}:
|
|
71
|
+
web_dirs.append(str(Path(root)))
|
|
72
|
+
for fn in files:
|
|
73
|
+
if fn in MANIFESTS:
|
|
74
|
+
manifests_found.append(str(Path(root) / fn))
|
|
75
|
+
if fn in CONFIG_MARKERS:
|
|
76
|
+
manifests_found.append(str(Path(root) / fn))
|
|
77
|
+
if fn.endswith(".lock") and fn in {"composer.lock", "package-lock.json", "yarn.lock", "pnpm-lock.yaml"}:
|
|
78
|
+
lockfiles.append(str(Path(root) / fn))
|
|
79
|
+
|
|
80
|
+
for m in manifests_found:
|
|
81
|
+
mp = Path(m)
|
|
82
|
+
lang = MANIFESTS.get(mp.name)
|
|
83
|
+
framework = None
|
|
84
|
+
if lang is None and mp.name in CONFIG_MARKERS:
|
|
85
|
+
framework = CONFIG_MARKERS[mp.name]
|
|
86
|
+
lang = "node"
|
|
87
|
+
if mp.name == "package.json":
|
|
88
|
+
framework = framework or _js_framework(mp)
|
|
89
|
+
if mp.name == "composer.json":
|
|
90
|
+
framework = _php_framework(mp)
|
|
91
|
+
|
|
92
|
+
stacks.append(
|
|
93
|
+
{
|
|
94
|
+
"language": lang or "unknown",
|
|
95
|
+
"framework": framework,
|
|
96
|
+
"manifest": m,
|
|
97
|
+
}
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
# Static site detection
|
|
101
|
+
if not stacks:
|
|
102
|
+
for candidate in ["index.html", "index.htm"]:
|
|
103
|
+
if (p / candidate).is_file() if p.is_dir() else False:
|
|
104
|
+
stacks.append({"language": "static", "framework": "html", "manifest": candidate})
|
|
105
|
+
|
|
106
|
+
return {
|
|
107
|
+
"path": str(p),
|
|
108
|
+
"is_archive": p.is_file() and _looks_archive(p.name),
|
|
109
|
+
"stacks": _dedupe_stacks(stacks),
|
|
110
|
+
"web_dirs": web_dirs[:20],
|
|
111
|
+
"lockfiles": lockfiles[:20],
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _js_framework(package_json: Path) -> str | None:
|
|
116
|
+
try:
|
|
117
|
+
data = json.loads(package_json.read_text(encoding="utf-8", errors="ignore"))
|
|
118
|
+
except Exception:
|
|
119
|
+
return None
|
|
120
|
+
deps = {}
|
|
121
|
+
deps.update(data.get("dependencies") or {})
|
|
122
|
+
deps.update(data.get("devDependencies") or {})
|
|
123
|
+
for key, fw in JS_FRAMEWORKS.items():
|
|
124
|
+
if key in deps:
|
|
125
|
+
return fw
|
|
126
|
+
return None
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _php_framework(composer_json: Path) -> str | None:
|
|
130
|
+
try:
|
|
131
|
+
data = json.loads(composer_json.read_text(encoding="utf-8", errors="ignore"))
|
|
132
|
+
except Exception:
|
|
133
|
+
return None
|
|
134
|
+
deps = {}
|
|
135
|
+
deps.update(data.get("require") or {})
|
|
136
|
+
deps.update(data.get("require-dev") or {})
|
|
137
|
+
for key, fw in PHP_FRAMEWORKS.items():
|
|
138
|
+
if key in deps:
|
|
139
|
+
return fw
|
|
140
|
+
return None
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _dedupe_stacks(stacks: list[dict]) -> list[dict]:
|
|
144
|
+
seen = set()
|
|
145
|
+
out = []
|
|
146
|
+
for s in stacks:
|
|
147
|
+
key = (s["language"], s["framework"])
|
|
148
|
+
if key in seen:
|
|
149
|
+
continue
|
|
150
|
+
seen.add(key)
|
|
151
|
+
out.append(s)
|
|
152
|
+
return out
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _looks_archive(name: str) -> bool:
|
|
156
|
+
lowered = name.lower()
|
|
157
|
+
return lowered.endswith(
|
|
158
|
+
(".tar", ".tar.gz", ".tgz", ".tar.bz2", ".tar.xz", ".zip", ".7z", ".rar", ".gz", ".bz2", ".xz")
|
|
159
|
+
)
|
grim/core/findings.py
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""Finding model, ranking, and dedupe for GRIM."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
from dataclasses import dataclass, field, asdict
|
|
8
|
+
from datetime import datetime, timezone
|
|
9
|
+
from typing import Any, Iterable
|
|
10
|
+
|
|
11
|
+
SEVERITY_ORDER = {"critical": 4, "high": 3, "medium": 2, "low": 1, "info": 0}
|
|
12
|
+
SEVERITY_EMOJI = {"critical": "!!", "high": "!", "medium": "+", "low": "-", "info": "."}
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _now() -> str:
|
|
16
|
+
return datetime.now(timezone.utc).isoformat(timespec="seconds")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass
|
|
20
|
+
class Finding:
|
|
21
|
+
id: str
|
|
22
|
+
severity: str
|
|
23
|
+
confidence: float
|
|
24
|
+
category: str
|
|
25
|
+
owasp: str
|
|
26
|
+
title: str
|
|
27
|
+
description: str = ""
|
|
28
|
+
location: dict[str, Any] = field(default_factory=dict)
|
|
29
|
+
evidence: str = ""
|
|
30
|
+
remediation: str = ""
|
|
31
|
+
references: list[str] = field(default_factory=list)
|
|
32
|
+
engine: str = "grim"
|
|
33
|
+
first_seen: str = field(default_factory=_now)
|
|
34
|
+
tags: list[str] = field(default_factory=list)
|
|
35
|
+
mitre: list[str] = field(default_factory=list)
|
|
36
|
+
|
|
37
|
+
def to_dict(self) -> dict[str, Any]:
|
|
38
|
+
d = asdict(self)
|
|
39
|
+
d["severity_rank"] = SEVERITY_ORDER.get(self.severity, 0)
|
|
40
|
+
return d
|
|
41
|
+
|
|
42
|
+
def fingerprint(self) -> str:
|
|
43
|
+
key = json.dumps(
|
|
44
|
+
[
|
|
45
|
+
self.severity,
|
|
46
|
+
self.category,
|
|
47
|
+
self.location.get("file", self.location.get("url", "")),
|
|
48
|
+
self.location.get("line", 0),
|
|
49
|
+
self.title,
|
|
50
|
+
],
|
|
51
|
+
sort_keys=True,
|
|
52
|
+
)
|
|
53
|
+
return hashlib.sha256(key.encode()).hexdigest()[:16]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
SEVERITY_CVSS_BANDS = [
|
|
57
|
+
(9.0, "critical"),
|
|
58
|
+
(7.0, "high"),
|
|
59
|
+
(4.0, "medium"),
|
|
60
|
+
(0.1, "low"),
|
|
61
|
+
(0.0, "info"),
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def cvss_to_severity(score: float) -> str:
|
|
66
|
+
for threshold, sev in SEVERITY_CVSS_BANDS:
|
|
67
|
+
if score >= threshold:
|
|
68
|
+
return sev
|
|
69
|
+
return "info"
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def rank(findings: Iterable[Finding]) -> list[Finding]:
|
|
73
|
+
"""Dedupe and sort by severity x confidence, then by location."""
|
|
74
|
+
seen: dict[str, Finding] = {}
|
|
75
|
+
for f in findings:
|
|
76
|
+
fp = f.fingerprint()
|
|
77
|
+
if fp in seen:
|
|
78
|
+
existing = seen[fp]
|
|
79
|
+
# keep the higher-confidence version, merge tags
|
|
80
|
+
if f.confidence > existing.confidence:
|
|
81
|
+
existing.confidence = f.confidence
|
|
82
|
+
existing.evidence = f.evidence or existing.evidence
|
|
83
|
+
for t in f.tags:
|
|
84
|
+
if t not in existing.tags:
|
|
85
|
+
existing.tags.append(t)
|
|
86
|
+
continue
|
|
87
|
+
seen[fp] = f
|
|
88
|
+
ordered = sorted(
|
|
89
|
+
seen.values(),
|
|
90
|
+
key=lambda f: (
|
|
91
|
+
-SEVERITY_ORDER.get(f.severity, 0),
|
|
92
|
+
-f.confidence,
|
|
93
|
+
str(f.location.get("file", f.location.get("url", ""))),
|
|
94
|
+
int(f.location.get("line", 0) or 0),
|
|
95
|
+
),
|
|
96
|
+
)
|
|
97
|
+
return ordered
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def summarize(findings: Iterable[Finding]) -> dict[str, Any]:
|
|
101
|
+
counts: dict[str, int] = {k: 0 for k in SEVERITY_ORDER}
|
|
102
|
+
total = 0
|
|
103
|
+
for f in findings:
|
|
104
|
+
counts[f.severity] = counts.get(f.severity, 0) + 1
|
|
105
|
+
total += 1
|
|
106
|
+
return {"total": total, "by_severity": counts}
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def make_id(prefix: str, title: str, location: str = "") -> str:
|
|
110
|
+
digest = hashlib.sha1(f"{prefix}|{title}|{location}".encode()).hexdigest()[:6].upper()
|
|
111
|
+
return f"GRIM-{prefix}-{digest}"
|