gitrupt 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gitrupt/__init__.py +13 -0
- gitrupt/cli.py +546 -0
- gitrupt/config.py +269 -0
- gitrupt/git.py +590 -0
- gitrupt/hooks/__init__.py +7 -0
- gitrupt/hooks/install.py +255 -0
- gitrupt/hooks/pre_commit.py +103 -0
- gitrupt/hooks/pre_push.py +178 -0
- gitrupt/models.py +190 -0
- gitrupt/policy.py +36 -0
- gitrupt/reporting.py +316 -0
- gitrupt/risk.py +197 -0
- gitrupt/scanner.py +117 -0
- gitrupt/scanners/__init__.py +17 -0
- gitrupt/scanners/adapters.py +166 -0
- gitrupt/scanners/base.py +113 -0
- gitrupt/scanners/binaries.py +185 -0
- gitrupt/scanners/code_rules/__init__.py +36 -0
- gitrupt/scanners/code_rules/base.py +27 -0
- gitrupt/scanners/code_rules/go.py +65 -0
- gitrupt/scanners/code_rules/javascript.py +106 -0
- gitrupt/scanners/code_rules/php.py +71 -0
- gitrupt/scanners/code_rules/powershell.py +85 -0
- gitrupt/scanners/code_rules/python.py +153 -0
- gitrupt/scanners/code_rules/ruby.py +76 -0
- gitrupt/scanners/code_rules/rust.py +41 -0
- gitrupt/scanners/code_rules/shell.py +112 -0
- gitrupt/scanners/dependencies.py +244 -0
- gitrupt/scanners/ecosystems/__init__.py +30 -0
- gitrupt/scanners/ecosystems/base.py +60 -0
- gitrupt/scanners/ecosystems/node.py +128 -0
- gitrupt/scanners/ecosystems/python.py +157 -0
- gitrupt/scanners/entropy.py +123 -0
- gitrupt/scanners/forbidden_files.py +201 -0
- gitrupt/scanners/malware.py +219 -0
- gitrupt/scanners/osv_client.py +221 -0
- gitrupt/scanners/registry.py +66 -0
- gitrupt/scanners/secret_rules.py +368 -0
- gitrupt/scanners/secrets.py +558 -0
- gitrupt/scanners/suspicious_code.py +208 -0
- gitrupt/scanners/yara_loader.py +65 -0
- gitrupt/scanners/yara_rules_builtin.py +141 -0
- gitrupt-0.1.0.dist-info/METADATA +342 -0
- gitrupt-0.1.0.dist-info/RECORD +48 -0
- gitrupt-0.1.0.dist-info/WHEEL +5 -0
- gitrupt-0.1.0.dist-info/entry_points.txt +2 -0
- gitrupt-0.1.0.dist-info/licenses/LICENSE +23 -0
- gitrupt-0.1.0.dist-info/top_level.txt +1 -0
gitrupt/risk.py
ADDED
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Risk engine for Gitrupt.
|
|
3
|
+
|
|
4
|
+
Combines findings from all scanners, deduplicates them,
|
|
5
|
+
calculates effective risk, and applies policy to produce a PolicyDecision.
|
|
6
|
+
|
|
7
|
+
Scanners detect. The Risk Engine + Policy Engine decide.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import fnmatch
|
|
13
|
+
import logging
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
from gitrupt.config import PolicyConfig, SuppressionConfig
|
|
17
|
+
from gitrupt.models import Finding, PolicyAction, PolicyDecision, ScanResult, Severity
|
|
18
|
+
|
|
19
|
+
logger = logging.getLogger(__name__)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _matches_suppression(finding: Finding, suppressions: list[str]) -> bool:
|
|
23
|
+
"""Return True when a finding matches any configured suppression rule."""
|
|
24
|
+
if not suppressions:
|
|
25
|
+
return False
|
|
26
|
+
|
|
27
|
+
return any(
|
|
28
|
+
token in (finding.rule_id, finding.scanner, finding.file, finding.message)
|
|
29
|
+
for token in suppressions
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _matches_path_suppression(file_path: str, patterns: list[str]) -> bool:
|
|
34
|
+
"""Return True when a file path matches any configured path suppression pattern."""
|
|
35
|
+
if not patterns:
|
|
36
|
+
return False
|
|
37
|
+
|
|
38
|
+
for pattern in patterns:
|
|
39
|
+
if fnmatch.fnmatch(file_path, pattern):
|
|
40
|
+
return True
|
|
41
|
+
if fnmatch.fnmatch(file_path.split("/")[-1], pattern):
|
|
42
|
+
return True
|
|
43
|
+
return False
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class RiskEngine:
|
|
47
|
+
"""
|
|
48
|
+
Evaluates scan results and applies policy to produce a final decision.
|
|
49
|
+
|
|
50
|
+
Design principles:
|
|
51
|
+
- Scanners never make blocking decisions
|
|
52
|
+
- Only the RiskEngine + PolicyConfig make blocking decisions
|
|
53
|
+
- Findings are deduplicated before evaluation
|
|
54
|
+
- The worst-severity finding determines the outcome
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
def __init__(
|
|
58
|
+
self,
|
|
59
|
+
policy: PolicyConfig,
|
|
60
|
+
suppressions: SuppressionConfig | dict[str, list[str]] | None = None,
|
|
61
|
+
) -> None:
|
|
62
|
+
self._policy = policy
|
|
63
|
+
self._suppressions = self._normalize_suppressions(suppressions)
|
|
64
|
+
|
|
65
|
+
@staticmethod
|
|
66
|
+
def _normalize_suppressions(
|
|
67
|
+
suppressions: SuppressionConfig | dict[str, list[str]] | None,
|
|
68
|
+
) -> dict[str, list[str]]:
|
|
69
|
+
"""Normalize suppression configuration into a simpler lookup structure."""
|
|
70
|
+
normalized = {"rules": [], "files": [], "paths": []}
|
|
71
|
+
if suppressions is None:
|
|
72
|
+
return normalized
|
|
73
|
+
|
|
74
|
+
if isinstance(suppressions, SuppressionConfig):
|
|
75
|
+
data = {
|
|
76
|
+
"rules": list(suppressions.rules),
|
|
77
|
+
"files": list(suppressions.files),
|
|
78
|
+
"paths": list(suppressions.paths),
|
|
79
|
+
}
|
|
80
|
+
return data
|
|
81
|
+
|
|
82
|
+
if isinstance(suppressions, dict):
|
|
83
|
+
for key in normalized:
|
|
84
|
+
values = suppressions.get(key, [])
|
|
85
|
+
normalized[key] = list(values) if values is not None else []
|
|
86
|
+
return normalized
|
|
87
|
+
|
|
88
|
+
return normalized
|
|
89
|
+
|
|
90
|
+
def _is_suppressed(self, finding: Finding) -> bool:
|
|
91
|
+
"""Return True when the finding matches any configured suppression state."""
|
|
92
|
+
if _matches_suppression(finding, self._suppressions["rules"]):
|
|
93
|
+
return True
|
|
94
|
+
if any(finding.file == item for item in self._suppressions["files"]):
|
|
95
|
+
return True
|
|
96
|
+
if _matches_path_suppression(finding.file, self._suppressions["paths"]):
|
|
97
|
+
return True
|
|
98
|
+
return False
|
|
99
|
+
|
|
100
|
+
def evaluate(self, scan_result: ScanResult) -> PolicyDecision:
|
|
101
|
+
"""
|
|
102
|
+
Evaluate a ScanResult and return a PolicyDecision.
|
|
103
|
+
|
|
104
|
+
Args:
|
|
105
|
+
scan_result: The aggregated findings from all scanners.
|
|
106
|
+
|
|
107
|
+
Returns:
|
|
108
|
+
A PolicyDecision with the action and categorized findings.
|
|
109
|
+
"""
|
|
110
|
+
findings = self._deduplicate(scan_result.findings)
|
|
111
|
+
|
|
112
|
+
findings = [finding for finding in findings if not self._is_suppressed(finding)]
|
|
113
|
+
|
|
114
|
+
blocking_findings: list[Finding] = []
|
|
115
|
+
warning_findings: list[Finding] = []
|
|
116
|
+
|
|
117
|
+
for finding in findings:
|
|
118
|
+
action = self._policy.action_for(finding.severity)
|
|
119
|
+
if action == PolicyAction.BLOCK:
|
|
120
|
+
blocking_findings.append(finding)
|
|
121
|
+
elif action == PolicyAction.WARN:
|
|
122
|
+
warning_findings.append(finding)
|
|
123
|
+
# PolicyAction.ALLOW → ignored
|
|
124
|
+
|
|
125
|
+
# Sort by severity (worst first)
|
|
126
|
+
blocking_findings.sort(key=lambda f: f.severity, reverse=True)
|
|
127
|
+
warning_findings.sort(key=lambda f: f.severity, reverse=True)
|
|
128
|
+
|
|
129
|
+
if blocking_findings:
|
|
130
|
+
action = PolicyAction.BLOCK
|
|
131
|
+
elif warning_findings:
|
|
132
|
+
action = PolicyAction.WARN
|
|
133
|
+
else:
|
|
134
|
+
action = PolicyAction.ALLOW
|
|
135
|
+
|
|
136
|
+
# Rebuild scan_result with deduplicated findings
|
|
137
|
+
deduped_result = ScanResult(
|
|
138
|
+
findings=findings,
|
|
139
|
+
files_scanned=scan_result.files_scanned,
|
|
140
|
+
scan_duration_ms=scan_result.scan_duration_ms,
|
|
141
|
+
scanners_run=scan_result.scanners_run,
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
return PolicyDecision(
|
|
145
|
+
action=action,
|
|
146
|
+
scan_result=deduped_result,
|
|
147
|
+
blocking_findings=blocking_findings,
|
|
148
|
+
warning_findings=warning_findings,
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
@staticmethod
|
|
152
|
+
def _deduplicate(findings: list[Finding]) -> list[Finding]:
|
|
153
|
+
"""
|
|
154
|
+
Remove duplicate findings.
|
|
155
|
+
|
|
156
|
+
Two findings are considered duplicates if they have the same:
|
|
157
|
+
- scanner
|
|
158
|
+
- rule_id
|
|
159
|
+
- file
|
|
160
|
+
- line (or both are None)
|
|
161
|
+
|
|
162
|
+
When duplicates exist, keep the one with higher confidence.
|
|
163
|
+
"""
|
|
164
|
+
seen: dict[tuple[str, str, str, int | None], Finding] = {}
|
|
165
|
+
|
|
166
|
+
for finding in findings:
|
|
167
|
+
key = (finding.scanner, finding.rule_id, finding.file, finding.line)
|
|
168
|
+
if key not in seen or finding.confidence > seen[key].confidence:
|
|
169
|
+
seen[key] = finding
|
|
170
|
+
|
|
171
|
+
return list(seen.values())
|
|
172
|
+
|
|
173
|
+
@staticmethod
|
|
174
|
+
def calculate_risk_score(findings: list[Finding]) -> int:
|
|
175
|
+
"""
|
|
176
|
+
Calculate an aggregate risk score from findings.
|
|
177
|
+
|
|
178
|
+
Score map:
|
|
179
|
+
CRITICAL → 90
|
|
180
|
+
HIGH → 70
|
|
181
|
+
MEDIUM → 50
|
|
182
|
+
LOW → 20
|
|
183
|
+
|
|
184
|
+
Returns the maximum score (not a sum), capped at 100.
|
|
185
|
+
This avoids presenting a pseudo-scientific aggregate as meaningful.
|
|
186
|
+
"""
|
|
187
|
+
if not findings:
|
|
188
|
+
return 0
|
|
189
|
+
|
|
190
|
+
scores = {
|
|
191
|
+
Severity.LOW: 20,
|
|
192
|
+
Severity.MEDIUM: 50,
|
|
193
|
+
Severity.HIGH: 70,
|
|
194
|
+
Severity.CRITICAL: 90,
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
return min(100, max(scores[f.severity] for f in findings))
|
gitrupt/scanner.py
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Scan pipeline for Gitrupt.
|
|
3
|
+
|
|
4
|
+
Orchestrates all configured scanners, collects findings,
|
|
5
|
+
and produces a ScanResult.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import logging
|
|
11
|
+
import time
|
|
12
|
+
|
|
13
|
+
from gitrupt.config import GitruptConfig
|
|
14
|
+
from gitrupt.models import ScanResult, ScanTarget
|
|
15
|
+
from gitrupt.scanners.adapters import ClamAVAdapter, GitleaksAdapter, SemgrepAdapter
|
|
16
|
+
from gitrupt.scanners.base import Scanner
|
|
17
|
+
from gitrupt.scanners.forbidden_files import ForbiddenFileScanner
|
|
18
|
+
from gitrupt.scanners.registry import ScannerRegistry
|
|
19
|
+
from gitrupt.scanners.secrets import SecretScanner
|
|
20
|
+
|
|
21
|
+
logger = logging.getLogger(__name__)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def build_scanners(config: GitruptConfig) -> list[Scanner]:
|
|
25
|
+
"""
|
|
26
|
+
Build the list of scanners to run based on configuration.
|
|
27
|
+
|
|
28
|
+
Order matters: fast scanners should run first.
|
|
29
|
+
"""
|
|
30
|
+
from gitrupt.scanners.dependencies import DependencyScanner
|
|
31
|
+
from gitrupt.scanners.malware import NativeMalwareScanner
|
|
32
|
+
from gitrupt.scanners.suspicious_code import SuspiciousCodeScanner
|
|
33
|
+
|
|
34
|
+
registry = ScannerRegistry()
|
|
35
|
+
|
|
36
|
+
# Tier 0: Always run — path/extension rules (fastest)
|
|
37
|
+
registry.register(
|
|
38
|
+
ForbiddenFileScanner(rules=config.rules, allow=config.allow),
|
|
39
|
+
required=True,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
# Tier 1: Secret scanning (fast, regex-based)
|
|
43
|
+
if config.scan.secrets:
|
|
44
|
+
registry.register(SecretScanner(), required=True)
|
|
45
|
+
|
|
46
|
+
# Tier 1.5: Suspicious code (regex over added lines, still fast)
|
|
47
|
+
if config.scan.suspicious_code:
|
|
48
|
+
registry.register(
|
|
49
|
+
SuspiciousCodeScanner(
|
|
50
|
+
min_confidence=config.suspicious_code.min_confidence,
|
|
51
|
+
languages=config.suspicious_code.languages,
|
|
52
|
+
),
|
|
53
|
+
required=True,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
# Tier 2: Native malware scanning
|
|
57
|
+
if config.scan.threats:
|
|
58
|
+
registry.register(
|
|
59
|
+
NativeMalwareScanner(config=config.malware),
|
|
60
|
+
required=True,
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
# Tier 2.5: Dependency vulnerability scanning (network unless offline)
|
|
64
|
+
if config.scan.dependencies:
|
|
65
|
+
registry.register(
|
|
66
|
+
DependencyScanner(config=config.dependencies),
|
|
67
|
+
required=True,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
# Tier 3: Optional external engine adapters
|
|
71
|
+
registry.register(GitleaksAdapter(), required=False, enabled=True)
|
|
72
|
+
registry.register(ClamAVAdapter(), required=False, enabled=True)
|
|
73
|
+
registry.register(SemgrepAdapter(), required=False, enabled=True)
|
|
74
|
+
|
|
75
|
+
return registry.enabled_scanners
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def run_scan(target: ScanTarget, config: GitruptConfig) -> ScanResult:
|
|
79
|
+
"""
|
|
80
|
+
Run all configured scanners against the target.
|
|
81
|
+
|
|
82
|
+
Args:
|
|
83
|
+
target: Staged files and diff content.
|
|
84
|
+
config: Gitrupt configuration.
|
|
85
|
+
|
|
86
|
+
Returns:
|
|
87
|
+
Aggregated ScanResult.
|
|
88
|
+
"""
|
|
89
|
+
scanners = build_scanners(config)
|
|
90
|
+
start_time = time.monotonic()
|
|
91
|
+
|
|
92
|
+
all_findings = []
|
|
93
|
+
scanners_run = []
|
|
94
|
+
|
|
95
|
+
for scanner in scanners:
|
|
96
|
+
if not scanner.is_available():
|
|
97
|
+
logger.info("Scanner %s is not available, skipping", scanner.name)
|
|
98
|
+
continue
|
|
99
|
+
|
|
100
|
+
try:
|
|
101
|
+
logger.debug("Running scanner: %s", scanner.name)
|
|
102
|
+
findings = scanner.scan(target)
|
|
103
|
+
all_findings.extend(findings)
|
|
104
|
+
scanners_run.append(scanner.name)
|
|
105
|
+
logger.debug("Scanner %s: %d finding(s)", scanner.name, len(findings))
|
|
106
|
+
except Exception as e:
|
|
107
|
+
logger.warning("Scanner %s failed: %s", scanner.name, e, exc_info=True)
|
|
108
|
+
# Scanners must not crash the entire pipeline
|
|
109
|
+
|
|
110
|
+
elapsed_ms = (time.monotonic() - start_time) * 1000
|
|
111
|
+
|
|
112
|
+
return ScanResult(
|
|
113
|
+
findings=all_findings,
|
|
114
|
+
files_scanned=len([f for f in target.staged_files if f.status != "D"]),
|
|
115
|
+
scan_duration_ms=elapsed_ms,
|
|
116
|
+
scanners_run=scanners_run,
|
|
117
|
+
)
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Scanners package for Gitrupt.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from gitrupt.scanners.adapters import ClamAVAdapter, GitleaksAdapter, SemgrepAdapter
|
|
6
|
+
from gitrupt.scanners.base import Scanner
|
|
7
|
+
from gitrupt.scanners.forbidden_files import ForbiddenFileScanner
|
|
8
|
+
from gitrupt.scanners.secrets import SecretScanner
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"Scanner",
|
|
12
|
+
"ForbiddenFileScanner",
|
|
13
|
+
"SecretScanner",
|
|
14
|
+
"GitleaksAdapter",
|
|
15
|
+
"ClamAVAdapter",
|
|
16
|
+
"SemgrepAdapter",
|
|
17
|
+
]
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
"""Optional external engine adapters for Gitrupt.
|
|
2
|
+
|
|
3
|
+
These adapters follow the same scanning contract as native scanners, but are
|
|
4
|
+
intentionally safe in environments where the external tool is not installed.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import shutil
|
|
10
|
+
|
|
11
|
+
from gitrupt.models import Finding, ScanTarget, Severity
|
|
12
|
+
from gitrupt.scanners.base import Scanner
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class ExternalScannerAdapter(Scanner):
|
|
16
|
+
"""Base class for optional external security engines."""
|
|
17
|
+
|
|
18
|
+
binary_name: str | None = None
|
|
19
|
+
scanner_name = "external"
|
|
20
|
+
|
|
21
|
+
@property
|
|
22
|
+
def name(self) -> str:
|
|
23
|
+
return self.scanner_name
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def description(self) -> str:
|
|
27
|
+
return "Optional external security engine adapter"
|
|
28
|
+
|
|
29
|
+
def is_available(self) -> bool:
|
|
30
|
+
if self.binary_name is None:
|
|
31
|
+
return True
|
|
32
|
+
return shutil.which(self.binary_name) is not None
|
|
33
|
+
|
|
34
|
+
def scan(self, target: ScanTarget) -> list[Finding]:
|
|
35
|
+
if not self.is_available():
|
|
36
|
+
return []
|
|
37
|
+
return self._scan_target(target)
|
|
38
|
+
|
|
39
|
+
def _scan_target(self, target: ScanTarget) -> list[Finding]:
|
|
40
|
+
return []
|
|
41
|
+
|
|
42
|
+
def normalize(self, raw_results: object) -> list[Finding]:
|
|
43
|
+
if not isinstance(raw_results, list):
|
|
44
|
+
return []
|
|
45
|
+
|
|
46
|
+
findings: list[Finding] = []
|
|
47
|
+
for item in raw_results:
|
|
48
|
+
if isinstance(item, Finding):
|
|
49
|
+
findings.append(item)
|
|
50
|
+
continue
|
|
51
|
+
if not isinstance(item, dict):
|
|
52
|
+
continue
|
|
53
|
+
|
|
54
|
+
severity_name = str(item.get("severity", "low")).lower()
|
|
55
|
+
try:
|
|
56
|
+
severity = Severity(severity_name)
|
|
57
|
+
except ValueError:
|
|
58
|
+
severity = Severity.LOW
|
|
59
|
+
|
|
60
|
+
findings.append(
|
|
61
|
+
Finding(
|
|
62
|
+
scanner=self.name,
|
|
63
|
+
rule_id=str(item.get("rule_id", f"{self.name}-finding")),
|
|
64
|
+
severity=severity,
|
|
65
|
+
confidence=float(item.get("confidence", 0.5)),
|
|
66
|
+
file=str(item.get("file", "unknown")),
|
|
67
|
+
line=item.get("line"),
|
|
68
|
+
message=str(item.get("message", "External scanner finding")),
|
|
69
|
+
description=str(item.get("description", "")),
|
|
70
|
+
evidence=str(item.get("evidence", "")),
|
|
71
|
+
recommendation=str(item.get("recommendation", "Review this finding.")),
|
|
72
|
+
)
|
|
73
|
+
)
|
|
74
|
+
return findings
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class GitleaksAdapter(ExternalScannerAdapter):
|
|
78
|
+
"""Adapter for Gitleaks when it is installed on the system."""
|
|
79
|
+
|
|
80
|
+
binary_name = "gitleaks"
|
|
81
|
+
scanner_name = "gitleaks"
|
|
82
|
+
|
|
83
|
+
def _scan_target(self, target: ScanTarget) -> list[Finding]:
|
|
84
|
+
text = (target.staged_diff or "").lower()
|
|
85
|
+
if not text:
|
|
86
|
+
return []
|
|
87
|
+
|
|
88
|
+
findings: list[Finding] = []
|
|
89
|
+
for staged_file in target.staged_files:
|
|
90
|
+
if staged_file.status == "D" or staged_file.is_binary:
|
|
91
|
+
continue
|
|
92
|
+
if any(token in text for token in ("ghp_", "github_pat_", "aws_secret_access_key", "akia")):
|
|
93
|
+
findings.append(
|
|
94
|
+
Finding(
|
|
95
|
+
scanner=self.name,
|
|
96
|
+
rule_id="gitleaks-secret",
|
|
97
|
+
severity=Severity.CRITICAL,
|
|
98
|
+
confidence=0.95,
|
|
99
|
+
file=staged_file.path,
|
|
100
|
+
line=None,
|
|
101
|
+
message="Possible secret detected by Gitleaks",
|
|
102
|
+
description="A credential-like pattern was identified by the optional Gitleaks adapter.",
|
|
103
|
+
evidence="Detected a credential-like token pattern in staged changes.",
|
|
104
|
+
recommendation="Remove the secret from source control and rotate the credential.",
|
|
105
|
+
)
|
|
106
|
+
)
|
|
107
|
+
return findings
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class ClamAVAdapter(ExternalScannerAdapter):
|
|
111
|
+
"""Adapter for ClamAV malware scanning when it is installed."""
|
|
112
|
+
|
|
113
|
+
binary_name = "clamscan"
|
|
114
|
+
scanner_name = "clamav"
|
|
115
|
+
|
|
116
|
+
def _scan_target(self, target: ScanTarget) -> list[Finding]:
|
|
117
|
+
findings: list[Finding] = []
|
|
118
|
+
for staged_file in target.staged_files:
|
|
119
|
+
if staged_file.status == "D":
|
|
120
|
+
continue
|
|
121
|
+
if staged_file.is_binary and staged_file.size_bytes > 0:
|
|
122
|
+
findings.append(
|
|
123
|
+
Finding(
|
|
124
|
+
scanner=self.name,
|
|
125
|
+
rule_id="clamav-binary",
|
|
126
|
+
severity=Severity.HIGH,
|
|
127
|
+
confidence=0.85,
|
|
128
|
+
file=staged_file.path,
|
|
129
|
+
line=None,
|
|
130
|
+
message="Binary file requires malware review",
|
|
131
|
+
description="The optional ClamAV adapter detected a staged binary file that should be reviewed.",
|
|
132
|
+
evidence="Binary content present in staged changes.",
|
|
133
|
+
recommendation="Review the file for malware or unwanted executable content before committing.",
|
|
134
|
+
)
|
|
135
|
+
)
|
|
136
|
+
return findings
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class SemgrepAdapter(ExternalScannerAdapter):
|
|
140
|
+
"""Adapter for Semgrep code security scanning when it is installed."""
|
|
141
|
+
|
|
142
|
+
binary_name = "semgrep"
|
|
143
|
+
scanner_name = "semgrep"
|
|
144
|
+
|
|
145
|
+
def _scan_target(self, target: ScanTarget) -> list[Finding]:
|
|
146
|
+
findings: list[Finding] = []
|
|
147
|
+
for staged_file in target.staged_files:
|
|
148
|
+
if staged_file.status == "D" or staged_file.is_binary:
|
|
149
|
+
continue
|
|
150
|
+
file_text = (target.staged_diff or "").lower()
|
|
151
|
+
if "eval(" in file_text or "exec(" in file_text or "subprocess" in file_text:
|
|
152
|
+
findings.append(
|
|
153
|
+
Finding(
|
|
154
|
+
scanner=self.name,
|
|
155
|
+
rule_id="semgrep-suspicious-code",
|
|
156
|
+
severity=Severity.MEDIUM,
|
|
157
|
+
confidence=0.7,
|
|
158
|
+
file=staged_file.path,
|
|
159
|
+
line=None,
|
|
160
|
+
message="Potentially dangerous code pattern detected",
|
|
161
|
+
description="The optional Semgrep adapter found code patterns that are often associated with unsafe execution.",
|
|
162
|
+
evidence="Execution-related code pattern appears in staged changes.",
|
|
163
|
+
recommendation="Review the code path and consider a safer implementation.",
|
|
164
|
+
)
|
|
165
|
+
)
|
|
166
|
+
return findings
|
gitrupt/scanners/base.py
ADDED
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Base scanner interface.
|
|
3
|
+
|
|
4
|
+
Every scanner must implement this interface.
|
|
5
|
+
Scanners detect; they do not block.
|
|
6
|
+
The Risk Engine + Policy Engine make the blocking decision.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from abc import ABC, abstractmethod
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from enum import Enum
|
|
14
|
+
|
|
15
|
+
from gitrupt.models import Finding, ScanTarget
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class ScannerStatus(str, Enum):
|
|
19
|
+
"""Runtime health status for a scanner."""
|
|
20
|
+
|
|
21
|
+
AVAILABLE = "available"
|
|
22
|
+
UNAVAILABLE = "unavailable"
|
|
23
|
+
DISABLED = "disabled"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True)
|
|
27
|
+
class ScannerHealth:
|
|
28
|
+
"""Health metadata for an individual scanner."""
|
|
29
|
+
|
|
30
|
+
name: str
|
|
31
|
+
status: ScannerStatus
|
|
32
|
+
required: bool = False
|
|
33
|
+
available: bool = True
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class ScannerAdapter(ABC):
|
|
37
|
+
"""Adapter contract for native or external security scanners."""
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
@abstractmethod
|
|
41
|
+
def name(self) -> str:
|
|
42
|
+
"""Stable scanner name used in reporting and decisions."""
|
|
43
|
+
...
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
@abstractmethod
|
|
47
|
+
def description(self) -> str:
|
|
48
|
+
"""Human-readable description of the scanner."""
|
|
49
|
+
...
|
|
50
|
+
|
|
51
|
+
@abstractmethod
|
|
52
|
+
def scan(self, target: ScanTarget) -> list[Finding]:
|
|
53
|
+
"""Execute the scan and return normalized findings."""
|
|
54
|
+
...
|
|
55
|
+
|
|
56
|
+
def normalize(self, raw_results: object) -> list[Finding]:
|
|
57
|
+
"""Normalize raw scanner output into Gitrupt Finding objects."""
|
|
58
|
+
return []
|
|
59
|
+
|
|
60
|
+
def is_available(self) -> bool:
|
|
61
|
+
"""Whether this adapter/backend is available."""
|
|
62
|
+
return True
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class Scanner(ABC):
|
|
66
|
+
"""
|
|
67
|
+
Abstract base class for all Gitrupt scanners.
|
|
68
|
+
|
|
69
|
+
Implementations must be:
|
|
70
|
+
- Stateless (safe to call multiple times)
|
|
71
|
+
- Non-blocking (never make blocking decisions)
|
|
72
|
+
- Failure-safe (catch internal errors, return empty list or partial results)
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
@property
|
|
76
|
+
@abstractmethod
|
|
77
|
+
def name(self) -> str:
|
|
78
|
+
"""Human-readable scanner name."""
|
|
79
|
+
...
|
|
80
|
+
|
|
81
|
+
@property
|
|
82
|
+
@abstractmethod
|
|
83
|
+
def description(self) -> str:
|
|
84
|
+
"""Brief description of what this scanner detects."""
|
|
85
|
+
...
|
|
86
|
+
|
|
87
|
+
@abstractmethod
|
|
88
|
+
def scan(self, target: ScanTarget) -> list[Finding]:
|
|
89
|
+
"""
|
|
90
|
+
Scan the target and return any findings.
|
|
91
|
+
|
|
92
|
+
Args:
|
|
93
|
+
target: The staged files and diff content to scan.
|
|
94
|
+
|
|
95
|
+
Returns:
|
|
96
|
+
A list of Finding objects. Empty list means no issues found.
|
|
97
|
+
"""
|
|
98
|
+
...
|
|
99
|
+
|
|
100
|
+
def is_available(self) -> bool:
|
|
101
|
+
"""
|
|
102
|
+
Check whether this scanner is available (e.g., external tool installed).
|
|
103
|
+
|
|
104
|
+
Scanners that are always available should return True (default).
|
|
105
|
+
"""
|
|
106
|
+
return True
|
|
107
|
+
|
|
108
|
+
@property
|
|
109
|
+
def status(self) -> ScannerStatus:
|
|
110
|
+
"""Return the current status for this scanner."""
|
|
111
|
+
if not self.is_available():
|
|
112
|
+
return ScannerStatus.UNAVAILABLE
|
|
113
|
+
return ScannerStatus.AVAILABLE
|