gitrupt 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. gitrupt/__init__.py +13 -0
  2. gitrupt/cli.py +546 -0
  3. gitrupt/config.py +269 -0
  4. gitrupt/git.py +590 -0
  5. gitrupt/hooks/__init__.py +7 -0
  6. gitrupt/hooks/install.py +255 -0
  7. gitrupt/hooks/pre_commit.py +103 -0
  8. gitrupt/hooks/pre_push.py +178 -0
  9. gitrupt/models.py +190 -0
  10. gitrupt/policy.py +36 -0
  11. gitrupt/reporting.py +316 -0
  12. gitrupt/risk.py +197 -0
  13. gitrupt/scanner.py +117 -0
  14. gitrupt/scanners/__init__.py +17 -0
  15. gitrupt/scanners/adapters.py +166 -0
  16. gitrupt/scanners/base.py +113 -0
  17. gitrupt/scanners/binaries.py +185 -0
  18. gitrupt/scanners/code_rules/__init__.py +36 -0
  19. gitrupt/scanners/code_rules/base.py +27 -0
  20. gitrupt/scanners/code_rules/go.py +65 -0
  21. gitrupt/scanners/code_rules/javascript.py +106 -0
  22. gitrupt/scanners/code_rules/php.py +71 -0
  23. gitrupt/scanners/code_rules/powershell.py +85 -0
  24. gitrupt/scanners/code_rules/python.py +153 -0
  25. gitrupt/scanners/code_rules/ruby.py +76 -0
  26. gitrupt/scanners/code_rules/rust.py +41 -0
  27. gitrupt/scanners/code_rules/shell.py +112 -0
  28. gitrupt/scanners/dependencies.py +244 -0
  29. gitrupt/scanners/ecosystems/__init__.py +30 -0
  30. gitrupt/scanners/ecosystems/base.py +60 -0
  31. gitrupt/scanners/ecosystems/node.py +128 -0
  32. gitrupt/scanners/ecosystems/python.py +157 -0
  33. gitrupt/scanners/entropy.py +123 -0
  34. gitrupt/scanners/forbidden_files.py +201 -0
  35. gitrupt/scanners/malware.py +219 -0
  36. gitrupt/scanners/osv_client.py +221 -0
  37. gitrupt/scanners/registry.py +66 -0
  38. gitrupt/scanners/secret_rules.py +368 -0
  39. gitrupt/scanners/secrets.py +558 -0
  40. gitrupt/scanners/suspicious_code.py +208 -0
  41. gitrupt/scanners/yara_loader.py +65 -0
  42. gitrupt/scanners/yara_rules_builtin.py +141 -0
  43. gitrupt-0.1.0.dist-info/METADATA +342 -0
  44. gitrupt-0.1.0.dist-info/RECORD +48 -0
  45. gitrupt-0.1.0.dist-info/WHEEL +5 -0
  46. gitrupt-0.1.0.dist-info/entry_points.txt +2 -0
  47. gitrupt-0.1.0.dist-info/licenses/LICENSE +23 -0
  48. gitrupt-0.1.0.dist-info/top_level.txt +1 -0
gitrupt/risk.py ADDED
@@ -0,0 +1,197 @@
1
+ """
2
+ Risk engine for Gitrupt.
3
+
4
+ Combines findings from all scanners, deduplicates them,
5
+ calculates effective risk, and applies policy to produce a PolicyDecision.
6
+
7
+ Scanners detect. The Risk Engine + Policy Engine decide.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import fnmatch
13
+ import logging
14
+ from typing import Any
15
+
16
+ from gitrupt.config import PolicyConfig, SuppressionConfig
17
+ from gitrupt.models import Finding, PolicyAction, PolicyDecision, ScanResult, Severity
18
+
19
+ logger = logging.getLogger(__name__)
20
+
21
+
22
+ def _matches_suppression(finding: Finding, suppressions: list[str]) -> bool:
23
+ """Return True when a finding matches any configured suppression rule."""
24
+ if not suppressions:
25
+ return False
26
+
27
+ return any(
28
+ token in (finding.rule_id, finding.scanner, finding.file, finding.message)
29
+ for token in suppressions
30
+ )
31
+
32
+
33
+ def _matches_path_suppression(file_path: str, patterns: list[str]) -> bool:
34
+ """Return True when a file path matches any configured path suppression pattern."""
35
+ if not patterns:
36
+ return False
37
+
38
+ for pattern in patterns:
39
+ if fnmatch.fnmatch(file_path, pattern):
40
+ return True
41
+ if fnmatch.fnmatch(file_path.split("/")[-1], pattern):
42
+ return True
43
+ return False
44
+
45
+
46
+ class RiskEngine:
47
+ """
48
+ Evaluates scan results and applies policy to produce a final decision.
49
+
50
+ Design principles:
51
+ - Scanners never make blocking decisions
52
+ - Only the RiskEngine + PolicyConfig make blocking decisions
53
+ - Findings are deduplicated before evaluation
54
+ - The worst-severity finding determines the outcome
55
+ """
56
+
57
+ def __init__(
58
+ self,
59
+ policy: PolicyConfig,
60
+ suppressions: SuppressionConfig | dict[str, list[str]] | None = None,
61
+ ) -> None:
62
+ self._policy = policy
63
+ self._suppressions = self._normalize_suppressions(suppressions)
64
+
65
+ @staticmethod
66
+ def _normalize_suppressions(
67
+ suppressions: SuppressionConfig | dict[str, list[str]] | None,
68
+ ) -> dict[str, list[str]]:
69
+ """Normalize suppression configuration into a simpler lookup structure."""
70
+ normalized = {"rules": [], "files": [], "paths": []}
71
+ if suppressions is None:
72
+ return normalized
73
+
74
+ if isinstance(suppressions, SuppressionConfig):
75
+ data = {
76
+ "rules": list(suppressions.rules),
77
+ "files": list(suppressions.files),
78
+ "paths": list(suppressions.paths),
79
+ }
80
+ return data
81
+
82
+ if isinstance(suppressions, dict):
83
+ for key in normalized:
84
+ values = suppressions.get(key, [])
85
+ normalized[key] = list(values) if values is not None else []
86
+ return normalized
87
+
88
+ return normalized
89
+
90
+ def _is_suppressed(self, finding: Finding) -> bool:
91
+ """Return True when the finding matches any configured suppression state."""
92
+ if _matches_suppression(finding, self._suppressions["rules"]):
93
+ return True
94
+ if any(finding.file == item for item in self._suppressions["files"]):
95
+ return True
96
+ if _matches_path_suppression(finding.file, self._suppressions["paths"]):
97
+ return True
98
+ return False
99
+
100
+ def evaluate(self, scan_result: ScanResult) -> PolicyDecision:
101
+ """
102
+ Evaluate a ScanResult and return a PolicyDecision.
103
+
104
+ Args:
105
+ scan_result: The aggregated findings from all scanners.
106
+
107
+ Returns:
108
+ A PolicyDecision with the action and categorized findings.
109
+ """
110
+ findings = self._deduplicate(scan_result.findings)
111
+
112
+ findings = [finding for finding in findings if not self._is_suppressed(finding)]
113
+
114
+ blocking_findings: list[Finding] = []
115
+ warning_findings: list[Finding] = []
116
+
117
+ for finding in findings:
118
+ action = self._policy.action_for(finding.severity)
119
+ if action == PolicyAction.BLOCK:
120
+ blocking_findings.append(finding)
121
+ elif action == PolicyAction.WARN:
122
+ warning_findings.append(finding)
123
+ # PolicyAction.ALLOW → ignored
124
+
125
+ # Sort by severity (worst first)
126
+ blocking_findings.sort(key=lambda f: f.severity, reverse=True)
127
+ warning_findings.sort(key=lambda f: f.severity, reverse=True)
128
+
129
+ if blocking_findings:
130
+ action = PolicyAction.BLOCK
131
+ elif warning_findings:
132
+ action = PolicyAction.WARN
133
+ else:
134
+ action = PolicyAction.ALLOW
135
+
136
+ # Rebuild scan_result with deduplicated findings
137
+ deduped_result = ScanResult(
138
+ findings=findings,
139
+ files_scanned=scan_result.files_scanned,
140
+ scan_duration_ms=scan_result.scan_duration_ms,
141
+ scanners_run=scan_result.scanners_run,
142
+ )
143
+
144
+ return PolicyDecision(
145
+ action=action,
146
+ scan_result=deduped_result,
147
+ blocking_findings=blocking_findings,
148
+ warning_findings=warning_findings,
149
+ )
150
+
151
+ @staticmethod
152
+ def _deduplicate(findings: list[Finding]) -> list[Finding]:
153
+ """
154
+ Remove duplicate findings.
155
+
156
+ Two findings are considered duplicates if they have the same:
157
+ - scanner
158
+ - rule_id
159
+ - file
160
+ - line (or both are None)
161
+
162
+ When duplicates exist, keep the one with higher confidence.
163
+ """
164
+ seen: dict[tuple[str, str, str, int | None], Finding] = {}
165
+
166
+ for finding in findings:
167
+ key = (finding.scanner, finding.rule_id, finding.file, finding.line)
168
+ if key not in seen or finding.confidence > seen[key].confidence:
169
+ seen[key] = finding
170
+
171
+ return list(seen.values())
172
+
173
+ @staticmethod
174
+ def calculate_risk_score(findings: list[Finding]) -> int:
175
+ """
176
+ Calculate an aggregate risk score from findings.
177
+
178
+ Score map:
179
+ CRITICAL → 90
180
+ HIGH → 70
181
+ MEDIUM → 50
182
+ LOW → 20
183
+
184
+ Returns the maximum score (not a sum), capped at 100.
185
+ This avoids presenting a pseudo-scientific aggregate as meaningful.
186
+ """
187
+ if not findings:
188
+ return 0
189
+
190
+ scores = {
191
+ Severity.LOW: 20,
192
+ Severity.MEDIUM: 50,
193
+ Severity.HIGH: 70,
194
+ Severity.CRITICAL: 90,
195
+ }
196
+
197
+ return min(100, max(scores[f.severity] for f in findings))
gitrupt/scanner.py ADDED
@@ -0,0 +1,117 @@
1
+ """
2
+ Scan pipeline for Gitrupt.
3
+
4
+ Orchestrates all configured scanners, collects findings,
5
+ and produces a ScanResult.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import logging
11
+ import time
12
+
13
+ from gitrupt.config import GitruptConfig
14
+ from gitrupt.models import ScanResult, ScanTarget
15
+ from gitrupt.scanners.adapters import ClamAVAdapter, GitleaksAdapter, SemgrepAdapter
16
+ from gitrupt.scanners.base import Scanner
17
+ from gitrupt.scanners.forbidden_files import ForbiddenFileScanner
18
+ from gitrupt.scanners.registry import ScannerRegistry
19
+ from gitrupt.scanners.secrets import SecretScanner
20
+
21
+ logger = logging.getLogger(__name__)
22
+
23
+
24
+ def build_scanners(config: GitruptConfig) -> list[Scanner]:
25
+ """
26
+ Build the list of scanners to run based on configuration.
27
+
28
+ Order matters: fast scanners should run first.
29
+ """
30
+ from gitrupt.scanners.dependencies import DependencyScanner
31
+ from gitrupt.scanners.malware import NativeMalwareScanner
32
+ from gitrupt.scanners.suspicious_code import SuspiciousCodeScanner
33
+
34
+ registry = ScannerRegistry()
35
+
36
+ # Tier 0: Always run — path/extension rules (fastest)
37
+ registry.register(
38
+ ForbiddenFileScanner(rules=config.rules, allow=config.allow),
39
+ required=True,
40
+ )
41
+
42
+ # Tier 1: Secret scanning (fast, regex-based)
43
+ if config.scan.secrets:
44
+ registry.register(SecretScanner(), required=True)
45
+
46
+ # Tier 1.5: Suspicious code (regex over added lines, still fast)
47
+ if config.scan.suspicious_code:
48
+ registry.register(
49
+ SuspiciousCodeScanner(
50
+ min_confidence=config.suspicious_code.min_confidence,
51
+ languages=config.suspicious_code.languages,
52
+ ),
53
+ required=True,
54
+ )
55
+
56
+ # Tier 2: Native malware scanning
57
+ if config.scan.threats:
58
+ registry.register(
59
+ NativeMalwareScanner(config=config.malware),
60
+ required=True,
61
+ )
62
+
63
+ # Tier 2.5: Dependency vulnerability scanning (network unless offline)
64
+ if config.scan.dependencies:
65
+ registry.register(
66
+ DependencyScanner(config=config.dependencies),
67
+ required=True,
68
+ )
69
+
70
+ # Tier 3: Optional external engine adapters
71
+ registry.register(GitleaksAdapter(), required=False, enabled=True)
72
+ registry.register(ClamAVAdapter(), required=False, enabled=True)
73
+ registry.register(SemgrepAdapter(), required=False, enabled=True)
74
+
75
+ return registry.enabled_scanners
76
+
77
+
78
+ def run_scan(target: ScanTarget, config: GitruptConfig) -> ScanResult:
79
+ """
80
+ Run all configured scanners against the target.
81
+
82
+ Args:
83
+ target: Staged files and diff content.
84
+ config: Gitrupt configuration.
85
+
86
+ Returns:
87
+ Aggregated ScanResult.
88
+ """
89
+ scanners = build_scanners(config)
90
+ start_time = time.monotonic()
91
+
92
+ all_findings = []
93
+ scanners_run = []
94
+
95
+ for scanner in scanners:
96
+ if not scanner.is_available():
97
+ logger.info("Scanner %s is not available, skipping", scanner.name)
98
+ continue
99
+
100
+ try:
101
+ logger.debug("Running scanner: %s", scanner.name)
102
+ findings = scanner.scan(target)
103
+ all_findings.extend(findings)
104
+ scanners_run.append(scanner.name)
105
+ logger.debug("Scanner %s: %d finding(s)", scanner.name, len(findings))
106
+ except Exception as e:
107
+ logger.warning("Scanner %s failed: %s", scanner.name, e, exc_info=True)
108
+ # Scanners must not crash the entire pipeline
109
+
110
+ elapsed_ms = (time.monotonic() - start_time) * 1000
111
+
112
+ return ScanResult(
113
+ findings=all_findings,
114
+ files_scanned=len([f for f in target.staged_files if f.status != "D"]),
115
+ scan_duration_ms=elapsed_ms,
116
+ scanners_run=scanners_run,
117
+ )
@@ -0,0 +1,17 @@
1
+ """
2
+ Scanners package for Gitrupt.
3
+ """
4
+
5
+ from gitrupt.scanners.adapters import ClamAVAdapter, GitleaksAdapter, SemgrepAdapter
6
+ from gitrupt.scanners.base import Scanner
7
+ from gitrupt.scanners.forbidden_files import ForbiddenFileScanner
8
+ from gitrupt.scanners.secrets import SecretScanner
9
+
10
+ __all__ = [
11
+ "Scanner",
12
+ "ForbiddenFileScanner",
13
+ "SecretScanner",
14
+ "GitleaksAdapter",
15
+ "ClamAVAdapter",
16
+ "SemgrepAdapter",
17
+ ]
@@ -0,0 +1,166 @@
1
+ """Optional external engine adapters for Gitrupt.
2
+
3
+ These adapters follow the same scanning contract as native scanners, but are
4
+ intentionally safe in environments where the external tool is not installed.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import shutil
10
+
11
+ from gitrupt.models import Finding, ScanTarget, Severity
12
+ from gitrupt.scanners.base import Scanner
13
+
14
+
15
+ class ExternalScannerAdapter(Scanner):
16
+ """Base class for optional external security engines."""
17
+
18
+ binary_name: str | None = None
19
+ scanner_name = "external"
20
+
21
+ @property
22
+ def name(self) -> str:
23
+ return self.scanner_name
24
+
25
+ @property
26
+ def description(self) -> str:
27
+ return "Optional external security engine adapter"
28
+
29
+ def is_available(self) -> bool:
30
+ if self.binary_name is None:
31
+ return True
32
+ return shutil.which(self.binary_name) is not None
33
+
34
+ def scan(self, target: ScanTarget) -> list[Finding]:
35
+ if not self.is_available():
36
+ return []
37
+ return self._scan_target(target)
38
+
39
+ def _scan_target(self, target: ScanTarget) -> list[Finding]:
40
+ return []
41
+
42
+ def normalize(self, raw_results: object) -> list[Finding]:
43
+ if not isinstance(raw_results, list):
44
+ return []
45
+
46
+ findings: list[Finding] = []
47
+ for item in raw_results:
48
+ if isinstance(item, Finding):
49
+ findings.append(item)
50
+ continue
51
+ if not isinstance(item, dict):
52
+ continue
53
+
54
+ severity_name = str(item.get("severity", "low")).lower()
55
+ try:
56
+ severity = Severity(severity_name)
57
+ except ValueError:
58
+ severity = Severity.LOW
59
+
60
+ findings.append(
61
+ Finding(
62
+ scanner=self.name,
63
+ rule_id=str(item.get("rule_id", f"{self.name}-finding")),
64
+ severity=severity,
65
+ confidence=float(item.get("confidence", 0.5)),
66
+ file=str(item.get("file", "unknown")),
67
+ line=item.get("line"),
68
+ message=str(item.get("message", "External scanner finding")),
69
+ description=str(item.get("description", "")),
70
+ evidence=str(item.get("evidence", "")),
71
+ recommendation=str(item.get("recommendation", "Review this finding.")),
72
+ )
73
+ )
74
+ return findings
75
+
76
+
77
+ class GitleaksAdapter(ExternalScannerAdapter):
78
+ """Adapter for Gitleaks when it is installed on the system."""
79
+
80
+ binary_name = "gitleaks"
81
+ scanner_name = "gitleaks"
82
+
83
+ def _scan_target(self, target: ScanTarget) -> list[Finding]:
84
+ text = (target.staged_diff or "").lower()
85
+ if not text:
86
+ return []
87
+
88
+ findings: list[Finding] = []
89
+ for staged_file in target.staged_files:
90
+ if staged_file.status == "D" or staged_file.is_binary:
91
+ continue
92
+ if any(token in text for token in ("ghp_", "github_pat_", "aws_secret_access_key", "akia")):
93
+ findings.append(
94
+ Finding(
95
+ scanner=self.name,
96
+ rule_id="gitleaks-secret",
97
+ severity=Severity.CRITICAL,
98
+ confidence=0.95,
99
+ file=staged_file.path,
100
+ line=None,
101
+ message="Possible secret detected by Gitleaks",
102
+ description="A credential-like pattern was identified by the optional Gitleaks adapter.",
103
+ evidence="Detected a credential-like token pattern in staged changes.",
104
+ recommendation="Remove the secret from source control and rotate the credential.",
105
+ )
106
+ )
107
+ return findings
108
+
109
+
110
+ class ClamAVAdapter(ExternalScannerAdapter):
111
+ """Adapter for ClamAV malware scanning when it is installed."""
112
+
113
+ binary_name = "clamscan"
114
+ scanner_name = "clamav"
115
+
116
+ def _scan_target(self, target: ScanTarget) -> list[Finding]:
117
+ findings: list[Finding] = []
118
+ for staged_file in target.staged_files:
119
+ if staged_file.status == "D":
120
+ continue
121
+ if staged_file.is_binary and staged_file.size_bytes > 0:
122
+ findings.append(
123
+ Finding(
124
+ scanner=self.name,
125
+ rule_id="clamav-binary",
126
+ severity=Severity.HIGH,
127
+ confidence=0.85,
128
+ file=staged_file.path,
129
+ line=None,
130
+ message="Binary file requires malware review",
131
+ description="The optional ClamAV adapter detected a staged binary file that should be reviewed.",
132
+ evidence="Binary content present in staged changes.",
133
+ recommendation="Review the file for malware or unwanted executable content before committing.",
134
+ )
135
+ )
136
+ return findings
137
+
138
+
139
+ class SemgrepAdapter(ExternalScannerAdapter):
140
+ """Adapter for Semgrep code security scanning when it is installed."""
141
+
142
+ binary_name = "semgrep"
143
+ scanner_name = "semgrep"
144
+
145
+ def _scan_target(self, target: ScanTarget) -> list[Finding]:
146
+ findings: list[Finding] = []
147
+ for staged_file in target.staged_files:
148
+ if staged_file.status == "D" or staged_file.is_binary:
149
+ continue
150
+ file_text = (target.staged_diff or "").lower()
151
+ if "eval(" in file_text or "exec(" in file_text or "subprocess" in file_text:
152
+ findings.append(
153
+ Finding(
154
+ scanner=self.name,
155
+ rule_id="semgrep-suspicious-code",
156
+ severity=Severity.MEDIUM,
157
+ confidence=0.7,
158
+ file=staged_file.path,
159
+ line=None,
160
+ message="Potentially dangerous code pattern detected",
161
+ description="The optional Semgrep adapter found code patterns that are often associated with unsafe execution.",
162
+ evidence="Execution-related code pattern appears in staged changes.",
163
+ recommendation="Review the code path and consider a safer implementation.",
164
+ )
165
+ )
166
+ return findings
@@ -0,0 +1,113 @@
1
+ """
2
+ Base scanner interface.
3
+
4
+ Every scanner must implement this interface.
5
+ Scanners detect; they do not block.
6
+ The Risk Engine + Policy Engine make the blocking decision.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from abc import ABC, abstractmethod
12
+ from dataclasses import dataclass
13
+ from enum import Enum
14
+
15
+ from gitrupt.models import Finding, ScanTarget
16
+
17
+
18
+ class ScannerStatus(str, Enum):
19
+ """Runtime health status for a scanner."""
20
+
21
+ AVAILABLE = "available"
22
+ UNAVAILABLE = "unavailable"
23
+ DISABLED = "disabled"
24
+
25
+
26
+ @dataclass(frozen=True)
27
+ class ScannerHealth:
28
+ """Health metadata for an individual scanner."""
29
+
30
+ name: str
31
+ status: ScannerStatus
32
+ required: bool = False
33
+ available: bool = True
34
+
35
+
36
+ class ScannerAdapter(ABC):
37
+ """Adapter contract for native or external security scanners."""
38
+
39
+ @property
40
+ @abstractmethod
41
+ def name(self) -> str:
42
+ """Stable scanner name used in reporting and decisions."""
43
+ ...
44
+
45
+ @property
46
+ @abstractmethod
47
+ def description(self) -> str:
48
+ """Human-readable description of the scanner."""
49
+ ...
50
+
51
+ @abstractmethod
52
+ def scan(self, target: ScanTarget) -> list[Finding]:
53
+ """Execute the scan and return normalized findings."""
54
+ ...
55
+
56
+ def normalize(self, raw_results: object) -> list[Finding]:
57
+ """Normalize raw scanner output into Gitrupt Finding objects."""
58
+ return []
59
+
60
+ def is_available(self) -> bool:
61
+ """Whether this adapter/backend is available."""
62
+ return True
63
+
64
+
65
+ class Scanner(ABC):
66
+ """
67
+ Abstract base class for all Gitrupt scanners.
68
+
69
+ Implementations must be:
70
+ - Stateless (safe to call multiple times)
71
+ - Non-blocking (never make blocking decisions)
72
+ - Failure-safe (catch internal errors, return empty list or partial results)
73
+ """
74
+
75
+ @property
76
+ @abstractmethod
77
+ def name(self) -> str:
78
+ """Human-readable scanner name."""
79
+ ...
80
+
81
+ @property
82
+ @abstractmethod
83
+ def description(self) -> str:
84
+ """Brief description of what this scanner detects."""
85
+ ...
86
+
87
+ @abstractmethod
88
+ def scan(self, target: ScanTarget) -> list[Finding]:
89
+ """
90
+ Scan the target and return any findings.
91
+
92
+ Args:
93
+ target: The staged files and diff content to scan.
94
+
95
+ Returns:
96
+ A list of Finding objects. Empty list means no issues found.
97
+ """
98
+ ...
99
+
100
+ def is_available(self) -> bool:
101
+ """
102
+ Check whether this scanner is available (e.g., external tool installed).
103
+
104
+ Scanners that are always available should return True (default).
105
+ """
106
+ return True
107
+
108
+ @property
109
+ def status(self) -> ScannerStatus:
110
+ """Return the current status for this scanner."""
111
+ if not self.is_available():
112
+ return ScannerStatus.UNAVAILABLE
113
+ return ScannerStatus.AVAILABLE