gitrupt 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gitrupt/__init__.py +13 -0
- gitrupt/cli.py +546 -0
- gitrupt/config.py +269 -0
- gitrupt/git.py +590 -0
- gitrupt/hooks/__init__.py +7 -0
- gitrupt/hooks/install.py +255 -0
- gitrupt/hooks/pre_commit.py +103 -0
- gitrupt/hooks/pre_push.py +178 -0
- gitrupt/models.py +190 -0
- gitrupt/policy.py +36 -0
- gitrupt/reporting.py +316 -0
- gitrupt/risk.py +197 -0
- gitrupt/scanner.py +117 -0
- gitrupt/scanners/__init__.py +17 -0
- gitrupt/scanners/adapters.py +166 -0
- gitrupt/scanners/base.py +113 -0
- gitrupt/scanners/binaries.py +185 -0
- gitrupt/scanners/code_rules/__init__.py +36 -0
- gitrupt/scanners/code_rules/base.py +27 -0
- gitrupt/scanners/code_rules/go.py +65 -0
- gitrupt/scanners/code_rules/javascript.py +106 -0
- gitrupt/scanners/code_rules/php.py +71 -0
- gitrupt/scanners/code_rules/powershell.py +85 -0
- gitrupt/scanners/code_rules/python.py +153 -0
- gitrupt/scanners/code_rules/ruby.py +76 -0
- gitrupt/scanners/code_rules/rust.py +41 -0
- gitrupt/scanners/code_rules/shell.py +112 -0
- gitrupt/scanners/dependencies.py +244 -0
- gitrupt/scanners/ecosystems/__init__.py +30 -0
- gitrupt/scanners/ecosystems/base.py +60 -0
- gitrupt/scanners/ecosystems/node.py +128 -0
- gitrupt/scanners/ecosystems/python.py +157 -0
- gitrupt/scanners/entropy.py +123 -0
- gitrupt/scanners/forbidden_files.py +201 -0
- gitrupt/scanners/malware.py +219 -0
- gitrupt/scanners/osv_client.py +221 -0
- gitrupt/scanners/registry.py +66 -0
- gitrupt/scanners/secret_rules.py +368 -0
- gitrupt/scanners/secrets.py +558 -0
- gitrupt/scanners/suspicious_code.py +208 -0
- gitrupt/scanners/yara_loader.py +65 -0
- gitrupt/scanners/yara_rules_builtin.py +141 -0
- gitrupt-0.1.0.dist-info/METADATA +342 -0
- gitrupt-0.1.0.dist-info/RECORD +48 -0
- gitrupt-0.1.0.dist-info/WHEEL +5 -0
- gitrupt-0.1.0.dist-info/entry_points.txt +2 -0
- gitrupt-0.1.0.dist-info/licenses/LICENSE +23 -0
- gitrupt-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Suspicious-code scanner.
|
|
3
|
+
|
|
4
|
+
Language-agnostic orchestration over per-language regex rules.
|
|
5
|
+
Parses the staged diff, runs language-specific rule sets against added lines,
|
|
6
|
+
and returns normalized Finding objects.
|
|
7
|
+
|
|
8
|
+
This scanner is a heuristic. Its findings carry a `confidence` value and
|
|
9
|
+
should be treated as signals, not certainties.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import logging
|
|
15
|
+
import re
|
|
16
|
+
from typing import Iterator
|
|
17
|
+
|
|
18
|
+
from gitrupt.models import Finding, ScanTarget, Severity
|
|
19
|
+
from gitrupt.scanners.base import Scanner
|
|
20
|
+
from gitrupt.scanners.code_rules import (
|
|
21
|
+
get_rules_for_language,
|
|
22
|
+
supported_languages,
|
|
23
|
+
)
|
|
24
|
+
from gitrupt.scanners.code_rules.base import CodeRule
|
|
25
|
+
|
|
26
|
+
logger = logging.getLogger(__name__)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# File extension → language id (matches rule `language` fields)
|
|
30
|
+
_EXT_TO_LANG: tuple[tuple[str, str], ...] = (
|
|
31
|
+
(".py", "python"),
|
|
32
|
+
(".pyi", "python"),
|
|
33
|
+
(".js", "javascript"),
|
|
34
|
+
(".mjs", "javascript"),
|
|
35
|
+
(".cjs", "javascript"),
|
|
36
|
+
(".jsx", "javascript"),
|
|
37
|
+
(".ts", "javascript"),
|
|
38
|
+
(".tsx", "javascript"),
|
|
39
|
+
(".sh", "shell"),
|
|
40
|
+
(".bash", "shell"),
|
|
41
|
+
(".zsh", "shell"),
|
|
42
|
+
(".ps1", "powershell"),
|
|
43
|
+
(".psm1", "powershell"),
|
|
44
|
+
(".php", "php"),
|
|
45
|
+
(".phtml", "php"),
|
|
46
|
+
(".rb", "ruby"),
|
|
47
|
+
(".go", "go"),
|
|
48
|
+
(".rs", "rust"),
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def detect_language(path: str) -> str | None:
|
|
53
|
+
"""Return a rule language id for a path, or None if unsupported."""
|
|
54
|
+
lowered = path.lower()
|
|
55
|
+
for ext, lang in _EXT_TO_LANG:
|
|
56
|
+
if lowered.endswith(ext):
|
|
57
|
+
return lang
|
|
58
|
+
return None
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
_HUNK_HEADER_RE = re.compile(r"\+(\d+)")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _iter_added_lines(diff: str) -> Iterator[tuple[str, int, str]]:
|
|
65
|
+
"""
|
|
66
|
+
Yield (file_path, line_number, line_content) for every added (+) line
|
|
67
|
+
in a unified diff.
|
|
68
|
+
|
|
69
|
+
Handles `+++ b/path`, `@@ -a,b +c,d @@`, deleted and context lines.
|
|
70
|
+
"""
|
|
71
|
+
if not diff:
|
|
72
|
+
return
|
|
73
|
+
|
|
74
|
+
current_file: str | None = None
|
|
75
|
+
current_line: int | None = None
|
|
76
|
+
|
|
77
|
+
for raw in diff.splitlines():
|
|
78
|
+
if raw.startswith("+++ "):
|
|
79
|
+
path = raw[4:].strip()
|
|
80
|
+
if path == "/dev/null":
|
|
81
|
+
current_file = None
|
|
82
|
+
elif path.startswith("b/"):
|
|
83
|
+
current_file = path[2:]
|
|
84
|
+
else:
|
|
85
|
+
current_file = path
|
|
86
|
+
current_line = None
|
|
87
|
+
|
|
88
|
+
elif raw.startswith("@@"):
|
|
89
|
+
m = _HUNK_HEADER_RE.search(raw)
|
|
90
|
+
current_line = int(m.group(1)) if m else None
|
|
91
|
+
|
|
92
|
+
elif raw.startswith("+++"):
|
|
93
|
+
# Defensive: unrecognized header
|
|
94
|
+
continue
|
|
95
|
+
|
|
96
|
+
elif raw.startswith("+"):
|
|
97
|
+
if current_file and current_line is not None:
|
|
98
|
+
yield current_file, current_line, raw[1:]
|
|
99
|
+
if current_line is not None:
|
|
100
|
+
current_line += 1
|
|
101
|
+
|
|
102
|
+
elif raw.startswith("---"):
|
|
103
|
+
continue
|
|
104
|
+
|
|
105
|
+
elif raw.startswith("-"):
|
|
106
|
+
# Deletion: does not advance new-file line counter
|
|
107
|
+
pass
|
|
108
|
+
|
|
109
|
+
elif raw.startswith(" "):
|
|
110
|
+
# Context: advances new-file line counter
|
|
111
|
+
if current_line is not None:
|
|
112
|
+
current_line += 1
|
|
113
|
+
|
|
114
|
+
# Anything else (diff --git, index, new file mode, etc.) is ignored
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class SuspiciousCodeScanner(Scanner):
|
|
118
|
+
"""
|
|
119
|
+
Detects suspicious code patterns on added lines.
|
|
120
|
+
|
|
121
|
+
Behavior:
|
|
122
|
+
- Parses target.staged_diff
|
|
123
|
+
- For each added line, looks up the file's language
|
|
124
|
+
- Runs the language's rule set
|
|
125
|
+
- Emits normalized Findings with `file` and `line` populated
|
|
126
|
+
|
|
127
|
+
Never blocks. The Risk Engine owns the decision.
|
|
128
|
+
"""
|
|
129
|
+
|
|
130
|
+
def __init__(
|
|
131
|
+
self,
|
|
132
|
+
min_confidence: float = 0.6,
|
|
133
|
+
languages: list[str] | None = None,
|
|
134
|
+
) -> None:
|
|
135
|
+
self._min_confidence = max(0.0, min(1.0, min_confidence))
|
|
136
|
+
self._allowed_languages: set[str] | None = set(languages) if languages else None
|
|
137
|
+
|
|
138
|
+
# ── Scanner interface ────────────────────────────────────────────────────
|
|
139
|
+
|
|
140
|
+
@property
|
|
141
|
+
def name(self) -> str:
|
|
142
|
+
return "suspicious_code"
|
|
143
|
+
|
|
144
|
+
@property
|
|
145
|
+
def description(self) -> str:
|
|
146
|
+
return "Detects suspicious code patterns (download-and-execute, obfuscation, shell exec)."
|
|
147
|
+
|
|
148
|
+
def scan(self, target: ScanTarget) -> list[Finding]:
|
|
149
|
+
if not target.staged_diff:
|
|
150
|
+
return []
|
|
151
|
+
|
|
152
|
+
findings: list[Finding] = []
|
|
153
|
+
for file_path, line_no, line_text in _iter_added_lines(target.staged_diff):
|
|
154
|
+
language = detect_language(file_path)
|
|
155
|
+
if language is None:
|
|
156
|
+
continue
|
|
157
|
+
if self._allowed_languages is not None and language not in self._allowed_languages:
|
|
158
|
+
continue
|
|
159
|
+
|
|
160
|
+
for rule in get_rules_for_language(language):
|
|
161
|
+
if rule.confidence < self._min_confidence:
|
|
162
|
+
continue
|
|
163
|
+
if rule.pattern.search(line_text):
|
|
164
|
+
findings.append(
|
|
165
|
+
self._to_finding(rule, file_path, line_no, line_text)
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
return findings
|
|
169
|
+
|
|
170
|
+
# ── Helpers ──────────────────────────────────────────────────────────────
|
|
171
|
+
|
|
172
|
+
@staticmethod
|
|
173
|
+
def _to_finding(
|
|
174
|
+
rule: CodeRule,
|
|
175
|
+
file_path: str,
|
|
176
|
+
line_no: int,
|
|
177
|
+
line_text: str,
|
|
178
|
+
) -> Finding:
|
|
179
|
+
return Finding(
|
|
180
|
+
scanner="suspicious_code",
|
|
181
|
+
rule_id=rule.rule_id,
|
|
182
|
+
severity=rule.severity,
|
|
183
|
+
confidence=rule.confidence,
|
|
184
|
+
file=file_path,
|
|
185
|
+
line=line_no,
|
|
186
|
+
message=rule.message,
|
|
187
|
+
description=rule.description,
|
|
188
|
+
evidence=_redact_line(line_text),
|
|
189
|
+
recommendation=rule.recommendation,
|
|
190
|
+
can_override=True,
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _redact_line(line: str, max_len: int = 120) -> str:
|
|
195
|
+
"""
|
|
196
|
+
Return a safe, truncated preview of a source line.
|
|
197
|
+
|
|
198
|
+
We never include whole lines verbatim in evidence — long lines could
|
|
199
|
+
leak adjacent secrets. Truncate and mark the cut.
|
|
200
|
+
"""
|
|
201
|
+
stripped = line.strip()
|
|
202
|
+
if len(stripped) <= max_len:
|
|
203
|
+
return stripped
|
|
204
|
+
return stripped[:max_len] + " …"
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
# Exposed for tests and for `gitrupt status`
|
|
208
|
+
__all__ = ["SuspiciousCodeScanner", "detect_language", "supported_languages"]
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""
|
|
2
|
+
YARA engine loader with graceful degradation.
|
|
3
|
+
|
|
4
|
+
If yara-python is not installed, load_rules() returns None and the
|
|
5
|
+
malware scanner falls back to its always-on signature pass.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import logging
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from gitrupt.scanners import yara_rules_builtin
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
try:
|
|
19
|
+
import yara # type: ignore[import-untyped]
|
|
20
|
+
YARA_AVAILABLE = True
|
|
21
|
+
except ImportError:
|
|
22
|
+
yara = None # type: ignore[assignment]
|
|
23
|
+
YARA_AVAILABLE = False
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def load_rules(custom_dirs: list[str] | None = None) -> Any | None:
|
|
27
|
+
"""
|
|
28
|
+
Compile built-in YARA rules plus any .yar files found in custom_dirs.
|
|
29
|
+
|
|
30
|
+
Returns a compiled yara.Rules object, or None if YARA is not installed
|
|
31
|
+
or compilation failed.
|
|
32
|
+
"""
|
|
33
|
+
if not YARA_AVAILABLE:
|
|
34
|
+
logger.info("yara-python not installed — malware scanner running in fallback mode")
|
|
35
|
+
return None
|
|
36
|
+
|
|
37
|
+
sources = yara_rules_builtin.all_rule_sources()
|
|
38
|
+
|
|
39
|
+
# Add external .yar files, namespaced by their filename stem.
|
|
40
|
+
# Compilation errors on a single external rule file are isolated —
|
|
41
|
+
# they don't take down the built-in rules.
|
|
42
|
+
if custom_dirs:
|
|
43
|
+
for directory in custom_dirs:
|
|
44
|
+
path = Path(directory)
|
|
45
|
+
if not path.is_dir():
|
|
46
|
+
logger.warning("Custom YARA dir does not exist: %s", directory)
|
|
47
|
+
continue
|
|
48
|
+
for yar_file in path.glob("*.yar"):
|
|
49
|
+
try:
|
|
50
|
+
sources[f"external_{yar_file.stem}"] = yar_file.read_text(
|
|
51
|
+
encoding="utf-8", errors="replace"
|
|
52
|
+
)
|
|
53
|
+
except OSError as e:
|
|
54
|
+
logger.warning("Could not read %s: %s", yar_file, e)
|
|
55
|
+
|
|
56
|
+
try:
|
|
57
|
+
return yara.compile(sources=sources)
|
|
58
|
+
except Exception as e:
|
|
59
|
+
logger.warning("YARA compilation failed: %s", e)
|
|
60
|
+
return None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def is_available() -> bool:
|
|
64
|
+
"""Whether the YARA engine is importable."""
|
|
65
|
+
return YARA_AVAILABLE
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Built-in YARA rules, embedded as Python strings.
|
|
3
|
+
|
|
4
|
+
Why embedded instead of .yar files:
|
|
5
|
+
- No packaging/package-data configuration needed
|
|
6
|
+
- Works identically on Windows, Linux, macOS
|
|
7
|
+
- Users can still add external rules via `malware.custom_yara_dirs`
|
|
8
|
+
|
|
9
|
+
The EICAR string is split into parts to avoid Gitrupt's own source
|
|
10
|
+
tripping file-based antivirus scanners during development.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
# ── EICAR ────────────────────────────────────────────────────────────────────
|
|
16
|
+
# The EICAR test string is inert and standard; split here for readability.
|
|
17
|
+
_EICAR_PARTS = [
|
|
18
|
+
"X5O!P%@AP[4\\",
|
|
19
|
+
"PZX54(P^)7CC)7}",
|
|
20
|
+
"$EICAR-STANDARD-",
|
|
21
|
+
"ANTIVIRUS-TEST-",
|
|
22
|
+
"FILE!$H+H*",
|
|
23
|
+
]
|
|
24
|
+
_EICAR_STR = "".join(_EICAR_PARTS)
|
|
25
|
+
|
|
26
|
+
EICAR = r"""
|
|
27
|
+
rule EICAR_Test_File
|
|
28
|
+
{
|
|
29
|
+
meta:
|
|
30
|
+
description = "EICAR antivirus test file"
|
|
31
|
+
author = "Gitrupt"
|
|
32
|
+
severity = "critical"
|
|
33
|
+
strings:
|
|
34
|
+
$eicar = "EICAR_PLACEHOLDER"
|
|
35
|
+
condition:
|
|
36
|
+
$eicar
|
|
37
|
+
}
|
|
38
|
+
""".replace("EICAR_PLACEHOLDER", _EICAR_STR)
|
|
39
|
+
|
|
40
|
+
# ── Packers ──────────────────────────────────────────────────────────────────
|
|
41
|
+
PACKERS = r"""
|
|
42
|
+
rule UPX_Packed_Executable
|
|
43
|
+
{
|
|
44
|
+
meta:
|
|
45
|
+
description = "UPX-packed executable"
|
|
46
|
+
author = "Gitrupt"
|
|
47
|
+
severity = "high"
|
|
48
|
+
strings:
|
|
49
|
+
$upx0 = "UPX0" ascii
|
|
50
|
+
$upx1 = "UPX1" ascii
|
|
51
|
+
$upx_sig = "$Info: This file is packed with the UPX executable packer" ascii
|
|
52
|
+
condition:
|
|
53
|
+
2 of them
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
rule MPRESS_Packed_Executable
|
|
57
|
+
{
|
|
58
|
+
meta:
|
|
59
|
+
description = "MPRESS-packed executable"
|
|
60
|
+
author = "Gitrupt"
|
|
61
|
+
severity = "high"
|
|
62
|
+
strings:
|
|
63
|
+
$s1 = ".MPRESS1" ascii
|
|
64
|
+
$s2 = ".MPRESS2" ascii
|
|
65
|
+
condition:
|
|
66
|
+
any of them
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
rule ASPack_Signature
|
|
70
|
+
{
|
|
71
|
+
meta:
|
|
72
|
+
description = "ASPack packer signature"
|
|
73
|
+
author = "Gitrupt"
|
|
74
|
+
severity = "high"
|
|
75
|
+
strings:
|
|
76
|
+
$s1 = ".aspack" ascii nocase
|
|
77
|
+
$s2 = ".adata" ascii
|
|
78
|
+
condition:
|
|
79
|
+
all of them
|
|
80
|
+
}
|
|
81
|
+
"""
|
|
82
|
+
|
|
83
|
+
# ── Office macros ────────────────────────────────────────────────────────────
|
|
84
|
+
OFFICE_MACROS = r"""
|
|
85
|
+
rule Office_VBA_Macro_Indicators
|
|
86
|
+
{
|
|
87
|
+
meta:
|
|
88
|
+
description = "Office document contains VBA macro indicators"
|
|
89
|
+
author = "Gitrupt"
|
|
90
|
+
severity = "medium"
|
|
91
|
+
strings:
|
|
92
|
+
$vba1 = "Attribute VB_Name" ascii
|
|
93
|
+
$vba2 = "Attribute VB_Base" ascii
|
|
94
|
+
$vba3 = "AutoOpen" ascii
|
|
95
|
+
$vba4 = "AutoExec" ascii
|
|
96
|
+
$vba5 = "Document_Open" ascii
|
|
97
|
+
$vba6 = "Workbook_Open" ascii
|
|
98
|
+
$vba7 = "Shell(" ascii
|
|
99
|
+
condition:
|
|
100
|
+
3 of them
|
|
101
|
+
}
|
|
102
|
+
"""
|
|
103
|
+
|
|
104
|
+
# ── Script obfuscation (higher-severity signals than Phase 5) ───────────────
|
|
105
|
+
SUSPICIOUS_SCRIPTS = r"""
|
|
106
|
+
rule Powershell_Download_Execute
|
|
107
|
+
{
|
|
108
|
+
meta:
|
|
109
|
+
description = "PowerShell download-and-execute one-liner"
|
|
110
|
+
author = "Gitrupt"
|
|
111
|
+
severity = "critical"
|
|
112
|
+
strings:
|
|
113
|
+
$a = /IEX[\t ]*\(?[\t ]*New-Object[\t ]+Net\.WebClient/ ascii nocase
|
|
114
|
+
$b = /DownloadString[\t ]*\(/ ascii nocase
|
|
115
|
+
condition:
|
|
116
|
+
all of them
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
rule Base64_Decode_Pipe_Shell
|
|
120
|
+
{
|
|
121
|
+
meta:
|
|
122
|
+
description = "Base64 decode piped directly to a shell"
|
|
123
|
+
author = "Gitrupt"
|
|
124
|
+
severity = "high"
|
|
125
|
+
strings:
|
|
126
|
+
$a = /base64[\t ]+(--decode|-d)/ ascii nocase
|
|
127
|
+
$b = /\|[\t ]*(ba|z)?sh\b/ ascii nocase
|
|
128
|
+
condition:
|
|
129
|
+
all of them
|
|
130
|
+
}
|
|
131
|
+
"""
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def all_rule_sources() -> dict[str, str]:
|
|
135
|
+
"""Return namespace → rule source mapping for yara.compile(sources=...)."""
|
|
136
|
+
return {
|
|
137
|
+
"eicar": EICAR,
|
|
138
|
+
"packers": PACKERS,
|
|
139
|
+
"office": OFFICE_MACROS,
|
|
140
|
+
"scripts": SUSPICIOUS_SCRIPTS,
|
|
141
|
+
}
|