gitrupt 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gitrupt/__init__.py +13 -0
- gitrupt/cli.py +546 -0
- gitrupt/config.py +269 -0
- gitrupt/git.py +590 -0
- gitrupt/hooks/__init__.py +7 -0
- gitrupt/hooks/install.py +255 -0
- gitrupt/hooks/pre_commit.py +103 -0
- gitrupt/hooks/pre_push.py +178 -0
- gitrupt/models.py +190 -0
- gitrupt/policy.py +36 -0
- gitrupt/reporting.py +316 -0
- gitrupt/risk.py +197 -0
- gitrupt/scanner.py +117 -0
- gitrupt/scanners/__init__.py +17 -0
- gitrupt/scanners/adapters.py +166 -0
- gitrupt/scanners/base.py +113 -0
- gitrupt/scanners/binaries.py +185 -0
- gitrupt/scanners/code_rules/__init__.py +36 -0
- gitrupt/scanners/code_rules/base.py +27 -0
- gitrupt/scanners/code_rules/go.py +65 -0
- gitrupt/scanners/code_rules/javascript.py +106 -0
- gitrupt/scanners/code_rules/php.py +71 -0
- gitrupt/scanners/code_rules/powershell.py +85 -0
- gitrupt/scanners/code_rules/python.py +153 -0
- gitrupt/scanners/code_rules/ruby.py +76 -0
- gitrupt/scanners/code_rules/rust.py +41 -0
- gitrupt/scanners/code_rules/shell.py +112 -0
- gitrupt/scanners/dependencies.py +244 -0
- gitrupt/scanners/ecosystems/__init__.py +30 -0
- gitrupt/scanners/ecosystems/base.py +60 -0
- gitrupt/scanners/ecosystems/node.py +128 -0
- gitrupt/scanners/ecosystems/python.py +157 -0
- gitrupt/scanners/entropy.py +123 -0
- gitrupt/scanners/forbidden_files.py +201 -0
- gitrupt/scanners/malware.py +219 -0
- gitrupt/scanners/osv_client.py +221 -0
- gitrupt/scanners/registry.py +66 -0
- gitrupt/scanners/secret_rules.py +368 -0
- gitrupt/scanners/secrets.py +558 -0
- gitrupt/scanners/suspicious_code.py +208 -0
- gitrupt/scanners/yara_loader.py +65 -0
- gitrupt/scanners/yara_rules_builtin.py +141 -0
- gitrupt-0.1.0.dist-info/METADATA +342 -0
- gitrupt-0.1.0.dist-info/RECORD +48 -0
- gitrupt-0.1.0.dist-info/WHEEL +5 -0
- gitrupt-0.1.0.dist-info/entry_points.txt +2 -0
- gitrupt-0.1.0.dist-info/licenses/LICENSE +23 -0
- gitrupt-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Base types for ecosystem adapters."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import Protocol
|
|
8
|
+
|
|
9
|
+
from gitrupt.models import Severity
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(frozen=True)
|
|
13
|
+
class Dependency:
|
|
14
|
+
"""A single dependency occurrence in a manifest or lockfile."""
|
|
15
|
+
|
|
16
|
+
osv_ecosystem: str # "PyPI", "npm" — matches OSV's ecosystem names
|
|
17
|
+
name: str
|
|
18
|
+
version: str
|
|
19
|
+
source_file: str
|
|
20
|
+
line: int | None = None
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class EcosystemAdapter(Protocol):
|
|
24
|
+
"""
|
|
25
|
+
Parses one ecosystem's manifest/lockfile into Dependency records.
|
|
26
|
+
|
|
27
|
+
Adapters are stateless. Add a language = add one module and register it
|
|
28
|
+
in __init__.py. No changes to the scanner, risk engine, or reporting.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
name: str # "python", "node" — config name
|
|
32
|
+
osv_ecosystem: str # "PyPI", "npm"
|
|
33
|
+
manifest_patterns: tuple[str, ...] # fnmatch patterns against basename
|
|
34
|
+
lockfile_patterns: tuple[str, ...]
|
|
35
|
+
|
|
36
|
+
def parse(self, path: str, content: str) -> list[Dependency]:
|
|
37
|
+
"""Return pinned dependencies found in this file. Empty if none."""
|
|
38
|
+
...
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
# ── Shared helpers ──────────────────────────────────────────────────────────
|
|
42
|
+
|
|
43
|
+
_VERSION_EXACT = re.compile(r"^\d+\.\d+(?:\.\d+)?(?:[-+][\w.\-]+)?$")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def is_exact_version(spec: str) -> bool:
|
|
47
|
+
"""True if `spec` pins a specific version (no ranges, no caret/tilde)."""
|
|
48
|
+
return bool(_VERSION_EXACT.match(spec.strip()))
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def find_line_number(text: str, needle: str) -> int | None:
|
|
52
|
+
"""First 1-based line containing `needle`. Heuristic — best effort."""
|
|
53
|
+
for i, line in enumerate(text.splitlines(), start=1):
|
|
54
|
+
if needle in line:
|
|
55
|
+
return i
|
|
56
|
+
return None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
# Default severity when OSV provides no usable signal.
|
|
60
|
+
DEFAULT_OSV_SEVERITY = Severity.MEDIUM
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Node.js ecosystem: package.json, package-lock.json."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
|
|
7
|
+
from gitrupt.scanners.ecosystems.base import (
|
|
8
|
+
Dependency,
|
|
9
|
+
find_line_number,
|
|
10
|
+
is_exact_version,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
_OSV = "npm"
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class NodeAdapter:
|
|
17
|
+
name = "node"
|
|
18
|
+
osv_ecosystem = _OSV
|
|
19
|
+
manifest_patterns = ("package.json",)
|
|
20
|
+
lockfile_patterns = ("package-lock.json",)
|
|
21
|
+
|
|
22
|
+
def parse(self, path: str, content: str) -> list[Dependency]:
|
|
23
|
+
lowered = path.lower()
|
|
24
|
+
if lowered.endswith("package-lock.json"):
|
|
25
|
+
return self._parse_lockfile(path, content)
|
|
26
|
+
if lowered.endswith("package.json"):
|
|
27
|
+
return self._parse_manifest(path, content)
|
|
28
|
+
return []
|
|
29
|
+
|
|
30
|
+
# ── package.json ────────────────────────────────────────────────────────
|
|
31
|
+
|
|
32
|
+
def _parse_manifest(self, path: str, content: str) -> list[Dependency]:
|
|
33
|
+
try:
|
|
34
|
+
data = json.loads(content)
|
|
35
|
+
except json.JSONDecodeError:
|
|
36
|
+
return []
|
|
37
|
+
|
|
38
|
+
deps: list[Dependency] = []
|
|
39
|
+
for section in ("dependencies", "devDependencies"):
|
|
40
|
+
block = data.get(section) or {}
|
|
41
|
+
if not isinstance(block, dict):
|
|
42
|
+
continue
|
|
43
|
+
for name, spec in block.items():
|
|
44
|
+
if not isinstance(spec, str):
|
|
45
|
+
continue
|
|
46
|
+
# Strip leading v so "v1.2.3" matches
|
|
47
|
+
candidate = spec.strip().lstrip("v")
|
|
48
|
+
if not is_exact_version(candidate):
|
|
49
|
+
continue
|
|
50
|
+
deps.append(
|
|
51
|
+
Dependency(
|
|
52
|
+
osv_ecosystem=_OSV,
|
|
53
|
+
name=name,
|
|
54
|
+
version=candidate,
|
|
55
|
+
source_file=path,
|
|
56
|
+
line=find_line_number(content, f'"{name}"'),
|
|
57
|
+
)
|
|
58
|
+
)
|
|
59
|
+
return deps
|
|
60
|
+
|
|
61
|
+
# ── package-lock.json ───────────────────────────────────────────────────
|
|
62
|
+
|
|
63
|
+
def _parse_lockfile(self, path: str, content: str) -> list[Dependency]:
|
|
64
|
+
try:
|
|
65
|
+
data = json.loads(content)
|
|
66
|
+
except json.JSONDecodeError:
|
|
67
|
+
return []
|
|
68
|
+
|
|
69
|
+
deps: list[Dependency] = []
|
|
70
|
+
seen: set[tuple[str, str]] = set()
|
|
71
|
+
|
|
72
|
+
# v2/v3: "packages" key
|
|
73
|
+
packages = data.get("packages")
|
|
74
|
+
if isinstance(packages, dict):
|
|
75
|
+
for key, entry in packages.items():
|
|
76
|
+
if not isinstance(entry, dict):
|
|
77
|
+
continue
|
|
78
|
+
version = entry.get("version")
|
|
79
|
+
if not isinstance(version, str):
|
|
80
|
+
continue
|
|
81
|
+
# key is like "" (root) or "node_modules/<name>"
|
|
82
|
+
name = entry.get("name") or key.rsplit("node_modules/", 1)[-1]
|
|
83
|
+
if not name or not is_exact_version(version):
|
|
84
|
+
continue
|
|
85
|
+
pair = (name, version)
|
|
86
|
+
if pair in seen:
|
|
87
|
+
continue
|
|
88
|
+
seen.add(pair)
|
|
89
|
+
deps.append(
|
|
90
|
+
Dependency(
|
|
91
|
+
osv_ecosystem=_OSV,
|
|
92
|
+
name=name,
|
|
93
|
+
version=version,
|
|
94
|
+
source_file=path,
|
|
95
|
+
line=find_line_number(content, f'"node_modules/{name}"')
|
|
96
|
+
or find_line_number(content, f'"{name}"'),
|
|
97
|
+
)
|
|
98
|
+
)
|
|
99
|
+
return deps
|
|
100
|
+
|
|
101
|
+
# v1: recursive "dependencies"
|
|
102
|
+
def walk(block: dict, prefix: str = "") -> None:
|
|
103
|
+
for name, entry in block.items():
|
|
104
|
+
if not isinstance(entry, dict):
|
|
105
|
+
continue
|
|
106
|
+
version = entry.get("version")
|
|
107
|
+
if isinstance(version, str) and is_exact_version(version):
|
|
108
|
+
pair = (name, version)
|
|
109
|
+
if pair not in seen:
|
|
110
|
+
seen.add(pair)
|
|
111
|
+
deps.append(
|
|
112
|
+
Dependency(
|
|
113
|
+
osv_ecosystem=_OSV,
|
|
114
|
+
name=name,
|
|
115
|
+
version=version,
|
|
116
|
+
source_file=path,
|
|
117
|
+
line=find_line_number(content, f'"{name}"'),
|
|
118
|
+
)
|
|
119
|
+
)
|
|
120
|
+
nested = entry.get("dependencies")
|
|
121
|
+
if isinstance(nested, dict):
|
|
122
|
+
walk(nested, prefix + name + "/")
|
|
123
|
+
|
|
124
|
+
root = data.get("dependencies")
|
|
125
|
+
if isinstance(root, dict):
|
|
126
|
+
walk(root)
|
|
127
|
+
|
|
128
|
+
return deps
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
"""Python ecosystem: requirements.txt, pyproject.toml, poetry.lock."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
import tomllib
|
|
7
|
+
|
|
8
|
+
from gitrupt.scanners.ecosystems.base import (
|
|
9
|
+
Dependency,
|
|
10
|
+
find_line_number,
|
|
11
|
+
is_exact_version,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
_OSV = "PyPI"
|
|
15
|
+
|
|
16
|
+
# requirements.txt: name[extras]==1.2.3
|
|
17
|
+
_REQ_PIN = re.compile(
|
|
18
|
+
r"^\s*([A-Za-z0-9][A-Za-z0-9_.\-]*)"
|
|
19
|
+
r"(?:\[[^\]]*\])?"
|
|
20
|
+
r"\s*==\s*"
|
|
21
|
+
r"([A-Za-z0-9_.\-+!]+)"
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
# PEP 508 pin inside a list string: "django==4.2.0"
|
|
25
|
+
_PEP508_PIN = re.compile(
|
|
26
|
+
r"^\s*([A-Za-z0-9][A-Za-z0-9_.\-]*)"
|
|
27
|
+
r"(?:\[[^\]]*\])?"
|
|
28
|
+
r"\s*==\s*"
|
|
29
|
+
r"([A-Za-z0-9_.\-+!]+)"
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class PythonAdapter:
|
|
34
|
+
name = "python"
|
|
35
|
+
osv_ecosystem = _OSV
|
|
36
|
+
manifest_patterns = ("requirements.txt", "requirements-*.txt", "pyproject.toml")
|
|
37
|
+
lockfile_patterns = ("poetry.lock", "Pipfile.lock")
|
|
38
|
+
|
|
39
|
+
def parse(self, path: str, content: str) -> list[Dependency]:
|
|
40
|
+
lowered = path.lower()
|
|
41
|
+
if lowered.endswith("requirements.txt") or "/requirements-" in lowered or lowered.startswith("requirements-"):
|
|
42
|
+
return self._parse_requirements(path, content)
|
|
43
|
+
if lowered.endswith("pyproject.toml"):
|
|
44
|
+
return self._parse_pyproject(path, content)
|
|
45
|
+
if lowered.endswith("poetry.lock"):
|
|
46
|
+
return self._parse_poetry_lock(path, content)
|
|
47
|
+
return []
|
|
48
|
+
|
|
49
|
+
# ── requirements.txt ────────────────────────────────────────────────────
|
|
50
|
+
|
|
51
|
+
def _parse_requirements(self, path: str, content: str) -> list[Dependency]:
|
|
52
|
+
deps: list[Dependency] = []
|
|
53
|
+
for i, raw in enumerate(content.splitlines(), start=1):
|
|
54
|
+
line = raw.split("#", 1)[0].strip()
|
|
55
|
+
if not line or line.startswith("-"):
|
|
56
|
+
continue
|
|
57
|
+
m = _REQ_PIN.match(line)
|
|
58
|
+
if not m:
|
|
59
|
+
continue
|
|
60
|
+
deps.append(
|
|
61
|
+
Dependency(
|
|
62
|
+
osv_ecosystem=_OSV,
|
|
63
|
+
name=m.group(1),
|
|
64
|
+
version=m.group(2),
|
|
65
|
+
source_file=path,
|
|
66
|
+
line=i,
|
|
67
|
+
)
|
|
68
|
+
)
|
|
69
|
+
return deps
|
|
70
|
+
|
|
71
|
+
# ── pyproject.toml ──────────────────────────────────────────────────────
|
|
72
|
+
|
|
73
|
+
def _parse_pyproject(self, path: str, content: str) -> list[Dependency]:
|
|
74
|
+
try:
|
|
75
|
+
data = tomllib.loads(content)
|
|
76
|
+
except tomllib.TOMLDecodeError:
|
|
77
|
+
return []
|
|
78
|
+
|
|
79
|
+
deps: list[Dependency] = []
|
|
80
|
+
|
|
81
|
+
# PEP 621: [project] dependencies = ["django==4.2.0", ...]
|
|
82
|
+
project = data.get("project", {})
|
|
83
|
+
for spec in project.get("dependencies", []) or []:
|
|
84
|
+
if not isinstance(spec, str):
|
|
85
|
+
continue
|
|
86
|
+
m = _PEP508_PIN.match(spec)
|
|
87
|
+
if not m:
|
|
88
|
+
continue
|
|
89
|
+
deps.append(
|
|
90
|
+
Dependency(
|
|
91
|
+
osv_ecosystem=_OSV,
|
|
92
|
+
name=m.group(1),
|
|
93
|
+
version=m.group(2),
|
|
94
|
+
source_file=path,
|
|
95
|
+
line=find_line_number(content, m.group(1)),
|
|
96
|
+
)
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
# Poetry: [tool.poetry.dependencies] name = "1.2.3"
|
|
100
|
+
poetry = (data.get("tool", {}) or {}).get("poetry", {}) or {}
|
|
101
|
+
for section in ("dependencies", "dev-dependencies"):
|
|
102
|
+
block = poetry.get(section, {}) or {}
|
|
103
|
+
for name, spec in block.items():
|
|
104
|
+
if name.lower() == "python":
|
|
105
|
+
continue
|
|
106
|
+
if isinstance(spec, str) and is_exact_version(spec):
|
|
107
|
+
deps.append(
|
|
108
|
+
Dependency(
|
|
109
|
+
osv_ecosystem=_OSV,
|
|
110
|
+
name=name,
|
|
111
|
+
version=spec.strip(),
|
|
112
|
+
source_file=path,
|
|
113
|
+
line=find_line_number(content, f'"{name}"')
|
|
114
|
+
or find_line_number(content, name),
|
|
115
|
+
)
|
|
116
|
+
)
|
|
117
|
+
elif isinstance(spec, dict):
|
|
118
|
+
ver = spec.get("version")
|
|
119
|
+
if isinstance(ver, str) and is_exact_version(ver):
|
|
120
|
+
deps.append(
|
|
121
|
+
Dependency(
|
|
122
|
+
osv_ecosystem=_OSV,
|
|
123
|
+
name=name,
|
|
124
|
+
version=ver.strip(),
|
|
125
|
+
source_file=path,
|
|
126
|
+
line=find_line_number(content, name),
|
|
127
|
+
)
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
return deps
|
|
131
|
+
|
|
132
|
+
# ── poetry.lock ─────────────────────────────────────────────────────────
|
|
133
|
+
|
|
134
|
+
def _parse_poetry_lock(self, path: str, content: str) -> list[Dependency]:
|
|
135
|
+
try:
|
|
136
|
+
data = tomllib.loads(content)
|
|
137
|
+
except tomllib.TOMLDecodeError:
|
|
138
|
+
return []
|
|
139
|
+
|
|
140
|
+
deps: list[Dependency] = []
|
|
141
|
+
for pkg in data.get("package", []) or []:
|
|
142
|
+
name = pkg.get("name")
|
|
143
|
+
version = pkg.get("version")
|
|
144
|
+
if not isinstance(name, str) or not isinstance(version, str):
|
|
145
|
+
continue
|
|
146
|
+
if not is_exact_version(version):
|
|
147
|
+
continue
|
|
148
|
+
deps.append(
|
|
149
|
+
Dependency(
|
|
150
|
+
osv_ecosystem=_OSV,
|
|
151
|
+
name=name,
|
|
152
|
+
version=version,
|
|
153
|
+
source_file=path,
|
|
154
|
+
line=find_line_number(content, f'name = "{name}"'),
|
|
155
|
+
)
|
|
156
|
+
)
|
|
157
|
+
return deps
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Entropy-based secret detection with calibration and allowlisting.
|
|
3
|
+
|
|
4
|
+
Design:
|
|
5
|
+
- Threshold: 4.5 bits/char (calibrated; see docs/entropy-calibration.md)
|
|
6
|
+
- Minimum length: 20 chars
|
|
7
|
+
- Allowlist: known-safe shapes (git SHA, UUID, semver, SRI hash, base64 PNG)
|
|
8
|
+
- Context bonus: lower the bar when a key/token/secret word is nearby
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import math
|
|
14
|
+
import re
|
|
15
|
+
from collections import Counter
|
|
16
|
+
from dataclasses import dataclass
|
|
17
|
+
|
|
18
|
+
from gitrupt.models import Severity
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
DEFAULT_THRESHOLD = 4.5
|
|
22
|
+
MIN_LENGTH = 20
|
|
23
|
+
CONTEXT_BONUS = 0.3 # lower threshold by this when context word is nearby
|
|
24
|
+
CONTEXT_WINDOW = 120
|
|
25
|
+
|
|
26
|
+
_CONTEXT_WORDS = (
|
|
27
|
+
"key", "token", "secret", "password", "passwd", "pwd",
|
|
28
|
+
"api_key", "apikey", "auth", "credential", "bearer",
|
|
29
|
+
"private", "access", "session",
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def shannon_entropy(s: str) -> float:
|
|
34
|
+
"""Byte-level Shannon entropy, 0.0 – ~6.5 for alphanumerics."""
|
|
35
|
+
if not s:
|
|
36
|
+
return 0.0
|
|
37
|
+
counts = Counter(s)
|
|
38
|
+
total = len(s)
|
|
39
|
+
return -sum((c / total) * math.log2(c / total) for c in counts.values())
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
# ── Allowlist: shapes that look random but are not secrets ──────────────────
|
|
43
|
+
|
|
44
|
+
_ALLOWLIST_PATTERNS: tuple[re.Pattern, ...] = (
|
|
45
|
+
# Git commit SHA (full or short)
|
|
46
|
+
re.compile(r"^[0-9a-f]{7,40}$"),
|
|
47
|
+
# UUID v4
|
|
48
|
+
re.compile(r"^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$"),
|
|
49
|
+
# Semver (with optional prerelease/build)
|
|
50
|
+
re.compile(r"^\d+\.\d+\.\d+(?:-[\w.]+)?(?:\+[\w.]+)?$"),
|
|
51
|
+
# SRI hash (sha256-/sha384-/sha512- + base64)
|
|
52
|
+
re.compile(r"^sha(256|384|512)-[A-Za-z0-9+/=]+$"),
|
|
53
|
+
# Hex-encoded hash (MD5, SHA1, SHA256)
|
|
54
|
+
re.compile(r"^[0-9a-f]{32}$|^[0-9a-f]{40}$|^[0-9a-f]{64}$"),
|
|
55
|
+
# Base64-encoded small payloads (PNG header, etc.)
|
|
56
|
+
re.compile(r"^iVBORw0KGgo[A-Za-z0-9+/=]+$"),
|
|
57
|
+
# IPv6
|
|
58
|
+
re.compile(r"^[0-9a-fA-F:]{8,}$"),
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def is_allowlisted(value: str) -> bool:
|
|
63
|
+
"""True if value matches a known-safe shape."""
|
|
64
|
+
return any(p.match(value) for p in _ALLOWLIST_PATTERNS)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
# ── Candidate extraction ────────────────────────────────────────────────────
|
|
68
|
+
|
|
69
|
+
# Strings of length >= MIN_LENGTH built from the token alphabet.
|
|
70
|
+
_CANDIDATE_RE = re.compile(
|
|
71
|
+
rf"[A-Za-z0-9_\-+/=]{{{MIN_LENGTH},}}"
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass(frozen=True)
|
|
76
|
+
class EntropyHit:
|
|
77
|
+
value: str
|
|
78
|
+
entropy: float
|
|
79
|
+
reason: str
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _context_present(line: str, start: int, end: int) -> bool:
|
|
83
|
+
window = line[max(0, start - CONTEXT_WINDOW): min(len(line), end + CONTEXT_WINDOW)]
|
|
84
|
+
lowered = window.lower()
|
|
85
|
+
return any(w in lowered for w in _CONTEXT_WORDS)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def find_entropy_secrets(
|
|
89
|
+
line: str,
|
|
90
|
+
threshold: float = DEFAULT_THRESHOLD,
|
|
91
|
+
) -> list[EntropyHit]:
|
|
92
|
+
"""
|
|
93
|
+
Scan a single line for high-entropy candidates.
|
|
94
|
+
|
|
95
|
+
Returns one EntropyHit per candidate that clears the bar.
|
|
96
|
+
"""
|
|
97
|
+
hits: list[EntropyHit] = []
|
|
98
|
+
|
|
99
|
+
for m in _CANDIDATE_RE.finditer(line):
|
|
100
|
+
candidate = m.group(0)
|
|
101
|
+
if len(candidate) < MIN_LENGTH:
|
|
102
|
+
continue
|
|
103
|
+
if is_allowlisted(candidate):
|
|
104
|
+
continue
|
|
105
|
+
|
|
106
|
+
ent = shannon_entropy(candidate)
|
|
107
|
+
|
|
108
|
+
# Lower the bar when the surrounding text hints at a secret.
|
|
109
|
+
effective_threshold = threshold
|
|
110
|
+
context = _context_present(line, m.start(), m.end())
|
|
111
|
+
if context:
|
|
112
|
+
effective_threshold -= CONTEXT_BONUS
|
|
113
|
+
|
|
114
|
+
if ent >= effective_threshold:
|
|
115
|
+
hits.append(
|
|
116
|
+
EntropyHit(
|
|
117
|
+
value=candidate,
|
|
118
|
+
entropy=ent,
|
|
119
|
+
reason="context-assisted" if context else "high-entropy",
|
|
120
|
+
)
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
return hits
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Forbidden file scanner.
|
|
3
|
+
|
|
4
|
+
Checks staged files against a list of forbidden path patterns.
|
|
5
|
+
Operates on file paths only — no content scanning needed.
|
|
6
|
+
Language-agnostic by construction.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import fnmatch
|
|
12
|
+
import logging
|
|
13
|
+
from pathlib import PurePosixPath
|
|
14
|
+
|
|
15
|
+
from gitrupt.config import AllowConfig, RulesConfig
|
|
16
|
+
from gitrupt.models import Finding, ScanTarget, Severity
|
|
17
|
+
from gitrupt.scanners.base import Scanner
|
|
18
|
+
|
|
19
|
+
logger = logging.getLogger(__name__)
|
|
20
|
+
|
|
21
|
+
# Default forbidden path patterns (always active, even without config file)
|
|
22
|
+
DEFAULT_FORBIDDEN_PATTERNS: list[tuple[str, str, Severity]] = [
|
|
23
|
+
# (pattern, reason, severity)
|
|
24
|
+
(".env", "Environment file — likely contains credentials", Severity.CRITICAL),
|
|
25
|
+
(".env.*", "Environment file — likely contains credentials", Severity.CRITICAL),
|
|
26
|
+
("*.pem", "PEM-encoded private key or certificate", Severity.CRITICAL),
|
|
27
|
+
("*.key", "Private key file", Severity.CRITICAL),
|
|
28
|
+
("*.p12", "PKCS#12 certificate store — may contain private keys", Severity.CRITICAL),
|
|
29
|
+
("*.pfx", "PFX certificate file — may contain private keys", Severity.CRITICAL),
|
|
30
|
+
("credentials.json", "Cloud credential file", Severity.CRITICAL),
|
|
31
|
+
("service-account.json", "Service account credential file", Severity.CRITICAL),
|
|
32
|
+
("*.secret", "Secret file", Severity.CRITICAL),
|
|
33
|
+
# High severity
|
|
34
|
+
(".htpasswd", "Apache password file", Severity.HIGH),
|
|
35
|
+
("*.jks", "Java KeyStore — may contain private keys", Severity.HIGH),
|
|
36
|
+
("*.keystore", "Key store file", Severity.HIGH),
|
|
37
|
+
# Suspicious executables
|
|
38
|
+
("*.scr", "Windows screensaver executable", Severity.HIGH),
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
# Paths that are always allowed regardless of rules
|
|
42
|
+
DEFAULT_ALLOW_PATTERNS: list[str] = [
|
|
43
|
+
".env.example",
|
|
44
|
+
".env.template",
|
|
45
|
+
".env.sample",
|
|
46
|
+
".env.test",
|
|
47
|
+
]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class ForbiddenFileScanner(Scanner):
|
|
51
|
+
"""
|
|
52
|
+
Scans staged files against forbidden path patterns.
|
|
53
|
+
|
|
54
|
+
Checks file names and paths using glob patterns.
|
|
55
|
+
Does not read file content — purely path-based.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
def __init__(
|
|
59
|
+
self,
|
|
60
|
+
rules: RulesConfig | None = None,
|
|
61
|
+
allow: AllowConfig | None = None,
|
|
62
|
+
) -> None:
|
|
63
|
+
self._rules = rules
|
|
64
|
+
self._allow = allow
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def name(self) -> str:
|
|
68
|
+
return "forbidden-files"
|
|
69
|
+
|
|
70
|
+
@property
|
|
71
|
+
def description(self) -> str:
|
|
72
|
+
return "Detects forbidden file paths (credentials, keys, dangerous executables)"
|
|
73
|
+
|
|
74
|
+
def scan(self, target: ScanTarget) -> list[Finding]:
|
|
75
|
+
findings: list[Finding] = []
|
|
76
|
+
|
|
77
|
+
# Build effective allow list
|
|
78
|
+
allow_patterns = list(DEFAULT_ALLOW_PATTERNS)
|
|
79
|
+
if self._allow:
|
|
80
|
+
allow_patterns.extend(self._allow.paths)
|
|
81
|
+
|
|
82
|
+
# Build effective forbidden patterns
|
|
83
|
+
forbidden_patterns = self._build_forbidden_patterns()
|
|
84
|
+
|
|
85
|
+
for staged_file in target.staged_files:
|
|
86
|
+
# Skip deleted files (they're being removed, not added)
|
|
87
|
+
if staged_file.status == "D":
|
|
88
|
+
continue
|
|
89
|
+
|
|
90
|
+
file_path = staged_file.path
|
|
91
|
+
|
|
92
|
+
# Check allow list first
|
|
93
|
+
if self._is_allowed(file_path, allow_patterns):
|
|
94
|
+
logger.debug("File %s is on the allow list, skipping", file_path)
|
|
95
|
+
continue
|
|
96
|
+
|
|
97
|
+
# Check forbidden patterns
|
|
98
|
+
match = self._matches_forbidden(file_path, forbidden_patterns)
|
|
99
|
+
if match:
|
|
100
|
+
pattern, reason, severity = match
|
|
101
|
+
findings.append(
|
|
102
|
+
Finding(
|
|
103
|
+
scanner=self.name,
|
|
104
|
+
rule_id=f"forbidden-file-{_pattern_to_rule_id(pattern)}",
|
|
105
|
+
severity=severity,
|
|
106
|
+
confidence=1.0,
|
|
107
|
+
file=file_path,
|
|
108
|
+
line=None,
|
|
109
|
+
message=f"Forbidden file: {file_path}",
|
|
110
|
+
description=reason,
|
|
111
|
+
evidence=f"Path matches pattern: {pattern}",
|
|
112
|
+
recommendation=(
|
|
113
|
+
f"Remove this file from staging:\n"
|
|
114
|
+
f" git restore --staged {file_path}\n"
|
|
115
|
+
f"If this file is intentional, add it to the 'allow.paths' "
|
|
116
|
+
f"section of .gitrupt.yml"
|
|
117
|
+
),
|
|
118
|
+
can_override=False,
|
|
119
|
+
)
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
return findings
|
|
123
|
+
|
|
124
|
+
def _build_forbidden_patterns(self) -> list[tuple[str, str, Severity]]:
|
|
125
|
+
"""Build the complete list of forbidden patterns."""
|
|
126
|
+
patterns = list(DEFAULT_FORBIDDEN_PATTERNS)
|
|
127
|
+
|
|
128
|
+
if self._rules:
|
|
129
|
+
# Add config-specified forbidden paths (default to HIGH severity)
|
|
130
|
+
for p in self._rules.forbidden_paths:
|
|
131
|
+
if not any(p == pat for pat, _, _ in DEFAULT_FORBIDDEN_PATTERNS):
|
|
132
|
+
patterns.append((p, f"Forbidden by policy: {p}", Severity.HIGH))
|
|
133
|
+
|
|
134
|
+
# Add config-specified forbidden extensions
|
|
135
|
+
for ext in self._rules.forbidden_extensions:
|
|
136
|
+
pattern = f"*{ext}"
|
|
137
|
+
patterns.append(
|
|
138
|
+
(pattern, f"Forbidden file extension: {ext}", Severity.HIGH)
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
return patterns
|
|
142
|
+
|
|
143
|
+
@staticmethod
|
|
144
|
+
def _is_allowed(file_path: str, allow_patterns: list[str]) -> bool:
|
|
145
|
+
"""Check whether a file path matches any allow pattern."""
|
|
146
|
+
# Check the full path and just the filename
|
|
147
|
+
basename = PurePosixPath(file_path).name
|
|
148
|
+
|
|
149
|
+
for pattern in allow_patterns:
|
|
150
|
+
# Exact match
|
|
151
|
+
if file_path == pattern or basename == pattern:
|
|
152
|
+
return True
|
|
153
|
+
# Glob match on full path
|
|
154
|
+
if fnmatch.fnmatch(file_path, pattern):
|
|
155
|
+
return True
|
|
156
|
+
# Glob match on basename
|
|
157
|
+
if fnmatch.fnmatch(basename, pattern):
|
|
158
|
+
return True
|
|
159
|
+
|
|
160
|
+
return False
|
|
161
|
+
|
|
162
|
+
@staticmethod
|
|
163
|
+
def _matches_forbidden(
|
|
164
|
+
file_path: str,
|
|
165
|
+
patterns: list[tuple[str, str, Severity]],
|
|
166
|
+
) -> tuple[str, str, Severity] | None:
|
|
167
|
+
"""
|
|
168
|
+
Return the first forbidden pattern that matches, or None.
|
|
169
|
+
|
|
170
|
+
Checks both the full path and the filename.
|
|
171
|
+
"""
|
|
172
|
+
basename = PurePosixPath(file_path).name
|
|
173
|
+
|
|
174
|
+
for pattern, reason, severity in patterns:
|
|
175
|
+
# Exact match on basename (most common case)
|
|
176
|
+
if basename == pattern:
|
|
177
|
+
return pattern, reason, severity
|
|
178
|
+
# Glob match on basename
|
|
179
|
+
if fnmatch.fnmatch(basename, pattern):
|
|
180
|
+
return pattern, reason, severity
|
|
181
|
+
# Glob match on full path (for path-specific rules like "config/secrets/*")
|
|
182
|
+
if fnmatch.fnmatch(file_path, pattern):
|
|
183
|
+
return pattern, reason, severity
|
|
184
|
+
# Also try matching against path components
|
|
185
|
+
path_parts = PurePosixPath(file_path).parts
|
|
186
|
+
for part in path_parts:
|
|
187
|
+
if fnmatch.fnmatch(part, pattern):
|
|
188
|
+
return pattern, reason, severity
|
|
189
|
+
|
|
190
|
+
return None
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _pattern_to_rule_id(pattern: str) -> str:
|
|
194
|
+
"""Convert a glob pattern to a safe rule ID."""
|
|
195
|
+
return (
|
|
196
|
+
pattern.replace("*", "glob")
|
|
197
|
+
.replace(".", "-")
|
|
198
|
+
.replace("/", "-")
|
|
199
|
+
.strip("-")
|
|
200
|
+
.lower()
|
|
201
|
+
)
|