gitrupt 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gitrupt-0.2.0 → gitrupt-0.2.1}/CHANGELOG.md +13 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/PKG-INFO +1 -1
- {gitrupt-0.2.0 → gitrupt-0.2.1}/pyproject.toml +1 -1
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/__init__.py +1 -1
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/forbidden_files.py +92 -5
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/secrets.py +182 -165
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt.egg-info/PKG-INFO +1 -1
- {gitrupt-0.2.0 → gitrupt-0.2.1}/.gitrupt.yml +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/CONTRIBUTING.md +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/LICENSE +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/MANIFEST.in +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/README.md +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/SECURITY.md +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/docs/architecture.md +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/docs/ci-setup.md +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/docs/implementation-plan.md +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/docs/plan.md +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/docs/roadmap.md +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/docs/templates/docker/Dockerfile.gitrupt +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/docs/templates/github-actions/gitrupt-for-self.yml +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/docs/templates/github-actions/gitrupt.yml +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/docs/templates/gitlab-ci/gitrupt.yml +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/setup.cfg +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/__main__.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/ci.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/cli.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/config.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/git.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/hooks/__init__.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/hooks/install.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/hooks/pre_commit.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/hooks/pre_push.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/models.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/policy.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/reporting.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/risk.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanner.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/__init__.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/adapters.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/base.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/binaries.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/code_rules/__init__.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/code_rules/base.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/code_rules/go.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/code_rules/javascript.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/code_rules/php.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/code_rules/powershell.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/code_rules/python.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/code_rules/ruby.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/code_rules/rust.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/code_rules/shell.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/dependencies.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/ecosystems/__init__.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/ecosystems/base.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/ecosystems/node.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/ecosystems/python.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/entropy.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/malware.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/osv_client.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/registry.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/secret_rules.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/suspicious_code.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/yara_loader.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt/scanners/yara_rules_builtin.py +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt.egg-info/SOURCES.txt +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt.egg-info/dependency_links.txt +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt.egg-info/entry_points.txt +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt.egg-info/requires.txt +0 -0
- {gitrupt-0.2.0 → gitrupt-0.2.1}/src/gitrupt.egg-info/top_level.txt +0 -0
|
@@ -8,7 +8,20 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
8
8
|
---
|
|
9
9
|
|
|
10
10
|
## [Unreleased]
|
|
11
|
+
## [0.2.1] - 2026-09-17
|
|
11
12
|
|
|
13
|
+
### Fixed
|
|
14
|
+
|
|
15
|
+
- **UTF-16 secret detection** — files written as UTF-16 by Windows PowerShell (`>` or `echo`) are now decoded and scanned instead of being skipped as binary. A file created with `'secret' > file.txt` in PowerShell 5.1 is now correctly flagged.
|
|
16
|
+
- **Binary text-file warning** — files with text extensions (`.py`, `.env`, `.json`, `.md`, etc.) that Git classifies as binary now produce a MEDIUM finding. This surfaces encoding issues even when the file would otherwise be skipped by content scanners.
|
|
17
|
+
|
|
18
|
+
### Added
|
|
19
|
+
|
|
20
|
+
- `binary-text-file` rule in the forbidden-files scanner.
|
|
21
|
+
|
|
22
|
+
### Security
|
|
23
|
+
|
|
24
|
+
- Windows PowerShell developers no longer have a silent bypass path via UTF-16 encoded files.
|
|
12
25
|
---
|
|
13
26
|
|
|
14
27
|
## [0.2.0] - 2026-09-16
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "gitrupt"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.1"
|
|
8
8
|
description = "A local Git security firewall for secrets, suspicious code, malware heuristics, and dependency vulnerability checks"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { file = "LICENSE" }
|
|
@@ -2,8 +2,13 @@
|
|
|
2
2
|
Forbidden file scanner.
|
|
3
3
|
|
|
4
4
|
Checks staged files against a list of forbidden path patterns.
|
|
5
|
-
Operates on file paths
|
|
5
|
+
Operates primarily on file paths — no content scanning needed.
|
|
6
6
|
Language-agnostic by construction.
|
|
7
|
+
|
|
8
|
+
Also flags text-extension files that Git classifies as binary. This often
|
|
9
|
+
means the file was written as UTF-16 (Windows PowerShell does this by
|
|
10
|
+
default) or has been corrupted, and it's a common way for secrets to slip
|
|
11
|
+
past content scanners that skip binary files.
|
|
7
12
|
"""
|
|
8
13
|
|
|
9
14
|
from __future__ import annotations
|
|
@@ -46,6 +51,29 @@ DEFAULT_ALLOW_PATTERNS: list[str] = [
|
|
|
46
51
|
".env.test",
|
|
47
52
|
]
|
|
48
53
|
|
|
54
|
+
# File extensions that indicate text content.
|
|
55
|
+
# If Git classifies one of these as binary, something is wrong —
|
|
56
|
+
# most commonly a UTF-16 encoding written by Windows PowerShell.
|
|
57
|
+
_TEXT_EXTENSIONS: frozenset[str] = frozenset({
|
|
58
|
+
# Python
|
|
59
|
+
".py", ".pyi",
|
|
60
|
+
# JavaScript / TypeScript
|
|
61
|
+
".js", ".mjs", ".cjs", ".ts", ".tsx", ".jsx",
|
|
62
|
+
# Config / data
|
|
63
|
+
".json", ".yaml", ".yml", ".toml", ".ini", ".cfg", ".conf", ".env",
|
|
64
|
+
# Docs / text
|
|
65
|
+
".txt", ".md", ".rst", ".log", ".csv", ".tsv",
|
|
66
|
+
# Shell
|
|
67
|
+
".sh", ".bash", ".zsh", ".fish", ".ps1", ".bat", ".cmd",
|
|
68
|
+
# Other languages
|
|
69
|
+
".rb", ".go", ".rs", ".java", ".kt", ".kts", ".cs", ".php",
|
|
70
|
+
".swift", ".c", ".h", ".cpp", ".hpp", ".m", ".mm",
|
|
71
|
+
# Web
|
|
72
|
+
".html", ".htm", ".css", ".scss", ".sass", ".less", ".xml", ".svg",
|
|
73
|
+
# Misc
|
|
74
|
+
".sql", ".gradle", ".properties",
|
|
75
|
+
})
|
|
76
|
+
|
|
49
77
|
|
|
50
78
|
class ForbiddenFileScanner(Scanner):
|
|
51
79
|
"""
|
|
@@ -53,6 +81,9 @@ class ForbiddenFileScanner(Scanner):
|
|
|
53
81
|
|
|
54
82
|
Checks file names and paths using glob patterns.
|
|
55
83
|
Does not read file content — purely path-based.
|
|
84
|
+
|
|
85
|
+
Additionally, flags any staged file with a text extension that Git
|
|
86
|
+
classifies as binary, since that often indicates a hidden encoding issue.
|
|
56
87
|
"""
|
|
57
88
|
|
|
58
89
|
def __init__(
|
|
@@ -69,7 +100,10 @@ class ForbiddenFileScanner(Scanner):
|
|
|
69
100
|
|
|
70
101
|
@property
|
|
71
102
|
def description(self) -> str:
|
|
72
|
-
return
|
|
103
|
+
return (
|
|
104
|
+
"Detects forbidden file paths (credentials, keys, dangerous "
|
|
105
|
+
"executables) and mis-encoded text files"
|
|
106
|
+
)
|
|
73
107
|
|
|
74
108
|
def scan(self, target: ScanTarget) -> list[Finding]:
|
|
75
109
|
findings: list[Finding] = []
|
|
@@ -89,7 +123,15 @@ class ForbiddenFileScanner(Scanner):
|
|
|
89
123
|
|
|
90
124
|
file_path = staged_file.path
|
|
91
125
|
|
|
92
|
-
#
|
|
126
|
+
# ── Binary text-file check ─────────────────────────────────────
|
|
127
|
+
# Fires regardless of the allow list — a file can be legitimately
|
|
128
|
+
# named (e.g. .env.example) and still have a suspicious encoding
|
|
129
|
+
# that hides content from the secret scanner.
|
|
130
|
+
binary_finding = self._check_binary_text_file(staged_file)
|
|
131
|
+
if binary_finding is not None:
|
|
132
|
+
findings.append(binary_finding)
|
|
133
|
+
|
|
134
|
+
# Check allow list — skip path-pattern checks if allowed
|
|
93
135
|
if self._is_allowed(file_path, allow_patterns):
|
|
94
136
|
logger.debug("File %s is on the allow list, skipping", file_path)
|
|
95
137
|
continue
|
|
@@ -121,6 +163,52 @@ class ForbiddenFileScanner(Scanner):
|
|
|
121
163
|
|
|
122
164
|
return findings
|
|
123
165
|
|
|
166
|
+
# ── Binary text-file detection ───────────────────────────────────────────
|
|
167
|
+
|
|
168
|
+
@staticmethod
|
|
169
|
+
def _check_binary_text_file(staged_file) -> Finding | None:
|
|
170
|
+
"""
|
|
171
|
+
Return a Finding if the file has a text extension but is binary.
|
|
172
|
+
|
|
173
|
+
Returns None otherwise. Never raises.
|
|
174
|
+
"""
|
|
175
|
+
if not staged_file.is_binary:
|
|
176
|
+
return None
|
|
177
|
+
|
|
178
|
+
path = staged_file.path
|
|
179
|
+
basename = PurePosixPath(path).name
|
|
180
|
+
if "." not in basename:
|
|
181
|
+
return None
|
|
182
|
+
|
|
183
|
+
suffix = "." + basename.rsplit(".", 1)[-1].lower()
|
|
184
|
+
if suffix not in _TEXT_EXTENSIONS:
|
|
185
|
+
return None
|
|
186
|
+
|
|
187
|
+
return Finding(
|
|
188
|
+
scanner="forbidden-files",
|
|
189
|
+
rule_id="binary-text-file",
|
|
190
|
+
severity=Severity.MEDIUM,
|
|
191
|
+
confidence=0.7,
|
|
192
|
+
file=path,
|
|
193
|
+
line=None,
|
|
194
|
+
message=f"'{path}' has a text extension but binary content",
|
|
195
|
+
description=(
|
|
196
|
+
"Text files should not contain NUL bytes. This usually means "
|
|
197
|
+
"the file was written as UTF-16 (Windows PowerShell does this "
|
|
198
|
+
"by default with `>` or `echo`) or has been corrupted. "
|
|
199
|
+
"Content scanners skip binary files, so a secret in this file "
|
|
200
|
+
"would not be detected."
|
|
201
|
+
),
|
|
202
|
+
evidence=f"extension={suffix}",
|
|
203
|
+
recommendation=(
|
|
204
|
+
"Re-save the file as UTF-8. On Windows PowerShell, use "
|
|
205
|
+
"`Set-Content -Encoding UTF8` instead of `>` or `echo`."
|
|
206
|
+
),
|
|
207
|
+
can_override=True,
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
# ── Pattern building / matching ──────────────────────────────────────────
|
|
211
|
+
|
|
124
212
|
def _build_forbidden_patterns(self) -> list[tuple[str, str, Severity]]:
|
|
125
213
|
"""Build the complete list of forbidden patterns."""
|
|
126
214
|
patterns = list(DEFAULT_FORBIDDEN_PATTERNS)
|
|
@@ -143,7 +231,6 @@ class ForbiddenFileScanner(Scanner):
|
|
|
143
231
|
@staticmethod
|
|
144
232
|
def _is_allowed(file_path: str, allow_patterns: list[str]) -> bool:
|
|
145
233
|
"""Check whether a file path matches any allow pattern."""
|
|
146
|
-
# Check the full path and just the filename
|
|
147
234
|
basename = PurePosixPath(file_path).name
|
|
148
235
|
|
|
149
236
|
for pattern in allow_patterns:
|
|
@@ -178,7 +265,7 @@ class ForbiddenFileScanner(Scanner):
|
|
|
178
265
|
# Glob match on basename
|
|
179
266
|
if fnmatch.fnmatch(basename, pattern):
|
|
180
267
|
return pattern, reason, severity
|
|
181
|
-
# Glob match on full path
|
|
268
|
+
# Glob match on full path
|
|
182
269
|
if fnmatch.fnmatch(file_path, pattern):
|
|
183
270
|
return pattern, reason, severity
|
|
184
271
|
# Also try matching against path components
|
|
@@ -2,13 +2,16 @@
|
|
|
2
2
|
Secret scanner.
|
|
3
3
|
|
|
4
4
|
Detects secrets in staged file content using:
|
|
5
|
-
- Curated legacy patterns (SECRET_PATTERNS)
|
|
5
|
+
- Curated legacy patterns (SECRET_PATTERNS)
|
|
6
6
|
- Private-key block detection (PRIVATE_KEY_RULE)
|
|
7
|
-
- Provider-specific rules (PROVIDER_RULES)
|
|
7
|
+
- Provider-specific rules (PROVIDER_RULES)
|
|
8
8
|
- Shannon-entropy analysis with context-aware scoring
|
|
9
9
|
|
|
10
10
|
Works on raw text content — language-agnostic.
|
|
11
11
|
|
|
12
|
+
Files that Git classifies as binary are decoded if they carry a text BOM
|
|
13
|
+
(UTF-16 LE/BE, UTF-8 with BOM) so PowerShell-generated files still scan.
|
|
14
|
+
|
|
12
15
|
Secrets are NEVER logged in full. All evidence is redacted.
|
|
13
16
|
"""
|
|
14
17
|
|
|
@@ -31,10 +34,7 @@ from gitrupt.scanners.secret_rules import (
|
|
|
31
34
|
|
|
32
35
|
logger = logging.getLogger(__name__)
|
|
33
36
|
|
|
34
|
-
# Maximum file size to scan for secrets (bytes). Large files are skipped.
|
|
35
37
|
MAX_CONTENT_SIZE = 5 * 1024 * 1024 # 5 MB
|
|
36
|
-
|
|
37
|
-
# Legacy entropy thresholds (kept for backward compatibility with earlier tests)
|
|
38
38
|
ENTROPY_THRESHOLD_HIGH = 4.5
|
|
39
39
|
ENTROPY_THRESHOLD_CRITICAL = 5.0
|
|
40
40
|
|
|
@@ -50,8 +50,6 @@ class SecretPattern:
|
|
|
50
50
|
confidence: float
|
|
51
51
|
description: str
|
|
52
52
|
recommendation: str
|
|
53
|
-
# Group index that contains the actual secret value (for redaction).
|
|
54
|
-
# 0 means use the whole match.
|
|
55
53
|
secret_group: int = 0
|
|
56
54
|
|
|
57
55
|
|
|
@@ -64,7 +62,6 @@ def _compile(pattern: str) -> re.Pattern[str]:
|
|
|
64
62
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
65
63
|
|
|
66
64
|
SECRET_PATTERNS: list[SecretPattern] = [
|
|
67
|
-
# ── Private Keys ─────────────────────────────────────────────────────────
|
|
68
65
|
SecretPattern(
|
|
69
66
|
rule_id="private-key-pem",
|
|
70
67
|
name="PEM Private Key",
|
|
@@ -72,7 +69,7 @@ SECRET_PATTERNS: list[SecretPattern] = [
|
|
|
72
69
|
severity=Severity.CRITICAL,
|
|
73
70
|
confidence=0.99,
|
|
74
71
|
description="PEM-encoded private key detected in file content",
|
|
75
|
-
recommendation="Remove the private key and rotate/revoke it immediately.
|
|
72
|
+
recommendation="Remove the private key and rotate/revoke it immediately.",
|
|
76
73
|
secret_group=0,
|
|
77
74
|
),
|
|
78
75
|
SecretPattern(
|
|
@@ -85,8 +82,6 @@ SECRET_PATTERNS: list[SecretPattern] = [
|
|
|
85
82
|
recommendation="Remove the private key from this file and use a secrets manager.",
|
|
86
83
|
secret_group=0,
|
|
87
84
|
),
|
|
88
|
-
|
|
89
|
-
# ── AWS Credentials ──────────────────────────────────────────────────────
|
|
90
85
|
SecretPattern(
|
|
91
86
|
rule_id="aws-access-key-id",
|
|
92
87
|
name="AWS Access Key ID",
|
|
@@ -110,8 +105,6 @@ SECRET_PATTERNS: list[SecretPattern] = [
|
|
|
110
105
|
recommendation="Revoke this AWS key immediately. Use IAM roles or environment variables.",
|
|
111
106
|
secret_group=1,
|
|
112
107
|
),
|
|
113
|
-
|
|
114
|
-
# ── GitHub Tokens ────────────────────────────────────────────────────────
|
|
115
108
|
SecretPattern(
|
|
116
109
|
rule_id="github-pat-classic",
|
|
117
110
|
name="GitHub Personal Access Token (Classic)",
|
|
@@ -152,8 +145,6 @@ SECRET_PATTERNS: list[SecretPattern] = [
|
|
|
152
145
|
recommendation="This token is short-lived but should not be committed",
|
|
153
146
|
secret_group=1,
|
|
154
147
|
),
|
|
155
|
-
|
|
156
|
-
# ── Generic API Keys & Tokens ────────────────────────────────────────────
|
|
157
148
|
SecretPattern(
|
|
158
149
|
rule_id="generic-api-key",
|
|
159
150
|
name="Generic API Key",
|
|
@@ -167,8 +158,6 @@ SECRET_PATTERNS: list[SecretPattern] = [
|
|
|
167
158
|
recommendation="Move this value to environment variables or a secrets manager.",
|
|
168
159
|
secret_group=1,
|
|
169
160
|
),
|
|
170
|
-
|
|
171
|
-
# ── Database URLs ────────────────────────────────────────────────────────
|
|
172
161
|
SecretPattern(
|
|
173
162
|
rule_id="database-url-with-password",
|
|
174
163
|
name="Database URL with Password",
|
|
@@ -179,11 +168,9 @@ SECRET_PATTERNS: list[SecretPattern] = [
|
|
|
179
168
|
severity=Severity.CRITICAL,
|
|
180
169
|
confidence=0.92,
|
|
181
170
|
description="Database connection URL with embedded password detected",
|
|
182
|
-
recommendation="Use environment variables for database credentials.
|
|
171
|
+
recommendation="Use environment variables for database credentials.",
|
|
183
172
|
secret_group=1,
|
|
184
173
|
),
|
|
185
|
-
|
|
186
|
-
# ── JWT Tokens ───────────────────────────────────────────────────────────
|
|
187
174
|
SecretPattern(
|
|
188
175
|
rule_id="jwt-token",
|
|
189
176
|
name="JWT Token",
|
|
@@ -191,11 +178,9 @@ SECRET_PATTERNS: list[SecretPattern] = [
|
|
|
191
178
|
severity=Severity.HIGH,
|
|
192
179
|
confidence=0.88,
|
|
193
180
|
description="JSON Web Token detected",
|
|
194
|
-
recommendation="JWT tokens may contain sensitive claims. Do not commit tokens
|
|
181
|
+
recommendation="JWT tokens may contain sensitive claims. Do not commit tokens.",
|
|
195
182
|
secret_group=1,
|
|
196
183
|
),
|
|
197
|
-
|
|
198
|
-
# ── Google Credentials ───────────────────────────────────────────────────
|
|
199
184
|
SecretPattern(
|
|
200
185
|
rule_id="google-api-key",
|
|
201
186
|
name="Google API Key",
|
|
@@ -216,8 +201,6 @@ SECRET_PATTERNS: list[SecretPattern] = [
|
|
|
216
201
|
recommendation="Revoke and rotate the OAuth client secret.",
|
|
217
202
|
secret_group=1,
|
|
218
203
|
),
|
|
219
|
-
|
|
220
|
-
# ── Slack ────────────────────────────────────────────────────────────────
|
|
221
204
|
SecretPattern(
|
|
222
205
|
rule_id="slack-token",
|
|
223
206
|
name="Slack Token",
|
|
@@ -238,8 +221,6 @@ SECRET_PATTERNS: list[SecretPattern] = [
|
|
|
238
221
|
recommendation="Regenerate this webhook URL",
|
|
239
222
|
secret_group=0,
|
|
240
223
|
),
|
|
241
|
-
|
|
242
|
-
# ── Stripe ───────────────────────────────────────────────────────────────
|
|
243
224
|
SecretPattern(
|
|
244
225
|
rule_id="stripe-secret-key",
|
|
245
226
|
name="Stripe Secret Key",
|
|
@@ -250,8 +231,6 @@ SECRET_PATTERNS: list[SecretPattern] = [
|
|
|
250
231
|
recommendation="Revoke this key at dashboard.stripe.com/apikeys",
|
|
251
232
|
secret_group=1,
|
|
252
233
|
),
|
|
253
|
-
|
|
254
|
-
# ── Generic password assignments ────────────────────────────────────────
|
|
255
234
|
SecretPattern(
|
|
256
235
|
rule_id="hardcoded-password",
|
|
257
236
|
name="Hardcoded Password",
|
|
@@ -261,12 +240,47 @@ SECRET_PATTERNS: list[SecretPattern] = [
|
|
|
261
240
|
severity=Severity.HIGH,
|
|
262
241
|
confidence=0.72,
|
|
263
242
|
description="Possible hardcoded password detected",
|
|
264
|
-
recommendation="Replace hardcoded passwords with environment variables
|
|
243
|
+
recommendation="Replace hardcoded passwords with environment variables.",
|
|
265
244
|
secret_group=1,
|
|
266
245
|
),
|
|
267
246
|
]
|
|
268
247
|
|
|
269
248
|
|
|
249
|
+
# ─────────────────────────────────────────────────────────────────────────────
|
|
250
|
+
# UTF-16 / BOM decoding — NEW
|
|
251
|
+
# ─────────────────────────────────────────────────────────────────────────────
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _decode_text_with_bom(raw: bytes) -> str | None:
|
|
255
|
+
"""
|
|
256
|
+
If `raw` starts with a text BOM, decode it as text. Otherwise return None.
|
|
257
|
+
|
|
258
|
+
Handles:
|
|
259
|
+
- UTF-16 LE (FF FE)
|
|
260
|
+
- UTF-16 BE (FE FF)
|
|
261
|
+
- UTF-8 with BOM (EF BB BF)
|
|
262
|
+
|
|
263
|
+
Returns the decoded string, or None if the bytes are not BOM-prefixed text.
|
|
264
|
+
Never raises.
|
|
265
|
+
"""
|
|
266
|
+
if not raw or len(raw) < 2:
|
|
267
|
+
return None
|
|
268
|
+
|
|
269
|
+
if raw[:2] in (b"\xff\xfe", b"\xfe\xff"):
|
|
270
|
+
try:
|
|
271
|
+
return raw.decode("utf-16")
|
|
272
|
+
except UnicodeDecodeError:
|
|
273
|
+
return None
|
|
274
|
+
|
|
275
|
+
if len(raw) >= 3 and raw[:3] == b"\xef\xbb\xbf":
|
|
276
|
+
try:
|
|
277
|
+
return raw[3:].decode("utf-8")
|
|
278
|
+
except UnicodeDecodeError:
|
|
279
|
+
return None
|
|
280
|
+
|
|
281
|
+
return None
|
|
282
|
+
|
|
283
|
+
|
|
270
284
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
271
285
|
# Scanner
|
|
272
286
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
@@ -276,13 +290,14 @@ class SecretScanner(Scanner):
|
|
|
276
290
|
"""
|
|
277
291
|
Scans staged file content for secrets and credentials.
|
|
278
292
|
|
|
279
|
-
Four detection passes per
|
|
280
|
-
1. SECRET_PATTERNS — curated legacy patterns
|
|
293
|
+
Four detection passes per line:
|
|
294
|
+
1. SECRET_PATTERNS — curated legacy patterns
|
|
281
295
|
2. PRIVATE_KEY_RULE — PEM private-key blocks
|
|
282
|
-
3. PROVIDER_RULES — AI
|
|
296
|
+
3. PROVIDER_RULES — AI, payments, comms, cloud, tooling
|
|
283
297
|
4. Entropy — Shannon entropy with context-aware scoring
|
|
284
298
|
|
|
285
|
-
|
|
299
|
+
Files classified by Git as binary are decoded if they carry a text BOM
|
|
300
|
+
(PowerShell writes UTF-16 by default), then scanned line-by-line.
|
|
286
301
|
"""
|
|
287
302
|
|
|
288
303
|
@property
|
|
@@ -299,8 +314,6 @@ class SecretScanner(Scanner):
|
|
|
299
314
|
for staged_file in target.staged_files:
|
|
300
315
|
if staged_file.status == "D":
|
|
301
316
|
continue
|
|
302
|
-
if staged_file.is_binary:
|
|
303
|
-
continue
|
|
304
317
|
if staged_file.size_bytes > MAX_CONTENT_SIZE:
|
|
305
318
|
logger.warning(
|
|
306
319
|
"Skipping secret scan of %s: file too large (%d bytes)",
|
|
@@ -309,134 +322,163 @@ class SecretScanner(Scanner):
|
|
|
309
322
|
)
|
|
310
323
|
continue
|
|
311
324
|
|
|
312
|
-
|
|
313
|
-
|
|
325
|
+
# ── Binary file path: try BOM decode ────────────────────────────
|
|
326
|
+
if staged_file.is_binary:
|
|
327
|
+
decoded = self._try_decode_binary(target, staged_file.path)
|
|
328
|
+
if decoded is None:
|
|
329
|
+
continue
|
|
330
|
+
findings.extend(
|
|
331
|
+
self._scan_text_lines(staged_file.path, decoded)
|
|
332
|
+
)
|
|
333
|
+
continue
|
|
334
|
+
|
|
335
|
+
# ── Normal text file: scan the diff for added lines ─────────────
|
|
336
|
+
findings.extend(
|
|
337
|
+
self._scan_diff_for_file(target.staged_diff, staged_file.path)
|
|
314
338
|
)
|
|
315
|
-
findings.extend(file_findings)
|
|
316
339
|
|
|
317
340
|
return findings
|
|
318
341
|
|
|
319
|
-
# ──
|
|
342
|
+
# ── BOM-decode path ──────────────────────────────────────────────────────
|
|
343
|
+
|
|
344
|
+
def _try_decode_binary(self, target: ScanTarget, path: str) -> str | None:
|
|
345
|
+
"""Fetch raw staged bytes; decode if BOM-prefixed text."""
|
|
346
|
+
from gitrupt.git import GitAdapter
|
|
347
|
+
|
|
348
|
+
raw = GitAdapter.get_staged_file_bytes(target.repo_root, path)
|
|
349
|
+
if not raw:
|
|
350
|
+
return None
|
|
351
|
+
return _decode_text_with_bom(raw)
|
|
352
|
+
|
|
353
|
+
def _scan_text_lines(self, file_path: str, text: str) -> list[Finding]:
|
|
354
|
+
"""
|
|
355
|
+
Scan decoded text line-by-line (no diff parsing — we scan every line).
|
|
356
|
+
"""
|
|
357
|
+
findings: list[Finding] = []
|
|
358
|
+
seen: set[tuple[str, str]] = set()
|
|
359
|
+
|
|
360
|
+
for line_number, line_content in enumerate(text.splitlines(), start=1):
|
|
361
|
+
findings.extend(
|
|
362
|
+
self._scan_one_line(file_path, line_number, line_content, seen)
|
|
363
|
+
)
|
|
364
|
+
|
|
365
|
+
return findings
|
|
366
|
+
|
|
367
|
+
# ── Diff path ────────────────────────────────────────────────────────────
|
|
320
368
|
|
|
321
369
|
def _scan_diff_for_file(self, diff: str, file_path: str) -> list[Finding]:
|
|
322
|
-
"""Extract added lines from the diff for a specific file and scan them."""
|
|
323
370
|
added_lines = _extract_added_lines(diff, file_path)
|
|
324
371
|
if not added_lines:
|
|
325
372
|
return []
|
|
326
373
|
|
|
327
374
|
findings: list[Finding] = []
|
|
328
|
-
# Dedup key is (rule_id, file_path) — one finding per rule per file,
|
|
329
|
-
# to avoid flooding when a config file contains many similar values.
|
|
330
375
|
seen: set[tuple[str, str]] = set()
|
|
331
376
|
|
|
332
377
|
for line_number, line_content in added_lines:
|
|
378
|
+
findings.extend(
|
|
379
|
+
self._scan_one_line(file_path, line_number, line_content, seen)
|
|
380
|
+
)
|
|
333
381
|
|
|
334
|
-
|
|
335
|
-
for sp in SECRET_PATTERNS:
|
|
336
|
-
match = sp.pattern.search(line_content)
|
|
337
|
-
if not match:
|
|
338
|
-
continue
|
|
339
|
-
|
|
340
|
-
key = (sp.rule_id, file_path)
|
|
341
|
-
if key in seen:
|
|
342
|
-
continue
|
|
382
|
+
return findings
|
|
343
383
|
|
|
344
|
-
|
|
345
|
-
secret_value = (
|
|
346
|
-
match.group(sp.secret_group)
|
|
347
|
-
if sp.secret_group > 0
|
|
348
|
-
else match.group(0)
|
|
349
|
-
)
|
|
350
|
-
except IndexError:
|
|
351
|
-
secret_value = match.group(0)
|
|
384
|
+
# ── Shared line scanner ──────────────────────────────────────────────────
|
|
352
385
|
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
line=line_number,
|
|
362
|
-
message=f"{sp.name} detected in {file_path}",
|
|
363
|
-
description=sp.description,
|
|
364
|
-
evidence=_redact(secret_value),
|
|
365
|
-
recommendation=sp.recommendation,
|
|
366
|
-
can_override=False,
|
|
367
|
-
)
|
|
368
|
-
)
|
|
386
|
+
def _scan_one_line(
|
|
387
|
+
self,
|
|
388
|
+
file_path: str,
|
|
389
|
+
line_number: int,
|
|
390
|
+
line_content: str,
|
|
391
|
+
seen: set[tuple[str, str]],
|
|
392
|
+
) -> list[Finding]:
|
|
393
|
+
findings: list[Finding] = []
|
|
369
394
|
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
PRIVATE_KEY_RULE,
|
|
379
|
-
file_path,
|
|
380
|
-
line_number,
|
|
381
|
-
line_content,
|
|
382
|
-
match=pk_match,
|
|
383
|
-
)
|
|
384
|
-
)
|
|
385
|
-
|
|
386
|
-
# ── Pass 3: provider-specific rules ─────────────────────────────
|
|
387
|
-
for rule in PROVIDER_RULES:
|
|
388
|
-
match = rule.pattern.search(line_content)
|
|
389
|
-
if not match:
|
|
390
|
-
continue
|
|
391
|
-
if not _context_ok(line_content, match, rule):
|
|
392
|
-
continue
|
|
395
|
+
# Pass 1 — curated legacy patterns
|
|
396
|
+
for sp in SECRET_PATTERNS:
|
|
397
|
+
match = sp.pattern.search(line_content)
|
|
398
|
+
if not match:
|
|
399
|
+
continue
|
|
400
|
+
key = (sp.rule_id, file_path)
|
|
401
|
+
if key in seen:
|
|
402
|
+
continue
|
|
393
403
|
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
404
|
+
try:
|
|
405
|
+
secret_value = (
|
|
406
|
+
match.group(sp.secret_group)
|
|
407
|
+
if sp.secret_group > 0
|
|
408
|
+
else match.group(0)
|
|
409
|
+
)
|
|
410
|
+
except IndexError:
|
|
411
|
+
secret_value = match.group(0)
|
|
412
|
+
|
|
413
|
+
seen.add(key)
|
|
414
|
+
findings.append(
|
|
415
|
+
Finding(
|
|
416
|
+
scanner=self.name,
|
|
417
|
+
rule_id=sp.rule_id,
|
|
418
|
+
severity=sp.severity,
|
|
419
|
+
confidence=sp.confidence,
|
|
420
|
+
file=file_path,
|
|
421
|
+
line=line_number,
|
|
422
|
+
message=f"{sp.name} detected in {file_path}",
|
|
423
|
+
description=sp.description,
|
|
424
|
+
evidence=_redact(secret_value),
|
|
425
|
+
recommendation=sp.recommendation,
|
|
426
|
+
can_override=False,
|
|
427
|
+
)
|
|
428
|
+
)
|
|
397
429
|
|
|
430
|
+
# Pass 2 — private key block
|
|
431
|
+
pk_match = PRIVATE_KEY_RULE.pattern.search(line_content)
|
|
432
|
+
if pk_match:
|
|
433
|
+
key = (PRIVATE_KEY_RULE.rule_id, file_path)
|
|
434
|
+
if key not in seen:
|
|
398
435
|
seen.add(key)
|
|
399
436
|
findings.append(
|
|
400
|
-
_rule_to_finding(
|
|
401
|
-
rule,
|
|
402
|
-
file_path,
|
|
403
|
-
line_number,
|
|
404
|
-
line_content,
|
|
405
|
-
match=match,
|
|
406
|
-
)
|
|
437
|
+
_rule_to_finding(PRIVATE_KEY_RULE, file_path, line_number, pk_match)
|
|
407
438
|
)
|
|
408
439
|
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
440
|
+
# Pass 3 — provider rules
|
|
441
|
+
for rule in PROVIDER_RULES:
|
|
442
|
+
match = rule.pattern.search(line_content)
|
|
443
|
+
if not match:
|
|
444
|
+
continue
|
|
445
|
+
if not _context_ok(line_content, match, rule):
|
|
446
|
+
continue
|
|
447
|
+
key = (rule.rule_id, file_path)
|
|
448
|
+
if key in seen:
|
|
449
|
+
continue
|
|
450
|
+
seen.add(key)
|
|
451
|
+
findings.append(
|
|
452
|
+
_rule_to_finding(rule, file_path, line_number, match)
|
|
453
|
+
)
|
|
414
454
|
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
)
|
|
455
|
+
# Pass 4 — entropy
|
|
456
|
+
for hit in find_entropy_secrets(line_content):
|
|
457
|
+
key = ("generic-high-entropy", file_path)
|
|
458
|
+
if key in seen:
|
|
459
|
+
continue
|
|
460
|
+
seen.add(key)
|
|
461
|
+
findings.append(
|
|
462
|
+
Finding(
|
|
463
|
+
scanner=self.name,
|
|
464
|
+
rule_id="generic-high-entropy",
|
|
465
|
+
severity=Severity.MEDIUM,
|
|
466
|
+
confidence=0.7 if hit.reason == "context-assisted" else 0.55,
|
|
467
|
+
file=file_path,
|
|
468
|
+
line=line_number,
|
|
469
|
+
message="High-entropy string — possible secret",
|
|
470
|
+
description=(
|
|
471
|
+
f"Entropy {hit.entropy:.2f} bits/char ({hit.reason}). "
|
|
472
|
+
"May be an unrecognised API key or token."
|
|
473
|
+
),
|
|
474
|
+
evidence=_redact(hit.value),
|
|
475
|
+
recommendation=(
|
|
476
|
+
"Verify whether this is a secret. If it is, move it to "
|
|
477
|
+
"an environment variable and allowlist the shape."
|
|
478
|
+
),
|
|
479
|
+
can_override=True,
|
|
439
480
|
)
|
|
481
|
+
)
|
|
440
482
|
|
|
441
483
|
return findings
|
|
442
484
|
|
|
@@ -447,10 +489,6 @@ class SecretScanner(Scanner):
|
|
|
447
489
|
|
|
448
490
|
|
|
449
491
|
def _context_ok(line: str, match: re.Match[str], rule: SecretRule) -> bool:
|
|
450
|
-
"""
|
|
451
|
-
True if the rule has no context requirement, or if one of its context
|
|
452
|
-
words appears within CONTEXT_WINDOW characters of the match.
|
|
453
|
-
"""
|
|
454
492
|
if not rule.context_words:
|
|
455
493
|
return True
|
|
456
494
|
start = max(0, match.start() - CONTEXT_WINDOW)
|
|
@@ -463,13 +501,9 @@ def _rule_to_finding(
|
|
|
463
501
|
rule: SecretRule,
|
|
464
502
|
file_path: str,
|
|
465
503
|
line_number: int,
|
|
466
|
-
|
|
467
|
-
match: re.Match[str] | None = None,
|
|
504
|
+
match: re.Match[str],
|
|
468
505
|
) -> Finding:
|
|
469
|
-
|
|
470
|
-
if match is None:
|
|
471
|
-
match = rule.pattern.search(line_content)
|
|
472
|
-
raw_value = match.group(0) if match else line_content
|
|
506
|
+
raw_value = match.group(0)
|
|
473
507
|
return Finding(
|
|
474
508
|
scanner="secrets",
|
|
475
509
|
rule_id=rule.rule_id,
|
|
@@ -486,12 +520,6 @@ def _rule_to_finding(
|
|
|
486
520
|
|
|
487
521
|
|
|
488
522
|
def _extract_added_lines(diff: str, file_path: str) -> list[tuple[int, str]]:
|
|
489
|
-
"""
|
|
490
|
-
Extract lines that were added (starting with '+') for a specific file
|
|
491
|
-
from a unified diff.
|
|
492
|
-
|
|
493
|
-
Returns a list of (line_number, content) tuples.
|
|
494
|
-
"""
|
|
495
523
|
if not diff:
|
|
496
524
|
return []
|
|
497
525
|
|
|
@@ -526,33 +554,22 @@ def _extract_added_lines(diff: str, file_path: str) -> list[tuple[int, str]]:
|
|
|
526
554
|
|
|
527
555
|
|
|
528
556
|
def _redact(secret: str) -> str:
|
|
529
|
-
"""
|
|
530
|
-
Redact a secret value for safe display.
|
|
531
|
-
|
|
532
|
-
Shows the first few characters followed by bullets.
|
|
533
|
-
Never shows the full value.
|
|
534
|
-
"""
|
|
535
557
|
if not secret:
|
|
536
558
|
return "••••••••"
|
|
537
|
-
|
|
538
559
|
visible = min(4, len(secret) // 4)
|
|
539
560
|
hidden = len(secret) - visible
|
|
540
561
|
return secret[:visible] + "•" * min(hidden, 20)
|
|
541
562
|
|
|
542
563
|
|
|
543
564
|
def shannon_entropy(data: str) -> float:
|
|
544
|
-
"""Calculate the Shannon entropy of a string."""
|
|
545
565
|
if not data:
|
|
546
566
|
return 0.0
|
|
547
|
-
|
|
548
567
|
freq: dict[str, int] = {}
|
|
549
568
|
for char in data:
|
|
550
569
|
freq[char] = freq.get(char, 0) + 1
|
|
551
|
-
|
|
552
570
|
entropy = 0.0
|
|
553
571
|
length = len(data)
|
|
554
572
|
for count in freq.values():
|
|
555
573
|
probability = count / length
|
|
556
574
|
entropy -= probability * math.log2(probability)
|
|
557
|
-
|
|
558
575
|
return entropy
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|