envleak-cli 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- envleak_cli-0.1.0/.github/workflows/scan.yml +10 -0
- envleak_cli-0.1.0/.gitignore +8 -0
- envleak_cli-0.1.0/LICENSE +21 -0
- envleak_cli-0.1.0/PKG-INFO +147 -0
- envleak_cli-0.1.0/README.md +129 -0
- envleak_cli-0.1.0/envleak/__init__.py +3 -0
- envleak_cli-0.1.0/envleak/__main__.py +4 -0
- envleak_cli-0.1.0/envleak/cli.py +72 -0
- envleak_cli-0.1.0/envleak/report.py +165 -0
- envleak_cli-0.1.0/envleak/rules.py +162 -0
- envleak_cli-0.1.0/envleak/scanner.py +268 -0
- envleak_cli-0.1.0/pyproject.toml +31 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Diego
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: envleak-cli
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Find exposed API keys and secrets before someone else does. 100% local.
|
|
5
|
+
Project-URL: Homepage, https://github.com/novasdiego1/envleak
|
|
6
|
+
Project-URL: Issues, https://github.com/novasdiego1/envleak/issues
|
|
7
|
+
License: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Keywords: ai-agents,api-keys,devsecops,llm,secrets,security
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Topic :: Security
|
|
16
|
+
Requires-Python: >=3.8
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
|
|
19
|
+
# envleak
|
|
20
|
+
|
|
21
|
+
**Find your exposed API keys before someone else does.**
|
|
22
|
+
|
|
23
|
+
One command. No config. Nothing ever leaves your machine.
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pipx run envleak
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
```
|
|
30
|
+
┌────────────────────────────────────────────┐
|
|
31
|
+
│ E N V L E A K S C A N │
|
|
32
|
+
└────────────────────────────────────────────┘
|
|
33
|
+
|
|
34
|
+
F █░░░░ Crítico
|
|
35
|
+
|
|
36
|
+
5 hallazgo(s) en 128 archivos
|
|
37
|
+
5 critical
|
|
38
|
+
|
|
39
|
+
⚡ 2 en la superficie de agentes/LLM
|
|
40
|
+
|
|
41
|
+
✗ ARCHIVOS DE ENTORNO RASTREADOS POR GIT:
|
|
42
|
+
.env
|
|
43
|
+
|
|
44
|
+
CRIT Anthropic API key
|
|
45
|
+
.env:2
|
|
46
|
+
ANTHROPIC_API_KEY=sk-a********o9Pq
|
|
47
|
+
|
|
48
|
+
CRIT GitHub token
|
|
49
|
+
.config/mcp.json:1
|
|
50
|
+
{"mcpServers":{"gh":{"env":{"GITHUB_TOKEN":"ghp_********3zA5"}}}}
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
---
|
|
54
|
+
|
|
55
|
+
## Why another secret scanner?
|
|
56
|
+
|
|
57
|
+
There are good ones already — `gitleaks` and `trufflehog` scan git history
|
|
58
|
+
and belong in your CI pipeline. `envleak` is for a different moment: **right
|
|
59
|
+
now, on your laptop, in ten seconds, with a grade you can screenshot.**
|
|
60
|
+
|
|
61
|
+
Two things it does that the others don't:
|
|
62
|
+
|
|
63
|
+
**1. It scans the agent surface.** Everyone is shipping AI agents in 2026,
|
|
64
|
+
and the credentials moved with them — MCP server configs, `claude_desktop_config.json`,
|
|
65
|
+
n8n and LangGraph workflows, notebooks, LLM provider keys pasted into JSON.
|
|
66
|
+
`envleak` knows what an OpenAI, Anthropic, Groq, LangSmith or Hugging Face key
|
|
67
|
+
looks like and where agent tooling hides them.
|
|
68
|
+
|
|
69
|
+
**2. It gives you a grade, not a JSON dump.** A 400-line report gets closed.
|
|
70
|
+
An `F` gets fixed.
|
|
71
|
+
|
|
72
|
+
## Install
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
pipx run envleak # no install, just run it
|
|
76
|
+
pip install envleak # or keep it around
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Python 3.8+. Zero dependencies.
|
|
80
|
+
|
|
81
|
+
## Use
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
envleak # scan current directory
|
|
85
|
+
envleak ~/code/myproject # scan somewhere else
|
|
86
|
+
envleak --markdown # scorecard ready to paste in an issue or PR
|
|
87
|
+
envleak --json # machine-readable, for pipelines
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
### In CI
|
|
91
|
+
|
|
92
|
+
Exits `1` when it finds anything `high` or worse:
|
|
93
|
+
|
|
94
|
+
```yaml
|
|
95
|
+
- run: pipx run envleak --fail-on critical
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
envleak --fail-on none # report only, never fail
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
## What it looks for
|
|
103
|
+
|
|
104
|
+
| Surface | Examples |
|
|
105
|
+
|---|---|
|
|
106
|
+
| **AI / agents** | OpenAI, Anthropic, Google AI, Groq, Mistral, Hugging Face, Replicate, LangSmith, Perplexity, MCP configs |
|
|
107
|
+
| **Cloud / infra** | AWS access keys, GCP service accounts, private key blocks, Docker registry auth |
|
|
108
|
+
| **Classic** | GitHub, GitLab, Stripe, Slack, Telegram, SendGrid, Twilio, JWTs, database URLs with passwords |
|
|
109
|
+
| **Generic** | High-entropy values assigned to `*_KEY`, `*_TOKEN`, `*_SECRET`, `*_PASSWORD` — in config files only |
|
|
110
|
+
|
|
111
|
+
It also flags the thing that actually causes breaches: **`.env` files tracked
|
|
112
|
+
by git.**
|
|
113
|
+
|
|
114
|
+
## Privacy
|
|
115
|
+
|
|
116
|
+
This is a security tool, so the guarantee matters more than the feature list:
|
|
117
|
+
|
|
118
|
+
- **It makes no network calls.** None. Read the source — it's a few hundred
|
|
119
|
+
lines with zero dependencies.
|
|
120
|
+
- **Secrets are redacted in output.** You see `sk-a********o9Pq`, never the
|
|
121
|
+
full key, so a screenshot is safe to post.
|
|
122
|
+
- Nothing is written anywhere except stdout.
|
|
123
|
+
|
|
124
|
+
## False positives
|
|
125
|
+
|
|
126
|
+
The generic entropy check runs **only in config files** (`.env`, `.json`,
|
|
127
|
+
`.yaml`, `.toml`, `.ini`). In source code it fires on ordinary expressions
|
|
128
|
+
like `key = key.upper()` and buries the real findings, so it doesn't run there.
|
|
129
|
+
|
|
130
|
+
Template files (`.env.example`, `*.sample`, `*.template`) are downgraded, and
|
|
131
|
+
documentation credentials (`user:password@localhost`) are ignored.
|
|
132
|
+
|
|
133
|
+
Found a false positive? [Open an issue](https://github.com/novasdiego1/envleak/issues)
|
|
134
|
+
with the pattern — that feedback is the whole roadmap.
|
|
135
|
+
|
|
136
|
+
## What it is not
|
|
137
|
+
|
|
138
|
+
`envleak` scans your **working tree**, not git history. If a key was committed
|
|
139
|
+
and later deleted, it's still in your history and still compromised — use
|
|
140
|
+
`trufflehog` for that, and rotate the key regardless.
|
|
141
|
+
|
|
142
|
+
Finding a key is step one. **Rotate it.** A key that appeared in this scan
|
|
143
|
+
should be considered burned.
|
|
144
|
+
|
|
145
|
+
## License
|
|
146
|
+
|
|
147
|
+
MIT
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
# envleak
|
|
2
|
+
|
|
3
|
+
**Find your exposed API keys before someone else does.**
|
|
4
|
+
|
|
5
|
+
One command. No config. Nothing ever leaves your machine.
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pipx run envleak
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
```
|
|
12
|
+
┌────────────────────────────────────────────┐
|
|
13
|
+
│ E N V L E A K S C A N │
|
|
14
|
+
└────────────────────────────────────────────┘
|
|
15
|
+
|
|
16
|
+
F █░░░░ Crítico
|
|
17
|
+
|
|
18
|
+
5 hallazgo(s) en 128 archivos
|
|
19
|
+
5 critical
|
|
20
|
+
|
|
21
|
+
⚡ 2 en la superficie de agentes/LLM
|
|
22
|
+
|
|
23
|
+
✗ ARCHIVOS DE ENTORNO RASTREADOS POR GIT:
|
|
24
|
+
.env
|
|
25
|
+
|
|
26
|
+
CRIT Anthropic API key
|
|
27
|
+
.env:2
|
|
28
|
+
ANTHROPIC_API_KEY=sk-a********o9Pq
|
|
29
|
+
|
|
30
|
+
CRIT GitHub token
|
|
31
|
+
.config/mcp.json:1
|
|
32
|
+
{"mcpServers":{"gh":{"env":{"GITHUB_TOKEN":"ghp_********3zA5"}}}}
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
---
|
|
36
|
+
|
|
37
|
+
## Why another secret scanner?
|
|
38
|
+
|
|
39
|
+
There are good ones already — `gitleaks` and `trufflehog` scan git history
|
|
40
|
+
and belong in your CI pipeline. `envleak` is for a different moment: **right
|
|
41
|
+
now, on your laptop, in ten seconds, with a grade you can screenshot.**
|
|
42
|
+
|
|
43
|
+
Two things it does that the others don't:
|
|
44
|
+
|
|
45
|
+
**1. It scans the agent surface.** Everyone is shipping AI agents in 2026,
|
|
46
|
+
and the credentials moved with them — MCP server configs, `claude_desktop_config.json`,
|
|
47
|
+
n8n and LangGraph workflows, notebooks, LLM provider keys pasted into JSON.
|
|
48
|
+
`envleak` knows what an OpenAI, Anthropic, Groq, LangSmith or Hugging Face key
|
|
49
|
+
looks like and where agent tooling hides them.
|
|
50
|
+
|
|
51
|
+
**2. It gives you a grade, not a JSON dump.** A 400-line report gets closed.
|
|
52
|
+
An `F` gets fixed.
|
|
53
|
+
|
|
54
|
+
## Install
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
pipx run envleak # no install, just run it
|
|
58
|
+
pip install envleak # or keep it around
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Python 3.8+. Zero dependencies.
|
|
62
|
+
|
|
63
|
+
## Use
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
envleak # scan current directory
|
|
67
|
+
envleak ~/code/myproject # scan somewhere else
|
|
68
|
+
envleak --markdown # scorecard ready to paste in an issue or PR
|
|
69
|
+
envleak --json # machine-readable, for pipelines
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
### In CI
|
|
73
|
+
|
|
74
|
+
Exits `1` when it finds anything `high` or worse:
|
|
75
|
+
|
|
76
|
+
```yaml
|
|
77
|
+
- run: pipx run envleak --fail-on critical
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
envleak --fail-on none # report only, never fail
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
## What it looks for
|
|
85
|
+
|
|
86
|
+
| Surface | Examples |
|
|
87
|
+
|---|---|
|
|
88
|
+
| **AI / agents** | OpenAI, Anthropic, Google AI, Groq, Mistral, Hugging Face, Replicate, LangSmith, Perplexity, MCP configs |
|
|
89
|
+
| **Cloud / infra** | AWS access keys, GCP service accounts, private key blocks, Docker registry auth |
|
|
90
|
+
| **Classic** | GitHub, GitLab, Stripe, Slack, Telegram, SendGrid, Twilio, JWTs, database URLs with passwords |
|
|
91
|
+
| **Generic** | High-entropy values assigned to `*_KEY`, `*_TOKEN`, `*_SECRET`, `*_PASSWORD` — in config files only |
|
|
92
|
+
|
|
93
|
+
It also flags the thing that actually causes breaches: **`.env` files tracked
|
|
94
|
+
by git.**
|
|
95
|
+
|
|
96
|
+
## Privacy
|
|
97
|
+
|
|
98
|
+
This is a security tool, so the guarantee matters more than the feature list:
|
|
99
|
+
|
|
100
|
+
- **It makes no network calls.** None. Read the source — it's a few hundred
|
|
101
|
+
lines with zero dependencies.
|
|
102
|
+
- **Secrets are redacted in output.** You see `sk-a********o9Pq`, never the
|
|
103
|
+
full key, so a screenshot is safe to post.
|
|
104
|
+
- Nothing is written anywhere except stdout.
|
|
105
|
+
|
|
106
|
+
## False positives
|
|
107
|
+
|
|
108
|
+
The generic entropy check runs **only in config files** (`.env`, `.json`,
|
|
109
|
+
`.yaml`, `.toml`, `.ini`). In source code it fires on ordinary expressions
|
|
110
|
+
like `key = key.upper()` and buries the real findings, so it doesn't run there.
|
|
111
|
+
|
|
112
|
+
Template files (`.env.example`, `*.sample`, `*.template`) are downgraded, and
|
|
113
|
+
documentation credentials (`user:password@localhost`) are ignored.
|
|
114
|
+
|
|
115
|
+
Found a false positive? [Open an issue](https://github.com/novasdiego1/envleak/issues)
|
|
116
|
+
with the pattern — that feedback is the whole roadmap.
|
|
117
|
+
|
|
118
|
+
## What it is not
|
|
119
|
+
|
|
120
|
+
`envleak` scans your **working tree**, not git history. If a key was committed
|
|
121
|
+
and later deleted, it's still in your history and still compromised — use
|
|
122
|
+
`trufflehog` for that, and rotate the key regardless.
|
|
123
|
+
|
|
124
|
+
Finding a key is step one. **Rotate it.** A key that appeared in this scan
|
|
125
|
+
should be considered burned.
|
|
126
|
+
|
|
127
|
+
## License
|
|
128
|
+
|
|
129
|
+
MIT
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""envleak — find exposed credentials before someone else does.
|
|
2
|
+
|
|
3
|
+
Runs entirely on your machine. Nothing is uploaded, ever.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import sys
|
|
8
|
+
|
|
9
|
+
from . import __version__
|
|
10
|
+
from .report import compute_score, render_json, render_markdown, render_terminal
|
|
11
|
+
from .scanner import scan
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def main(argv=None) -> int:
|
|
15
|
+
p = argparse.ArgumentParser(
|
|
16
|
+
prog="envleak",
|
|
17
|
+
description="Escanea credenciales expuestas. 100%% local, no sube nada.",
|
|
18
|
+
)
|
|
19
|
+
p.add_argument("path", nargs="?", default=".", help="directorio a escanear (default: .)")
|
|
20
|
+
p.add_argument("--json", action="store_true", help="salida JSON")
|
|
21
|
+
p.add_argument("--markdown", "--md", action="store_true", dest="markdown",
|
|
22
|
+
help="scorecard en markdown, listo para pegar")
|
|
23
|
+
p.add_argument("--no-color", action="store_true", help="sin colores ANSI")
|
|
24
|
+
p.add_argument("--fail-on", choices=["none", "medium", "high", "critical"],
|
|
25
|
+
default="high",
|
|
26
|
+
help="código de salida 1 si hay hallazgos de este nivel o peor "
|
|
27
|
+
"(default: high). Útil en CI.")
|
|
28
|
+
p.add_argument("--max-files", type=int, default=20000)
|
|
29
|
+
p.add_argument("--version", action="version", version=f"envleak {__version__}")
|
|
30
|
+
args = p.parse_args(argv)
|
|
31
|
+
|
|
32
|
+
try:
|
|
33
|
+
result = scan(args.path, max_files=args.max_files)
|
|
34
|
+
except FileNotFoundError:
|
|
35
|
+
print(f"envleak: no existe la ruta {args.path}", file=sys.stderr)
|
|
36
|
+
return 2
|
|
37
|
+
except PermissionError:
|
|
38
|
+
print(f"envleak: sin permiso para leer {args.path}", file=sys.stderr)
|
|
39
|
+
return 2
|
|
40
|
+
|
|
41
|
+
if args.json:
|
|
42
|
+
print(render_json(result))
|
|
43
|
+
elif args.markdown:
|
|
44
|
+
print(render_markdown(result))
|
|
45
|
+
else:
|
|
46
|
+
color = not args.no_color and sys.stdout.isatty()
|
|
47
|
+
print(render_terminal(result, color=color))
|
|
48
|
+
|
|
49
|
+
if args.fail_on == "none":
|
|
50
|
+
return 0
|
|
51
|
+
order = {"medium": 0, "high": 1, "critical": 2}
|
|
52
|
+
threshold = order[args.fail_on]
|
|
53
|
+
worst = max((order[f.severity] for f in result.findings), default=-1)
|
|
54
|
+
return 1 if worst >= threshold else 0
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _entry() -> int:
|
|
58
|
+
"""Console entry point that survives `| head` without a traceback."""
|
|
59
|
+
try:
|
|
60
|
+
return main()
|
|
61
|
+
except BrokenPipeError:
|
|
62
|
+
try:
|
|
63
|
+
sys.stdout.close()
|
|
64
|
+
except Exception:
|
|
65
|
+
pass
|
|
66
|
+
return 0
|
|
67
|
+
except KeyboardInterrupt:
|
|
68
|
+
return 130
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
if __name__ == "__main__":
|
|
72
|
+
sys.exit(_entry())
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
"""Scoring and the shareable scorecard."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections import Counter
|
|
5
|
+
|
|
6
|
+
WEIGHTS = {"critical": 25, "high": 10, "medium": 3}
|
|
7
|
+
HOT_FILE_MULTIPLIER = 1.5 # a secret sitting in .env / config is worse
|
|
8
|
+
GIT_TRACKED_PENALTY = 20 # .env committed to git
|
|
9
|
+
|
|
10
|
+
GRADES = [
|
|
11
|
+
(0, "A", "Limpio"),
|
|
12
|
+
(1, "B", "Menor"),
|
|
13
|
+
(25, "C", "Atención"),
|
|
14
|
+
(60, "D", "Riesgo alto"),
|
|
15
|
+
(100, "F", "Crítico"),
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
RESET = "\033[0m"
|
|
19
|
+
BOLD = "\033[1m"
|
|
20
|
+
DIM = "\033[2m"
|
|
21
|
+
COLORS = {
|
|
22
|
+
"A": "\033[92m", "B": "\033[92m", "C": "\033[93m",
|
|
23
|
+
"D": "\033[91m", "F": "\033[91m",
|
|
24
|
+
}
|
|
25
|
+
SEV_COLOR = {"critical": "\033[91m", "high": "\033[93m", "medium": "\033[94m"}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def compute_score(result) -> tuple:
|
|
29
|
+
"""Return (penalty_points, grade_letter, label)."""
|
|
30
|
+
points = 0.0
|
|
31
|
+
for f in result.findings:
|
|
32
|
+
w = WEIGHTS.get(f.severity, 3)
|
|
33
|
+
if f.in_hot_file:
|
|
34
|
+
w *= HOT_FILE_MULTIPLIER
|
|
35
|
+
if f.gitignored:
|
|
36
|
+
w *= 0.5 # ignored by git = less likely to leak outward
|
|
37
|
+
points += w
|
|
38
|
+
points += GIT_TRACKED_PENALTY * len(result.env_in_git)
|
|
39
|
+
|
|
40
|
+
grade, label = "A", "Limpio"
|
|
41
|
+
for threshold, letter, text in GRADES:
|
|
42
|
+
if points >= threshold:
|
|
43
|
+
grade, label = letter, text
|
|
44
|
+
return round(points), grade, label
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _bar(grade: str) -> str:
|
|
48
|
+
filled = {"A": 5, "B": 4, "C": 3, "D": 2, "F": 1}[grade]
|
|
49
|
+
return "█" * filled + "░" * (5 - filled)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def render_terminal(result, color: bool = True) -> str:
|
|
53
|
+
points, grade, label = compute_score(result)
|
|
54
|
+
c = (lambda s, code: f"{code}{s}{RESET}") if color else (lambda s, code: s)
|
|
55
|
+
gcol = COLORS[grade]
|
|
56
|
+
|
|
57
|
+
sev = Counter(f.severity for f in result.findings)
|
|
58
|
+
groups = Counter(f.group for f in result.findings)
|
|
59
|
+
|
|
60
|
+
out = []
|
|
61
|
+
out.append("")
|
|
62
|
+
out.append(c(" ┌────────────────────────────────────────────┐", DIM))
|
|
63
|
+
out.append(c(" │", DIM) + c(" E N V L E A K S C A N ", BOLD)
|
|
64
|
+
+ c(" │", DIM))
|
|
65
|
+
out.append(c(" └────────────────────────────────────────────┘", DIM))
|
|
66
|
+
out.append("")
|
|
67
|
+
out.append(f" {c(grade, gcol + BOLD)} {c(_bar(grade), gcol)} {c(label, BOLD)}")
|
|
68
|
+
out.append("")
|
|
69
|
+
out.append(f" {len(result.findings)} hallazgo(s) en {result.files_scanned} archivos")
|
|
70
|
+
if result.findings:
|
|
71
|
+
parts = []
|
|
72
|
+
for s in ("critical", "high", "medium"):
|
|
73
|
+
if sev[s]:
|
|
74
|
+
parts.append(c(f"{sev[s]} {s}", SEV_COLOR[s]))
|
|
75
|
+
out.append(" " + " · ".join(parts))
|
|
76
|
+
out.append("")
|
|
77
|
+
|
|
78
|
+
if groups.get("agent"):
|
|
79
|
+
out.append(c(f" ⚡ {groups['agent']} en la superficie de agentes/LLM", "\033[95m"))
|
|
80
|
+
out.append("")
|
|
81
|
+
|
|
82
|
+
if result.env_in_git:
|
|
83
|
+
out.append(c(" ✗ ARCHIVOS DE ENTORNO RASTREADOS POR GIT:", "\033[91m" + BOLD))
|
|
84
|
+
for p in result.env_in_git[:5]:
|
|
85
|
+
out.append(c(f" {p}", "\033[91m"))
|
|
86
|
+
out.append("")
|
|
87
|
+
|
|
88
|
+
if result.findings:
|
|
89
|
+
out.append(c(" Hallazgos", BOLD))
|
|
90
|
+
out.append("")
|
|
91
|
+
shown = sorted(
|
|
92
|
+
result.findings,
|
|
93
|
+
key=lambda f: (
|
|
94
|
+
{"critical": 0, "high": 1, "medium": 2}[f.severity],
|
|
95
|
+
not f.in_hot_file,
|
|
96
|
+
),
|
|
97
|
+
)
|
|
98
|
+
for f in shown[:15]:
|
|
99
|
+
tag = c(f.severity.upper()[:4], SEV_COLOR[f.severity])
|
|
100
|
+
out.append(f" {tag} {c(f.rule_name, BOLD)}")
|
|
101
|
+
out.append(c(f" {f.path}:{f.line_no}", DIM))
|
|
102
|
+
out.append(c(f" {f.snippet}", DIM))
|
|
103
|
+
out.append("")
|
|
104
|
+
if len(shown) > 15:
|
|
105
|
+
out.append(c(f" … y {len(shown) - 15} más\n", DIM))
|
|
106
|
+
else:
|
|
107
|
+
out.append(c(" Sin credenciales expuestas. Bien ahí.\n", "\033[92m"))
|
|
108
|
+
|
|
109
|
+
if not result.has_gitignore and result.env_files_found:
|
|
110
|
+
out.append(c(" ! No hay .gitignore y existen archivos .env\n", "\033[93m"))
|
|
111
|
+
|
|
112
|
+
out.append(c(" Nada de esto salió de tu máquina.", DIM))
|
|
113
|
+
out.append("")
|
|
114
|
+
return "\n".join(out)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def render_markdown(result) -> str:
|
|
118
|
+
points, grade, label = compute_score(result)
|
|
119
|
+
sev = Counter(f.severity for f in result.findings)
|
|
120
|
+
groups = Counter(f.group for f in result.findings)
|
|
121
|
+
|
|
122
|
+
lines = [
|
|
123
|
+
"```",
|
|
124
|
+
" E N V L E A K S C A N",
|
|
125
|
+
"",
|
|
126
|
+
f" {grade} {_bar(grade)} {label}",
|
|
127
|
+
"",
|
|
128
|
+
f" {len(result.findings)} hallazgo(s) · {result.files_scanned} archivos",
|
|
129
|
+
]
|
|
130
|
+
if result.findings:
|
|
131
|
+
parts = [f"{sev[s]} {s}" for s in ("critical", "high", "medium") if sev[s]]
|
|
132
|
+
lines.append(" " + " · ".join(parts))
|
|
133
|
+
if groups.get("agent"):
|
|
134
|
+
lines.append(f" {groups['agent']} en la superficie de agentes/LLM")
|
|
135
|
+
if result.env_in_git:
|
|
136
|
+
lines.append(f" {len(result.env_in_git)} archivo(s) .env rastreados por git")
|
|
137
|
+
lines += ["```", "", "_Escaneado localmente con `pipx run envleak`. "
|
|
138
|
+
"Ningún dato salió de la máquina._"]
|
|
139
|
+
return "\n".join(lines)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def render_json(result) -> str:
|
|
143
|
+
points, grade, label = compute_score(result)
|
|
144
|
+
return json.dumps({
|
|
145
|
+
"grade": grade,
|
|
146
|
+
"label": label,
|
|
147
|
+
"penalty_points": points,
|
|
148
|
+
"files_scanned": result.files_scanned,
|
|
149
|
+
"env_files_tracked_by_git": result.env_in_git,
|
|
150
|
+
"agent_config_files": result.agent_files_found,
|
|
151
|
+
"findings": [
|
|
152
|
+
{
|
|
153
|
+
"rule": f.rule_id,
|
|
154
|
+
"name": f.rule_name,
|
|
155
|
+
"severity": f.severity,
|
|
156
|
+
"group": f.group,
|
|
157
|
+
"path": f.path,
|
|
158
|
+
"line": f.line_no,
|
|
159
|
+
"snippet": f.snippet,
|
|
160
|
+
"in_env_or_config_file": f.in_hot_file,
|
|
161
|
+
"gitignored": f.gitignored,
|
|
162
|
+
}
|
|
163
|
+
for f in result.findings
|
|
164
|
+
],
|
|
165
|
+
}, indent=2, ensure_ascii=False)
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""Detection rules for envleak.
|
|
2
|
+
|
|
3
|
+
Each rule finds one kind of credential. Rules are ordered: the agent/LLM
|
|
4
|
+
surface comes first because that's the surface nobody else scans.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import math
|
|
8
|
+
import re
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(frozen=True)
|
|
13
|
+
class Rule:
|
|
14
|
+
id: str
|
|
15
|
+
name: str
|
|
16
|
+
pattern: re.Pattern
|
|
17
|
+
severity: str # critical | high | medium
|
|
18
|
+
group: str # "agent" | "cloud" | "classic"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _c(p: str) -> re.Pattern:
|
|
22
|
+
return re.compile(p)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
# --- The agent / LLM surface (the differentiator) -------------------------
|
|
26
|
+
AGENT_RULES = [
|
|
27
|
+
Rule("openai-key", "OpenAI API key",
|
|
28
|
+
_c(r"\bsk-(?!ant-)(?:proj-)?[A-Za-z0-9_-]{20,}\b"), "critical", "agent"),
|
|
29
|
+
Rule("anthropic-key", "Anthropic API key",
|
|
30
|
+
_c(r"\bsk-ant-[A-Za-z0-9_-]{20,}\b"), "critical", "agent"),
|
|
31
|
+
Rule("google-ai-key", "Google AI / Gemini key",
|
|
32
|
+
_c(r"\bAIza[0-9A-Za-z_-]{35}\b"), "critical", "agent"),
|
|
33
|
+
Rule("groq-key", "Groq API key",
|
|
34
|
+
_c(r"\bgsk_[A-Za-z0-9]{40,}\b"), "critical", "agent"),
|
|
35
|
+
Rule("hf-token", "Hugging Face token",
|
|
36
|
+
_c(r"\bhf_[A-Za-z0-9]{30,}\b"), "high", "agent"),
|
|
37
|
+
Rule("replicate-token", "Replicate token",
|
|
38
|
+
_c(r"\br8_[A-Za-z0-9]{30,}\b"), "high", "agent"),
|
|
39
|
+
Rule("langsmith-key", "LangSmith API key",
|
|
40
|
+
_c(r"\blsv2_(?:pt|sk)_[A-Za-z0-9]{20,}"), "high", "agent"),
|
|
41
|
+
Rule("perplexity-key", "Perplexity API key",
|
|
42
|
+
_c(r"\bpplx-[A-Za-z0-9]{32,}\b"), "high", "agent"),
|
|
43
|
+
Rule("mistral-key", "Mistral API key",
|
|
44
|
+
_c(r"(?i)\bmistral[_-]?api[_-]?key\b\s*[:=]\s*['\"]?([A-Za-z0-9]{24,})"),
|
|
45
|
+
"high", "agent"),
|
|
46
|
+
Rule("generic-llm-key", "LLM provider key in config",
|
|
47
|
+
_c(r"(?i)\b(?:llm|openai|anthropic|together|fireworks|deepseek|xai)"
|
|
48
|
+
r"[_-]?(?:api[_-]?)?key\b\s*[:=]\s*['\"]?([A-Za-z0-9_\-]{20,})"),
|
|
49
|
+
"high", "agent"),
|
|
50
|
+
]
|
|
51
|
+
|
|
52
|
+
# --- Cloud / infra --------------------------------------------------------
|
|
53
|
+
CLOUD_RULES = [
|
|
54
|
+
Rule("aws-access-key", "AWS access key ID",
|
|
55
|
+
_c(r"\b(?:AKIA|ASIA)[0-9A-Z]{16}\b"), "critical", "cloud"),
|
|
56
|
+
Rule("aws-secret", "AWS secret access key",
|
|
57
|
+
_c(r"(?i)aws[_-]?secret[_-]?access[_-]?key\b\s*[:=]\s*['\"]?([A-Za-z0-9/+=]{40})"),
|
|
58
|
+
"critical", "cloud"),
|
|
59
|
+
Rule("gcp-sa", "GCP service-account private key",
|
|
60
|
+
_c(r'"type"\s*:\s*"service_account"'), "critical", "cloud"),
|
|
61
|
+
Rule("private-key", "Private key block",
|
|
62
|
+
_c(r"-----BEGIN (?:RSA |EC |DSA |OPENSSH |PGP )?PRIVATE KEY-----"),
|
|
63
|
+
"critical", "cloud"),
|
|
64
|
+
Rule("docker-auth", "Docker registry auth",
|
|
65
|
+
_c(r'"auths"\s*:\s*\{[^}]*"auth"\s*:\s*"[A-Za-z0-9+/=]{16,}"'),
|
|
66
|
+
"high", "cloud"),
|
|
67
|
+
]
|
|
68
|
+
|
|
69
|
+
# --- Classic third-party --------------------------------------------------
|
|
70
|
+
CLASSIC_RULES = [
|
|
71
|
+
Rule("github-token", "GitHub token",
|
|
72
|
+
_c(r"\bgh[pousr]_[A-Za-z0-9]{36,}\b"), "critical", "classic"),
|
|
73
|
+
Rule("gitlab-token", "GitLab token",
|
|
74
|
+
_c(r"\bglpat-[A-Za-z0-9_-]{20,}\b"), "critical", "classic"),
|
|
75
|
+
Rule("stripe-key", "Stripe secret key",
|
|
76
|
+
_c(r"\b(?:sk|rk)_live_[A-Za-z0-9]{20,}\b"), "critical", "classic"),
|
|
77
|
+
Rule("slack-token", "Slack token",
|
|
78
|
+
_c(r"\bxox[abprs]-[A-Za-z0-9-]{10,}\b"), "high", "classic"),
|
|
79
|
+
Rule("telegram-bot", "Telegram bot token",
|
|
80
|
+
_c(r"\b\d{8,10}:AA[A-Za-z0-9_-]{32,}\b"), "high", "classic"),
|
|
81
|
+
Rule("sendgrid-key", "SendGrid API key",
|
|
82
|
+
_c(r"\bSG\.[A-Za-z0-9_-]{20,}\.[A-Za-z0-9_-]{20,}\b"), "high", "classic"),
|
|
83
|
+
Rule("twilio-sid", "Twilio account SID",
|
|
84
|
+
_c(r"\bAC[0-9a-fA-F]{32}\b"), "medium", "classic"),
|
|
85
|
+
Rule("jwt", "JSON Web Token",
|
|
86
|
+
_c(r"\beyJ[A-Za-z0-9_-]{10,}\.eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\b"),
|
|
87
|
+
"medium", "classic"),
|
|
88
|
+
Rule("db-url", "Database URL with password",
|
|
89
|
+
_c(r"\b(?:postgres(?:ql)?|mysql|mongodb(?:\+srv)?|redis|amqp)://"
|
|
90
|
+
r"[^\s:'\"]+:[^\s@'\"]{3,}@[^\s/'\"]+"), "critical", "classic"),
|
|
91
|
+
Rule("crypto-priv", "Crypto private key / seed",
|
|
92
|
+
_c(r"(?i)\b(?:private[_-]?key|mnemonic|seed[_-]?phrase)\b\s*[:=]\s*"
|
|
93
|
+
r"['\"]?(0x[a-fA-F0-9]{64}|(?:[a-z]+\s+){11,23}[a-z]+)"),
|
|
94
|
+
"critical", "classic"),
|
|
95
|
+
]
|
|
96
|
+
|
|
97
|
+
ALL_RULES = AGENT_RULES + CLOUD_RULES + CLASSIC_RULES
|
|
98
|
+
|
|
99
|
+
# Generic high-entropy assignment (catches what the rules above miss)
|
|
100
|
+
ASSIGNMENT = re.compile(
|
|
101
|
+
r"(?i)\b([A-Z0-9_]*(?:KEY|TOKEN|SECRET|PASSWORD|PASSWD|PWD|CREDENTIAL)[A-Z0-9_]*)"
|
|
102
|
+
r"\s*[:=]\s*['\"]?([^\s'\"#,;]{16,})"
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
PLACEHOLDERS = {
|
|
106
|
+
"changeme", "your_key_here", "xxx", "todo", "placeholder", "example",
|
|
107
|
+
"null", "none", "true", "false", "undefined", "secret", "password",
|
|
108
|
+
"username", "user", "admin", "root", "test", "localhost",
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
# Credential pairs that only ever appear in docs and templates.
|
|
112
|
+
DOC_CREDENTIALS = (
|
|
113
|
+
"user:password@", "username:password@", "admin:admin@", "root:root@",
|
|
114
|
+
"user:pass@", "foo:bar@", "test:test@", ":password@", ":changeme@",
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
LOCAL_HOSTS = ("@localhost", "@127.0.0.1", "@0.0.0.0", "@db:", "@host.docker")
|
|
118
|
+
|
|
119
|
+
# A value that is really an expression, not a literal secret.
|
|
120
|
+
CODE_CHARS = set("()[]{}<>+*/\\ ")
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def shannon_entropy(s: str) -> float:
|
|
124
|
+
if not s:
|
|
125
|
+
return 0.0
|
|
126
|
+
freq = {}
|
|
127
|
+
for ch in s:
|
|
128
|
+
freq[ch] = freq.get(ch, 0) + 1
|
|
129
|
+
n = len(s)
|
|
130
|
+
return -sum((c / n) * math.log2(c / n) for c in freq.values())
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def looks_like_code(value: str) -> bool:
|
|
134
|
+
"""True when the "secret" is actually a code expression."""
|
|
135
|
+
v = value.strip().strip("'\"")
|
|
136
|
+
if any(ch in CODE_CHARS for ch in v):
|
|
137
|
+
return True
|
|
138
|
+
if v.endswith((".upper", ".lower", ".strip", ".format", ".get")):
|
|
139
|
+
return True
|
|
140
|
+
if v.count(".") >= 1 and "_" not in v and not any(c.isdigit() for c in v):
|
|
141
|
+
return True # attribute access like key.upper
|
|
142
|
+
return False
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def looks_like_placeholder(value: str) -> bool:
|
|
146
|
+
v = value.strip().strip("'\"").lower()
|
|
147
|
+
if any(d in v for d in DOC_CREDENTIALS):
|
|
148
|
+
return True
|
|
149
|
+
if any(h in v for h in LOCAL_HOSTS):
|
|
150
|
+
return True
|
|
151
|
+
if len(v) < 16:
|
|
152
|
+
return True
|
|
153
|
+
if v in PLACEHOLDERS:
|
|
154
|
+
return True
|
|
155
|
+
if v.startswith(("${", "{{", "<", "$(")): # templated
|
|
156
|
+
return True
|
|
157
|
+
if any(p in v for p in ("your-", "your_", "example", "placeholder",
|
|
158
|
+
"changeme", "dummy", "fake", "sample", "xxxx")):
|
|
159
|
+
return True
|
|
160
|
+
if len(set(v)) <= 3: # aaaaaaaa, 00000000
|
|
161
|
+
return True
|
|
162
|
+
return False
|
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
"""Filesystem scanner. Everything here runs locally; nothing leaves the machine."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import re
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from .rules import (
|
|
9
|
+
ALL_RULES, ASSIGNMENT, looks_like_code, looks_like_placeholder,
|
|
10
|
+
shannon_entropy,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
MAX_FILE_BYTES = 2 * 1024 * 1024 # skip anything bigger than 2 MB
|
|
14
|
+
MAX_LINE_LEN = 4000
|
|
15
|
+
|
|
16
|
+
SKIP_DIRS = {
|
|
17
|
+
".git", "node_modules", "__pycache__", ".venv", "venv", "env",
|
|
18
|
+
"dist", "build", ".next", ".nuxt", "target", "vendor", ".tox",
|
|
19
|
+
".mypy_cache", ".pytest_cache", ".ruff_cache", "site-packages",
|
|
20
|
+
".terraform", "coverage", ".idea", ".gradle",
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
SKIP_EXT = {
|
|
24
|
+
".png", ".jpg", ".jpeg", ".gif", ".webp", ".svg", ".ico", ".pdf",
|
|
25
|
+
".zip", ".gz", ".tar", ".bz2", ".xz", ".7z", ".rar", ".jar",
|
|
26
|
+
".mp3", ".mp4", ".mov", ".avi", ".woff", ".woff2", ".ttf", ".otf",
|
|
27
|
+
".so", ".dylib", ".dll", ".exe", ".bin", ".pyc", ".pyo", ".class",
|
|
28
|
+
".lock", ".min.js", ".map",
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
# Config-shaped files. The generic entropy pass runs ONLY here: in source
|
|
32
|
+
# code it fires on ordinary expressions (key = key.upper()) and destroys
|
|
33
|
+
# the signal-to-noise ratio.
|
|
34
|
+
CONFIG_EXT = {".env", ".json", ".yaml", ".yml", ".toml", ".ini", ".cfg",
|
|
35
|
+
".conf", ".properties", ".tfvars"}
|
|
36
|
+
|
|
37
|
+
# Template files ship fake credentials on purpose.
|
|
38
|
+
EXAMPLE_MARKERS = (".example", ".sample", ".template", ".dist", "sample.",
|
|
39
|
+
"example.", "template.")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def is_example_file(name: str) -> bool:
|
|
43
|
+
n = name.lower()
|
|
44
|
+
return any(m in n for m in EXAMPLE_MARKERS)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
# Files that are, by their nature, credential-bearing. Finding a live secret
|
|
48
|
+
# in one of these is worse because they're the ones that get committed or
|
|
49
|
+
# copied into images by accident.
|
|
50
|
+
HOT_FILES = {
|
|
51
|
+
".env", ".env.local", ".env.production", ".env.prod", ".env.dev",
|
|
52
|
+
".env.development", ".env.staging", ".envrc",
|
|
53
|
+
"credentials", "config.json", "secrets.json", "settings.json",
|
|
54
|
+
"docker-compose.yml", "docker-compose.yaml",
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
# The agent surface: MCP servers, orchestration configs, notebooks.
|
|
58
|
+
AGENT_FILE_HINTS = (
|
|
59
|
+
"mcp.json", "mcp_config", "claude_desktop_config.json",
|
|
60
|
+
"langgraph.json", "crew", "agents.yaml", "agent.yaml",
|
|
61
|
+
".n8n", "workflows", "flowise", "dify",
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
TEXT_HINT_EXT = {
|
|
65
|
+
".env", ".json", ".yaml", ".yml", ".toml", ".ini", ".cfg", ".conf",
|
|
66
|
+
".py", ".js", ".ts", ".jsx", ".tsx", ".sh", ".bash", ".zsh", ".fish",
|
|
67
|
+
".rb", ".go", ".rs", ".java", ".php", ".cs", ".txt", ".md", ".properties",
|
|
68
|
+
".tf", ".tfvars", ".xml", ".sql", ".ipynb", ".example", ".sample", "",
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass
|
|
73
|
+
class Finding:
|
|
74
|
+
rule_id: str
|
|
75
|
+
rule_name: str
|
|
76
|
+
severity: str
|
|
77
|
+
group: str
|
|
78
|
+
path: str
|
|
79
|
+
line_no: int
|
|
80
|
+
snippet: str
|
|
81
|
+
in_hot_file: bool = False
|
|
82
|
+
gitignored: bool = False
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@dataclass
|
|
86
|
+
class ScanResult:
|
|
87
|
+
root: str
|
|
88
|
+
findings: list = field(default_factory=list)
|
|
89
|
+
files_scanned: int = 0
|
|
90
|
+
env_files_found: list = field(default_factory=list)
|
|
91
|
+
agent_files_found: list = field(default_factory=list)
|
|
92
|
+
env_in_git: list = field(default_factory=list)
|
|
93
|
+
has_gitignore: bool = False
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _redact(value: str) -> str:
|
|
97
|
+
"""Show enough to identify the key, never enough to use it."""
|
|
98
|
+
v = value.strip().strip("'\"")
|
|
99
|
+
if len(v) <= 10:
|
|
100
|
+
return v[:2] + "*" * max(len(v) - 2, 0)
|
|
101
|
+
return f"{v[:4]}{'*' * 8}{v[-4:]}"
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _redact_line(line: str, match_text: str) -> str:
|
|
105
|
+
line = line.strip()[:MAX_LINE_LEN]
|
|
106
|
+
return line.replace(match_text, _redact(match_text))[:200]
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _is_probably_text(path: Path) -> bool:
|
|
110
|
+
if path.suffix.lower() in SKIP_EXT:
|
|
111
|
+
return False
|
|
112
|
+
if path.suffix.lower() in TEXT_HINT_EXT or path.name.startswith("."):
|
|
113
|
+
return True
|
|
114
|
+
return path.suffix == ""
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _read_gitignore(root: Path) -> set:
|
|
118
|
+
patterns = set()
|
|
119
|
+
gi = root / ".gitignore"
|
|
120
|
+
if gi.exists():
|
|
121
|
+
try:
|
|
122
|
+
for line in gi.read_text(errors="ignore").splitlines():
|
|
123
|
+
line = line.strip()
|
|
124
|
+
if line and not line.startswith("#"):
|
|
125
|
+
patterns.add(line.rstrip("/"))
|
|
126
|
+
except OSError:
|
|
127
|
+
pass
|
|
128
|
+
return patterns
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _matches_gitignore(rel: str, patterns: set) -> bool:
|
|
132
|
+
name = os.path.basename(rel)
|
|
133
|
+
for p in patterns:
|
|
134
|
+
if p in (name, rel):
|
|
135
|
+
return True
|
|
136
|
+
if p.startswith("*") and name.endswith(p[1:]):
|
|
137
|
+
return True
|
|
138
|
+
if p.endswith("*") and name.startswith(p[:-1]):
|
|
139
|
+
return True
|
|
140
|
+
if rel.startswith(p + "/"):
|
|
141
|
+
return True
|
|
142
|
+
return False
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def scan(root: str, max_files: int = 20000) -> ScanResult:
|
|
146
|
+
root_path = Path(root).resolve()
|
|
147
|
+
result = ScanResult(root=str(root_path))
|
|
148
|
+
gitignore = _read_gitignore(root_path)
|
|
149
|
+
result.has_gitignore = bool(gitignore)
|
|
150
|
+
tracked = _git_tracked_files(root_path)
|
|
151
|
+
|
|
152
|
+
for dirpath, dirnames, filenames in os.walk(root_path):
|
|
153
|
+
dirnames[:] = [d for d in dirnames if d not in SKIP_DIRS]
|
|
154
|
+
for fname in filenames:
|
|
155
|
+
if result.files_scanned >= max_files:
|
|
156
|
+
return result
|
|
157
|
+
fpath = Path(dirpath) / fname
|
|
158
|
+
try:
|
|
159
|
+
if fpath.is_symlink() or fpath.stat().st_size > MAX_FILE_BYTES:
|
|
160
|
+
continue
|
|
161
|
+
except OSError:
|
|
162
|
+
continue
|
|
163
|
+
if not _is_probably_text(fpath):
|
|
164
|
+
continue
|
|
165
|
+
|
|
166
|
+
rel = str(fpath.relative_to(root_path))
|
|
167
|
+
lower = rel.lower()
|
|
168
|
+
|
|
169
|
+
is_example = is_example_file(fname)
|
|
170
|
+
is_hot = (fname in HOT_FILES or fname.startswith(".env")) \
|
|
171
|
+
and not is_example
|
|
172
|
+
if is_hot:
|
|
173
|
+
result.env_files_found.append(rel)
|
|
174
|
+
if tracked is not None and rel in tracked:
|
|
175
|
+
result.env_in_git.append(rel)
|
|
176
|
+
|
|
177
|
+
if any(h in lower for h in AGENT_FILE_HINTS):
|
|
178
|
+
result.agent_files_found.append(rel)
|
|
179
|
+
|
|
180
|
+
try:
|
|
181
|
+
text = fpath.read_text(errors="ignore")
|
|
182
|
+
except (OSError, UnicodeDecodeError):
|
|
183
|
+
continue
|
|
184
|
+
result.files_scanned += 1
|
|
185
|
+
|
|
186
|
+
ignored = _matches_gitignore(rel, gitignore)
|
|
187
|
+
is_config = (fpath.suffix.lower() in CONFIG_EXT
|
|
188
|
+
or fname.startswith(".env"))
|
|
189
|
+
_scan_text(text, rel, is_hot, ignored, result,
|
|
190
|
+
is_config=is_config, is_example=is_example)
|
|
191
|
+
|
|
192
|
+
return result
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _git_tracked_files(root: Path):
|
|
196
|
+
"""Return set of git-tracked paths, or None if not a repo / git missing."""
|
|
197
|
+
if not (root / ".git").exists():
|
|
198
|
+
return None
|
|
199
|
+
try:
|
|
200
|
+
import subprocess
|
|
201
|
+
out = subprocess.run(
|
|
202
|
+
["git", "-C", str(root), "ls-files"],
|
|
203
|
+
capture_output=True, text=True, timeout=20,
|
|
204
|
+
)
|
|
205
|
+
if out.returncode != 0:
|
|
206
|
+
return None
|
|
207
|
+
return set(out.stdout.splitlines())
|
|
208
|
+
except Exception:
|
|
209
|
+
return None
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _scan_text(text: str, rel: str, is_hot: bool, ignored: bool,
|
|
213
|
+
result: ScanResult, is_config: bool = False,
|
|
214
|
+
is_example: bool = False) -> None:
|
|
215
|
+
seen = set()
|
|
216
|
+
claimed = set()
|
|
217
|
+
lines = text.splitlines()
|
|
218
|
+
|
|
219
|
+
for i, line in enumerate(lines, 1):
|
|
220
|
+
if len(line) > MAX_LINE_LEN:
|
|
221
|
+
continue
|
|
222
|
+
stripped = line.strip()
|
|
223
|
+
if stripped.startswith(("#", "//", "*", "<!--")) and "=" not in stripped:
|
|
224
|
+
continue
|
|
225
|
+
|
|
226
|
+
for rule in ALL_RULES:
|
|
227
|
+
for m in rule.pattern.finditer(line):
|
|
228
|
+
hit = m.group(1) if m.groups() else m.group(0)
|
|
229
|
+
# One credential, one finding: the first (most specific) rule
|
|
230
|
+
# to claim a value wins. ALL_RULES is ordered specific-first.
|
|
231
|
+
claim = hit.strip().strip("'\"")[:60]
|
|
232
|
+
if claim in claimed:
|
|
233
|
+
continue
|
|
234
|
+
if looks_like_placeholder(hit):
|
|
235
|
+
continue
|
|
236
|
+
claimed.add(claim)
|
|
237
|
+
sev = "medium" if is_example else rule.severity
|
|
238
|
+
result.findings.append(Finding(
|
|
239
|
+
rule_id=rule.id, rule_name=rule.name,
|
|
240
|
+
severity=sev, group=rule.group,
|
|
241
|
+
path=rel, line_no=i,
|
|
242
|
+
snippet=_redact_line(line, hit),
|
|
243
|
+
in_hot_file=is_hot, gitignored=ignored,
|
|
244
|
+
))
|
|
245
|
+
|
|
246
|
+
# Generic entropy pass — config files only. In source code this
|
|
247
|
+
# fires on ordinary expressions and buries the real findings.
|
|
248
|
+
if not is_config:
|
|
249
|
+
continue
|
|
250
|
+
for m in ASSIGNMENT.finditer(line):
|
|
251
|
+
varname, value = m.group(1), m.group(2)
|
|
252
|
+
if looks_like_placeholder(value) or looks_like_code(value):
|
|
253
|
+
continue
|
|
254
|
+
if shannon_entropy(value) < 3.5:
|
|
255
|
+
continue
|
|
256
|
+
claim = value.strip().strip("'\"")[:60]
|
|
257
|
+
if claim in claimed:
|
|
258
|
+
continue
|
|
259
|
+
if any(claim in c or c in claim for c in claimed):
|
|
260
|
+
continue
|
|
261
|
+
claimed.add(claim)
|
|
262
|
+
result.findings.append(Finding(
|
|
263
|
+
rule_id="high-entropy", rule_name=f"High-entropy value in {varname}",
|
|
264
|
+
severity="medium", group="classic",
|
|
265
|
+
path=rel, line_no=i,
|
|
266
|
+
snippet=_redact_line(line, value),
|
|
267
|
+
in_hot_file=is_hot, gitignored=ignored,
|
|
268
|
+
))
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "envleak-cli"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Find exposed API keys and secrets before someone else does. 100% local."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { text = "MIT" }
|
|
11
|
+
requires-python = ">=3.8"
|
|
12
|
+
keywords = ["security", "secrets", "api-keys", "ai-agents", "llm", "devsecops"]
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Development Status :: 4 - Beta",
|
|
15
|
+
"Environment :: Console",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Topic :: Security",
|
|
20
|
+
]
|
|
21
|
+
dependencies = []
|
|
22
|
+
|
|
23
|
+
[project.urls]
|
|
24
|
+
Homepage = "https://github.com/novasdiego1/envleak"
|
|
25
|
+
Issues = "https://github.com/novasdiego1/envleak/issues"
|
|
26
|
+
|
|
27
|
+
[project.scripts]
|
|
28
|
+
envleak = "envleak.cli:_entry"
|
|
29
|
+
|
|
30
|
+
[tool.hatch.build.targets.wheel]
|
|
31
|
+
packages = ["envleak"]
|