narvy-cli 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- narvy/__init__.py +3 -0
- narvy/android/__init__.py +0 -0
- narvy/android/source_analyzer.py +268 -0
- narvy/android/split_bundle.py +475 -0
- narvy/android_rule_context.py +196 -0
- narvy/apk_memory_preflight.py +248 -0
- narvy/auth.py +40 -0
- narvy/ci_templates/bitbucket.yml +25 -0
- narvy/ci_templates/github.yml +60 -0
- narvy/ci_templates/gitlab.yml +32 -0
- narvy/cloud/__init__.py +0 -0
- narvy/cloud/aws_scan.py +1188 -0
- narvy/comment_filter.py +134 -0
- narvy/crypto_taint_lite.py +82 -0
- narvy/decompiler.py +337 -0
- narvy/doctor.py +244 -0
- narvy/host/__init__.py +0 -0
- narvy/host/audit.py +157 -0
- narvy/host/host_knowledge.py +982 -0
- narvy/host/lynis_bootstrap.py +400 -0
- narvy/host/lynis_parser.py +358 -0
- narvy/host/narvy_checks.py +789 -0
- narvy/host/report.py +69 -0
- narvy/host/ssh_exec.py +219 -0
- narvy/ios/__init__.py +0 -0
- narvy/ios/binary_analyzer.py +1201 -0
- narvy/ios/plist_checks.py +249 -0
- narvy/ios/source_analyzer.py +301 -0
- narvy/ios/third_party_filter.py +242 -0
- narvy/ios/trust_all_context.py +66 -0
- narvy/main.py +1983 -0
- narvy/native_hardening.py +503 -0
- narvy/reporter.py +151 -0
- narvy/rule_engine.py +57 -0
- narvy/rules/android/config.yml +77 -0
- narvy/rules/android/crypto.yml +34 -0
- narvy/rules/android/secrets.yml +161 -0
- narvy/rules/android/storage.yml +24 -0
- narvy/rules/android/webview.yml +23 -0
- narvy/rules/android.yml +219 -0
- narvy/rules/ios/objc/crypto.yml +56 -0
- narvy/rules/ios/objc/network.yml +67 -0
- narvy/rules/ios/objc/secrets.yml +79 -0
- narvy/rules/ios/objc/storage.yml +45 -0
- narvy/rules/ios/objc/webview.yml +45 -0
- narvy/rules/ios_swift.yml +327 -0
- narvy/rules/web/go.yml +2837 -0
- narvy/rules/web/java.yml +1576 -0
- narvy/rules/web/javascript.yml +3683 -0
- narvy/rules/web/kotlin.yml +413 -0
- narvy/rules/web/local/csharp_narvy/config.yml +61 -0
- narvy/rules/web/local/csharp_narvy/crypto.yml +64 -0
- narvy/rules/web/local/csharp_narvy/deserialization.yml +59 -0
- narvy/rules/web/local/csharp_narvy/injection.yml +122 -0
- narvy/rules/web/local/csharp_narvy/xxe.yml +48 -0
- narvy/rules/web/local/java_narvy/auth_jwt.yml +134 -0
- narvy/rules/web/local/java_narvy/deserialization.yml +108 -0
- narvy/rules/web/local/java_narvy/mybatis.yml +39 -0
- narvy/rules/web/local/java_narvy/snakeyaml.yml +34 -0
- narvy/rules/web/local/java_narvy/spring_authz.yml +32 -0
- narvy/rules/web/local/java_narvy/spring_config.yml +67 -0
- narvy/rules/web/local/java_narvy/spring_hardening.yml +291 -0
- narvy/rules/web/local/java_narvy/sqli.yml +235 -0
- narvy/rules/web/local/java_narvy/xxe.yml +212 -0
- narvy/rules/web/php.yml +1644 -0
- narvy/rules/web/python.yml +3967 -0
- narvy/rules/web/ruby.yml +703 -0
- narvy/rules/web/rust.yml +258 -0
- narvy/rules/web/secrets.yml +1420 -0
- narvy/rules/web/secrets_supplement.yml +383 -0
- narvy/sca/__init__.py +1 -0
- narvy/sca/android_deps.py +349 -0
- narvy/sca/ios_deps.py +578 -0
- narvy/sca/osv_client.py +617 -0
- narvy/sca/web_deps.py +955 -0
- narvy/scope_config.py +195 -0
- narvy/semgrep_engine.py +219 -0
- narvy/stack_protector_evidence.py +96 -0
- narvy/third_party_filter.py +211 -0
- narvy/uploader.py +92 -0
- narvy/weak_prng_context.py +270 -0
- narvy/web/__init__.py +0 -0
- narvy/web/nuclei_binary.py +105 -0
- narvy/web/scan_blocklist.py +96 -0
- narvy/web/scanner.py +842 -0
- narvy/web/source_analyzer.py +714 -0
- narvy/web/ssrf_guard.py +374 -0
- narvy_cli-1.0.0.dist-info/METADATA +165 -0
- narvy_cli-1.0.0.dist-info/RECORD +93 -0
- narvy_cli-1.0.0.dist-info/WHEEL +5 -0
- narvy_cli-1.0.0.dist-info/entry_points.txt +2 -0
- narvy_cli-1.0.0.dist-info/licenses/LICENSE +202 -0
- narvy_cli-1.0.0.dist-info/top_level.txt +1 -0
narvy/scope_config.py
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
"""Loads and validates `.narvy-scope.yml`, the first-party/third-party override.
|
|
2
|
+
|
|
3
|
+
Parses the `own`/`vendor` lists. Never raises: a bad or missing file gives EMPTY.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
import sys
|
|
9
|
+
from typing import Dict, FrozenSet, NamedTuple, Optional
|
|
10
|
+
|
|
11
|
+
try:
|
|
12
|
+
import yaml
|
|
13
|
+
except ImportError: # pragma: no cover
|
|
14
|
+
yaml = None
|
|
15
|
+
|
|
16
|
+
SCOPE_CONFIG_FILENAME = ".narvy-scope.yml"
|
|
17
|
+
SCOPE_CONFIG_FILENAMES = (SCOPE_CONFIG_FILENAME,)
|
|
18
|
+
|
|
19
|
+
MAX_ENTRIES_PER_LIST = 500
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class ScopeConfig(NamedTuple):
|
|
23
|
+
"""Validated config: lower-cased patterns, exact unless they end in `*`."""
|
|
24
|
+
own: FrozenSet[str] = frozenset()
|
|
25
|
+
vendor: FrozenSet[str] = frozenset()
|
|
26
|
+
reasons: Dict[str, str] = {}
|
|
27
|
+
source_path: Optional[str] = None
|
|
28
|
+
|
|
29
|
+
def __bool__(self) -> bool:
|
|
30
|
+
return bool(self.own) or bool(self.vendor)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
EMPTY = ScopeConfig()
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _warn(msg: str) -> None:
|
|
37
|
+
print(f"[narvy] WARNING: {msg}", file=sys.stderr)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def entry_matches(entry: str, candidate: str) -> bool:
|
|
41
|
+
"""Match an undelimited token: exact, or prefix if entry ends in '*'. Args lower-cased."""
|
|
42
|
+
if entry.endswith("*"):
|
|
43
|
+
return candidate.startswith(entry[:-1])
|
|
44
|
+
return candidate == entry
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def segment_prefix_match(entry: str, candidate: str, delimiter: str = ".") -> bool:
|
|
48
|
+
"""Whole-segment prefix match for delimiter-separated identifiers. Args lower-cased."""
|
|
49
|
+
e = entry[:-1] if entry.endswith("*") else entry
|
|
50
|
+
e = e.rstrip(delimiter)
|
|
51
|
+
if not e:
|
|
52
|
+
return False
|
|
53
|
+
entry_segs = e.split(delimiter)
|
|
54
|
+
cand_segs = candidate.split(delimiter)
|
|
55
|
+
return cand_segs[:len(entry_segs)] == entry_segs
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def find_scope_config_path(target_path: str) -> Optional[str]:
|
|
59
|
+
"""First `.narvy-scope.yml` in the cwd, then next to target_path."""
|
|
60
|
+
target_dir = target_path if os.path.isdir(target_path) else os.path.dirname(
|
|
61
|
+
os.path.abspath(target_path))
|
|
62
|
+
target_dir = target_dir or "."
|
|
63
|
+
|
|
64
|
+
# cwd is checked before the target dir.
|
|
65
|
+
for base in (os.getcwd(), target_dir):
|
|
66
|
+
for name in SCOPE_CONFIG_FILENAMES:
|
|
67
|
+
candidate = os.path.join(base, name)
|
|
68
|
+
if os.path.isfile(candidate):
|
|
69
|
+
return candidate
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _parse_entry(item, key: str, path: str):
|
|
74
|
+
"""Return (name, reason) for one list item, or None if it is malformed."""
|
|
75
|
+
if isinstance(item, str):
|
|
76
|
+
name = item.strip()
|
|
77
|
+
if not name:
|
|
78
|
+
_warn(f"{path}: '{key}' contains an empty string entry - skipping it.")
|
|
79
|
+
return None
|
|
80
|
+
return name.lower(), None
|
|
81
|
+
if isinstance(item, dict):
|
|
82
|
+
name = item.get("name")
|
|
83
|
+
if not isinstance(name, str) or not name.strip():
|
|
84
|
+
_warn(f"{path}: '{key}' has a mapping entry with no valid 'name' "
|
|
85
|
+
f"({item!r}) - skipping it.")
|
|
86
|
+
return None
|
|
87
|
+
reason = item.get("reason")
|
|
88
|
+
reason = reason.strip() if isinstance(reason, str) and reason.strip() else None
|
|
89
|
+
return name.strip().lower(), reason
|
|
90
|
+
_warn(f"{path}: '{key}' contains an entry that is neither a string nor a "
|
|
91
|
+
f"{{name, reason}} mapping ({item!r}) - skipping it.")
|
|
92
|
+
return None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _as_pattern_list(data: dict, key: str, path: str):
|
|
96
|
+
"""(name, reason) tuples in file order, or None if data[key] is malformed."""
|
|
97
|
+
val = data.get(key)
|
|
98
|
+
if val is None:
|
|
99
|
+
return []
|
|
100
|
+
if not isinstance(val, list):
|
|
101
|
+
_warn(f"{path}: '{key}' should be a YAML list, got "
|
|
102
|
+
f"{type(val).__name__} - ignoring this config file, continuing "
|
|
103
|
+
f"with no overrides.")
|
|
104
|
+
return None
|
|
105
|
+
if len(val) > MAX_ENTRIES_PER_LIST:
|
|
106
|
+
_warn(f"{path}: '{key}' has {len(val)} entries, more than the "
|
|
107
|
+
f"{MAX_ENTRIES_PER_LIST} cap - using only the first "
|
|
108
|
+
f"{MAX_ENTRIES_PER_LIST} (in file order).")
|
|
109
|
+
val = val[:MAX_ENTRIES_PER_LIST]
|
|
110
|
+
out = []
|
|
111
|
+
for item in val:
|
|
112
|
+
parsed = _parse_entry(item, key, path)
|
|
113
|
+
if parsed is not None:
|
|
114
|
+
out.append(parsed)
|
|
115
|
+
return out
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _strip_star(pattern: str) -> str:
|
|
119
|
+
return pattern[:-1] if pattern.endswith("*") else pattern
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _patterns_overlap(a: str, b: str) -> bool:
|
|
123
|
+
"""Load-time heuristic: could an own and a vendor pattern both match?"""
|
|
124
|
+
sa, sb = _strip_star(a), _strip_star(b)
|
|
125
|
+
return sa == sb or sa.startswith(sb) or sb.startswith(sa)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def load_scope_config(target_path: str) -> ScopeConfig:
|
|
129
|
+
"""Auto-detect and parse the scope config. Never raises: degrades to EMPTY."""
|
|
130
|
+
path = find_scope_config_path(target_path)
|
|
131
|
+
if path is None:
|
|
132
|
+
return EMPTY
|
|
133
|
+
|
|
134
|
+
if yaml is None: # pragma: no cover
|
|
135
|
+
_warn(f"found {path} but PyYAML is not installed - overrides ignored.")
|
|
136
|
+
return EMPTY
|
|
137
|
+
|
|
138
|
+
try:
|
|
139
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
140
|
+
data = yaml.safe_load(f)
|
|
141
|
+
except Exception as e:
|
|
142
|
+
_warn(f"could not read/parse {path} ({e.__class__.__name__}: {e}) - "
|
|
143
|
+
f"continuing with no overrides.")
|
|
144
|
+
return EMPTY
|
|
145
|
+
|
|
146
|
+
if data is None:
|
|
147
|
+
return EMPTY
|
|
148
|
+
if not isinstance(data, dict):
|
|
149
|
+
_warn(f"{path} does not look like a valid scope config (expected a "
|
|
150
|
+
f"YAML mapping with 'own'/'vendor' keys, got "
|
|
151
|
+
f"{type(data).__name__}) - continuing with no overrides.")
|
|
152
|
+
return EMPTY
|
|
153
|
+
|
|
154
|
+
own_entries = _as_pattern_list(data, "own", path)
|
|
155
|
+
vendor_entries = _as_pattern_list(data, "vendor", path)
|
|
156
|
+
if own_entries is None or vendor_entries is None:
|
|
157
|
+
return EMPTY
|
|
158
|
+
|
|
159
|
+
own_names = {name for name, _ in own_entries}
|
|
160
|
+
vendor_names = {name for name, _ in vendor_entries}
|
|
161
|
+
|
|
162
|
+
# Identical entry in both lists: ambiguous, drop from both.
|
|
163
|
+
conflicts = own_names & vendor_names
|
|
164
|
+
if conflicts:
|
|
165
|
+
plural = len(conflicts) != 1
|
|
166
|
+
_warn(
|
|
167
|
+
f"{path}: entr{'ies' if plural else 'y'} "
|
|
168
|
+
f"{', '.join(sorted(conflicts))!r} listed in BOTH 'own' and "
|
|
169
|
+
f"'vendor' - ambiguous, falling back to automatic classification "
|
|
170
|
+
f"for {'them' if plural else 'it'}."
|
|
171
|
+
)
|
|
172
|
+
own_names -= conflicts
|
|
173
|
+
vendor_names -= conflicts
|
|
174
|
+
|
|
175
|
+
# Two patterns that could both match: warn, drop neither ('own' wins at match time).
|
|
176
|
+
overlap_pairs = sorted({
|
|
177
|
+
(o, v) for o in own_names for v in vendor_names if _patterns_overlap(o, v)
|
|
178
|
+
})
|
|
179
|
+
for o, v in overlap_pairs:
|
|
180
|
+
_warn(
|
|
181
|
+
f"{path}: 'own' pattern {o!r} and 'vendor' pattern {v!r} could "
|
|
182
|
+
f"both match the same name - 'own' wins for anything matched by "
|
|
183
|
+
f"both, per policy."
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
if not own_names and not vendor_names:
|
|
187
|
+
return EMPTY
|
|
188
|
+
|
|
189
|
+
reasons = {name: reason for name, reason in (own_entries + vendor_entries) if reason}
|
|
190
|
+
return ScopeConfig(
|
|
191
|
+
own=frozenset(own_names),
|
|
192
|
+
vendor=frozenset(vendor_names),
|
|
193
|
+
reasons=reasons,
|
|
194
|
+
source_path=path,
|
|
195
|
+
)
|
narvy/semgrep_engine.py
ADDED
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
"""Semgrep-backed detection engine for the Android rule pack: runs Semgrep CE as a
|
|
2
|
+
subprocess over the jadx-decompiled tree. rule_engine.py stays the fallback for manifest/XML rules."""
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import shutil
|
|
8
|
+
import subprocess
|
|
9
|
+
from typing import Any, Dict, List, Optional
|
|
10
|
+
|
|
11
|
+
import yaml
|
|
12
|
+
|
|
13
|
+
from .third_party_filter import THIRD_PARTY_PREFIXES
|
|
14
|
+
from .scope_config import ScopeConfig
|
|
15
|
+
from . import weak_prng_context
|
|
16
|
+
|
|
17
|
+
RULES_PATH = os.path.join(os.path.dirname(__file__), 'rules', 'android.yml')
|
|
18
|
+
# Third-party prefixes as --exclude globs; leading '**/' de-anchors from the scan root.
|
|
19
|
+
_EXCLUDE_ARGS = []
|
|
20
|
+
for _p in THIRD_PARTY_PREFIXES:
|
|
21
|
+
_EXCLUDE_ARGS += ['--exclude', f"**/{_p.replace('.', '/').rstrip('/')}/**"]
|
|
22
|
+
|
|
23
|
+
SEVERITY_MAP = {
|
|
24
|
+
'ERROR': 'CRITICAL',
|
|
25
|
+
'WARNING': 'HIGH',
|
|
26
|
+
'INFO': 'MEDIUM',
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
# Whole-run timeout scales with tree size; a timeout is recorded in LAST_RUN.
|
|
30
|
+
SEMGREP_BASE_TIMEOUT_SEC = 600
|
|
31
|
+
SEMGREP_MAX_TIMEOUT_SEC = 1800
|
|
32
|
+
SEMGREP_TIMEOUT_PER_1K_FILES = 120
|
|
33
|
+
|
|
34
|
+
# Callers read this to tell "0 findings" apart from "never finished".
|
|
35
|
+
LAST_RUN: Dict[str, Any] = {
|
|
36
|
+
'status': 'not_run', # ok | timeout | error | unavailable | not_run
|
|
37
|
+
'timeout_s': None,
|
|
38
|
+
'file_count': None,
|
|
39
|
+
'degraded': False,
|
|
40
|
+
'message': None,
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def count_source_files(source_dir: str) -> int:
|
|
45
|
+
"""Cheap file count used to size the semgrep timeout."""
|
|
46
|
+
skip = {'.git', 'node_modules', 'build', '.gradle', '.idea', 'Pods',
|
|
47
|
+
'Carthage', '__pycache__', '.venv', 'venv', 'DerivedData'}
|
|
48
|
+
n = 0
|
|
49
|
+
try:
|
|
50
|
+
for dirpath, dirnames, filenames in os.walk(source_dir):
|
|
51
|
+
dirnames[:] = [d for d in dirnames if d not in skip]
|
|
52
|
+
n += len(filenames)
|
|
53
|
+
except OSError:
|
|
54
|
+
return 0
|
|
55
|
+
return n
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def compute_semgrep_timeout(source_dir: str,
|
|
59
|
+
base: int = SEMGREP_BASE_TIMEOUT_SEC) -> int:
|
|
60
|
+
"""Scale the whole-run timeout with tree size. Never below `base`."""
|
|
61
|
+
files = count_source_files(source_dir)
|
|
62
|
+
if files < 2000:
|
|
63
|
+
return base
|
|
64
|
+
extra = ((files - 2000) // 1000) * SEMGREP_TIMEOUT_PER_1K_FILES
|
|
65
|
+
return min(base + extra, SEMGREP_MAX_TIMEOUT_SEC)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _record_run(status: str, timeout_s: Optional[int] = None,
|
|
69
|
+
file_count: Optional[int] = None,
|
|
70
|
+
message: Optional[str] = None) -> None:
|
|
71
|
+
LAST_RUN.update({
|
|
72
|
+
'status': status,
|
|
73
|
+
'timeout_s': timeout_s,
|
|
74
|
+
'file_count': file_count,
|
|
75
|
+
'degraded': status in ('timeout', 'error'),
|
|
76
|
+
'message': message,
|
|
77
|
+
})
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def degraded_coverage_note() -> Optional[str]:
|
|
81
|
+
"""User-facing sentence when the last pass did not complete, else None."""
|
|
82
|
+
if not LAST_RUN.get('degraded'):
|
|
83
|
+
return None
|
|
84
|
+
if LAST_RUN.get('status') == 'timeout':
|
|
85
|
+
return (
|
|
86
|
+
"PARTIAL COVERAGE: the deep structural (Semgrep) pass did NOT "
|
|
87
|
+
f"finish within its {LAST_RUN.get('timeout_s')}s budget on this "
|
|
88
|
+
f"target ({LAST_RUN.get('file_count')} files). Zero findings from "
|
|
89
|
+
"that pass means 'not fully scanned', NOT 'clean'. The regex, "
|
|
90
|
+
"taint and SCA passes still ran to completion and their findings "
|
|
91
|
+
"are complete. Re-run with NARVY_SEMGREP_TIMEOUT set higher "
|
|
92
|
+
"(seconds), or scan a narrower path."
|
|
93
|
+
)
|
|
94
|
+
return (
|
|
95
|
+
"PARTIAL COVERAGE: the deep structural (Semgrep) pass failed "
|
|
96
|
+
f"({LAST_RUN.get('message')}). Zero findings from that pass means "
|
|
97
|
+
"'not fully scanned', NOT 'clean'."
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def is_available() -> bool:
|
|
102
|
+
return shutil.which('semgrep') is not None
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def load_rule_defs(rules_path: str = RULES_PATH) -> List[Dict[str, Any]]:
|
|
106
|
+
"""Synthesize a rule-def dict per rule id in the YAML pack, for SARIF."""
|
|
107
|
+
try:
|
|
108
|
+
with open(rules_path, 'r') as f:
|
|
109
|
+
data = yaml.safe_load(f)
|
|
110
|
+
except OSError:
|
|
111
|
+
return []
|
|
112
|
+
defs = []
|
|
113
|
+
for rule in (data or {}).get('rules', []):
|
|
114
|
+
meta = rule.get('metadata', {}) or {}
|
|
115
|
+
defs.append({
|
|
116
|
+
'id': rule['id'],
|
|
117
|
+
'name': rule.get('message', rule['id'])[:120],
|
|
118
|
+
'severity': SEVERITY_MAP.get(rule.get('severity', 'INFO'), 'MEDIUM'),
|
|
119
|
+
'details': {
|
|
120
|
+
'cwe': meta.get('cwe', ''),
|
|
121
|
+
'masvs': meta.get('masvs', ''),
|
|
122
|
+
'description': rule.get('message', ''),
|
|
123
|
+
'recommendation': rule.get('message', ''),
|
|
124
|
+
},
|
|
125
|
+
})
|
|
126
|
+
return defs
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def run_semgrep(source_dir: str, timeout: Optional[int] = None,
|
|
130
|
+
own_roots: Optional[set] = None,
|
|
131
|
+
override_config: Optional[ScopeConfig] = None,
|
|
132
|
+
include_path_prefixes: Optional[List[str]] = None) -> Optional[List[Dict[str, Any]]]:
|
|
133
|
+
"""Run the Semgrep rule pack against source_dir; findings, or None if semgrep is unavailable. own_roots scopes --include; include_path_prefixes replaces the 'sources/' anchor."""
|
|
134
|
+
if not is_available():
|
|
135
|
+
_record_run('unavailable')
|
|
136
|
+
return None
|
|
137
|
+
|
|
138
|
+
file_count = count_source_files(source_dir)
|
|
139
|
+
if timeout is None:
|
|
140
|
+
env_to = os.environ.get('NARVY_SEMGREP_TIMEOUT', '').strip()
|
|
141
|
+
timeout = (int(env_to) if env_to.isdigit()
|
|
142
|
+
else compute_semgrep_timeout(source_dir))
|
|
143
|
+
|
|
144
|
+
override_own = {e for e in (override_config.own if override_config else ())
|
|
145
|
+
if not e.endswith('*')}
|
|
146
|
+
override_vendor = {e for e in (override_config.vendor if override_config else ())
|
|
147
|
+
if not e.endswith('*')}
|
|
148
|
+
|
|
149
|
+
if own_roots or override_own:
|
|
150
|
+
scope_args = []
|
|
151
|
+
prefixes = include_path_prefixes if include_path_prefixes is not None else ["sources/"]
|
|
152
|
+
anchor = "**/" if include_path_prefixes is not None else ""
|
|
153
|
+
for root in (set(own_roots or ()) | override_own):
|
|
154
|
+
root_path = root.replace('.', '/')
|
|
155
|
+
for prefix in prefixes:
|
|
156
|
+
scope_args += ['--include', f"{anchor}{prefix}{root_path}/**"]
|
|
157
|
+
else:
|
|
158
|
+
scope_args = list(_EXCLUDE_ARGS)
|
|
159
|
+
|
|
160
|
+
for entry in override_vendor:
|
|
161
|
+
# --exclude wins over --include in semgrep's path filtering, so this
|
|
162
|
+
# suppresses a vendor entry nested inside an included own_root.
|
|
163
|
+
scope_args += ['--exclude', f"**/{entry.replace('.', '/')}/**"]
|
|
164
|
+
|
|
165
|
+
cmd = (
|
|
166
|
+
['semgrep', '--config', RULES_PATH, source_dir,
|
|
167
|
+
'--json', '--quiet', '--no-git-ignore',
|
|
168
|
+
'--timeout', '60'] # per-file timeout, not the whole run
|
|
169
|
+
+ scope_args
|
|
170
|
+
)
|
|
171
|
+
try:
|
|
172
|
+
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout)
|
|
173
|
+
except subprocess.TimeoutExpired:
|
|
174
|
+
# LAST_RUN stops the caller presenting the empty list as a clean pass.
|
|
175
|
+
_record_run('timeout', timeout_s=timeout, file_count=file_count)
|
|
176
|
+
return []
|
|
177
|
+
|
|
178
|
+
if not proc.stdout:
|
|
179
|
+
if proc.returncode not in (0, 1):
|
|
180
|
+
_record_run('error', timeout_s=timeout, file_count=file_count,
|
|
181
|
+
message=f"semgrep exited rc={proc.returncode}")
|
|
182
|
+
else:
|
|
183
|
+
_record_run('ok', timeout_s=timeout, file_count=file_count)
|
|
184
|
+
return []
|
|
185
|
+
|
|
186
|
+
try:
|
|
187
|
+
data = json.loads(proc.stdout)
|
|
188
|
+
except json.JSONDecodeError:
|
|
189
|
+
_record_run('error', timeout_s=timeout, file_count=file_count,
|
|
190
|
+
message="semgrep output was not valid JSON")
|
|
191
|
+
return []
|
|
192
|
+
|
|
193
|
+
_record_run('ok', timeout_s=timeout, file_count=file_count)
|
|
194
|
+
findings = []
|
|
195
|
+
for r in data.get('results', []):
|
|
196
|
+
rel_path = os.path.relpath(r.get('path', ''), source_dir)
|
|
197
|
+
extra = r.get('extra', {})
|
|
198
|
+
meta = extra.get('metadata', {})
|
|
199
|
+
# Strip semgrep's dotted config namespace; our own ids never contain dots.
|
|
200
|
+
check_id = r.get('check_id', 'semgrep-finding').rsplit('.', 1)[-1]
|
|
201
|
+
findings.append({
|
|
202
|
+
'rule_id': check_id,
|
|
203
|
+
'file_path': rel_path,
|
|
204
|
+
'name': extra.get('message', r.get('check_id', ''))[:120],
|
|
205
|
+
'severity': SEVERITY_MAP.get(extra.get('severity', 'INFO'), 'MEDIUM'),
|
|
206
|
+
'line': r.get('start', {}).get('line', 1),
|
|
207
|
+
'details': {
|
|
208
|
+
'cwe': meta.get('cwe', ''),
|
|
209
|
+
'masvs': meta.get('masvs', ''),
|
|
210
|
+
'description': extra.get('message', ''),
|
|
211
|
+
'recommendation': extra.get('message', ''),
|
|
212
|
+
},
|
|
213
|
+
'engine': 'semgrep',
|
|
214
|
+
})
|
|
215
|
+
|
|
216
|
+
# Dedupe before grading: one source line can hold several Random calls.
|
|
217
|
+
findings = weak_prng_context.dedupe_by_location(findings)
|
|
218
|
+
weak_prng_context.apply(findings, source_dir)
|
|
219
|
+
return findings
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""Decide whether a missing `__stack_chk_fail` symbol really means the stack
|
|
2
|
+
protector was off, by detecting whether the ELF holds any function that
|
|
3
|
+
`-fstack-protector-strong` would have instrumented.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import logging
|
|
8
|
+
import struct
|
|
9
|
+
from typing import Optional
|
|
10
|
+
|
|
11
|
+
logger = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
try:
|
|
14
|
+
import lief
|
|
15
|
+
LIEF_AVAILABLE = True
|
|
16
|
+
except ImportError: # pragma: no cover
|
|
17
|
+
LIEF_AVAILABLE = False
|
|
18
|
+
|
|
19
|
+
# Cap pure-Python decoding on very large .text sections.
|
|
20
|
+
_MAX_TEXT_SCAN_BYTES = 8 * 1024 * 1024
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _aarch64_takes_stack_address(code: bytes) -> bool:
|
|
24
|
+
"""AArch64 is fixed-width, so every 4-byte aligned word is an instruction."""
|
|
25
|
+
for off in range(0, min(len(code), _MAX_TEXT_SCAN_BYTES) // 4 * 4, 4):
|
|
26
|
+
w = struct.unpack_from("<I", code, off)[0]
|
|
27
|
+
# ADD (immediate), 64-bit; mask covers both shift forms.
|
|
28
|
+
if (w & 0xFF800000) == 0x91000000:
|
|
29
|
+
rd, rn = w & 0x1F, (w >> 5) & 0x1F
|
|
30
|
+
# Rd 31/29 are sp/frame-pointer setup, not an escaping address.
|
|
31
|
+
if rd not in (29, 31) and rn in (29, 31):
|
|
32
|
+
return True
|
|
33
|
+
# SUB (immediate), 64-bit: a slot below the frame pointer.
|
|
34
|
+
elif (w & 0xFF800000) == 0xD1000000:
|
|
35
|
+
rd, rn = w & 0x1F, (w >> 5) & 0x1F
|
|
36
|
+
if rd not in (29, 31) and rn == 29:
|
|
37
|
+
return True
|
|
38
|
+
return False
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _arm32_takes_stack_address(code: bytes) -> bool:
|
|
42
|
+
"""ARM32 mixes A32 and Thumb-2 in one .text, so both decodings are scanned."""
|
|
43
|
+
limit = min(len(code), _MAX_TEXT_SCAN_BYTES)
|
|
44
|
+
# A32
|
|
45
|
+
for off in range(0, limit // 4 * 4, 4):
|
|
46
|
+
w = struct.unpack_from("<I", code, off)[0]
|
|
47
|
+
rd = (w >> 12) & 0xF
|
|
48
|
+
rn = (w >> 16) & 0xF
|
|
49
|
+
if rd == 13:
|
|
50
|
+
continue
|
|
51
|
+
if (w & 0x0FE00000) == 0x02800000 and rn == 13: # ADD Rd, sp, #imm
|
|
52
|
+
return True
|
|
53
|
+
if (w & 0x0FE00000) == 0x02400000 and rn == 11: # SUB Rd, r11(fp), #imm
|
|
54
|
+
return True
|
|
55
|
+
if (w & 0x0FEF0FFF) == 0x01A0000D: # MOV Rd, sp
|
|
56
|
+
return True
|
|
57
|
+
# Thumb-2
|
|
58
|
+
for off in range(0, limit - 1, 2):
|
|
59
|
+
h = struct.unpack_from("<H", code, off)[0]
|
|
60
|
+
if 0xA800 <= h <= 0xAFFF: # ADD Rd, SP, #imm8*4
|
|
61
|
+
return True
|
|
62
|
+
if (h & 0xFF78) == 0x4668: # MOV Rd, SP
|
|
63
|
+
return True
|
|
64
|
+
# ADD.W Rd, SP, #const (T3) and ADDW Rd, SP, #imm12 (T4).
|
|
65
|
+
if h in (0xF10D, 0xF50D, 0xF20D, 0xF60D) and off + 3 < limit:
|
|
66
|
+
h2 = struct.unpack_from("<H", code, off + 2)[0]
|
|
67
|
+
if ((h2 >> 8) & 0xF) != 13:
|
|
68
|
+
return True
|
|
69
|
+
return False
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def has_instrumentable_stack_code(binary) -> Optional[bool]:
|
|
73
|
+
"""Does this ELF hold a function `-fstack-protector-strong` would instrument? True: absent `__stack_chk_fail` means protector off; False: proves nothing; None: no analysis. Never raises."""
|
|
74
|
+
if not LIEF_AVAILABLE or binary is None:
|
|
75
|
+
return None
|
|
76
|
+
try:
|
|
77
|
+
machine = binary.header.machine_type
|
|
78
|
+
if machine == lief.ELF.ARCH.AARCH64:
|
|
79
|
+
decode = _aarch64_takes_stack_address
|
|
80
|
+
elif machine == lief.ELF.ARCH.ARM:
|
|
81
|
+
decode = _arm32_takes_stack_address
|
|
82
|
+
else:
|
|
83
|
+
return None
|
|
84
|
+
|
|
85
|
+
code = None
|
|
86
|
+
for section in binary.sections:
|
|
87
|
+
if section.name == ".text":
|
|
88
|
+
code = bytes(section.content)
|
|
89
|
+
break
|
|
90
|
+
if not code:
|
|
91
|
+
# A missing .text is "no analysis"; an empty one is a real verdict.
|
|
92
|
+
return None if code is None else False
|
|
93
|
+
return decode(code)
|
|
94
|
+
except Exception as e: # pragma: no cover
|
|
95
|
+
logger.debug(f"[stack-protector-evidence] analysis failed: {e}")
|
|
96
|
+
return None
|