narvy-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. narvy/__init__.py +3 -0
  2. narvy/android/__init__.py +0 -0
  3. narvy/android/source_analyzer.py +268 -0
  4. narvy/android/split_bundle.py +475 -0
  5. narvy/android_rule_context.py +196 -0
  6. narvy/apk_memory_preflight.py +248 -0
  7. narvy/auth.py +40 -0
  8. narvy/ci_templates/bitbucket.yml +25 -0
  9. narvy/ci_templates/github.yml +60 -0
  10. narvy/ci_templates/gitlab.yml +32 -0
  11. narvy/cloud/__init__.py +0 -0
  12. narvy/cloud/aws_scan.py +1188 -0
  13. narvy/comment_filter.py +134 -0
  14. narvy/crypto_taint_lite.py +82 -0
  15. narvy/decompiler.py +337 -0
  16. narvy/doctor.py +244 -0
  17. narvy/host/__init__.py +0 -0
  18. narvy/host/audit.py +157 -0
  19. narvy/host/host_knowledge.py +982 -0
  20. narvy/host/lynis_bootstrap.py +400 -0
  21. narvy/host/lynis_parser.py +358 -0
  22. narvy/host/narvy_checks.py +789 -0
  23. narvy/host/report.py +69 -0
  24. narvy/host/ssh_exec.py +219 -0
  25. narvy/ios/__init__.py +0 -0
  26. narvy/ios/binary_analyzer.py +1201 -0
  27. narvy/ios/plist_checks.py +249 -0
  28. narvy/ios/source_analyzer.py +301 -0
  29. narvy/ios/third_party_filter.py +242 -0
  30. narvy/ios/trust_all_context.py +66 -0
  31. narvy/main.py +1983 -0
  32. narvy/native_hardening.py +503 -0
  33. narvy/reporter.py +151 -0
  34. narvy/rule_engine.py +57 -0
  35. narvy/rules/android/config.yml +77 -0
  36. narvy/rules/android/crypto.yml +34 -0
  37. narvy/rules/android/secrets.yml +161 -0
  38. narvy/rules/android/storage.yml +24 -0
  39. narvy/rules/android/webview.yml +23 -0
  40. narvy/rules/android.yml +219 -0
  41. narvy/rules/ios/objc/crypto.yml +56 -0
  42. narvy/rules/ios/objc/network.yml +67 -0
  43. narvy/rules/ios/objc/secrets.yml +79 -0
  44. narvy/rules/ios/objc/storage.yml +45 -0
  45. narvy/rules/ios/objc/webview.yml +45 -0
  46. narvy/rules/ios_swift.yml +327 -0
  47. narvy/rules/web/go.yml +2837 -0
  48. narvy/rules/web/java.yml +1576 -0
  49. narvy/rules/web/javascript.yml +3683 -0
  50. narvy/rules/web/kotlin.yml +413 -0
  51. narvy/rules/web/local/csharp_narvy/config.yml +61 -0
  52. narvy/rules/web/local/csharp_narvy/crypto.yml +64 -0
  53. narvy/rules/web/local/csharp_narvy/deserialization.yml +59 -0
  54. narvy/rules/web/local/csharp_narvy/injection.yml +122 -0
  55. narvy/rules/web/local/csharp_narvy/xxe.yml +48 -0
  56. narvy/rules/web/local/java_narvy/auth_jwt.yml +134 -0
  57. narvy/rules/web/local/java_narvy/deserialization.yml +108 -0
  58. narvy/rules/web/local/java_narvy/mybatis.yml +39 -0
  59. narvy/rules/web/local/java_narvy/snakeyaml.yml +34 -0
  60. narvy/rules/web/local/java_narvy/spring_authz.yml +32 -0
  61. narvy/rules/web/local/java_narvy/spring_config.yml +67 -0
  62. narvy/rules/web/local/java_narvy/spring_hardening.yml +291 -0
  63. narvy/rules/web/local/java_narvy/sqli.yml +235 -0
  64. narvy/rules/web/local/java_narvy/xxe.yml +212 -0
  65. narvy/rules/web/php.yml +1644 -0
  66. narvy/rules/web/python.yml +3967 -0
  67. narvy/rules/web/ruby.yml +703 -0
  68. narvy/rules/web/rust.yml +258 -0
  69. narvy/rules/web/secrets.yml +1420 -0
  70. narvy/rules/web/secrets_supplement.yml +383 -0
  71. narvy/sca/__init__.py +1 -0
  72. narvy/sca/android_deps.py +349 -0
  73. narvy/sca/ios_deps.py +578 -0
  74. narvy/sca/osv_client.py +617 -0
  75. narvy/sca/web_deps.py +955 -0
  76. narvy/scope_config.py +195 -0
  77. narvy/semgrep_engine.py +219 -0
  78. narvy/stack_protector_evidence.py +96 -0
  79. narvy/third_party_filter.py +211 -0
  80. narvy/uploader.py +92 -0
  81. narvy/weak_prng_context.py +270 -0
  82. narvy/web/__init__.py +0 -0
  83. narvy/web/nuclei_binary.py +105 -0
  84. narvy/web/scan_blocklist.py +96 -0
  85. narvy/web/scanner.py +842 -0
  86. narvy/web/source_analyzer.py +714 -0
  87. narvy/web/ssrf_guard.py +374 -0
  88. narvy_cli-1.0.0.dist-info/METADATA +165 -0
  89. narvy_cli-1.0.0.dist-info/RECORD +93 -0
  90. narvy_cli-1.0.0.dist-info/WHEEL +5 -0
  91. narvy_cli-1.0.0.dist-info/entry_points.txt +2 -0
  92. narvy_cli-1.0.0.dist-info/licenses/LICENSE +202 -0
  93. narvy_cli-1.0.0.dist-info/top_level.txt +1 -0
narvy/scope_config.py ADDED
@@ -0,0 +1,195 @@
1
+ """Loads and validates `.narvy-scope.yml`, the first-party/third-party override.
2
+
3
+ Parses the `own`/`vendor` lists. Never raises: a bad or missing file gives EMPTY.
4
+ """
5
+ from __future__ import annotations
6
+
7
+ import os
8
+ import sys
9
+ from typing import Dict, FrozenSet, NamedTuple, Optional
10
+
11
+ try:
12
+ import yaml
13
+ except ImportError: # pragma: no cover
14
+ yaml = None
15
+
16
+ SCOPE_CONFIG_FILENAME = ".narvy-scope.yml"
17
+ SCOPE_CONFIG_FILENAMES = (SCOPE_CONFIG_FILENAME,)
18
+
19
+ MAX_ENTRIES_PER_LIST = 500
20
+
21
+
22
+ class ScopeConfig(NamedTuple):
23
+ """Validated config: lower-cased patterns, exact unless they end in `*`."""
24
+ own: FrozenSet[str] = frozenset()
25
+ vendor: FrozenSet[str] = frozenset()
26
+ reasons: Dict[str, str] = {}
27
+ source_path: Optional[str] = None
28
+
29
+ def __bool__(self) -> bool:
30
+ return bool(self.own) or bool(self.vendor)
31
+
32
+
33
+ EMPTY = ScopeConfig()
34
+
35
+
36
+ def _warn(msg: str) -> None:
37
+ print(f"[narvy] WARNING: {msg}", file=sys.stderr)
38
+
39
+
40
+ def entry_matches(entry: str, candidate: str) -> bool:
41
+ """Match an undelimited token: exact, or prefix if entry ends in '*'. Args lower-cased."""
42
+ if entry.endswith("*"):
43
+ return candidate.startswith(entry[:-1])
44
+ return candidate == entry
45
+
46
+
47
+ def segment_prefix_match(entry: str, candidate: str, delimiter: str = ".") -> bool:
48
+ """Whole-segment prefix match for delimiter-separated identifiers. Args lower-cased."""
49
+ e = entry[:-1] if entry.endswith("*") else entry
50
+ e = e.rstrip(delimiter)
51
+ if not e:
52
+ return False
53
+ entry_segs = e.split(delimiter)
54
+ cand_segs = candidate.split(delimiter)
55
+ return cand_segs[:len(entry_segs)] == entry_segs
56
+
57
+
58
+ def find_scope_config_path(target_path: str) -> Optional[str]:
59
+ """First `.narvy-scope.yml` in the cwd, then next to target_path."""
60
+ target_dir = target_path if os.path.isdir(target_path) else os.path.dirname(
61
+ os.path.abspath(target_path))
62
+ target_dir = target_dir or "."
63
+
64
+ # cwd is checked before the target dir.
65
+ for base in (os.getcwd(), target_dir):
66
+ for name in SCOPE_CONFIG_FILENAMES:
67
+ candidate = os.path.join(base, name)
68
+ if os.path.isfile(candidate):
69
+ return candidate
70
+ return None
71
+
72
+
73
+ def _parse_entry(item, key: str, path: str):
74
+ """Return (name, reason) for one list item, or None if it is malformed."""
75
+ if isinstance(item, str):
76
+ name = item.strip()
77
+ if not name:
78
+ _warn(f"{path}: '{key}' contains an empty string entry - skipping it.")
79
+ return None
80
+ return name.lower(), None
81
+ if isinstance(item, dict):
82
+ name = item.get("name")
83
+ if not isinstance(name, str) or not name.strip():
84
+ _warn(f"{path}: '{key}' has a mapping entry with no valid 'name' "
85
+ f"({item!r}) - skipping it.")
86
+ return None
87
+ reason = item.get("reason")
88
+ reason = reason.strip() if isinstance(reason, str) and reason.strip() else None
89
+ return name.strip().lower(), reason
90
+ _warn(f"{path}: '{key}' contains an entry that is neither a string nor a "
91
+ f"{{name, reason}} mapping ({item!r}) - skipping it.")
92
+ return None
93
+
94
+
95
+ def _as_pattern_list(data: dict, key: str, path: str):
96
+ """(name, reason) tuples in file order, or None if data[key] is malformed."""
97
+ val = data.get(key)
98
+ if val is None:
99
+ return []
100
+ if not isinstance(val, list):
101
+ _warn(f"{path}: '{key}' should be a YAML list, got "
102
+ f"{type(val).__name__} - ignoring this config file, continuing "
103
+ f"with no overrides.")
104
+ return None
105
+ if len(val) > MAX_ENTRIES_PER_LIST:
106
+ _warn(f"{path}: '{key}' has {len(val)} entries, more than the "
107
+ f"{MAX_ENTRIES_PER_LIST} cap - using only the first "
108
+ f"{MAX_ENTRIES_PER_LIST} (in file order).")
109
+ val = val[:MAX_ENTRIES_PER_LIST]
110
+ out = []
111
+ for item in val:
112
+ parsed = _parse_entry(item, key, path)
113
+ if parsed is not None:
114
+ out.append(parsed)
115
+ return out
116
+
117
+
118
+ def _strip_star(pattern: str) -> str:
119
+ return pattern[:-1] if pattern.endswith("*") else pattern
120
+
121
+
122
+ def _patterns_overlap(a: str, b: str) -> bool:
123
+ """Load-time heuristic: could an own and a vendor pattern both match?"""
124
+ sa, sb = _strip_star(a), _strip_star(b)
125
+ return sa == sb or sa.startswith(sb) or sb.startswith(sa)
126
+
127
+
128
+ def load_scope_config(target_path: str) -> ScopeConfig:
129
+ """Auto-detect and parse the scope config. Never raises: degrades to EMPTY."""
130
+ path = find_scope_config_path(target_path)
131
+ if path is None:
132
+ return EMPTY
133
+
134
+ if yaml is None: # pragma: no cover
135
+ _warn(f"found {path} but PyYAML is not installed - overrides ignored.")
136
+ return EMPTY
137
+
138
+ try:
139
+ with open(path, "r", encoding="utf-8") as f:
140
+ data = yaml.safe_load(f)
141
+ except Exception as e:
142
+ _warn(f"could not read/parse {path} ({e.__class__.__name__}: {e}) - "
143
+ f"continuing with no overrides.")
144
+ return EMPTY
145
+
146
+ if data is None:
147
+ return EMPTY
148
+ if not isinstance(data, dict):
149
+ _warn(f"{path} does not look like a valid scope config (expected a "
150
+ f"YAML mapping with 'own'/'vendor' keys, got "
151
+ f"{type(data).__name__}) - continuing with no overrides.")
152
+ return EMPTY
153
+
154
+ own_entries = _as_pattern_list(data, "own", path)
155
+ vendor_entries = _as_pattern_list(data, "vendor", path)
156
+ if own_entries is None or vendor_entries is None:
157
+ return EMPTY
158
+
159
+ own_names = {name for name, _ in own_entries}
160
+ vendor_names = {name for name, _ in vendor_entries}
161
+
162
+ # Identical entry in both lists: ambiguous, drop from both.
163
+ conflicts = own_names & vendor_names
164
+ if conflicts:
165
+ plural = len(conflicts) != 1
166
+ _warn(
167
+ f"{path}: entr{'ies' if plural else 'y'} "
168
+ f"{', '.join(sorted(conflicts))!r} listed in BOTH 'own' and "
169
+ f"'vendor' - ambiguous, falling back to automatic classification "
170
+ f"for {'them' if plural else 'it'}."
171
+ )
172
+ own_names -= conflicts
173
+ vendor_names -= conflicts
174
+
175
+ # Two patterns that could both match: warn, drop neither ('own' wins at match time).
176
+ overlap_pairs = sorted({
177
+ (o, v) for o in own_names for v in vendor_names if _patterns_overlap(o, v)
178
+ })
179
+ for o, v in overlap_pairs:
180
+ _warn(
181
+ f"{path}: 'own' pattern {o!r} and 'vendor' pattern {v!r} could "
182
+ f"both match the same name - 'own' wins for anything matched by "
183
+ f"both, per policy."
184
+ )
185
+
186
+ if not own_names and not vendor_names:
187
+ return EMPTY
188
+
189
+ reasons = {name: reason for name, reason in (own_entries + vendor_entries) if reason}
190
+ return ScopeConfig(
191
+ own=frozenset(own_names),
192
+ vendor=frozenset(vendor_names),
193
+ reasons=reasons,
194
+ source_path=path,
195
+ )
@@ -0,0 +1,219 @@
1
+ """Semgrep-backed detection engine for the Android rule pack: runs Semgrep CE as a
2
+ subprocess over the jadx-decompiled tree. rule_engine.py stays the fallback for manifest/XML rules."""
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import os
7
+ import shutil
8
+ import subprocess
9
+ from typing import Any, Dict, List, Optional
10
+
11
+ import yaml
12
+
13
+ from .third_party_filter import THIRD_PARTY_PREFIXES
14
+ from .scope_config import ScopeConfig
15
+ from . import weak_prng_context
16
+
17
+ RULES_PATH = os.path.join(os.path.dirname(__file__), 'rules', 'android.yml')
18
+ # Third-party prefixes as --exclude globs; leading '**/' de-anchors from the scan root.
19
+ _EXCLUDE_ARGS = []
20
+ for _p in THIRD_PARTY_PREFIXES:
21
+ _EXCLUDE_ARGS += ['--exclude', f"**/{_p.replace('.', '/').rstrip('/')}/**"]
22
+
23
+ SEVERITY_MAP = {
24
+ 'ERROR': 'CRITICAL',
25
+ 'WARNING': 'HIGH',
26
+ 'INFO': 'MEDIUM',
27
+ }
28
+
29
+ # Whole-run timeout scales with tree size; a timeout is recorded in LAST_RUN.
30
+ SEMGREP_BASE_TIMEOUT_SEC = 600
31
+ SEMGREP_MAX_TIMEOUT_SEC = 1800
32
+ SEMGREP_TIMEOUT_PER_1K_FILES = 120
33
+
34
+ # Callers read this to tell "0 findings" apart from "never finished".
35
+ LAST_RUN: Dict[str, Any] = {
36
+ 'status': 'not_run', # ok | timeout | error | unavailable | not_run
37
+ 'timeout_s': None,
38
+ 'file_count': None,
39
+ 'degraded': False,
40
+ 'message': None,
41
+ }
42
+
43
+
44
+ def count_source_files(source_dir: str) -> int:
45
+ """Cheap file count used to size the semgrep timeout."""
46
+ skip = {'.git', 'node_modules', 'build', '.gradle', '.idea', 'Pods',
47
+ 'Carthage', '__pycache__', '.venv', 'venv', 'DerivedData'}
48
+ n = 0
49
+ try:
50
+ for dirpath, dirnames, filenames in os.walk(source_dir):
51
+ dirnames[:] = [d for d in dirnames if d not in skip]
52
+ n += len(filenames)
53
+ except OSError:
54
+ return 0
55
+ return n
56
+
57
+
58
+ def compute_semgrep_timeout(source_dir: str,
59
+ base: int = SEMGREP_BASE_TIMEOUT_SEC) -> int:
60
+ """Scale the whole-run timeout with tree size. Never below `base`."""
61
+ files = count_source_files(source_dir)
62
+ if files < 2000:
63
+ return base
64
+ extra = ((files - 2000) // 1000) * SEMGREP_TIMEOUT_PER_1K_FILES
65
+ return min(base + extra, SEMGREP_MAX_TIMEOUT_SEC)
66
+
67
+
68
+ def _record_run(status: str, timeout_s: Optional[int] = None,
69
+ file_count: Optional[int] = None,
70
+ message: Optional[str] = None) -> None:
71
+ LAST_RUN.update({
72
+ 'status': status,
73
+ 'timeout_s': timeout_s,
74
+ 'file_count': file_count,
75
+ 'degraded': status in ('timeout', 'error'),
76
+ 'message': message,
77
+ })
78
+
79
+
80
+ def degraded_coverage_note() -> Optional[str]:
81
+ """User-facing sentence when the last pass did not complete, else None."""
82
+ if not LAST_RUN.get('degraded'):
83
+ return None
84
+ if LAST_RUN.get('status') == 'timeout':
85
+ return (
86
+ "PARTIAL COVERAGE: the deep structural (Semgrep) pass did NOT "
87
+ f"finish within its {LAST_RUN.get('timeout_s')}s budget on this "
88
+ f"target ({LAST_RUN.get('file_count')} files). Zero findings from "
89
+ "that pass means 'not fully scanned', NOT 'clean'. The regex, "
90
+ "taint and SCA passes still ran to completion and their findings "
91
+ "are complete. Re-run with NARVY_SEMGREP_TIMEOUT set higher "
92
+ "(seconds), or scan a narrower path."
93
+ )
94
+ return (
95
+ "PARTIAL COVERAGE: the deep structural (Semgrep) pass failed "
96
+ f"({LAST_RUN.get('message')}). Zero findings from that pass means "
97
+ "'not fully scanned', NOT 'clean'."
98
+ )
99
+
100
+
101
+ def is_available() -> bool:
102
+ return shutil.which('semgrep') is not None
103
+
104
+
105
+ def load_rule_defs(rules_path: str = RULES_PATH) -> List[Dict[str, Any]]:
106
+ """Synthesize a rule-def dict per rule id in the YAML pack, for SARIF."""
107
+ try:
108
+ with open(rules_path, 'r') as f:
109
+ data = yaml.safe_load(f)
110
+ except OSError:
111
+ return []
112
+ defs = []
113
+ for rule in (data or {}).get('rules', []):
114
+ meta = rule.get('metadata', {}) or {}
115
+ defs.append({
116
+ 'id': rule['id'],
117
+ 'name': rule.get('message', rule['id'])[:120],
118
+ 'severity': SEVERITY_MAP.get(rule.get('severity', 'INFO'), 'MEDIUM'),
119
+ 'details': {
120
+ 'cwe': meta.get('cwe', ''),
121
+ 'masvs': meta.get('masvs', ''),
122
+ 'description': rule.get('message', ''),
123
+ 'recommendation': rule.get('message', ''),
124
+ },
125
+ })
126
+ return defs
127
+
128
+
129
+ def run_semgrep(source_dir: str, timeout: Optional[int] = None,
130
+ own_roots: Optional[set] = None,
131
+ override_config: Optional[ScopeConfig] = None,
132
+ include_path_prefixes: Optional[List[str]] = None) -> Optional[List[Dict[str, Any]]]:
133
+ """Run the Semgrep rule pack against source_dir; findings, or None if semgrep is unavailable. own_roots scopes --include; include_path_prefixes replaces the 'sources/' anchor."""
134
+ if not is_available():
135
+ _record_run('unavailable')
136
+ return None
137
+
138
+ file_count = count_source_files(source_dir)
139
+ if timeout is None:
140
+ env_to = os.environ.get('NARVY_SEMGREP_TIMEOUT', '').strip()
141
+ timeout = (int(env_to) if env_to.isdigit()
142
+ else compute_semgrep_timeout(source_dir))
143
+
144
+ override_own = {e for e in (override_config.own if override_config else ())
145
+ if not e.endswith('*')}
146
+ override_vendor = {e for e in (override_config.vendor if override_config else ())
147
+ if not e.endswith('*')}
148
+
149
+ if own_roots or override_own:
150
+ scope_args = []
151
+ prefixes = include_path_prefixes if include_path_prefixes is not None else ["sources/"]
152
+ anchor = "**/" if include_path_prefixes is not None else ""
153
+ for root in (set(own_roots or ()) | override_own):
154
+ root_path = root.replace('.', '/')
155
+ for prefix in prefixes:
156
+ scope_args += ['--include', f"{anchor}{prefix}{root_path}/**"]
157
+ else:
158
+ scope_args = list(_EXCLUDE_ARGS)
159
+
160
+ for entry in override_vendor:
161
+ # --exclude wins over --include in semgrep's path filtering, so this
162
+ # suppresses a vendor entry nested inside an included own_root.
163
+ scope_args += ['--exclude', f"**/{entry.replace('.', '/')}/**"]
164
+
165
+ cmd = (
166
+ ['semgrep', '--config', RULES_PATH, source_dir,
167
+ '--json', '--quiet', '--no-git-ignore',
168
+ '--timeout', '60'] # per-file timeout, not the whole run
169
+ + scope_args
170
+ )
171
+ try:
172
+ proc = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout)
173
+ except subprocess.TimeoutExpired:
174
+ # LAST_RUN stops the caller presenting the empty list as a clean pass.
175
+ _record_run('timeout', timeout_s=timeout, file_count=file_count)
176
+ return []
177
+
178
+ if not proc.stdout:
179
+ if proc.returncode not in (0, 1):
180
+ _record_run('error', timeout_s=timeout, file_count=file_count,
181
+ message=f"semgrep exited rc={proc.returncode}")
182
+ else:
183
+ _record_run('ok', timeout_s=timeout, file_count=file_count)
184
+ return []
185
+
186
+ try:
187
+ data = json.loads(proc.stdout)
188
+ except json.JSONDecodeError:
189
+ _record_run('error', timeout_s=timeout, file_count=file_count,
190
+ message="semgrep output was not valid JSON")
191
+ return []
192
+
193
+ _record_run('ok', timeout_s=timeout, file_count=file_count)
194
+ findings = []
195
+ for r in data.get('results', []):
196
+ rel_path = os.path.relpath(r.get('path', ''), source_dir)
197
+ extra = r.get('extra', {})
198
+ meta = extra.get('metadata', {})
199
+ # Strip semgrep's dotted config namespace; our own ids never contain dots.
200
+ check_id = r.get('check_id', 'semgrep-finding').rsplit('.', 1)[-1]
201
+ findings.append({
202
+ 'rule_id': check_id,
203
+ 'file_path': rel_path,
204
+ 'name': extra.get('message', r.get('check_id', ''))[:120],
205
+ 'severity': SEVERITY_MAP.get(extra.get('severity', 'INFO'), 'MEDIUM'),
206
+ 'line': r.get('start', {}).get('line', 1),
207
+ 'details': {
208
+ 'cwe': meta.get('cwe', ''),
209
+ 'masvs': meta.get('masvs', ''),
210
+ 'description': extra.get('message', ''),
211
+ 'recommendation': extra.get('message', ''),
212
+ },
213
+ 'engine': 'semgrep',
214
+ })
215
+
216
+ # Dedupe before grading: one source line can hold several Random calls.
217
+ findings = weak_prng_context.dedupe_by_location(findings)
218
+ weak_prng_context.apply(findings, source_dir)
219
+ return findings
@@ -0,0 +1,96 @@
1
+ """Decide whether a missing `__stack_chk_fail` symbol really means the stack
2
+ protector was off, by detecting whether the ELF holds any function that
3
+ `-fstack-protector-strong` would have instrumented.
4
+ """
5
+ from __future__ import annotations
6
+
7
+ import logging
8
+ import struct
9
+ from typing import Optional
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+ try:
14
+ import lief
15
+ LIEF_AVAILABLE = True
16
+ except ImportError: # pragma: no cover
17
+ LIEF_AVAILABLE = False
18
+
19
+ # Cap pure-Python decoding on very large .text sections.
20
+ _MAX_TEXT_SCAN_BYTES = 8 * 1024 * 1024
21
+
22
+
23
+ def _aarch64_takes_stack_address(code: bytes) -> bool:
24
+ """AArch64 is fixed-width, so every 4-byte aligned word is an instruction."""
25
+ for off in range(0, min(len(code), _MAX_TEXT_SCAN_BYTES) // 4 * 4, 4):
26
+ w = struct.unpack_from("<I", code, off)[0]
27
+ # ADD (immediate), 64-bit; mask covers both shift forms.
28
+ if (w & 0xFF800000) == 0x91000000:
29
+ rd, rn = w & 0x1F, (w >> 5) & 0x1F
30
+ # Rd 31/29 are sp/frame-pointer setup, not an escaping address.
31
+ if rd not in (29, 31) and rn in (29, 31):
32
+ return True
33
+ # SUB (immediate), 64-bit: a slot below the frame pointer.
34
+ elif (w & 0xFF800000) == 0xD1000000:
35
+ rd, rn = w & 0x1F, (w >> 5) & 0x1F
36
+ if rd not in (29, 31) and rn == 29:
37
+ return True
38
+ return False
39
+
40
+
41
+ def _arm32_takes_stack_address(code: bytes) -> bool:
42
+ """ARM32 mixes A32 and Thumb-2 in one .text, so both decodings are scanned."""
43
+ limit = min(len(code), _MAX_TEXT_SCAN_BYTES)
44
+ # A32
45
+ for off in range(0, limit // 4 * 4, 4):
46
+ w = struct.unpack_from("<I", code, off)[0]
47
+ rd = (w >> 12) & 0xF
48
+ rn = (w >> 16) & 0xF
49
+ if rd == 13:
50
+ continue
51
+ if (w & 0x0FE00000) == 0x02800000 and rn == 13: # ADD Rd, sp, #imm
52
+ return True
53
+ if (w & 0x0FE00000) == 0x02400000 and rn == 11: # SUB Rd, r11(fp), #imm
54
+ return True
55
+ if (w & 0x0FEF0FFF) == 0x01A0000D: # MOV Rd, sp
56
+ return True
57
+ # Thumb-2
58
+ for off in range(0, limit - 1, 2):
59
+ h = struct.unpack_from("<H", code, off)[0]
60
+ if 0xA800 <= h <= 0xAFFF: # ADD Rd, SP, #imm8*4
61
+ return True
62
+ if (h & 0xFF78) == 0x4668: # MOV Rd, SP
63
+ return True
64
+ # ADD.W Rd, SP, #const (T3) and ADDW Rd, SP, #imm12 (T4).
65
+ if h in (0xF10D, 0xF50D, 0xF20D, 0xF60D) and off + 3 < limit:
66
+ h2 = struct.unpack_from("<H", code, off + 2)[0]
67
+ if ((h2 >> 8) & 0xF) != 13:
68
+ return True
69
+ return False
70
+
71
+
72
+ def has_instrumentable_stack_code(binary) -> Optional[bool]:
73
+ """Does this ELF hold a function `-fstack-protector-strong` would instrument? True: absent `__stack_chk_fail` means protector off; False: proves nothing; None: no analysis. Never raises."""
74
+ if not LIEF_AVAILABLE or binary is None:
75
+ return None
76
+ try:
77
+ machine = binary.header.machine_type
78
+ if machine == lief.ELF.ARCH.AARCH64:
79
+ decode = _aarch64_takes_stack_address
80
+ elif machine == lief.ELF.ARCH.ARM:
81
+ decode = _arm32_takes_stack_address
82
+ else:
83
+ return None
84
+
85
+ code = None
86
+ for section in binary.sections:
87
+ if section.name == ".text":
88
+ code = bytes(section.content)
89
+ break
90
+ if not code:
91
+ # A missing .text is "no analysis"; an empty one is a real verdict.
92
+ return None if code is None else False
93
+ return decode(code)
94
+ except Exception as e: # pragma: no cover
95
+ logger.debug(f"[stack-protector-evidence] analysis failed: {e}")
96
+ return None