narvy-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. narvy/__init__.py +3 -0
  2. narvy/android/__init__.py +0 -0
  3. narvy/android/source_analyzer.py +268 -0
  4. narvy/android/split_bundle.py +475 -0
  5. narvy/android_rule_context.py +196 -0
  6. narvy/apk_memory_preflight.py +248 -0
  7. narvy/auth.py +40 -0
  8. narvy/ci_templates/bitbucket.yml +25 -0
  9. narvy/ci_templates/github.yml +60 -0
  10. narvy/ci_templates/gitlab.yml +32 -0
  11. narvy/cloud/__init__.py +0 -0
  12. narvy/cloud/aws_scan.py +1188 -0
  13. narvy/comment_filter.py +134 -0
  14. narvy/crypto_taint_lite.py +82 -0
  15. narvy/decompiler.py +337 -0
  16. narvy/doctor.py +244 -0
  17. narvy/host/__init__.py +0 -0
  18. narvy/host/audit.py +157 -0
  19. narvy/host/host_knowledge.py +982 -0
  20. narvy/host/lynis_bootstrap.py +400 -0
  21. narvy/host/lynis_parser.py +358 -0
  22. narvy/host/narvy_checks.py +789 -0
  23. narvy/host/report.py +69 -0
  24. narvy/host/ssh_exec.py +219 -0
  25. narvy/ios/__init__.py +0 -0
  26. narvy/ios/binary_analyzer.py +1201 -0
  27. narvy/ios/plist_checks.py +249 -0
  28. narvy/ios/source_analyzer.py +301 -0
  29. narvy/ios/third_party_filter.py +242 -0
  30. narvy/ios/trust_all_context.py +66 -0
  31. narvy/main.py +1983 -0
  32. narvy/native_hardening.py +503 -0
  33. narvy/reporter.py +151 -0
  34. narvy/rule_engine.py +57 -0
  35. narvy/rules/android/config.yml +77 -0
  36. narvy/rules/android/crypto.yml +34 -0
  37. narvy/rules/android/secrets.yml +161 -0
  38. narvy/rules/android/storage.yml +24 -0
  39. narvy/rules/android/webview.yml +23 -0
  40. narvy/rules/android.yml +219 -0
  41. narvy/rules/ios/objc/crypto.yml +56 -0
  42. narvy/rules/ios/objc/network.yml +67 -0
  43. narvy/rules/ios/objc/secrets.yml +79 -0
  44. narvy/rules/ios/objc/storage.yml +45 -0
  45. narvy/rules/ios/objc/webview.yml +45 -0
  46. narvy/rules/ios_swift.yml +327 -0
  47. narvy/rules/web/go.yml +2837 -0
  48. narvy/rules/web/java.yml +1576 -0
  49. narvy/rules/web/javascript.yml +3683 -0
  50. narvy/rules/web/kotlin.yml +413 -0
  51. narvy/rules/web/local/csharp_narvy/config.yml +61 -0
  52. narvy/rules/web/local/csharp_narvy/crypto.yml +64 -0
  53. narvy/rules/web/local/csharp_narvy/deserialization.yml +59 -0
  54. narvy/rules/web/local/csharp_narvy/injection.yml +122 -0
  55. narvy/rules/web/local/csharp_narvy/xxe.yml +48 -0
  56. narvy/rules/web/local/java_narvy/auth_jwt.yml +134 -0
  57. narvy/rules/web/local/java_narvy/deserialization.yml +108 -0
  58. narvy/rules/web/local/java_narvy/mybatis.yml +39 -0
  59. narvy/rules/web/local/java_narvy/snakeyaml.yml +34 -0
  60. narvy/rules/web/local/java_narvy/spring_authz.yml +32 -0
  61. narvy/rules/web/local/java_narvy/spring_config.yml +67 -0
  62. narvy/rules/web/local/java_narvy/spring_hardening.yml +291 -0
  63. narvy/rules/web/local/java_narvy/sqli.yml +235 -0
  64. narvy/rules/web/local/java_narvy/xxe.yml +212 -0
  65. narvy/rules/web/php.yml +1644 -0
  66. narvy/rules/web/python.yml +3967 -0
  67. narvy/rules/web/ruby.yml +703 -0
  68. narvy/rules/web/rust.yml +258 -0
  69. narvy/rules/web/secrets.yml +1420 -0
  70. narvy/rules/web/secrets_supplement.yml +383 -0
  71. narvy/sca/__init__.py +1 -0
  72. narvy/sca/android_deps.py +349 -0
  73. narvy/sca/ios_deps.py +578 -0
  74. narvy/sca/osv_client.py +617 -0
  75. narvy/sca/web_deps.py +955 -0
  76. narvy/scope_config.py +195 -0
  77. narvy/semgrep_engine.py +219 -0
  78. narvy/stack_protector_evidence.py +96 -0
  79. narvy/third_party_filter.py +211 -0
  80. narvy/uploader.py +92 -0
  81. narvy/weak_prng_context.py +270 -0
  82. narvy/web/__init__.py +0 -0
  83. narvy/web/nuclei_binary.py +105 -0
  84. narvy/web/scan_blocklist.py +96 -0
  85. narvy/web/scanner.py +842 -0
  86. narvy/web/source_analyzer.py +714 -0
  87. narvy/web/ssrf_guard.py +374 -0
  88. narvy_cli-1.0.0.dist-info/METADATA +165 -0
  89. narvy_cli-1.0.0.dist-info/RECORD +93 -0
  90. narvy_cli-1.0.0.dist-info/WHEEL +5 -0
  91. narvy_cli-1.0.0.dist-info/entry_points.txt +2 -0
  92. narvy_cli-1.0.0.dist-info/licenses/LICENSE +202 -0
  93. narvy_cli-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,714 @@
1
+ """Web/backend source-repo SAST entry point for JS/TS, Python, PHP, Go, Ruby,
2
+ Rust, Java/Kotlin and C#/.NET, driving vendored semgrep rule packs.
3
+ """
4
+ from __future__ import annotations
5
+
6
+ import json
7
+ import os
8
+ import re
9
+ import subprocess
10
+ from typing import Any, Dict, List, Optional, Set, Tuple
11
+
12
+ import yaml
13
+
14
+ from narvy import semgrep_engine
15
+
16
+ _HERE = os.path.dirname(__file__)
17
+ _WEB_RULES_DIR = os.path.join(_HERE, "..", "rules", "web")
18
+ _WEB_LOCAL_RULES_DIR = os.path.join(_WEB_RULES_DIR, "local")
19
+ _REPO_ROOT = os.path.abspath(os.path.join(_HERE, "..", ".."))
20
+
21
+
22
+ def _find_pro_rules_root() -> Optional[str]:
23
+ """Locate the optional extended rule tree via NARVY_PRO_RULES_DIR; None when unset."""
24
+ env_dir = os.environ.get("NARVY_PRO_RULES_DIR", "").strip()
25
+ if env_dir and os.path.isdir(os.path.join(env_dir, "web")):
26
+ return env_dir
27
+ return None
28
+
29
+
30
+ _PRO_RULES_ROOT = _find_pro_rules_root()
31
+ _PRO_WEB_RULES_DIR = (
32
+ os.path.join(_PRO_RULES_ROOT, "web") if _PRO_RULES_ROOT else ""
33
+ )
34
+ _PRO_WEB_LOCAL_RULES_DIR = (
35
+ os.path.join(_PRO_WEB_RULES_DIR, "local") if _PRO_RULES_ROOT else ""
36
+ )
37
+
38
+
39
+ def pro_rules_available() -> bool:
40
+ return bool(_PRO_WEB_RULES_DIR) and os.path.isdir(_PRO_WEB_RULES_DIR)
41
+
42
+
43
+ # C#/.NET ships a base rule pack, so it is always a supported web-source language.
44
+ SUPPORTED_LANGUAGES_LABEL = (
45
+ "JavaScript/TypeScript, Python, PHP, Go, Ruby, Rust, Java/Kotlin, C#/.NET"
46
+ )
47
+
48
+ _ALWAYS_ON_PACKS = ["secrets", "secrets_supplement"]
49
+
50
+ WEB_PACK_MAP: Dict[str, List[str]] = {
51
+ # javascript.yml is tagged for both languages, so one pack covers both labels.
52
+ "javascript": ["javascript"] + _ALWAYS_ON_PACKS,
53
+ "typescript": ["javascript"] + _ALWAYS_ON_PACKS,
54
+ "python": ["python"] + _ALWAYS_ON_PACKS,
55
+ "php": ["php", "javascript"] + _ALWAYS_ON_PACKS,
56
+ "go": ["go"] + _ALWAYS_ON_PACKS,
57
+ "ruby": ["ruby"] + _ALWAYS_ON_PACKS,
58
+ "rust": ["rust"] + _ALWAYS_ON_PACKS,
59
+ "java": ["java", "kotlin"] + _ALWAYS_ON_PACKS,
60
+ # C# has no <name>.yml: its rules live in a local pack wired through _LOCAL_PACK_DIRS.
61
+ "csharp": list(_ALWAYS_ON_PACKS),
62
+ }
63
+
64
+ _LOW_COVERAGE_STACKS: Dict[str, str] = {
65
+ "rust": (
66
+ "Rust rule coverage is the thinnest of the supported stacks "
67
+ "(12 rules): command injection, MD5/SHA-1, sqlx/diesel/rusqlite/"
68
+ "postgres raw SQL, bincode deserialization, and axum/actix "
69
+ "extractor-to-sink taint for path traversal and process spawning. "
70
+ "TLS-verification-bypass and Trojan-Source bidi-character checks "
71
+ "are NOT in the current pack. Memory safety (use-after-free, "
72
+ "aliasing bugs in "
73
+ "`unsafe`) is NOT covered by any static pattern here and needs "
74
+ "Miri or a manual review; dependency CVEs come from this CLI's "
75
+ "own SCA pass, not from these rules."
76
+ ),
77
+ "csharp": (
78
+ "C#/.NET rule coverage is starter-level (10 rules): SQL injection "
79
+ "via ADO.NET command-text concatenation, path traversal from "
80
+ "HttpRequest input into System.IO file APIs, command injection from "
81
+ "request input into Process.Start, XXE via XmlDocument's default "
82
+ "resolver, BinaryFormatter and Json.NET TypeNameHandling "
83
+ "deserialization, weak hashes (MD5/SHA-1), weak/legacy ciphers "
84
+ "(DES/3DES/RC2), and two web.config/attribute checks (request "
85
+ "validation disabled, debug compilation left on). NOT covered: "
86
+ "Dapper/EF Core raw-SQL sinks, model-bound action-method parameters "
87
+ "as a taint source (only Request.Query/Form/Headers/Cookies/"
88
+ "RouteValues indexers are), Razor/Blazor template XSS, "
89
+ "authorization-attribute analysis, TLS/certificate-validation "
90
+ "settings, and connection-string secrets. Dependency CVEs come from "
91
+ "this CLI's own NuGet SCA pass, not from these rules."
92
+ ),
93
+ }
94
+
95
+ if not pro_rules_available():
96
+ _LOW_COVERAGE_STACKS["rust"] = (
97
+ "Rust rule coverage is the thinnest of the supported stacks "
98
+ "(7 rules): command injection, MD5/SHA-1 as a password hash, sqlx "
99
+ "raw SQL, and bincode deserialization. Memory safety "
100
+ "(use-after-free, aliasing bugs in `unsafe`) is NOT covered by any "
101
+ "static pattern here and needs Miri or a manual review; dependency "
102
+ "CVEs come from this CLI's own SCA pass, not from these rules."
103
+ )
104
+
105
+ _LOCAL_PACK_DIRS: Dict[str, Tuple[str, ...]] = {
106
+ "javascript": ("typescript_narvy",),
107
+ "typescript": ("typescript_narvy",),
108
+ "php": (),
109
+ "python": (),
110
+ "go": (),
111
+ "java": ("java_narvy",),
112
+ "ruby": (),
113
+ "rust": ("rust_narvy",),
114
+ "csharp": ("csharp_narvy",),
115
+ }
116
+
117
+ # Recognized stacks with no dedicated rule pack (zero semgrep configs).
118
+ _UNPORTED_STACK_LABELS = {
119
+ "dotnet": "C#/.NET backend (*.csproj / *.sln)",
120
+ }
121
+
122
+ _UNPORTED_STACK_PARTIAL_NOTE = {
123
+ "dotnet": (
124
+ "Its .cs files were still passed through the always-on cross-language "
125
+ "packs, but those carry ZERO C#-tagged rules - only their "
126
+ "language-agnostic hardcoded-secret patterns apply, which match any "
127
+ "file type. Treat that as partial, best-effort coverage (secrets "
128
+ "only), NOT a .NET security review: no ASP.NET Core framework rules, "
129
+ "no C# injection or deserialization rules, no C# taint analysis."
130
+ ),
131
+ }
132
+
133
+ _VENDOR_DIR_NAMES = frozenset({
134
+ "node_modules", "vendor", "bower_components",
135
+ ".venv", "venv", ".tox", "__pycache__",
136
+ "Pods", "Carthage", "DerivedData",
137
+ ".git", ".gradle", ".idea", ".vscode", ".vs",
138
+ "build", "dist", "target", "out", ".next", ".nuxt", ".cache",
139
+ "coverage", ".pytest_cache", "egg-info",
140
+ })
141
+
142
+ _LANG_EXTS: Dict[str, Tuple[str, ...]] = {
143
+ "javascript": (".js", ".jsx", ".mjs", ".cjs"),
144
+ "typescript": (".ts", ".tsx"),
145
+ "python": (".py",),
146
+ "php": (".php",),
147
+ "go": (".go",),
148
+ # Template extensions (.erb, .cshtml, .razor, .jsp) excluded: semgrep has no analyzer for them.
149
+ "ruby": (".rb", ".rake", ".gemspec"),
150
+ "rust": (".rs",),
151
+ "java": (".java", ".kt", ".kts"),
152
+ "csharp": (".cs",),
153
+ }
154
+
155
+ _DENSITY_THRESHOLD = {
156
+ "javascript": 5, "typescript": 5, "python": 5, "php": 2, "go": 5,
157
+ "ruby": 5, "rust": 5,
158
+ "java": 5,
159
+ "csharp": 5,
160
+ }
161
+
162
+ # Excluded from density counting so stray served assets do not promote an extra rule pack.
163
+ _SERVED_ASSET_DIR_NAMES = frozenset({"public", "static", "assets", "templates", "downloads"})
164
+
165
+ _MAX_DETECT_DEPTH = 8
166
+
167
+ # Must stay below main.py's _WEB_SOURCE_DETECT_MAX_DEPTH (3).
168
+ _NODE_APP_MAX_DEPTH = 2
169
+
170
+ # Go toolchain generated-code header convention (protoc-gen-go, codegen, mockgen, stringer).
171
+ _GENERATED_CODE_HEADER_RE = re.compile(
172
+ r"^\s*(?://|#)\s*Code generated .* DO NOT EDIT\.\s*$", re.MULTILINE
173
+ )
174
+ _GENERATED_HEADER_SCAN_BYTES = 2000
175
+
176
+
177
+ # Rule ids that flag the SAME bug (hardcoded secret into a JWT sign/verify call);
178
+ # multiple hits on one (file,line) are one finding double-counted.
179
+ _JWT_ALIAS_RULE_IDS = frozenset({
180
+ "js-jsonwebtoken-hardcoded-secret",
181
+ "hardcoded-jwt-sign-secret",
182
+ })
183
+
184
+ _SEVERITY_RANK = {"CRITICAL": 5, "HIGH": 4, "MEDIUM": 3, "LOW": 2, "INFO": 1}
185
+
186
+ # Markers of the hardcoded-secret family, the only category down-weighted in test paths.
187
+ _SECRET_RULE_MARKERS = (
188
+ "secret", "hardcoded", "api-key", "apikey", "api_key",
189
+ "token", "credential", "password", "private-key", "privatekey",
190
+ "aws", "gcp", "google-", "stripe", "slack", "twilio",
191
+ )
192
+
193
+ # Path fragments and filename patterns that mark test / fixture / mock code.
194
+ _TEST_PATH_DIR_FRAGMENTS = (
195
+ "/test/", "/tests/", "/__tests__/", "/spec/", "/specs/",
196
+ "/mocks/", "/__mocks__/", "/fixtures/", "/fixture/", "/testdata/",
197
+ "/e2e/", "/cypress/", "/testing/",
198
+ )
199
+ # Basename only; token must sit on a boundary so "latest.js"/"contest.js" are not tests.
200
+ _TEST_FILENAME_RE = re.compile(
201
+ r"(?:^|[._-])(?:test|tests|spec|specs|mock|mocks|fixture|fixtures)(?:[._-]|$)",
202
+ re.IGNORECASE,
203
+ )
204
+
205
+ LAST_RUN_JWT_DUPLICATES_DROPPED = 0
206
+ LAST_RUN_TEST_SECRETS_DOWNWEIGHTED = 0
207
+
208
+
209
+ def _is_secret_finding(finding: Dict[str, Any]) -> bool:
210
+ rid = finding.get("rule_id", "").lower()
211
+ return any(m in rid for m in _SECRET_RULE_MARKERS)
212
+
213
+
214
+ def _is_test_path(rel_path: str) -> bool:
215
+ p = "/" + rel_path.replace("\\", "/").lstrip("/")
216
+ if any(frag in p.lower() for frag in _TEST_PATH_DIR_FRAGMENTS):
217
+ return True
218
+ base = p.rsplit("/", 1)[-1]
219
+ return bool(_TEST_FILENAME_RE.search(base))
220
+
221
+
222
+ def _collapse_jwt_aliases(findings: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
223
+ """Drop the lower-severity alias when >1 JWT-secret rule hits one (file,line)."""
224
+ global LAST_RUN_JWT_DUPLICATES_DROPPED
225
+ # Index of alias-rule findings per (file,line).
226
+ by_loc: Dict[Tuple[str, int], List[int]] = {}
227
+ for idx, f in enumerate(findings):
228
+ if f.get("rule_id") in _JWT_ALIAS_RULE_IDS:
229
+ by_loc.setdefault((f.get("file_path"), f.get("line")), []).append(idx)
230
+
231
+ drop: Set[int] = set()
232
+ for _loc, idxs in by_loc.items():
233
+ if len(idxs) < 2:
234
+ continue
235
+ # Keep the highest-severity finding; drop the rest of the alias group.
236
+ keep = max(idxs, key=lambda i: _SEVERITY_RANK.get(findings[i].get("severity"), 0))
237
+ for i in idxs:
238
+ if i != keep:
239
+ drop.add(i)
240
+ LAST_RUN_JWT_DUPLICATES_DROPPED = len(drop)
241
+ return [f for i, f in enumerate(findings) if i not in drop]
242
+
243
+
244
+ def _downweight_test_secrets(findings: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
245
+ """Downgrade (never drop) hardcoded-secret findings in test/fixture paths to LOW and tag them."""
246
+ global LAST_RUN_TEST_SECRETS_DOWNWEIGHTED
247
+ n = 0
248
+ for f in findings:
249
+ if (
250
+ _is_secret_finding(f)
251
+ and f.get("severity") in ("CRITICAL", "HIGH", "MEDIUM")
252
+ and _is_test_path(f.get("file_path", ""))
253
+ ):
254
+ f["severity"] = "LOW"
255
+ details = f.setdefault("details", {})
256
+ note = (
257
+ " [Narvy: this hit is in a test/fixture path, where a hardcoded "
258
+ "secret is usually a fixture rather than a production credential, "
259
+ "so severity is lowered - confirm it is not a real leaked secret.]"
260
+ )
261
+ if note.strip() not in (details.get("description") or ""):
262
+ details["description"] = (details.get("description") or "") + note
263
+ n += 1
264
+ LAST_RUN_TEST_SECRETS_DOWNWEIGHTED = n
265
+ return findings
266
+
267
+
268
+ def _should_prune_dir(name: str) -> bool:
269
+ return name in _VENDOR_DIR_NAMES or name.startswith(".")
270
+
271
+
272
+ def _find_marker(source_dir: str, filenames: Tuple[str, ...], max_depth: int = 1) -> bool:
273
+ """True if any of `filenames` exists up to `max_depth` levels deep."""
274
+ base_depth = source_dir.rstrip(os.sep).count(os.sep)
275
+ for root, dirs, files in os.walk(source_dir):
276
+ depth = root.rstrip(os.sep).count(os.sep) - base_depth
277
+ if depth > max_depth:
278
+ dirs[:] = []
279
+ continue
280
+ dirs[:] = [d for d in dirs if not _should_prune_dir(d)]
281
+ if any(f in filenames for f in files):
282
+ return True
283
+ return False
284
+
285
+
286
+ def _find_marker_suffix(source_dir: str, suffixes: Tuple[str, ...], max_depth: int = 1) -> bool:
287
+ base_depth = source_dir.rstrip(os.sep).count(os.sep)
288
+ for root, dirs, files in os.walk(source_dir):
289
+ depth = root.rstrip(os.sep).count(os.sep) - base_depth
290
+ if depth > max_depth:
291
+ dirs[:] = []
292
+ continue
293
+ dirs[:] = [d for d in dirs if not _should_prune_dir(d)]
294
+ if any(f.endswith(suffixes) for f in files):
295
+ return True
296
+ return False
297
+
298
+
299
+ def _count_density(source_dir: str, exts: Tuple[str, ...]) -> int:
300
+ n = 0
301
+ for root, dirs, files in os.walk(source_dir):
302
+ dirs[:] = [d for d in dirs if not _should_prune_dir(d) and d not in _SERVED_ASSET_DIR_NAMES]
303
+ for f in files:
304
+ if f.endswith(exts):
305
+ n += 1
306
+ return n
307
+
308
+
309
+ def _has_real_node_app(source_dir: str, max_depth: int = _NODE_APP_MAX_DEPTH) -> bool:
310
+ """package.json within `max_depth` levels carrying a real dependency key."""
311
+ candidates: List[str] = []
312
+ base_depth = source_dir.rstrip(os.sep).count(os.sep)
313
+ for root, dirs, files in os.walk(source_dir):
314
+ depth = root.rstrip(os.sep).count(os.sep) - base_depth
315
+ if depth > max_depth:
316
+ dirs[:] = []
317
+ continue
318
+ dirs[:] = [d for d in dirs if not _should_prune_dir(d)]
319
+ if "package.json" in files:
320
+ candidates.append(os.path.join(root, "package.json"))
321
+ for candidate in candidates:
322
+ if not os.path.isfile(candidate):
323
+ continue
324
+ try:
325
+ with open(candidate, "r", encoding="utf-8", errors="ignore") as f:
326
+ data = json.load(f)
327
+ except (OSError, json.JSONDecodeError):
328
+ continue
329
+ if isinstance(data, dict) and any(
330
+ k in data for k in ("dependencies", "devDependencies", "scripts")
331
+ ):
332
+ return True
333
+ return False
334
+
335
+
336
+ def detect_web_stacks(source_dir: str) -> Tuple[Set[str], Set[str]]:
337
+ """Returns (ported_stacks, unported_stacks); either may be empty."""
338
+ ported: Set[str] = set()
339
+ unported: Set[str] = set()
340
+
341
+ if (
342
+ _find_marker(source_dir, ("requirements.txt", "setup.py", "pyproject.toml", "Pipfile"))
343
+ or _count_density(source_dir, _LANG_EXTS["python"]) >= _DENSITY_THRESHOLD["python"]
344
+ ):
345
+ ported.add("python")
346
+
347
+ if (
348
+ os.path.isfile(os.path.join(source_dir, "composer.json"))
349
+ or _count_density(source_dir, _LANG_EXTS["php"]) >= _DENSITY_THRESHOLD["php"]
350
+ ):
351
+ ported.add("php")
352
+
353
+ if (
354
+ _has_real_node_app(source_dir)
355
+ or _count_density(source_dir, _LANG_EXTS["javascript"]) >= _DENSITY_THRESHOLD["javascript"]
356
+ or _count_density(source_dir, _LANG_EXTS["typescript"]) >= _DENSITY_THRESHOLD["typescript"]
357
+ ):
358
+ # Both labels resolve to the same pack; this only sets the label in notes.
359
+ if _count_density(source_dir, _LANG_EXTS["typescript"]) > 0:
360
+ ported.add("typescript")
361
+ else:
362
+ ported.add("javascript")
363
+
364
+ if (
365
+ _find_marker(source_dir, ("go.mod", "go.sum"))
366
+ or _count_density(source_dir, _LANG_EXTS["go"]) >= _DENSITY_THRESHOLD["go"]
367
+ ):
368
+ ported.add("go")
369
+
370
+ if (
371
+ _find_marker(source_dir, ("Gemfile", "Gemfile.lock"))
372
+ or _find_marker_suffix(source_dir, (".gemspec",))
373
+ or _count_density(source_dir, _LANG_EXTS["ruby"]) >= _DENSITY_THRESHOLD["ruby"]
374
+ ):
375
+ ported.add("ruby")
376
+
377
+ if (
378
+ _find_marker(source_dir, ("Cargo.toml", "Cargo.lock"))
379
+ or _count_density(source_dir, _LANG_EXTS["rust"]) >= _DENSITY_THRESHOLD["rust"]
380
+ ):
381
+ ported.add("rust")
382
+
383
+ # Android projects are already excluded upstream, so a Gradle/Maven project here is JVM backend/library.
384
+ if (
385
+ _find_marker(source_dir, ("pom.xml", "build.gradle", "build.gradle.kts",
386
+ "settings.gradle", "settings.gradle.kts"), max_depth=2)
387
+ or _count_density(source_dir, _LANG_EXTS["java"]) >= _DENSITY_THRESHOLD["java"]
388
+ ):
389
+ ported.add("java")
390
+
391
+ # Depth 3: a .NET project file is typically src/<Project>/<Project>.csproj, not the root.
392
+ if (
393
+ _find_marker_suffix(source_dir, (".csproj", ".fsproj", ".vbproj", ".sln", ".slnx"), max_depth=3)
394
+ or _find_marker(source_dir, ("packages.config", "Directory.Packages.props"), max_depth=3)
395
+ or _count_density(source_dir, _LANG_EXTS["csharp"]) >= _DENSITY_THRESHOLD["csharp"]
396
+ ):
397
+ ported.add("csharp")
398
+
399
+ return ported, unported
400
+
401
+
402
+ def _pack_path(pack_name: str) -> Optional[str]:
403
+ p = os.path.join(_WEB_RULES_DIR, f"{pack_name}.yml")
404
+ return p if os.path.isfile(p) else None
405
+
406
+
407
+ def _pro_pack_path(pack_name: str) -> Optional[str]:
408
+ if not _PRO_WEB_RULES_DIR:
409
+ return None
410
+ p = os.path.join(_PRO_WEB_RULES_DIR, f"{pack_name}_pro.yml")
411
+ return p if os.path.isfile(p) else None
412
+
413
+
414
+ def _local_pack_path(local_dir_name: str) -> Optional[str]:
415
+ """Base tree wins over the extended tree when both carry the name."""
416
+ for parent in (_WEB_LOCAL_RULES_DIR, _PRO_WEB_LOCAL_RULES_DIR):
417
+ if not parent:
418
+ continue
419
+ p = os.path.join(parent, local_dir_name)
420
+ if os.path.isdir(p):
421
+ return p
422
+ return None
423
+
424
+
425
+ def _resolve_configs(ported_stacks: Set[str]) -> List[str]:
426
+ """--config paths for the detected stacks: language packs then local packs."""
427
+ wanted: List[str] = []
428
+ seen: Set[str] = set()
429
+ for stack in sorted(ported_stacks):
430
+ for pack in WEB_PACK_MAP.get(stack, []):
431
+ if pack not in seen:
432
+ wanted.append(pack)
433
+ seen.add(pack)
434
+
435
+ configs: List[str] = []
436
+ for pack in wanted:
437
+ path = _pack_path(pack)
438
+ if path:
439
+ configs.append(path)
440
+ pro_path = _pro_pack_path(pack)
441
+ if pro_path:
442
+ configs.append(pro_path)
443
+
444
+ local_dirs_added: Set[str] = set()
445
+ for stack in sorted(ported_stacks):
446
+ for local_dir_name in _LOCAL_PACK_DIRS.get(stack, ()):
447
+ if local_dir_name in local_dirs_added:
448
+ continue
449
+ local_path = _local_pack_path(local_dir_name)
450
+ if local_path:
451
+ configs.append(local_path)
452
+ local_dirs_added.add(local_dir_name)
453
+
454
+ return configs
455
+
456
+
457
+ def _load_rule_defs_from_yml(path: str) -> List[Dict[str, Any]]:
458
+ try:
459
+ with open(path, "r") as f:
460
+ data = yaml.safe_load(f)
461
+ except (OSError, yaml.YAMLError):
462
+ return []
463
+ defs = []
464
+ for rule in (data or {}).get("rules", []):
465
+ meta = rule.get("metadata", {}) or {}
466
+ defs.append({
467
+ "id": rule["id"],
468
+ "name": rule.get("message", rule["id"])[:120],
469
+ "severity": semgrep_engine.SEVERITY_MAP.get(rule.get("severity", "INFO"), "MEDIUM"),
470
+ "details": {
471
+ "cwe": meta.get("cwe", ""),
472
+ "masvs": meta.get("masvs", ""),
473
+ "description": rule.get("message", ""),
474
+ "recommendation": rule.get("message", ""),
475
+ },
476
+ })
477
+ return defs
478
+
479
+
480
+ def _load_all_rule_defs(configs: List[str]) -> List[Dict[str, Any]]:
481
+ defs: List[Dict[str, Any]] = []
482
+ for cfg in configs:
483
+ if os.path.isdir(cfg):
484
+ for fname in sorted(os.listdir(cfg)):
485
+ if fname.endswith((".yml", ".yaml")):
486
+ defs.extend(_load_rule_defs_from_yml(os.path.join(cfg, fname)))
487
+ else:
488
+ defs.extend(_load_rule_defs_from_yml(cfg))
489
+ return defs
490
+
491
+
492
+ def _rule_def_from_finding(finding: Dict[str, Any]) -> Dict[str, Any]:
493
+ """Fallback rule def for a rule that fired but was not in any loaded pack."""
494
+ return {
495
+ "id": finding["rule_id"],
496
+ "name": finding["name"],
497
+ "severity": finding["severity"],
498
+ "details": dict(finding["details"]),
499
+ }
500
+
501
+
502
+ def _is_generated_file(abs_path: str, _cache: Dict[str, bool] = {}) -> bool:
503
+ if abs_path in _cache:
504
+ return _cache[abs_path]
505
+ try:
506
+ with open(abs_path, "r", encoding="utf-8", errors="ignore") as f:
507
+ head = f.read(_GENERATED_HEADER_SCAN_BYTES)
508
+ except OSError:
509
+ _cache[abs_path] = False
510
+ return False
511
+ result = bool(_GENERATED_CODE_HEADER_RE.search(head))
512
+ _cache[abs_path] = result
513
+ return result
514
+
515
+
516
+ LAST_RUN_SKIPPED_GENERATED = 0
517
+ LAST_RUN_DUPLICATES_DROPPED = 0
518
+
519
+
520
+ def _run_web_semgrep(source_dir: str, configs: List[str],
521
+ timeout: Optional[int] = None) -> Optional[List[Dict[str, Any]]]:
522
+ global LAST_RUN_SKIPPED_GENERATED, LAST_RUN_DUPLICATES_DROPPED
523
+ LAST_RUN_SKIPPED_GENERATED = 0
524
+ LAST_RUN_DUPLICATES_DROPPED = 0
525
+ if not configs:
526
+ semgrep_engine._record_run("error", message="no rule packs resolved for detected stack(s)")
527
+ return None
528
+ if not semgrep_engine.is_available():
529
+ semgrep_engine._record_run("unavailable")
530
+ return None
531
+
532
+ file_count = semgrep_engine.count_source_files(source_dir)
533
+ if timeout is None:
534
+ env_to = os.environ.get("NARVY_SEMGREP_TIMEOUT", "").strip()
535
+ timeout = int(env_to) if env_to.isdigit() else semgrep_engine.compute_semgrep_timeout(source_dir)
536
+
537
+ cmd = ["semgrep"]
538
+ for cfg in configs:
539
+ cmd += ["--config", cfg]
540
+ cmd += [source_dir, "--json", "--quiet", "--no-git-ignore", "--timeout", "60"]
541
+ # Otherwise semgrep prefixes each rule id with the install-dependent config path.
542
+ cmd += ["--no-rewrite-rule-ids"]
543
+ for vendor_dir in sorted(_VENDOR_DIR_NAMES):
544
+ cmd += ["--exclude", vendor_dir]
545
+
546
+ try:
547
+ proc = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout)
548
+ except subprocess.TimeoutExpired:
549
+ semgrep_engine._record_run("timeout", timeout_s=timeout, file_count=file_count)
550
+ return []
551
+
552
+ if not proc.stdout:
553
+ if proc.returncode not in (0, 1):
554
+ semgrep_engine._record_run("error", timeout_s=timeout, file_count=file_count,
555
+ message=f"semgrep exited rc={proc.returncode}: {proc.stderr[:300]}")
556
+ else:
557
+ semgrep_engine._record_run("ok", timeout_s=timeout, file_count=file_count)
558
+ return []
559
+
560
+ try:
561
+ data = json.loads(proc.stdout)
562
+ except json.JSONDecodeError:
563
+ semgrep_engine._record_run("error", timeout_s=timeout, file_count=file_count,
564
+ message="semgrep output was not valid JSON")
565
+ return []
566
+
567
+ semgrep_engine._record_run("ok", timeout_s=timeout, file_count=file_count)
568
+
569
+ findings = []
570
+ skipped_generated = 0
571
+ # Semgrep can emit one rule at one byte range several times (multiple binding sets).
572
+ seen_coords: Set[Tuple[str, str, int, int]] = set()
573
+ # Second pass on the collapsed rule id: two configs can carry the same rule.
574
+ seen_keys: Set[Tuple[str, str, int]] = set()
575
+ duplicates_dropped = 0
576
+ for r in data.get("results", []):
577
+ abs_path = r.get("path", "")
578
+ coord = (
579
+ r.get("check_id", ""),
580
+ abs_path,
581
+ r.get("start", {}).get("line", 0),
582
+ r.get("start", {}).get("col", 0),
583
+ )
584
+ if coord in seen_coords:
585
+ continue
586
+ seen_coords.add(coord)
587
+ if _is_generated_file(abs_path):
588
+ skipped_generated += 1
589
+ continue
590
+ rel_path = os.path.relpath(abs_path, source_dir)
591
+ extra = r.get("extra", {})
592
+ meta = extra.get("metadata", {})
593
+ check_id = r.get("check_id", "semgrep-finding").rsplit(".", 1)[-1]
594
+ line_no = r.get("start", {}).get("line", 1)
595
+ key = (check_id, rel_path, line_no)
596
+ if key in seen_keys:
597
+ duplicates_dropped += 1
598
+ continue
599
+ seen_keys.add(key)
600
+ findings.append({
601
+ "rule_id": check_id,
602
+ "file_path": rel_path,
603
+ "name": extra.get("message", r.get("check_id", ""))[:120],
604
+ "severity": semgrep_engine.SEVERITY_MAP.get(extra.get("severity", "INFO"), "MEDIUM"),
605
+ "line": line_no,
606
+ "details": {
607
+ "cwe": meta.get("cwe", ""),
608
+ "masvs": meta.get("masvs", ""),
609
+ "description": extra.get("message", ""),
610
+ "recommendation": extra.get("message", ""),
611
+ },
612
+ "engine": "semgrep",
613
+ })
614
+ LAST_RUN_SKIPPED_GENERATED = skipped_generated
615
+ LAST_RUN_DUPLICATES_DROPPED = duplicates_dropped
616
+
617
+ # Collapse overlapping JWT-secret rules, then down-weight secret hits in test paths.
618
+ findings = _collapse_jwt_aliases(findings)
619
+ findings = _downweight_test_secrets(findings)
620
+ return findings
621
+
622
+
623
+ def _count_scanned_files(source_dir: str, ported_stacks: Set[str]) -> int:
624
+ exts: Set[str] = set()
625
+ for stack in ported_stacks:
626
+ exts.update(_LANG_EXTS.get(stack, ()))
627
+ if not exts:
628
+ return 0
629
+ return _count_density(source_dir, tuple(exts))
630
+
631
+
632
+ def analyze_source(source_dir: str) -> Dict[str, Any]:
633
+ """Run web/backend source-mode SAST over source_dir, read-only."""
634
+ empty_stats = {
635
+ "stacks_detected": [], "stacks_unsupported": [], "files_scanned": 0,
636
+ }
637
+ if not source_dir or not os.path.isdir(source_dir):
638
+ return {
639
+ "ok": False, "error": f"Not a directory: {source_dir}",
640
+ "findings": [], "rule_defs": [], "notes": [], **empty_stats,
641
+ }
642
+
643
+ notes: List[str] = []
644
+ all_findings: List[Dict[str, Any]] = []
645
+
646
+ ported_stacks, unported_stacks = detect_web_stacks(source_dir)
647
+
648
+ if unported_stacks:
649
+ labels = ", ".join(sorted(_UNPORTED_STACK_LABELS[s] for s in unported_stacks))
650
+ detail = " ".join(
651
+ _UNPORTED_STACK_PARTIAL_NOTE.get(s, "") for s in sorted(unported_stacks)
652
+ ).strip() if ported_stacks else (
653
+ "That part of the codebase was not scanned by this pass."
654
+ )
655
+ notes.append(
656
+ f"Also detected {labels} in this repo - no DEDICATED rule pack has "
657
+ f"been ported for that yet ({SUPPORTED_LANGUAGES_LABEL} are fully "
658
+ f"covered). {detail}"
659
+ )
660
+
661
+ for stack in sorted(ported_stacks):
662
+ if stack in _LOW_COVERAGE_STACKS:
663
+ notes.append(f"LIMITED COVERAGE: {_LOW_COVERAGE_STACKS[stack]}")
664
+
665
+ if not ported_stacks:
666
+ notes.append(
667
+ f"No supported web-source language ({SUPPORTED_LANGUAGES_LABEL}) "
668
+ "was detected in this directory - nothing to scan with the "
669
+ "current rule packs."
670
+ )
671
+ rule_defs = []
672
+ else:
673
+ configs = _resolve_configs(ported_stacks)
674
+ rule_defs = _load_all_rule_defs(configs)
675
+
676
+ if semgrep_engine.is_available():
677
+ findings = _run_web_semgrep(source_dir, configs)
678
+ if findings:
679
+ all_findings.extend(findings)
680
+ known_rule_ids = {d["id"] for d in rule_defs}
681
+ for finding in findings:
682
+ rid = finding["rule_id"]
683
+ if rid not in known_rule_ids:
684
+ rule_defs.append(_rule_def_from_finding(finding))
685
+ known_rule_ids.add(rid)
686
+ if LAST_RUN_SKIPPED_GENERATED:
687
+ notes.append(
688
+ f"Skipped {LAST_RUN_SKIPPED_GENERATED} finding(s) inside "
689
+ "auto-generated code (files carrying a 'Code generated "
690
+ "... DO NOT EDIT.' header - e.g. protobuf/OpenAPI/mock "
691
+ "codegen output) - not actionable, not hand-written app "
692
+ "code."
693
+ )
694
+ sg_note = semgrep_engine.degraded_coverage_note()
695
+ if sg_note:
696
+ notes.append(sg_note)
697
+ else:
698
+ notes.append(
699
+ "semgrep not found - install with `pip install semgrep` for "
700
+ "web source-code analysis. No findings from this pass."
701
+ )
702
+
703
+ files_scanned = _count_scanned_files(source_dir, ported_stacks)
704
+
705
+ return {
706
+ "ok": True,
707
+ "error": None,
708
+ "findings": all_findings,
709
+ "rule_defs": rule_defs,
710
+ "stacks_detected": sorted(ported_stacks),
711
+ "stacks_unsupported": sorted(unported_stacks),
712
+ "files_scanned": files_scanned,
713
+ "notes": notes,
714
+ }