cloudg 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cloudg/__init__.py +21 -0
- cloudg/api.py +613 -0
- cloudg/cli.py +934 -0
- cloudg/collectors/__init__.py +5 -0
- cloudg/collectors/aws.py +830 -0
- cloudg/collectors/azure.py +362 -0
- cloudg/collectors/base.py +44 -0
- cloudg/collectors/gcp.py +170 -0
- cloudg/collectors/multi.py +329 -0
- cloudg/config.py +261 -0
- cloudg/coverage.py +93 -0
- cloudg/credentials.py +451 -0
- cloudg/graph/__init__.py +6 -0
- cloudg/graph/builder.py +336 -0
- cloudg/graph/ontology.py +914 -0
- cloudg/graph/rag_export.py +533 -0
- cloudg/graph/reachability.py +237 -0
- cloudg/normaliser.py +390 -0
- cloudg/policies/custodian-aws-security.yml +139 -0
- cloudg/policies/custodian-azure.yml +95 -0
- cloudg/policies/custodian-gcp.yml +87 -0
- cloudg/policies/custodian.yml +131 -0
- cloudg/region_discovery.py +330 -0
- cloudg/registry.py +142 -0
- cloudg/renderers/__init__.py +1 -0
- cloudg/renderers/html_report.py +430 -0
- cloudg/renderers/json_export.py +65 -0
- cloudg/renderers/svg.py +689 -0
- cloudg/renderers/terraform_export.py +666 -0
- cloudg/retry.py +103 -0
- cloudg/rules/cis_aws_v3.yaml +148 -0
- cloudg/rules/cis_azure_v2.yaml +142 -0
- cloudg/rules/cis_gcp_v3.yaml +133 -0
- cloudg/rules/frameworks/aws_foundational_security_best_practices_aws.yaml +1286 -0
- cloudg/rules/frameworks/cis_5.0_aws.yaml +280 -0
- cloudg/rules/frameworks/cis_5.0_azure.yaml +475 -0
- cloudg/rules/frameworks/cis_5.0_gcp.yaml +327 -0
- cloudg/rules/frameworks/gdpr_aws.yaml +95 -0
- cloudg/rules/frameworks/hipaa_aws.yaml +461 -0
- cloudg/rules/frameworks/hipaa_azure.yaml +504 -0
- cloudg/rules/frameworks/hipaa_gcp.yaml +211 -0
- cloudg/rules/frameworks/iso27001_2022_aws.yaml +789 -0
- cloudg/rules/frameworks/iso27001_2022_azure.yaml +477 -0
- cloudg/rules/frameworks/iso27001_2022_gcp.yaml +240 -0
- cloudg/rules/frameworks/mitre_attack_aws.yaml +596 -0
- cloudg/rules/frameworks/mitre_attack_azure.yaml +418 -0
- cloudg/rules/frameworks/mitre_attack_gcp.yaml +300 -0
- cloudg/rules/frameworks/nist_800_53_revision_5_aws.yaml +3234 -0
- cloudg/rules/frameworks/nist_csf_2.0_aws.yaml +828 -0
- cloudg/rules/frameworks/pci_4.0_aws.yaml +7108 -0
- cloudg/rules/frameworks/pci_4.0_azure.yaml +4696 -0
- cloudg/rules/frameworks/pci_4.0_gcp.yaml +5405 -0
- cloudg/rules/frameworks/soc2_aws.yaml +421 -0
- cloudg/rules/frameworks/soc2_azure.yaml +477 -0
- cloudg/rules/frameworks/soc2_gcp.yaml +342 -0
- cloudg/rules/gdpr.yaml +56 -0
- cloudg/rules/hipaa_security_rule.yaml +83 -0
- cloudg/rules/iso_27001_2022.yaml +88 -0
- cloudg/rules/nist_800_53.yaml +134 -0
- cloudg/rules/pci_dss_v4.yaml +108 -0
- cloudg/rules/soc2_tsc.yaml +93 -0
- cloudg/scanners/__init__.py +1 -0
- cloudg/scanners/checkov.py +174 -0
- cloudg/scanners/iam_linter.py +200 -0
- cloudg/scanners/prowler.py +237 -0
- cloudg/scanners/scoutsuite.py +163 -0
- cloudg/scanners/trivy.py +339 -0
- cloudg/schema/__init__.py +25 -0
- cloudg/schema/models.py +260 -0
- cloudg/templates/report.html.j2 +828 -0
- cloudg-0.3.0.dist-info/METADATA +330 -0
- cloudg-0.3.0.dist-info/RECORD +75 -0
- cloudg-0.3.0.dist-info/WHEEL +4 -0
- cloudg-0.3.0.dist-info/entry_points.txt +14 -0
- cloudg-0.3.0.dist-info/licenses/LICENSE +21 -0
cloudg/cli.py
ADDED
|
@@ -0,0 +1,934 @@
|
|
|
1
|
+
"""CloudG CLI — Click-based command-line interface."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import json
|
|
7
|
+
import logging
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
import click
|
|
13
|
+
from rich.console import Console
|
|
14
|
+
from rich.logging import RichHandler
|
|
15
|
+
from rich.table import Table
|
|
16
|
+
|
|
17
|
+
from cloudg import __version__
|
|
18
|
+
|
|
19
|
+
console = Console()
|
|
20
|
+
|
|
21
|
+
# Global config reference (set by CLI group)
|
|
22
|
+
_config = None
|
|
23
|
+
|
|
24
|
+
_BANNER = r"""
|
|
25
|
+
_ _
|
|
26
|
+
___| | ___ _ _ __| | __ _
|
|
27
|
+
/ __| |/ _ \| | | |/ _` |/ _` |
|
|
28
|
+
| (__| | (_) | |_| | (_| | (_| |
|
|
29
|
+
\___|_|\___/ \__,_|\__,_|\__, |
|
|
30
|
+
|___/
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def print_banner() -> None:
|
|
35
|
+
"""Print the cloudg ASCII banner."""
|
|
36
|
+
console.print(f"[bold cyan]{_BANNER}[/]", highlight=False)
|
|
37
|
+
console.print(" [dim]cloud graphing — map, graph and audit AWS / Azure / GCP[/]\n")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def setup_logging(verbose: bool = False, log_file: str | None = None) -> None:
|
|
41
|
+
"""Configure structured logging with Rich + optional file output."""
|
|
42
|
+
level = logging.DEBUG if verbose else logging.INFO
|
|
43
|
+
handlers: list[logging.Handler] = [RichHandler(rich_tracebacks=True, console=console)]
|
|
44
|
+
if log_file:
|
|
45
|
+
file_handler = logging.FileHandler(log_file)
|
|
46
|
+
file_handler.setFormatter(
|
|
47
|
+
logging.Formatter("%(asctime)s [%(levelname)s] %(name)s: %(message)s")
|
|
48
|
+
)
|
|
49
|
+
handlers.append(file_handler)
|
|
50
|
+
logging.basicConfig(level=level, format="%(message)s", datefmt="[%X]", handlers=handlers)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@click.group()
|
|
54
|
+
@click.version_option(version=__version__, prog_name="cloudg")
|
|
55
|
+
@click.option("-v", "--verbose", is_flag=True, help="Enable debug logging")
|
|
56
|
+
@click.option("-c", "--config", "config_path", default=None, help="Path to config.yaml")
|
|
57
|
+
@click.option("--log-file", default=None, help="Path to log file")
|
|
58
|
+
def cli(verbose: bool, config_path: str | None, log_file: str | None) -> None:
|
|
59
|
+
"""☁️ cloudg — cloud graphing: infrastructure mapping and security intelligence."""
|
|
60
|
+
global _config
|
|
61
|
+
from cloudg.config import load_config
|
|
62
|
+
|
|
63
|
+
print_banner()
|
|
64
|
+
_config = load_config(config_path)
|
|
65
|
+
effective_verbose = verbose or _config.verbose
|
|
66
|
+
effective_log = log_file or _config.log_file
|
|
67
|
+
setup_logging(effective_verbose, effective_log)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
# ─────────────────────────────────────────────────────────────────────
|
|
71
|
+
# COLLECT command
|
|
72
|
+
# ─────────────────────────────────────────────────────────────────────
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@cli.command()
|
|
76
|
+
@click.option(
|
|
77
|
+
"-p",
|
|
78
|
+
"--provider",
|
|
79
|
+
type=click.Choice(["aws", "azure", "gcp"], case_sensitive=False),
|
|
80
|
+
required=True,
|
|
81
|
+
help="Cloud provider to collect from",
|
|
82
|
+
)
|
|
83
|
+
@click.option("--profile", default=None, help="AWS profile name")
|
|
84
|
+
@click.option("--region", default="us-east-1", help="AWS region")
|
|
85
|
+
@click.option("--subscription-id", default=None, help="Azure subscription ID")
|
|
86
|
+
@click.option("--project-id", default=None, help="GCP project ID")
|
|
87
|
+
@click.option(
|
|
88
|
+
"-o",
|
|
89
|
+
"--output",
|
|
90
|
+
default="./reports",
|
|
91
|
+
help="Output directory",
|
|
92
|
+
)
|
|
93
|
+
def collect(
|
|
94
|
+
provider: str,
|
|
95
|
+
profile: str | None,
|
|
96
|
+
region: str,
|
|
97
|
+
subscription_id: str | None,
|
|
98
|
+
project_id: str | None,
|
|
99
|
+
output: str,
|
|
100
|
+
) -> None:
|
|
101
|
+
"""Collect cloud assets from the specified provider."""
|
|
102
|
+
console.print(f"[bold cyan]☁️ Collecting assets from {provider.upper()}...[/]")
|
|
103
|
+
|
|
104
|
+
from cloudg.credentials import CredentialResolver
|
|
105
|
+
|
|
106
|
+
resolver = CredentialResolver()
|
|
107
|
+
output_dir = Path(output)
|
|
108
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
109
|
+
|
|
110
|
+
async def _collect() -> dict[str, Any]:
|
|
111
|
+
if provider == "aws":
|
|
112
|
+
from cloudg.collectors.aws import AsyncAWSCollector
|
|
113
|
+
|
|
114
|
+
creds = resolver.resolve_aws(profile=profile, region=region)
|
|
115
|
+
collector = AsyncAWSCollector(
|
|
116
|
+
session=creds.session, region=creds.region, account_id=creds.account_id
|
|
117
|
+
)
|
|
118
|
+
elif provider == "azure":
|
|
119
|
+
from cloudg.collectors.azure import AzureCollector
|
|
120
|
+
|
|
121
|
+
creds = resolver.resolve_azure(subscription_id=subscription_id)
|
|
122
|
+
collector = AzureCollector(
|
|
123
|
+
credential=creds.credential,
|
|
124
|
+
subscription_id=creds.subscription_id,
|
|
125
|
+
)
|
|
126
|
+
elif provider == "gcp":
|
|
127
|
+
from cloudg.collectors.gcp import GCPCollector
|
|
128
|
+
|
|
129
|
+
creds = resolver.resolve_gcp(project_id=project_id)
|
|
130
|
+
collector = GCPCollector(project_id=creds.project_id, credentials=creds.credentials)
|
|
131
|
+
else:
|
|
132
|
+
raise click.BadParameter(f"Unknown provider: {provider}")
|
|
133
|
+
|
|
134
|
+
assets, edges = await collector.run()
|
|
135
|
+
return {
|
|
136
|
+
"assets": [a.model_dump(mode="json") for a in assets],
|
|
137
|
+
"edges": [e.model_dump(mode="json") for e in edges],
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
try:
|
|
141
|
+
result = asyncio.run(_collect())
|
|
142
|
+
except Exception as exc:
|
|
143
|
+
console.print(f"[bold red]✗ Collection failed:[/] {exc}")
|
|
144
|
+
sys.exit(1)
|
|
145
|
+
|
|
146
|
+
# Save results
|
|
147
|
+
inventory_path = output_dir / f"inventory-{provider}.json"
|
|
148
|
+
with open(inventory_path, "w") as f:
|
|
149
|
+
json.dump(result, f, indent=2, default=str)
|
|
150
|
+
|
|
151
|
+
# Summary table
|
|
152
|
+
table = Table(title="Collection Summary")
|
|
153
|
+
table.add_column("Metric", style="cyan")
|
|
154
|
+
table.add_column("Count", style="bold green")
|
|
155
|
+
table.add_row("Assets", str(len(result["assets"])))
|
|
156
|
+
table.add_row("Edges", str(len(result["edges"])))
|
|
157
|
+
console.print(table)
|
|
158
|
+
console.print(f"[green]✓ Saved to {inventory_path}[/]")
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
# ─────────────────────────────────────────────────────────────────────
|
|
162
|
+
# SCAN command
|
|
163
|
+
# ─────────────────────────────────────────────────────────────────────
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
@cli.command()
|
|
167
|
+
@click.option("-p", "--provider", default="aws", help="Cloud provider")
|
|
168
|
+
@click.option("--profile", default=None, help="AWS profile name")
|
|
169
|
+
@click.option("--iac-dir", default=None, help="IaC directory for Checkov (defaults to '.')")
|
|
170
|
+
@click.option("--images", default=None, help="Comma-separated container images for Trivy")
|
|
171
|
+
@click.option(
|
|
172
|
+
"-o",
|
|
173
|
+
"--output",
|
|
174
|
+
default="./reports",
|
|
175
|
+
help="Output directory",
|
|
176
|
+
)
|
|
177
|
+
@click.option(
|
|
178
|
+
"--scanners",
|
|
179
|
+
default="prowler,checkov",
|
|
180
|
+
help="Comma-separated list of scanners to run (prowler,scoutsuite,checkov,trivy,iam)",
|
|
181
|
+
)
|
|
182
|
+
def scan(
|
|
183
|
+
provider: str,
|
|
184
|
+
profile: str | None,
|
|
185
|
+
iac_dir: str | None,
|
|
186
|
+
images: str | None,
|
|
187
|
+
output: str,
|
|
188
|
+
scanners: str,
|
|
189
|
+
) -> None:
|
|
190
|
+
"""Run security scanners and generate findings."""
|
|
191
|
+
import concurrent.futures
|
|
192
|
+
|
|
193
|
+
console.print("[bold cyan]🔍 Running security scans...[/]")
|
|
194
|
+
|
|
195
|
+
output_dir = Path(output)
|
|
196
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
197
|
+
scanner_list = [s.strip().lower() for s in scanners.split(",")]
|
|
198
|
+
all_findings: list[Any] = []
|
|
199
|
+
|
|
200
|
+
resolved_iac_dir = iac_dir or "."
|
|
201
|
+
resolved_images = [i.strip() for i in images.split(",")] if images else []
|
|
202
|
+
|
|
203
|
+
console.print(f" [bold]Scanners:[/] {', '.join(scanner_list)}")
|
|
204
|
+
|
|
205
|
+
def _run_prowler() -> list[Any]:
|
|
206
|
+
from cloudg.scanners.prowler import ProwlerScanner
|
|
207
|
+
|
|
208
|
+
console.print(" → Running Prowler...")
|
|
209
|
+
s = ProwlerScanner(
|
|
210
|
+
provider=provider, profile=profile, output_dir=str(output_dir / "prowler")
|
|
211
|
+
)
|
|
212
|
+
return s.run()
|
|
213
|
+
|
|
214
|
+
def _run_scoutsuite() -> list[Any]:
|
|
215
|
+
from cloudg.scanners.scoutsuite import ScoutSuiteScanner
|
|
216
|
+
|
|
217
|
+
console.print(" → Running ScoutSuite...")
|
|
218
|
+
s = ScoutSuiteScanner(
|
|
219
|
+
provider=provider, profile=profile, report_dir=str(output_dir / "scoutsuite")
|
|
220
|
+
)
|
|
221
|
+
return s.run()
|
|
222
|
+
|
|
223
|
+
def _run_checkov() -> list[Any]:
|
|
224
|
+
from cloudg.scanners.checkov import CheckovScanner
|
|
225
|
+
|
|
226
|
+
console.print(f" → Running Checkov (target: {resolved_iac_dir})...")
|
|
227
|
+
s = CheckovScanner(target_dir=resolved_iac_dir)
|
|
228
|
+
return s.run()
|
|
229
|
+
|
|
230
|
+
def _run_trivy() -> list[Any]:
|
|
231
|
+
from cloudg.scanners.trivy import TrivyScanner
|
|
232
|
+
|
|
233
|
+
console.print(f" → Running Trivy ({len(resolved_images)} images)...")
|
|
234
|
+
s = TrivyScanner()
|
|
235
|
+
return s.scan_images(resolved_images)
|
|
236
|
+
|
|
237
|
+
def _run_trivy_fs() -> list[Any]:
|
|
238
|
+
from cloudg.scanners.trivy import TrivyScanner
|
|
239
|
+
|
|
240
|
+
console.print(f" → Running Trivy filesystem scan (target: {resolved_iac_dir})...")
|
|
241
|
+
s = TrivyScanner()
|
|
242
|
+
return s.scan_filesystem([resolved_iac_dir])
|
|
243
|
+
|
|
244
|
+
with concurrent.futures.ThreadPoolExecutor(max_workers=len(scanner_list) + 1) as executor:
|
|
245
|
+
future_to_name: dict[concurrent.futures.Future, str] = {}
|
|
246
|
+
|
|
247
|
+
if "prowler" in scanner_list:
|
|
248
|
+
future_to_name[executor.submit(_run_prowler)] = "Prowler"
|
|
249
|
+
|
|
250
|
+
if "scoutsuite" in scanner_list:
|
|
251
|
+
future_to_name[executor.submit(_run_scoutsuite)] = "ScoutSuite"
|
|
252
|
+
|
|
253
|
+
if "checkov" in scanner_list:
|
|
254
|
+
future_to_name[executor.submit(_run_checkov)] = "Checkov"
|
|
255
|
+
|
|
256
|
+
if "trivy" in scanner_list:
|
|
257
|
+
if resolved_images:
|
|
258
|
+
future_to_name[executor.submit(_run_trivy)] = "Trivy"
|
|
259
|
+
else:
|
|
260
|
+
console.print(
|
|
261
|
+
" [yellow]⊘ Trivy: no images specified, falling back to filesystem scan[/yellow]"
|
|
262
|
+
)
|
|
263
|
+
future_to_name[executor.submit(_run_trivy_fs)] = "Trivy (filesystem)"
|
|
264
|
+
|
|
265
|
+
for future in concurrent.futures.as_completed(future_to_name):
|
|
266
|
+
name = future_to_name[future]
|
|
267
|
+
try:
|
|
268
|
+
findings = future.result(timeout=3600)
|
|
269
|
+
all_findings.extend(findings)
|
|
270
|
+
console.print(f" [green]{len(findings)} findings from {name}[/]")
|
|
271
|
+
except Exception as exc:
|
|
272
|
+
console.print(f" [red]✗ {name} failed: {exc}[/]")
|
|
273
|
+
|
|
274
|
+
# Save raw findings
|
|
275
|
+
findings_path = output_dir / "raw-findings.json"
|
|
276
|
+
with open(findings_path, "w") as f:
|
|
277
|
+
json.dump(
|
|
278
|
+
[f.model_dump(mode="json") for f in all_findings],
|
|
279
|
+
f,
|
|
280
|
+
indent=2,
|
|
281
|
+
default=str,
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
console.print(f"\n[green]✓ Total: {len(all_findings)} findings saved to {findings_path}[/]")
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
# ─────────────────────────────────────────────────────────────────────
|
|
288
|
+
# REPORT command
|
|
289
|
+
# ─────────────────────────────────────────────────────────────────────
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
@cli.command()
|
|
293
|
+
@click.option(
|
|
294
|
+
"-i",
|
|
295
|
+
"--input",
|
|
296
|
+
"input_file",
|
|
297
|
+
required=True,
|
|
298
|
+
help="Path to findings.json from a previous run",
|
|
299
|
+
)
|
|
300
|
+
@click.option(
|
|
301
|
+
"-o",
|
|
302
|
+
"--output",
|
|
303
|
+
default="./reports",
|
|
304
|
+
help="Output directory",
|
|
305
|
+
)
|
|
306
|
+
@click.option(
|
|
307
|
+
"--format",
|
|
308
|
+
"fmt",
|
|
309
|
+
type=click.Choice(["html", "json", "svg", "all"]),
|
|
310
|
+
default="all",
|
|
311
|
+
help="Output format",
|
|
312
|
+
)
|
|
313
|
+
def report(input_file: str, output: str, fmt: str) -> None:
|
|
314
|
+
"""Generate reports from existing scan data."""
|
|
315
|
+
console.print("[bold cyan]📊 Generating reports...[/]")
|
|
316
|
+
|
|
317
|
+
input_path = Path(input_file)
|
|
318
|
+
if not input_path.exists():
|
|
319
|
+
console.print(f"[bold red]✗ Input file not found: {input_path}[/]")
|
|
320
|
+
sys.exit(1)
|
|
321
|
+
|
|
322
|
+
output_dir = Path(output)
|
|
323
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
324
|
+
|
|
325
|
+
with open(input_path) as f:
|
|
326
|
+
data = json.load(f)
|
|
327
|
+
|
|
328
|
+
from cloudg.schema.models import CloudAsset, Finding, ScanResult
|
|
329
|
+
|
|
330
|
+
# Reconstruct ScanResult
|
|
331
|
+
assets = [CloudAsset.model_validate(a) for a in data.get("assets", [])]
|
|
332
|
+
findings = [Finding.model_validate(f) for f in data.get("findings", [])]
|
|
333
|
+
graph_data = data.get("graph", {"nodes": [], "links": []})
|
|
334
|
+
|
|
335
|
+
scan_result = ScanResult(assets=assets, findings=findings)
|
|
336
|
+
|
|
337
|
+
if fmt in ("json", "all"):
|
|
338
|
+
from cloudg.renderers.json_export import JSONExporter
|
|
339
|
+
|
|
340
|
+
exporter = JSONExporter(output_dir=str(output_dir))
|
|
341
|
+
path = exporter.export(scan_result, graph_json=graph_data)
|
|
342
|
+
console.print(f" [green]✓ JSON: {path}[/]")
|
|
343
|
+
|
|
344
|
+
if fmt in ("svg", "all"):
|
|
345
|
+
from cloudg.renderers.svg import SVGRenderer
|
|
346
|
+
|
|
347
|
+
renderer = SVGRenderer(output_dir=str(output_dir))
|
|
348
|
+
path = renderer.render(assets, scan_result.edges)
|
|
349
|
+
console.print(f" [green]✓ SVG: {path}[/]")
|
|
350
|
+
|
|
351
|
+
if fmt in ("html", "all"):
|
|
352
|
+
from cloudg.renderers.html_report import HTMLReportGenerator
|
|
353
|
+
|
|
354
|
+
generator = HTMLReportGenerator(output_dir=str(output_dir))
|
|
355
|
+
path = generator.generate(scan_result, graph_json=graph_data)
|
|
356
|
+
console.print(f" [green]✓ HTML: {path}[/]")
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
# ─────────────────────────────────────────────────────────────────────
|
|
360
|
+
# RUN command (full pipeline)
|
|
361
|
+
# ─────────────────────────────────────────────────────────────────────
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
@cli.command()
|
|
365
|
+
@click.option(
|
|
366
|
+
"-p",
|
|
367
|
+
"--provider",
|
|
368
|
+
type=click.Choice(["aws", "azure", "gcp", "all"], case_sensitive=False),
|
|
369
|
+
multiple=True,
|
|
370
|
+
required=True,
|
|
371
|
+
help="Cloud provider(s) to scan. Use multiple times or 'all' for simultaneous scanning.",
|
|
372
|
+
)
|
|
373
|
+
@click.option("--profile", default=None, help="AWS profile name (fallback if no direct keys)")
|
|
374
|
+
@click.option("--aws-key", default=None, help="AWS access key ID (direct credential)")
|
|
375
|
+
@click.option("--aws-secret", default=None, help="AWS secret access key (direct credential)")
|
|
376
|
+
@click.option("--aws-session-token", default=None, help="AWS session token (temporary credentials)")
|
|
377
|
+
@click.option(
|
|
378
|
+
"--aws-role-arn",
|
|
379
|
+
default=None,
|
|
380
|
+
help="Role ARN to assume via STS (or OIDC target with --aws-web-identity-token-file)",
|
|
381
|
+
)
|
|
382
|
+
@click.option(
|
|
383
|
+
"--aws-external-id",
|
|
384
|
+
default=None,
|
|
385
|
+
help="ExternalId for AssumeRole (third-party auditor pattern)",
|
|
386
|
+
)
|
|
387
|
+
@click.option(
|
|
388
|
+
"--aws-web-identity-token-file",
|
|
389
|
+
default=None,
|
|
390
|
+
help="OIDC token file for AssumeRoleWithWebIdentity (GitHub Actions, EKS)",
|
|
391
|
+
)
|
|
392
|
+
@click.option("--region", default=None, help="AWS region (ignored if --regions is set)")
|
|
393
|
+
@click.option("--subscription-id", default=None, help="Azure subscription ID")
|
|
394
|
+
@click.option(
|
|
395
|
+
"--azure-tenant-id",
|
|
396
|
+
default=None,
|
|
397
|
+
help="Azure AD tenant ID (service principal / workload identity)",
|
|
398
|
+
)
|
|
399
|
+
@click.option(
|
|
400
|
+
"--azure-client-id", default=None, help="Azure service principal or workload identity client ID"
|
|
401
|
+
)
|
|
402
|
+
@click.option("--azure-client-secret", default=None, help="Azure service principal client secret")
|
|
403
|
+
@click.option("--azure-cert-path", default=None, help="Azure service principal certificate path")
|
|
404
|
+
@click.option(
|
|
405
|
+
"--azure-federated-token-file",
|
|
406
|
+
default=None,
|
|
407
|
+
help="Federated OIDC token file (Azure workload identity)",
|
|
408
|
+
)
|
|
409
|
+
@click.option(
|
|
410
|
+
"--azure-managed-identity",
|
|
411
|
+
is_flag=True,
|
|
412
|
+
default=False,
|
|
413
|
+
help="Authenticate with the host's Azure managed identity",
|
|
414
|
+
)
|
|
415
|
+
@click.option("--project-id", default=None, help="GCP project ID")
|
|
416
|
+
@click.option(
|
|
417
|
+
"--gcp-credentials-file",
|
|
418
|
+
default=None,
|
|
419
|
+
help="GCP service account key JSON or workload identity federation config",
|
|
420
|
+
)
|
|
421
|
+
@click.option("--gcp-impersonate-sa", default=None, help="GCP service account email to impersonate")
|
|
422
|
+
@click.option("--iac-dir", default=None, help="IaC directory for Checkov")
|
|
423
|
+
@click.option("--images", default=None, help="Container images for Trivy (comma-separated)")
|
|
424
|
+
@click.option(
|
|
425
|
+
"-o",
|
|
426
|
+
"--output",
|
|
427
|
+
default="./reports",
|
|
428
|
+
help="Output directory",
|
|
429
|
+
)
|
|
430
|
+
@click.option(
|
|
431
|
+
"--scanners",
|
|
432
|
+
default=None,
|
|
433
|
+
help="Scanners to run (comma-separated: prowler,scoutsuite,checkov,trivy,iam). Defaults to config.yaml scanners.enabled.",
|
|
434
|
+
)
|
|
435
|
+
@click.option("--ontology/--no-ontology", default=True, help="Build semantic ontology graph")
|
|
436
|
+
@click.option("--rag-export/--no-rag-export", default=True, help="Generate RAG-ready chunks")
|
|
437
|
+
@click.option(
|
|
438
|
+
"--terraform/--no-terraform", default=False, help="Generate Terraform .tf.json recreation files"
|
|
439
|
+
)
|
|
440
|
+
@click.option(
|
|
441
|
+
"--regions",
|
|
442
|
+
"scan_regions",
|
|
443
|
+
default=None,
|
|
444
|
+
help="Regions to scan: 'all' for auto-discovery, or comma-separated list (e.g. 'us-east-1,eu-west-1')",
|
|
445
|
+
)
|
|
446
|
+
def run(
|
|
447
|
+
provider: tuple[str, ...],
|
|
448
|
+
profile: str | None,
|
|
449
|
+
aws_key: str | None,
|
|
450
|
+
aws_secret: str | None,
|
|
451
|
+
aws_session_token: str | None,
|
|
452
|
+
aws_role_arn: str | None,
|
|
453
|
+
aws_external_id: str | None,
|
|
454
|
+
aws_web_identity_token_file: str | None,
|
|
455
|
+
region: str | None,
|
|
456
|
+
subscription_id: str | None,
|
|
457
|
+
azure_tenant_id: str | None,
|
|
458
|
+
azure_client_id: str | None,
|
|
459
|
+
azure_client_secret: str | None,
|
|
460
|
+
azure_cert_path: str | None,
|
|
461
|
+
azure_federated_token_file: str | None,
|
|
462
|
+
azure_managed_identity: bool,
|
|
463
|
+
project_id: str | None,
|
|
464
|
+
gcp_credentials_file: str | None,
|
|
465
|
+
gcp_impersonate_sa: str | None,
|
|
466
|
+
iac_dir: str | None,
|
|
467
|
+
images: str | None,
|
|
468
|
+
output: str,
|
|
469
|
+
scanners: str | None,
|
|
470
|
+
ontology: bool,
|
|
471
|
+
rag_export: bool,
|
|
472
|
+
terraform: bool,
|
|
473
|
+
scan_regions: str | None,
|
|
474
|
+
) -> None:
|
|
475
|
+
"""Run the full pipeline: collect → scan → normalise → render.
|
|
476
|
+
|
|
477
|
+
Supports multi-provider scanning:
|
|
478
|
+
cloudg run -p aws -p azure
|
|
479
|
+
cloudg run -p all
|
|
480
|
+
cloudg run -p aws --regions all
|
|
481
|
+
cloudg run -p aws --aws-key AKIAXX --aws-secret yyy
|
|
482
|
+
"""
|
|
483
|
+
console.print("[bold cyan]🚀 CloudG Full Pipeline[/]")
|
|
484
|
+
console.print("=" * 50)
|
|
485
|
+
|
|
486
|
+
output_dir = Path(output)
|
|
487
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
488
|
+
|
|
489
|
+
from cloudg.config import CloudGConfig
|
|
490
|
+
from cloudg.graph.builder import GraphBuilder
|
|
491
|
+
from cloudg.graph.reachability import ReachabilityAnalyzer
|
|
492
|
+
from cloudg.normaliser import FindingsNormaliser
|
|
493
|
+
from cloudg.renderers.html_report import HTMLReportGenerator
|
|
494
|
+
from cloudg.renderers.json_export import JSONExporter
|
|
495
|
+
from cloudg.renderers.svg import SVGRenderer
|
|
496
|
+
from cloudg.scanners.iam_linter import IAMLinter
|
|
497
|
+
|
|
498
|
+
# Use global config if loaded, else defaults
|
|
499
|
+
cfg: CloudGConfig = _config or CloudGConfig()
|
|
500
|
+
|
|
501
|
+
# Resolve providers from CLI flags
|
|
502
|
+
providers_list = list(provider)
|
|
503
|
+
if "all" in providers_list:
|
|
504
|
+
providers_list = ["aws", "azure", "gcp"]
|
|
505
|
+
cfg.providers = providers_list
|
|
506
|
+
|
|
507
|
+
# Resolve regions from CLI flag
|
|
508
|
+
if scan_regions:
|
|
509
|
+
if scan_regions.lower() == "all":
|
|
510
|
+
region_list = ["ALL"]
|
|
511
|
+
else:
|
|
512
|
+
region_list = [r.strip() for r in scan_regions.split(",")]
|
|
513
|
+
cfg.aws.regions = region_list
|
|
514
|
+
cfg.azure.regions = region_list
|
|
515
|
+
cfg.gcp.regions = region_list
|
|
516
|
+
elif region:
|
|
517
|
+
cfg.aws.regions = [region]
|
|
518
|
+
|
|
519
|
+
# Inject credentials and IDs from CLI flags
|
|
520
|
+
if subscription_id:
|
|
521
|
+
cfg.azure.subscription_ids = [subscription_id]
|
|
522
|
+
if project_id:
|
|
523
|
+
cfg.gcp.project_ids = [project_id]
|
|
524
|
+
if profile:
|
|
525
|
+
cfg.aws.profile = profile
|
|
526
|
+
if aws_key:
|
|
527
|
+
cfg.aws.access_key_id = aws_key
|
|
528
|
+
if aws_secret:
|
|
529
|
+
cfg.aws.secret_access_key = aws_secret
|
|
530
|
+
if aws_session_token:
|
|
531
|
+
cfg.aws.session_token = aws_session_token
|
|
532
|
+
if aws_role_arn:
|
|
533
|
+
cfg.aws.role_arn = aws_role_arn
|
|
534
|
+
if aws_external_id:
|
|
535
|
+
cfg.aws.external_id = aws_external_id
|
|
536
|
+
if aws_web_identity_token_file:
|
|
537
|
+
cfg.aws.web_identity_token_file = aws_web_identity_token_file
|
|
538
|
+
if azure_tenant_id:
|
|
539
|
+
cfg.azure.tenant_id = azure_tenant_id
|
|
540
|
+
if azure_client_id:
|
|
541
|
+
cfg.azure.client_id = azure_client_id
|
|
542
|
+
if azure_client_secret:
|
|
543
|
+
cfg.azure.client_secret = azure_client_secret
|
|
544
|
+
if azure_cert_path:
|
|
545
|
+
cfg.azure.certificate_path = azure_cert_path
|
|
546
|
+
if azure_federated_token_file:
|
|
547
|
+
cfg.azure.federated_token_file = azure_federated_token_file
|
|
548
|
+
if azure_managed_identity:
|
|
549
|
+
cfg.azure.use_managed_identity = True
|
|
550
|
+
if gcp_credentials_file:
|
|
551
|
+
cfg.gcp.credentials_file = gcp_credentials_file
|
|
552
|
+
if gcp_impersonate_sa:
|
|
553
|
+
cfg.gcp.impersonate_service_account = gcp_impersonate_sa
|
|
554
|
+
|
|
555
|
+
coverage_records = []
|
|
556
|
+
|
|
557
|
+
# ── Resolve scanner list from CLI flag or config ──
|
|
558
|
+
if scanners is not None:
|
|
559
|
+
scanner_list = [s.strip().lower() for s in scanners.split(",")]
|
|
560
|
+
else:
|
|
561
|
+
scanner_list = [s.strip().lower() for s in cfg.scanners.enabled]
|
|
562
|
+
console.print(f"\n[bold]Providers:[/] {', '.join(cfg.providers)}")
|
|
563
|
+
console.print(f"[bold]Scanners:[/] {', '.join(scanner_list)}")
|
|
564
|
+
console.print(f"[bold]AWS regions:[/] {cfg.aws.regions}")
|
|
565
|
+
if "azure" in cfg.providers:
|
|
566
|
+
console.print(f"[bold]Azure regions:[/] {cfg.azure.regions}")
|
|
567
|
+
if "gcp" in cfg.providers:
|
|
568
|
+
console.print(f"[bold]GCP regions:[/] {cfg.gcp.regions}")
|
|
569
|
+
|
|
570
|
+
# ── Resolve IaC directories: CLI flag → config → default "." ──
|
|
571
|
+
resolved_iac_dirs: list[str] = []
|
|
572
|
+
if iac_dir:
|
|
573
|
+
resolved_iac_dirs = [iac_dir]
|
|
574
|
+
elif cfg.scanners.iac_directories:
|
|
575
|
+
resolved_iac_dirs = list(cfg.scanners.iac_directories)
|
|
576
|
+
else:
|
|
577
|
+
resolved_iac_dirs = ["."]
|
|
578
|
+
|
|
579
|
+
# ── Resolve container images: CLI flag → config ──
|
|
580
|
+
resolved_images: list[str] = []
|
|
581
|
+
if images:
|
|
582
|
+
resolved_images = [i.strip() for i in images.split(",")]
|
|
583
|
+
elif cfg.scanners.trivy_images:
|
|
584
|
+
resolved_images = list(cfg.scanners.trivy_images)
|
|
585
|
+
|
|
586
|
+
# Phase 1: Asset Collection (always uses multi-provider orchestrator)
|
|
587
|
+
console.print("\n[bold]Phase 1: Asset Collection[/]")
|
|
588
|
+
|
|
589
|
+
from cloudg.collectors.multi import MultiAccountCollector
|
|
590
|
+
|
|
591
|
+
multi_collector = MultiAccountCollector(cfg)
|
|
592
|
+
|
|
593
|
+
try:
|
|
594
|
+
assets, edges, coverage_records = asyncio.run(multi_collector.collect_all())
|
|
595
|
+
console.print(f" [green]✓ {len(assets)} assets, {len(edges)} edges[/]")
|
|
596
|
+
if hasattr(multi_collector, "_resolved_regions"):
|
|
597
|
+
for prov, regs in multi_collector._resolved_regions.items():
|
|
598
|
+
console.print(f" {prov}: {len(regs)} regions")
|
|
599
|
+
except Exception as exc:
|
|
600
|
+
console.print(f" [red]✗ Collection failed: {exc}[/]")
|
|
601
|
+
assets, edges = [], []
|
|
602
|
+
|
|
603
|
+
# Phase 2: Graph Analysis
|
|
604
|
+
console.print("\n[bold]Phase 2: Graph Analysis[/]")
|
|
605
|
+
graph_builder = GraphBuilder()
|
|
606
|
+
graph = graph_builder.build(assets, edges)
|
|
607
|
+
graph_json = graph_builder.to_d3_json()
|
|
608
|
+
|
|
609
|
+
# Persist graph as GraphML
|
|
610
|
+
graphml_path = output_dir / "topology.graphml"
|
|
611
|
+
graph_builder.save_graphml(graphml_path)
|
|
612
|
+
console.print(f" [green]✓ GraphML: {graphml_path}[/]")
|
|
613
|
+
|
|
614
|
+
# Cytoscape export
|
|
615
|
+
cytoscape_path = output_dir / "topology-cytoscape.json"
|
|
616
|
+
with open(cytoscape_path, "w") as f:
|
|
617
|
+
json.dump(graph_builder.to_cytoscape_json(), f, indent=2, default=str)
|
|
618
|
+
console.print(f" [green]✓ Cytoscape: {cytoscape_path}[/]")
|
|
619
|
+
|
|
620
|
+
analyzer = ReachabilityAnalyzer(graph)
|
|
621
|
+
reachability_findings = analyzer.generate_findings()
|
|
622
|
+
console.print(
|
|
623
|
+
f" [green]✓ Graph: {graph.number_of_nodes()} nodes, {graph.number_of_edges()} edges[/]"
|
|
624
|
+
)
|
|
625
|
+
console.print(f" [green]✓ Reachability findings: {len(reachability_findings)}[/]")
|
|
626
|
+
|
|
627
|
+
# Attack paths
|
|
628
|
+
if cfg.graph.compute_attack_paths:
|
|
629
|
+
lateral_paths = graph_builder.find_lateral_movement_paths()
|
|
630
|
+
if lateral_paths:
|
|
631
|
+
console.print(f" [yellow]⚠ {len(lateral_paths)} lateral movement paths detected[/]")
|
|
632
|
+
|
|
633
|
+
# Phase 2b: Semantic Ontology — deferred to after scanner phase
|
|
634
|
+
# (so security/compliance findings can be included in the ontology)
|
|
635
|
+
|
|
636
|
+
# Phase 2c: RAG Export
|
|
637
|
+
if rag_export and cfg.rag.enabled:
|
|
638
|
+
console.print("\n[bold]Phase 2c: RAG Export[/]")
|
|
639
|
+
try:
|
|
640
|
+
from cloudg.graph.rag_export import RAGExporter
|
|
641
|
+
|
|
642
|
+
rag = RAGExporter(max_chunk_tokens=cfg.rag.max_chunk_tokens)
|
|
643
|
+
rag_paths = rag.export_all(
|
|
644
|
+
assets,
|
|
645
|
+
edges,
|
|
646
|
+
graph,
|
|
647
|
+
findings=reachability_findings,
|
|
648
|
+
output_dir=output_dir,
|
|
649
|
+
)
|
|
650
|
+
console.print(f" [green]✓ RAG chunks: {rag_paths['chunks']}[/]")
|
|
651
|
+
console.print(f" [green]✓ RAG index: {rag_paths['index']}[/]")
|
|
652
|
+
except Exception as exc:
|
|
653
|
+
console.print(f" [red]✗ RAG export failed: {exc}[/]")
|
|
654
|
+
|
|
655
|
+
# Phase 2d: Terraform Recreation
|
|
656
|
+
if terraform or cfg.terraform.enabled:
|
|
657
|
+
console.print("\n[bold]Phase 2d: Terraform Recreation[/]")
|
|
658
|
+
try:
|
|
659
|
+
from cloudg.renderers.terraform_export import TerraformExporter
|
|
660
|
+
|
|
661
|
+
tf_dir = cfg.terraform.output_dir or str(output_dir / "terraform")
|
|
662
|
+
tf_exporter = TerraformExporter(output_dir=tf_dir)
|
|
663
|
+
preview = tf_exporter.preview(assets)
|
|
664
|
+
console.print(
|
|
665
|
+
f" [dim]Preview: {preview['total_mapped']} resources mappable, "
|
|
666
|
+
f"{preview['total_unmapped']} unmapped[/]"
|
|
667
|
+
)
|
|
668
|
+
|
|
669
|
+
tf_paths = tf_exporter.export(assets, edges)
|
|
670
|
+
console.print(f" [green]✓ Provider: {tf_paths['provider']}[/]")
|
|
671
|
+
console.print(f" [green]✓ Variables: {tf_paths['variables']}[/]")
|
|
672
|
+
console.print(f" [green]✓ Main: {tf_paths['main']}[/]")
|
|
673
|
+
console.print(f" [green]✓ Import: {tf_paths['import_commands']}[/]")
|
|
674
|
+
except Exception as exc:
|
|
675
|
+
console.print(f" [red]✗ Terraform export failed: {exc}[/]")
|
|
676
|
+
|
|
677
|
+
import concurrent.futures
|
|
678
|
+
|
|
679
|
+
# Phase 3: Security Scanning (all scanners in parallel)
|
|
680
|
+
console.print("\n[bold]Phase 3: Security Scanning[/bold] (Running in parallel)")
|
|
681
|
+
scanner_findings: list[Any] = []
|
|
682
|
+
|
|
683
|
+
def run_prowler(prov: str) -> list[Any]:
|
|
684
|
+
from cloudg.scanners.prowler import ProwlerScanner
|
|
685
|
+
|
|
686
|
+
console.print(f" → [cyan]Prowler ({prov})[/cyan] started...")
|
|
687
|
+
prowler_region = cfg.aws.regions[0] if cfg.aws.regions else None
|
|
688
|
+
s = ProwlerScanner(
|
|
689
|
+
provider=prov,
|
|
690
|
+
profile=profile,
|
|
691
|
+
output_dir=str(output_dir / "prowler" / prov),
|
|
692
|
+
extra_args=cfg.scanners.prowler_extra_args or [],
|
|
693
|
+
aws_access_key_id=cfg.aws.access_key_id if prov == "aws" else None,
|
|
694
|
+
aws_secret_access_key=cfg.aws.secret_access_key if prov == "aws" else None,
|
|
695
|
+
aws_region=prowler_region if prov == "aws" else None,
|
|
696
|
+
)
|
|
697
|
+
findings = s.run()
|
|
698
|
+
console.print(f" [green]✓ Prowler ({prov}):[/green] {len(findings)} findings")
|
|
699
|
+
return findings
|
|
700
|
+
|
|
701
|
+
def run_scoutsuite(prov: str) -> list[Any]:
|
|
702
|
+
from cloudg.scanners.scoutsuite import ScoutSuiteScanner
|
|
703
|
+
|
|
704
|
+
console.print(f" → [cyan]ScoutSuite ({prov})[/cyan] started...")
|
|
705
|
+
s = ScoutSuiteScanner(
|
|
706
|
+
provider=prov,
|
|
707
|
+
profile=profile if prov == "aws" else None,
|
|
708
|
+
report_dir=str(output_dir / "scoutsuite" / prov),
|
|
709
|
+
extra_args=cfg.scanners.scoutsuite_extra_args or [],
|
|
710
|
+
)
|
|
711
|
+
findings = s.run()
|
|
712
|
+
console.print(f" [green]✓ ScoutSuite ({prov}):[/green] {len(findings)} findings")
|
|
713
|
+
return findings
|
|
714
|
+
|
|
715
|
+
def run_checkov(target_dir: str) -> list[Any]:
|
|
716
|
+
from cloudg.scanners.checkov import CheckovScanner
|
|
717
|
+
|
|
718
|
+
frameworks = cfg.scanners.checkov_frameworks or []
|
|
719
|
+
fw_label = ", ".join(frameworks) if frameworks else "auto-detect"
|
|
720
|
+
console.print(
|
|
721
|
+
f" → [cyan]Checkov[/cyan] started (target: {target_dir}, frameworks: {fw_label})..."
|
|
722
|
+
)
|
|
723
|
+
s = CheckovScanner(
|
|
724
|
+
target_dir=target_dir,
|
|
725
|
+
frameworks=frameworks if frameworks else None,
|
|
726
|
+
extra_args=cfg.scanners.checkov_extra_args or [],
|
|
727
|
+
)
|
|
728
|
+
findings = s.run()
|
|
729
|
+
console.print(f" [green]✓ Checkov:[/green] {len(findings)} findings")
|
|
730
|
+
return findings
|
|
731
|
+
|
|
732
|
+
def run_trivy(image_list: list[str]) -> list[Any]:
|
|
733
|
+
from cloudg.scanners.trivy import TrivyScanner
|
|
734
|
+
|
|
735
|
+
console.print(f" → [cyan]Trivy[/cyan] started ({len(image_list)} images)...")
|
|
736
|
+
s = TrivyScanner(extra_args=cfg.scanners.trivy_extra_args or [])
|
|
737
|
+
findings = s.scan_images(image_list)
|
|
738
|
+
console.print(f" [green]✓ Trivy (images):[/green] {len(findings)} findings")
|
|
739
|
+
return findings
|
|
740
|
+
|
|
741
|
+
def run_trivy_fs(target_dirs: list[str]) -> list[Any]:
|
|
742
|
+
from cloudg.scanners.trivy import TrivyScanner
|
|
743
|
+
|
|
744
|
+
console.print(
|
|
745
|
+
f" → [cyan]Trivy (filesystem)[/cyan] started ({len(target_dirs)} directories)..."
|
|
746
|
+
)
|
|
747
|
+
s = TrivyScanner(extra_args=cfg.scanners.trivy_extra_args or [])
|
|
748
|
+
findings = s.scan_filesystem(target_dirs)
|
|
749
|
+
console.print(f" [green]✓ Trivy (filesystem):[/green] {len(findings)} findings")
|
|
750
|
+
return findings
|
|
751
|
+
|
|
752
|
+
def run_iam_linter() -> list[Any]:
|
|
753
|
+
console.print(" → [cyan]IAM Lint[/cyan] started...")
|
|
754
|
+
iam_linter = IAMLinter()
|
|
755
|
+
findings = iam_linter.analyze_policies(assets)
|
|
756
|
+
console.print(f" [green]✓ IAM Lint:[/green] {len(findings)} findings")
|
|
757
|
+
return findings
|
|
758
|
+
|
|
759
|
+
# Execute ALL enabled scanners concurrently
|
|
760
|
+
iam_findings: list[Any] = []
|
|
761
|
+
max_workers = len(scanner_list) + len(cfg.providers) + 1 # +1 for IAM linter
|
|
762
|
+
with concurrent.futures.ThreadPoolExecutor(max_workers=max(max_workers, 2)) as executor:
|
|
763
|
+
future_to_scanner: dict[concurrent.futures.Future, str] = {}
|
|
764
|
+
|
|
765
|
+
# Prowler — one instance per provider
|
|
766
|
+
if "prowler" in scanner_list:
|
|
767
|
+
for prov in cfg.providers:
|
|
768
|
+
if prov in ("aws", "azure", "gcp"):
|
|
769
|
+
future_to_scanner[executor.submit(run_prowler, prov)] = f"Prowler ({prov})"
|
|
770
|
+
else:
|
|
771
|
+
console.print(" [dim]⊘ Prowler: not enabled[/dim]")
|
|
772
|
+
|
|
773
|
+
# ScoutSuite — one instance per provider
|
|
774
|
+
if "scoutsuite" in scanner_list:
|
|
775
|
+
for prov in cfg.providers:
|
|
776
|
+
if prov in ("aws", "azure", "gcp"):
|
|
777
|
+
future_to_scanner[executor.submit(run_scoutsuite, prov)] = (
|
|
778
|
+
f"ScoutSuite ({prov})"
|
|
779
|
+
)
|
|
780
|
+
else:
|
|
781
|
+
console.print(" [dim]⊘ ScoutSuite: not enabled[/dim]")
|
|
782
|
+
|
|
783
|
+
# Checkov — always runs against resolved IaC directories
|
|
784
|
+
if "checkov" in scanner_list:
|
|
785
|
+
for d in resolved_iac_dirs:
|
|
786
|
+
future_to_scanner[executor.submit(run_checkov, d)] = f"Checkov ({d})"
|
|
787
|
+
else:
|
|
788
|
+
console.print(" [dim]⊘ Checkov: not enabled[/dim]")
|
|
789
|
+
|
|
790
|
+
# Trivy — runs against container images if available, otherwise falls back to filesystem scan
|
|
791
|
+
if "trivy" in scanner_list:
|
|
792
|
+
if resolved_images:
|
|
793
|
+
future_to_scanner[executor.submit(run_trivy, resolved_images)] = "Trivy (images)"
|
|
794
|
+
else:
|
|
795
|
+
console.print(
|
|
796
|
+
" [yellow]⊘ Trivy: no images configured, falling back to filesystem scan[/yellow]"
|
|
797
|
+
)
|
|
798
|
+
future_to_scanner[executor.submit(run_trivy_fs, resolved_iac_dirs)] = (
|
|
799
|
+
"Trivy (filesystem)"
|
|
800
|
+
)
|
|
801
|
+
else:
|
|
802
|
+
console.print(" [dim]⊘ Trivy: not enabled[/dim]")
|
|
803
|
+
|
|
804
|
+
# IAM Linter — always runs internally to analyze collected assets
|
|
805
|
+
if "iam" in scanner_list or assets:
|
|
806
|
+
future_to_scanner[executor.submit(run_iam_linter)] = "IAM Linter"
|
|
807
|
+
|
|
808
|
+
for future in concurrent.futures.as_completed(future_to_scanner):
|
|
809
|
+
scanner_name = future_to_scanner[future]
|
|
810
|
+
try:
|
|
811
|
+
findings = future.result(timeout=cfg.scanners.timeout_seconds)
|
|
812
|
+
if scanner_name == "IAM Linter":
|
|
813
|
+
iam_findings.extend(findings)
|
|
814
|
+
else:
|
|
815
|
+
scanner_findings.extend(findings)
|
|
816
|
+
except concurrent.futures.TimeoutError:
|
|
817
|
+
console.print(
|
|
818
|
+
f" [red]✗ {scanner_name} timed out after {cfg.scanners.timeout_seconds}s[/red]"
|
|
819
|
+
)
|
|
820
|
+
except Exception as exc:
|
|
821
|
+
console.print(f" [red]✗ {scanner_name} failed:[/red] {exc}")
|
|
822
|
+
|
|
823
|
+
# Combine all findings for downstream phases
|
|
824
|
+
all_security_findings = scanner_findings + iam_findings + reachability_findings
|
|
825
|
+
console.print(
|
|
826
|
+
f"\n [bold green]✓ Phase 3 complete:[/bold green] {len(all_security_findings)} total findings"
|
|
827
|
+
)
|
|
828
|
+
|
|
829
|
+
# Phase 3b: Semantic Ontology (runs AFTER scanners so findings are included)
|
|
830
|
+
if ontology and cfg.ontology.enabled:
|
|
831
|
+
console.print("\n[bold]Phase 3b: Semantic Ontology (with security findings)[/]")
|
|
832
|
+
try:
|
|
833
|
+
from cloudg.graph.ontology import CloudOntology
|
|
834
|
+
|
|
835
|
+
cloud_ontology = CloudOntology()
|
|
836
|
+
cloud_ontology.build(assets, edges, findings=all_security_findings)
|
|
837
|
+
stats = cloud_ontology.stats()
|
|
838
|
+
console.print(
|
|
839
|
+
f" [green]✓ Ontology: {stats['total_triples']} triples, "
|
|
840
|
+
f"{stats['classes_used']} classes, {stats['individuals']} individuals[/]"
|
|
841
|
+
)
|
|
842
|
+
console.print(
|
|
843
|
+
f" [green]✓ Security findings in ontology: {len(all_security_findings)}[/]"
|
|
844
|
+
)
|
|
845
|
+
|
|
846
|
+
for fmt in cfg.ontology.export_formats:
|
|
847
|
+
ext_map = {"turtle": "ttl", "json-ld": "jsonld", "xml": "rdf", "nt": "nt"}
|
|
848
|
+
ext = ext_map.get(fmt, "ttl")
|
|
849
|
+
onto_path = cloud_ontology.save(output_dir / f"ontology.{ext}", fmt=fmt)
|
|
850
|
+
console.print(f" [green]✓ Ontology ({fmt}): {onto_path}[/]")
|
|
851
|
+
|
|
852
|
+
# Group summary
|
|
853
|
+
for group, count in stats["relation_group_counts"].items():
|
|
854
|
+
console.print(f" {group}: {count} relations")
|
|
855
|
+
except Exception as exc:
|
|
856
|
+
console.print(f" [red]✗ Ontology build failed: {exc}[/]")
|
|
857
|
+
|
|
858
|
+
# Also update RAG export with all findings
|
|
859
|
+
if rag_export and cfg.rag.enabled:
|
|
860
|
+
try:
|
|
861
|
+
from cloudg.graph.rag_export import RAGExporter
|
|
862
|
+
|
|
863
|
+
rag = RAGExporter(max_chunk_tokens=cfg.rag.max_chunk_tokens)
|
|
864
|
+
rag_paths = rag.export_all(
|
|
865
|
+
assets,
|
|
866
|
+
edges,
|
|
867
|
+
graph,
|
|
868
|
+
findings=all_security_findings,
|
|
869
|
+
output_dir=output_dir,
|
|
870
|
+
)
|
|
871
|
+
except Exception:
|
|
872
|
+
pass # RAG already ran in Phase 2c, this is an update pass
|
|
873
|
+
|
|
874
|
+
# Phase 4: Normalise (with external rulesets)
|
|
875
|
+
console.print("\n[bold]Phase 4: Normalisation[/]")
|
|
876
|
+
normaliser = FindingsNormaliser(rules_dir=cfg.rulesets.rules_dir)
|
|
877
|
+
scan_result = normaliser.normalise(
|
|
878
|
+
reachability_findings, scanner_findings, iam_findings, assets=assets
|
|
879
|
+
)
|
|
880
|
+
scan_result.edges = edges
|
|
881
|
+
console.print(f" [green]✓ {len(scan_result.findings)} normalised findings[/]")
|
|
882
|
+
|
|
883
|
+
# Phase 5: Render
|
|
884
|
+
console.print("\n[bold]Phase 5: Report Generation[/]")
|
|
885
|
+
|
|
886
|
+
exporter = JSONExporter(output_dir=str(output_dir))
|
|
887
|
+
json_path = exporter.export(scan_result, graph_json=graph_json)
|
|
888
|
+
console.print(f" [green]✓ JSON: {json_path}[/]")
|
|
889
|
+
|
|
890
|
+
svg_renderer = SVGRenderer(output_dir=str(output_dir))
|
|
891
|
+
svg_path = svg_renderer.render(assets, edges)
|
|
892
|
+
console.print(f" [green]✓ SVG: {svg_path}[/]")
|
|
893
|
+
|
|
894
|
+
html_gen = HTMLReportGenerator(output_dir=str(output_dir))
|
|
895
|
+
html_path = html_gen.generate(scan_result, graph_json=graph_json)
|
|
896
|
+
console.print(f" [green]✓ HTML: {html_path}[/]")
|
|
897
|
+
|
|
898
|
+
# Summary
|
|
899
|
+
console.print("\n" + "=" * 50)
|
|
900
|
+
summary = scan_result.summary
|
|
901
|
+
table = Table(title="🏁 Pipeline Summary")
|
|
902
|
+
table.add_column("Metric", style="cyan")
|
|
903
|
+
table.add_column("Value", style="bold")
|
|
904
|
+
table.add_row("Assets", str(summary["total_assets"]))
|
|
905
|
+
table.add_row("Findings", str(summary["total_findings"]))
|
|
906
|
+
for sev, count in summary["severity_breakdown"].items():
|
|
907
|
+
color = {"CRITICAL": "red", "HIGH": "yellow", "MEDIUM": "blue"}.get(sev, "white")
|
|
908
|
+
table.add_row(sev, f"[{color}]{count}[/]")
|
|
909
|
+
table.add_row("Frameworks", ", ".join(summary["compliance_frameworks"]))
|
|
910
|
+
console.print(table)
|
|
911
|
+
|
|
912
|
+
# Coverage summary
|
|
913
|
+
if coverage_records:
|
|
914
|
+
cov_table = Table(title="📊 Collection Coverage")
|
|
915
|
+
cov_table.add_column("Region", style="cyan")
|
|
916
|
+
cov_table.add_column("Account", style="dim")
|
|
917
|
+
cov_table.add_column("Coverage", style="bold")
|
|
918
|
+
cov_table.add_column("Failures", style="red")
|
|
919
|
+
for cov in coverage_records:
|
|
920
|
+
summary_data = cov.to_summary()
|
|
921
|
+
failures = ", ".join(f["service"] for f in summary_data["failures"]) or "—"
|
|
922
|
+
cov_table.add_row(
|
|
923
|
+
summary_data["region"] or "—",
|
|
924
|
+
summary_data["account_id"] or "—",
|
|
925
|
+
f"{summary_data['coverage_pct']}%",
|
|
926
|
+
failures,
|
|
927
|
+
)
|
|
928
|
+
console.print(cov_table)
|
|
929
|
+
|
|
930
|
+
console.print(f"\n[bold green]✓ All reports saved to {output_dir}[/]")
|
|
931
|
+
|
|
932
|
+
|
|
933
|
+
if __name__ == "__main__":
|
|
934
|
+
cli()
|