cloudmap 1.1.0__tar.gz → 1.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cloudmap-1.3.0/CONTRIBUTING.md +55 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/PKG-INFO +33 -6
- {cloudmap-1.1.0 → cloudmap-1.3.0}/README.md +32 -5
- cloudmap-1.3.0/SECURITY.md +46 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/__init__.py +1 -1
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/adapters/__init__.py +7 -1
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/cli.py +151 -42
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/extract/extractors.py +303 -25
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/ingest/azure.py +64 -16
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/interactive.py +43 -30
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/render/html.py +8 -0
- cloudmap-1.3.0/tests/data/azure_resource_types.txt +4692 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_arg_rows.py +82 -5
- cloudmap-1.3.0/tests/test_azure.py +45 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_cli_exports.py +101 -0
- cloudmap-1.3.0/tests/test_common_types.py +357 -0
- cloudmap-1.3.0/tests/test_containment_vs_association.py +233 -0
- cloudmap-1.3.0/tests/test_enrich.py +197 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_interactive_wizard.py +20 -15
- cloudmap-1.3.0/tests/test_render_text.py +70 -0
- cloudmap-1.3.0/tests/test_type_agnostic.py +262 -0
- cloudmap-1.1.0/tests/test_azure.py +0 -20
- cloudmap-1.1.0/tests/test_enrich.py +0 -95
- {cloudmap-1.1.0 → cloudmap-1.3.0}/.github/workflows/ci.yml +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/.github/workflows/publish.yml +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/.gitignore +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/ARCHITECTURE.md +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/FORMAT.md +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/LICENSE +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/PLAN.md +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/__main__.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/ask/__init__.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/ask/intent.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/ask/narration.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/ask/queries.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/extract/__init__.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/extract/llm.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/graph.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/ingest/__init__.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/ingest/fixture.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/local_model.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/model.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/render/__init__.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/render/azure_icons.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/render/csv_export.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/render/drawio.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/render/json_out.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/render/mermaid.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/cloudmap/scrub.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/docs/social-preview.html +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/docs/social-preview.png +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/estate-viewer.png +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/fixtures/acme_orders.json +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/fixtures/contoso.json +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/fixtures/estate.json +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/pyproject.toml +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/complex_mock_demo/01_input_complex_random.json +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/complex_mock_demo/04_scrubbed_output.json +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/complex_mock_demo/06_trace_output.json +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/complex_mock_demo/07_trace_output.html +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/complex_mock_demo/08_trace_output.csv +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/complex_mock_demo/09_trace_output.drawio +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/complex_mock_demo/10_input_enterprise_architecture.json +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/complex_mock_demo/12_enterprise_trace.html +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/complex_mock_demo/13_enterprise_trace.json +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/complex_mock_demo/14_enterprise_scrubbed.json +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/complex_mock_demo/README.md +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_adapters.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_ask.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_containerapps.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_drawio_xml.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_estate.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_fixtures_safe.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_golden_orders.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_graph.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_html.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_ingest_paging.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_llm.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_local_model.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_scrub.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/tests/test_trust.py +0 -0
- {cloudmap-1.1.0 → cloudmap-1.3.0}/uv.lock +0 -0
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
Thanks for looking. The project is small on purpose; contributions that keep it
|
|
4
|
+
small are the easiest to merge.
|
|
5
|
+
|
|
6
|
+
## Setup
|
|
7
|
+
|
|
8
|
+
```
|
|
9
|
+
git clone https://github.com/KatsaounisThanasis/cloudmap && cd cloudmap
|
|
10
|
+
pip install -e ".[dev]"
|
|
11
|
+
pytest && ruff check .
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
Python 3.9+ (CI runs 3.9 and 3.13). No test touches a real cloud: `subprocess`
|
|
15
|
+
is faked everywhere, and several test files have an autouse fixture that turns
|
|
16
|
+
any real subprocess call into a failure - keep it that way.
|
|
17
|
+
|
|
18
|
+
## The rules that are not up for debate
|
|
19
|
+
|
|
20
|
+
These are the project's identity; PRs that violate them will be declined even
|
|
21
|
+
if the feature is useful:
|
|
22
|
+
|
|
23
|
+
1. **Read-only.** No `az` command that mutates anything.
|
|
24
|
+
2. **Local-first.** No telemetry, no upload, no outbound call except the
|
|
25
|
+
optional user-configured local model endpoint.
|
|
26
|
+
3. **Every edge carries evidence.** An edge without a provable source
|
|
27
|
+
(property path, config reference, RBAC assignment) does not go on the map.
|
|
28
|
+
4. **A model proposes, code verifies.** LLM output may only suggest candidates
|
|
29
|
+
that a deterministic check confirms against scanned resources. A guess that
|
|
30
|
+
cannot be verified is dropped, not shown.
|
|
31
|
+
5. **Incompleteness is declared.** If a scan was truncated or a read failed,
|
|
32
|
+
the artifact says so. Never let an empty result pass for "nothing depends
|
|
33
|
+
on this".
|
|
34
|
+
6. **Nothing sensitive in the repo.** Fixtures are synthetic or scrubbed;
|
|
35
|
+
`tests/test_fixtures_safe.py` enforces it in CI.
|
|
36
|
+
|
|
37
|
+
## Practical notes
|
|
38
|
+
|
|
39
|
+
- Match the existing style: docstrings explain *why*, tests are named as
|
|
40
|
+
behaviour claims (`test_a_truncated_capture_stays_truncated_when_retraced`).
|
|
41
|
+
- A bug fix needs a regression test that fails without the fix.
|
|
42
|
+
- New resource-type support usually means: a Resolver index entry, an
|
|
43
|
+
extractor rule (or nothing, if the generic ARM-reference pass covers it),
|
|
44
|
+
a FRIENDLY name, and a fixture-based test.
|
|
45
|
+
- Run `ruff check .` before pushing - CI treats lint as a failure.
|
|
46
|
+
|
|
47
|
+
## Reporting bugs
|
|
48
|
+
|
|
49
|
+
Open an issue with the command you ran and the output. If the map itself is
|
|
50
|
+
wrong (missing or bogus edge), the perfect report includes a minimal synthetic
|
|
51
|
+
fixture that reproduces it - see `fixtures/contoso.json` for the shape. Never
|
|
52
|
+
paste a real capture; scrub it first (`cloudmap scrub`) and read it before
|
|
53
|
+
posting.
|
|
54
|
+
|
|
55
|
+
Security issues: see [SECURITY.md](SECURITY.md) - do not open a public issue.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cloudmap
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0
|
|
4
4
|
Summary: Trace the blast radius of an Azure resource: one name in, a verified dependency graph out.
|
|
5
5
|
Project-URL: Homepage, https://github.com/KatsaounisThanasis/cloudmap
|
|
6
6
|
Project-URL: Repository, https://github.com/KatsaounisThanasis/cloudmap
|
|
@@ -202,10 +202,26 @@ identity RBAC on it, with the role assignments as proof.
|
|
|
202
202
|
|
|
203
203
|
## What it maps
|
|
204
204
|
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
205
|
+
**Any Azure resource type can be a seed.** The scan is not filtered by type, and
|
|
206
|
+
resources are mapped at two levels:
|
|
207
|
+
|
|
208
|
+
- **Typed rules** for the services where the relationship has a specific meaning:
|
|
209
|
+
App Service / Functions, Container Apps (+ environments), AKS, App Gateway, API
|
|
210
|
+
Management, Key Vault, Storage, SQL / PostgreSQL / MySQL / Cosmos, Redis, Service
|
|
211
|
+
Bus, Event Hub, Cognitive Search, Azure OpenAI, Container Registry, ML workspaces,
|
|
212
|
+
Log Analytics, App Insights, VNets, Private Endpoints and managed identities.
|
|
213
|
+
These produce edges like `hosted-on`, `reads-secret`, `pulls-image`, `routes-to`.
|
|
214
|
+
- **A generic ARM-reference pass** for everything else: any resolvable resource id
|
|
215
|
+
found in a resource's properties becomes a `references` edge, with the property
|
|
216
|
+
path as proof. So a type cloudmap has never heard of is still mapped, still
|
|
217
|
+
deterministically, still with evidence.
|
|
218
|
+
|
|
219
|
+
RBAC edges (`role: Key Vault Secrets User`) are extracted tenant-wide, so
|
|
220
|
+
"who has access to this" works for any resource that can be a role scope.
|
|
221
|
+
|
|
222
|
+
Depth is honest about itself: apps and clusters have rich outbound edges because
|
|
223
|
+
their config names other resources. Infrastructure resources are usually leaves
|
|
224
|
+
going outward, and their value is the reverse view (`--direction up`).
|
|
209
225
|
|
|
210
226
|
## How it works
|
|
211
227
|
|
|
@@ -276,6 +292,14 @@ is the deliberate switch, and cloudmap reads whatever subscription `az` is point
|
|
|
276
292
|
at. The read is read-only, but it is a read of live infrastructure, so point it on
|
|
277
293
|
purpose. **Do not point this at data you are not authorized to read.**
|
|
278
294
|
|
|
295
|
+
One asterisk on "read-only": AKS manifests are read through `az aks command
|
|
296
|
+
invoke`, which Azure implements by starting a short-lived pod in the cluster to
|
|
297
|
+
run the (read-only) `kubectl get` commands. Nothing of yours is modified, but an
|
|
298
|
+
audit of the cluster's control plane will see that ephemeral pod.
|
|
299
|
+
|
|
300
|
+
Tenant-wide enrichment runs its `az` reads concurrently; `CLOUDMAP_ENRICH_WORKERS`
|
|
301
|
+
(default 12) tunes how many at once.
|
|
302
|
+
|
|
279
303
|
```
|
|
280
304
|
cloudmap trace my-app --live --allow-live
|
|
281
305
|
```
|
|
@@ -298,7 +322,10 @@ everything".
|
|
|
298
322
|
subscription in the tenant)
|
|
299
323
|
--enrich MODE which web apps to deep-enrich for the dependencies that only
|
|
300
324
|
exist in app config. auto (default) = the seed alone when the
|
|
301
|
-
seed is a
|
|
325
|
+
seed is a workload; every app in scope when the seed is a data
|
|
326
|
+
service whose dependents hide in app config (Key Vault,
|
|
327
|
+
storage, SQL, Redis, ...); the seed alone for compute and
|
|
328
|
+
network resources, whose relationships ARM already returns;
|
|
302
329
|
all | seed | none
|
|
303
330
|
--resolve-secrets read KV secret values in-memory to see through KV-backed
|
|
304
331
|
connection strings (never printed or written)
|
|
@@ -168,10 +168,26 @@ identity RBAC on it, with the role assignments as proof.
|
|
|
168
168
|
|
|
169
169
|
## What it maps
|
|
170
170
|
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
171
|
+
**Any Azure resource type can be a seed.** The scan is not filtered by type, and
|
|
172
|
+
resources are mapped at two levels:
|
|
173
|
+
|
|
174
|
+
- **Typed rules** for the services where the relationship has a specific meaning:
|
|
175
|
+
App Service / Functions, Container Apps (+ environments), AKS, App Gateway, API
|
|
176
|
+
Management, Key Vault, Storage, SQL / PostgreSQL / MySQL / Cosmos, Redis, Service
|
|
177
|
+
Bus, Event Hub, Cognitive Search, Azure OpenAI, Container Registry, ML workspaces,
|
|
178
|
+
Log Analytics, App Insights, VNets, Private Endpoints and managed identities.
|
|
179
|
+
These produce edges like `hosted-on`, `reads-secret`, `pulls-image`, `routes-to`.
|
|
180
|
+
- **A generic ARM-reference pass** for everything else: any resolvable resource id
|
|
181
|
+
found in a resource's properties becomes a `references` edge, with the property
|
|
182
|
+
path as proof. So a type cloudmap has never heard of is still mapped, still
|
|
183
|
+
deterministically, still with evidence.
|
|
184
|
+
|
|
185
|
+
RBAC edges (`role: Key Vault Secrets User`) are extracted tenant-wide, so
|
|
186
|
+
"who has access to this" works for any resource that can be a role scope.
|
|
187
|
+
|
|
188
|
+
Depth is honest about itself: apps and clusters have rich outbound edges because
|
|
189
|
+
their config names other resources. Infrastructure resources are usually leaves
|
|
190
|
+
going outward, and their value is the reverse view (`--direction up`).
|
|
175
191
|
|
|
176
192
|
## How it works
|
|
177
193
|
|
|
@@ -242,6 +258,14 @@ is the deliberate switch, and cloudmap reads whatever subscription `az` is point
|
|
|
242
258
|
at. The read is read-only, but it is a read of live infrastructure, so point it on
|
|
243
259
|
purpose. **Do not point this at data you are not authorized to read.**
|
|
244
260
|
|
|
261
|
+
One asterisk on "read-only": AKS manifests are read through `az aks command
|
|
262
|
+
invoke`, which Azure implements by starting a short-lived pod in the cluster to
|
|
263
|
+
run the (read-only) `kubectl get` commands. Nothing of yours is modified, but an
|
|
264
|
+
audit of the cluster's control plane will see that ephemeral pod.
|
|
265
|
+
|
|
266
|
+
Tenant-wide enrichment runs its `az` reads concurrently; `CLOUDMAP_ENRICH_WORKERS`
|
|
267
|
+
(default 12) tunes how many at once.
|
|
268
|
+
|
|
245
269
|
```
|
|
246
270
|
cloudmap trace my-app --live --allow-live
|
|
247
271
|
```
|
|
@@ -264,7 +288,10 @@ everything".
|
|
|
264
288
|
subscription in the tenant)
|
|
265
289
|
--enrich MODE which web apps to deep-enrich for the dependencies that only
|
|
266
290
|
exist in app config. auto (default) = the seed alone when the
|
|
267
|
-
seed is a
|
|
291
|
+
seed is a workload; every app in scope when the seed is a data
|
|
292
|
+
service whose dependents hide in app config (Key Vault,
|
|
293
|
+
storage, SQL, Redis, ...); the seed alone for compute and
|
|
294
|
+
network resources, whose relationships ARM already returns;
|
|
268
295
|
all | seed | none
|
|
269
296
|
--resolve-secrets read KV secret values in-memory to see through KV-backed
|
|
270
297
|
connection strings (never printed or written)
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# Security
|
|
2
|
+
|
|
3
|
+
cloudmap reads cloud infrastructure and handles the output, so security reports
|
|
4
|
+
get priority over everything else.
|
|
5
|
+
|
|
6
|
+
## Reporting a vulnerability
|
|
7
|
+
|
|
8
|
+
Use [GitHub private vulnerability reporting](https://github.com/KatsaounisThanasis/cloudmap/security/advisories/new)
|
|
9
|
+
(Security tab → Report a vulnerability). Please do not open a public issue for
|
|
10
|
+
anything that could expose a user's infrastructure data before a fix exists.
|
|
11
|
+
|
|
12
|
+
You can expect an acknowledgement within a few days. There is no bounty - this
|
|
13
|
+
is a solo open-source project - but reports are credited in the release notes
|
|
14
|
+
unless you ask otherwise.
|
|
15
|
+
|
|
16
|
+
## What counts as a vulnerability here
|
|
17
|
+
|
|
18
|
+
The interesting failure modes for a tool like this:
|
|
19
|
+
|
|
20
|
+
- **Credential leakage into artifacts**: any way a secret, key, token or
|
|
21
|
+
password ends up in a `.drawio` / `.json` / `.html` / `.csv` / capture file.
|
|
22
|
+
The scrubber (`cloudmap/scrub.py`) redacts credentials and pseudonymises
|
|
23
|
+
identifiers; a value that survives it is a bug of the highest priority -
|
|
24
|
+
one shipped before (base64 padding, fixed in `2b1e06d`) and the regression
|
|
25
|
+
test suite grew from it.
|
|
26
|
+
- **Scope escalation**: cloudmap must only ever read what the caller's own
|
|
27
|
+
`az login` token can read, and must never perform a write operation against
|
|
28
|
+
the tenant.
|
|
29
|
+
- **Injection through cloud-controlled data**: resource names, tags and
|
|
30
|
+
properties are attacker-influenceable in shared tenants; anything that lets
|
|
31
|
+
them break out of an `az` argument list, the HTML viewer, or the draw.io XML
|
|
32
|
+
is in scope.
|
|
33
|
+
|
|
34
|
+
## Design promises you can hold us to
|
|
35
|
+
|
|
36
|
+
- Read-only: no `az` mutation commands, ever.
|
|
37
|
+
- Local-first: the only outbound network call in the codebase is to a
|
|
38
|
+
user-configured local model endpoint (`cloudmap/local_model.py`), off by
|
|
39
|
+
default. Resource data never leaves the machine.
|
|
40
|
+
- Secrets stay in memory: `--resolve-secrets` substitutes Key Vault values
|
|
41
|
+
in-memory for edge extraction and never prints or writes them.
|
|
42
|
+
- `tests/test_fixtures_safe.py` fails CI if a committed fixture ever carries
|
|
43
|
+
a credential or a real-looking identifier.
|
|
44
|
+
|
|
45
|
+
If you find code contradicting any of these, that is a valid report even if
|
|
46
|
+
you cannot demonstrate an exploit.
|
|
@@ -88,4 +88,10 @@ def load_graph(path):
|
|
|
88
88
|
data = json.load(f)
|
|
89
89
|
if _looks_neutral(data):
|
|
90
90
|
return graph_from_neutral(data)
|
|
91
|
-
|
|
91
|
+
graph = AzureAdapter.to_graph(data)
|
|
92
|
+
# A capture's own honesty flags (truncated, enriched, scrubbed) ride along:
|
|
93
|
+
# re-tracing a truncated export must not produce a map that claims to be
|
|
94
|
+
# complete just because the file was re-read from disk.
|
|
95
|
+
if isinstance(data, dict) and isinstance(data.get("meta"), dict):
|
|
96
|
+
graph.meta.update(data["meta"])
|
|
97
|
+
return graph
|
|
@@ -15,6 +15,14 @@ from .render.mermaid import to_mermaid
|
|
|
15
15
|
def main(argv=None):
|
|
16
16
|
# Interactive mode if no arguments are provided
|
|
17
17
|
if (argv is None and len(sys.argv) == 1) or (argv is not None and len(argv) == 0):
|
|
18
|
+
# The wizard hands the terminal to questionary; in a pipe / cron / CI
|
|
19
|
+
# there is no terminal to hand over, so fail with a pointer instead of
|
|
20
|
+
# letting the prompt library misbehave against a non-tty stream.
|
|
21
|
+
if not (sys.stdin.isatty() and sys.stdout.isatty()):
|
|
22
|
+
print("cloudmap: running with no arguments starts the interactive wizard, "
|
|
23
|
+
"which needs a terminal. Use `cloudmap --help` for the scriptable "
|
|
24
|
+
"commands.", file=sys.stderr)
|
|
25
|
+
return 2
|
|
18
26
|
from .interactive import interactive_main
|
|
19
27
|
return interactive_main()
|
|
20
28
|
|
|
@@ -43,9 +51,11 @@ def main(argv=None):
|
|
|
43
51
|
t.add_argument("--enrich", choices=["auto", "seed", "all", "none"], default="auto",
|
|
44
52
|
help="live: which web apps to deep-enrich for the dependencies that live "
|
|
45
53
|
"in app config (Key Vault refs, connection strings, RBAC). "
|
|
46
|
-
"auto = the seed alone when the seed is a
|
|
47
|
-
"when
|
|
48
|
-
"
|
|
54
|
+
"auto = the seed alone when the seed is a workload; every app in scope "
|
|
55
|
+
"when the seed is a data service whose dependents hide in app config "
|
|
56
|
+
"(Key Vault, storage, SQL, Redis, ...); the seed alone for compute and "
|
|
57
|
+
"network resources, whose relationships ARM already returns. "
|
|
58
|
+
"all = every app in scope; none = ARM topology only")
|
|
49
59
|
t.add_argument("--level", choices=["high", "detail"], default="high",
|
|
50
60
|
help="high = architecture view grouped by resource type (default); "
|
|
51
61
|
"detail = every instance with its real name")
|
|
@@ -114,8 +124,14 @@ def _cmd_trace(args):
|
|
|
114
124
|
graph = build_graph(resources)
|
|
115
125
|
else:
|
|
116
126
|
from .adapters import load_graph
|
|
117
|
-
resources
|
|
127
|
+
resources = []
|
|
118
128
|
graph = load_graph(args.fixture) # auto-detects raw export vs neutral cloudmap graph
|
|
129
|
+
# A re-traced capture keeps its own honesty flags: a scan that was
|
|
130
|
+
# truncated at capture time must not become a map that claims
|
|
131
|
+
# completeness just because it was re-read from disk.
|
|
132
|
+
truncated = bool(graph.meta.get("truncated", False))
|
|
133
|
+
read_gaps = list(graph.meta.get("read_gaps") or [])
|
|
134
|
+
blind_spots = list(graph.meta.get("blind_spots") or [])
|
|
119
135
|
|
|
120
136
|
seeds = find_seeds(graph, args.name)
|
|
121
137
|
if not seeds:
|
|
@@ -212,7 +228,9 @@ def _export_outputs(sub, seed, args, meta):
|
|
|
212
228
|
with open(args.csv_out, "w", encoding="utf-8") as f:
|
|
213
229
|
f.write(to_csv(sub, seed, meta=meta))
|
|
214
230
|
|
|
215
|
-
_print_summary(sub, seed, out, truncated=meta.get("truncated", False),
|
|
231
|
+
_print_summary(sub, seed, out, truncated=meta.get("truncated", False),
|
|
232
|
+
blind_spots=meta.get("blind_spots", []),
|
|
233
|
+
single_sub=bool(getattr(args, "single_sub", False)))
|
|
216
234
|
|
|
217
235
|
|
|
218
236
|
_WEBAPP = "microsoft.web/sites"
|
|
@@ -223,6 +241,35 @@ _AKS = "microsoft.containerservice/managedclusters"
|
|
|
223
241
|
# template, so there is nothing extra to fetch.
|
|
224
242
|
_CONFIG_WORKLOADS = (_WEBAPP, "microsoft.app/containerapps", _AKS)
|
|
225
243
|
|
|
244
|
+
# Seeds whose DEPENDENTS are typically named only inside application config, so
|
|
245
|
+
# reading every app in scope is the only way to find them. A tenant-wide pass
|
|
246
|
+
# costs one `az` round-trip per app, so it is spent here and nowhere else.
|
|
247
|
+
#
|
|
248
|
+
# Compute and network resources are deliberately absent: a VM's or a VNet's
|
|
249
|
+
# dependents are already in Resource Graph (a NIC's subnet, an app's
|
|
250
|
+
# virtualNetworkSubnetId), so enriching hundreds of apps to trace one VM buys
|
|
251
|
+
# nothing. Found the hard way - tracing a VM used to enrich 289 web apps and run
|
|
252
|
+
# for minutes before printing edges that were all in the ARM data already.
|
|
253
|
+
_CONFIG_REFERENCED_TYPES = frozenset({
|
|
254
|
+
"microsoft.keyvault/vaults",
|
|
255
|
+
"microsoft.storage/storageaccounts",
|
|
256
|
+
"microsoft.sql/servers",
|
|
257
|
+
"microsoft.sql/servers/databases",
|
|
258
|
+
"microsoft.dbforpostgresql/flexibleservers",
|
|
259
|
+
"microsoft.dbformysql/flexibleservers",
|
|
260
|
+
"microsoft.documentdb/databaseaccounts",
|
|
261
|
+
"microsoft.cache/redis",
|
|
262
|
+
"microsoft.servicebus/namespaces",
|
|
263
|
+
"microsoft.eventhub/namespaces",
|
|
264
|
+
"microsoft.search/searchservices",
|
|
265
|
+
"microsoft.cognitiveservices/accounts",
|
|
266
|
+
"microsoft.containerregistry/registries",
|
|
267
|
+
"microsoft.insights/components",
|
|
268
|
+
"microsoft.operationalinsights/workspaces",
|
|
269
|
+
"microsoft.web/sites",
|
|
270
|
+
"microsoft.app/containerapps",
|
|
271
|
+
})
|
|
272
|
+
|
|
226
273
|
|
|
227
274
|
def _enrichment_targets(graph, seed, mode, direction):
|
|
228
275
|
"""Which workloads to deep-enrich, and which stay a blind spot.
|
|
@@ -230,22 +277,25 @@ def _enrichment_targets(graph, seed, mode, direction):
|
|
|
230
277
|
An app's config-level dependencies exist nowhere until that app is enriched,
|
|
231
278
|
but enriching a whole tenant on every trace costs an `az` round-trip per app.
|
|
232
279
|
So `auto` spends the calls where the answer actually needs them: a web-app
|
|
233
|
-
seed's own config already yields its downward view, whereas a shared
|
|
234
|
-
|
|
235
|
-
config of the apps pointing at it.
|
|
280
|
+
seed's own config already yields its downward view, whereas a shared data
|
|
281
|
+
service (Key Vault, database, registry) can only learn its dependents from
|
|
282
|
+
the config of the apps pointing at it.
|
|
236
283
|
"""
|
|
237
284
|
workloads = [n for n in graph.nodes.values() if n.type in (_WEBAPP, _AKS)]
|
|
238
285
|
seed_only = [n for n in workloads if n.id == seed]
|
|
239
|
-
|
|
286
|
+
seed_type = graph.nodes[seed].type if seed in graph.nodes else ""
|
|
287
|
+
|
|
240
288
|
if mode == "none":
|
|
241
289
|
chosen = []
|
|
242
290
|
elif mode == "all":
|
|
243
291
|
chosen = workloads
|
|
244
|
-
elif mode == "seed" or seed_only and direction != "up":
|
|
292
|
+
elif mode == "seed" or (seed_only and direction != "up"):
|
|
245
293
|
chosen = seed_only
|
|
294
|
+
elif seed_type in _CONFIG_REFERENCED_TYPES:
|
|
295
|
+
chosen = workloads # its dependents hide in app config
|
|
246
296
|
else:
|
|
247
|
-
chosen =
|
|
248
|
-
|
|
297
|
+
chosen = seed_only # compute/network: ARM already has it
|
|
298
|
+
|
|
249
299
|
picked = {n.id for n in chosen}
|
|
250
300
|
return chosen, [n for n in workloads if n.id not in picked]
|
|
251
301
|
|
|
@@ -281,6 +331,13 @@ def _enrich_live(args, graph, seed, resources):
|
|
|
281
331
|
print("Read gaps while enriching (edges below may be INCOMPLETE):", file=sys.stderr)
|
|
282
332
|
for msg in read_gaps:
|
|
283
333
|
print(f" ! could not read {msg}", file=sys.stderr)
|
|
334
|
+
# The artifact is meant to be shared; az stderr carries correlation
|
|
335
|
+
# ids and GUIDs. The terminal (above) keeps the full text, the map
|
|
336
|
+
# records a short classified reason.
|
|
337
|
+
from .ingest.azure import classify_gap
|
|
338
|
+
read_gaps = [f"{m.partition(': ')[0]}: {classify_gap(m.partition(': ')[2])}"
|
|
339
|
+
if ": " in m else classify_gap(m)
|
|
340
|
+
for m in read_gaps]
|
|
284
341
|
|
|
285
342
|
graph = build_graph(resources) # re-extract, now that config is present
|
|
286
343
|
resolver = Resolver(graph.nodes)
|
|
@@ -308,12 +365,24 @@ def _enrich_live(args, graph, seed, resources):
|
|
|
308
365
|
# An un-enriched app is a known class of missing edge - say so rather than let
|
|
309
366
|
# an empty upward view read as "nothing depends on this".
|
|
310
367
|
if skipped and args.direction != "down":
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
368
|
+
# Only claim this matters when it actually does. A Key Vault's dependents
|
|
369
|
+
# really do hide in app config; a VNet's are already in the ARM data, so
|
|
370
|
+
# telling its user to enrich 289 apps would send them off for minutes to
|
|
371
|
+
# learn nothing.
|
|
372
|
+
if graph.nodes[seed].type in _CONFIG_REFERENCED_TYPES:
|
|
373
|
+
blind_spots.append(
|
|
374
|
+
f"{len(skipped)} workload(s) in scope were not deep-enriched, so anything that "
|
|
375
|
+
f"depends on this resource through config (Key Vault references, connection "
|
|
376
|
+
f"strings, hostnames) cannot appear as an inbound edge. "
|
|
377
|
+
f"Re-run with --enrich all to close this gap."
|
|
378
|
+
)
|
|
379
|
+
else:
|
|
380
|
+
blind_spots.append(
|
|
381
|
+
f"{len(skipped)} workload(s) in scope were not deep-enriched. For this "
|
|
382
|
+
f"resource type that is usually fine - its relationships are in the ARM "
|
|
383
|
+
f"data - but an application naming it in free-text config would be missed. "
|
|
384
|
+
f"Use --enrich all if you need that certainty."
|
|
385
|
+
)
|
|
317
386
|
return graph, read_gaps, blind_spots
|
|
318
387
|
|
|
319
388
|
|
|
@@ -354,6 +423,19 @@ def _cmd_capture(args):
|
|
|
354
423
|
from .scrub import scrub
|
|
355
424
|
resources, stats = scrub(resources)
|
|
356
425
|
|
|
426
|
+
# Kubernetes secret material is read in-memory so the extractors can spot a
|
|
427
|
+
# connection string inside a secret, and it is dropped here unconditionally -
|
|
428
|
+
# including under --no-scrub. A capture is a file people commit; the one
|
|
429
|
+
# field that carries decoded secrets has no business in it, and the scrub
|
|
430
|
+
# pass deleting it too is belt and braces rather than the only guard.
|
|
431
|
+
dropped = 0
|
|
432
|
+
for r in resources:
|
|
433
|
+
if isinstance(r, dict) and r.pop("kubernetes_text", None) is not None:
|
|
434
|
+
dropped += 1
|
|
435
|
+
if dropped:
|
|
436
|
+
print(f"Dropped Kubernetes manifest text from {dropped} cluster(s): it can carry "
|
|
437
|
+
f"decoded secret values and is never written to an export.", file=sys.stderr)
|
|
438
|
+
|
|
357
439
|
_write_export(args.out, resources, scrubbed=not args.no_scrub,
|
|
358
440
|
meta={"truncated": truncated, "enriched": args.enrich == "all"})
|
|
359
441
|
print(f"Captured {len(resources)} rows -> {args.out}")
|
|
@@ -462,7 +544,7 @@ def _ensure_parent(path):
|
|
|
462
544
|
os.makedirs(parent, exist_ok=True)
|
|
463
545
|
|
|
464
546
|
|
|
465
|
-
def _print_summary(graph, seed, out, truncated=False, blind_spots=()):
|
|
547
|
+
def _print_summary(graph, seed, out, truncated=False, blind_spots=(), single_sub=False):
|
|
466
548
|
try:
|
|
467
549
|
from rich.console import Console
|
|
468
550
|
from rich.panel import Panel
|
|
@@ -489,35 +571,62 @@ def _print_summary(graph, seed, out, truncated=False, blind_spots=()):
|
|
|
489
571
|
if "vault" in t: return "🔐"
|
|
490
572
|
return "📦"
|
|
491
573
|
|
|
492
|
-
def
|
|
493
|
-
node = graph.nodes[node_id]
|
|
574
|
+
def _label(node):
|
|
494
575
|
color = "cyan" if not node.external else "dim white"
|
|
495
576
|
txt = f"[{color}]{_get_icon(node.type)} {node.name}[/{color}]"
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
577
|
+
return txt + (" [italic dim](external)[/italic dim]" if node.external else "")
|
|
578
|
+
|
|
579
|
+
def _kind_colour(kind):
|
|
580
|
+
if "secret" in kind: return "yellow"
|
|
581
|
+
if "connects" in kind: return "green"
|
|
582
|
+
if "auth" in kind: return "magenta"
|
|
583
|
+
if "observes" in kind: return "dim white"
|
|
584
|
+
return "blue"
|
|
585
|
+
|
|
586
|
+
def _build_tree(node_id, seen, upward):
|
|
587
|
+
"""Walk one direction only. Upward reads "what depends on me", so its
|
|
588
|
+
arrows are drawn pointing back at the parent - printing them like
|
|
589
|
+
downstream edges would state the dependency backwards."""
|
|
590
|
+
tree = Tree(_label(graph.nodes[node_id]))
|
|
499
591
|
seen.add(node_id)
|
|
500
|
-
|
|
501
592
|
for e in graph.edges:
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
else:
|
|
514
|
-
branch = _build_tree(e.target, set(seen))
|
|
515
|
-
branch.label = f"{lbl} " + str(branch.label)
|
|
516
|
-
tree.add(branch)
|
|
593
|
+
nxt = e.source if upward else e.target
|
|
594
|
+
if (e.target if upward else e.source) != node_id:
|
|
595
|
+
continue
|
|
596
|
+
lbl = (f"[{_kind_colour(e.kind)}]<--{e.kind}--[/{_kind_colour(e.kind)}]" if upward
|
|
597
|
+
else f"[{_kind_colour(e.kind)}]--{e.kind}-->[/{_kind_colour(e.kind)}]")
|
|
598
|
+
if nxt in seen:
|
|
599
|
+
tree.add(f"{lbl} [dim]{graph.nodes[nxt].name} (cycle)[/dim]")
|
|
600
|
+
else:
|
|
601
|
+
branch = _build_tree(nxt, set(seen), upward)
|
|
602
|
+
branch.label = f"{lbl} " + str(branch.label)
|
|
603
|
+
tree.add(branch)
|
|
517
604
|
return tree
|
|
518
605
|
|
|
519
|
-
|
|
520
|
-
|
|
606
|
+
# Both directions get their own panel. Tracing shared infrastructure (a
|
|
607
|
+
# VNet, a Key Vault, a plan) puts everything in the upward half, so a
|
|
608
|
+
# single downward tree used to print two branches under a headline that
|
|
609
|
+
# counted seven resources.
|
|
610
|
+
down = [e for e in graph.edges if e.source == seed]
|
|
611
|
+
up = [e for e in graph.edges if e.target == seed]
|
|
612
|
+
if not graph.edges:
|
|
613
|
+
# A lonely one-node map is a real answer, but an unexplained one is
|
|
614
|
+
# indistinguishable from a broken tool. Say what could hide it.
|
|
615
|
+
console.print(
|
|
616
|
+
"[bold yellow]No dependencies found.[/bold yellow] Nothing in the scanned "
|
|
617
|
+
"scope references this resource and it references nothing scanned. Worth "
|
|
618
|
+
"checking: whether its dependents live in another subscription (this scan "
|
|
619
|
+
"was scoped), and whether --enrich all would reveal a config-only reference."
|
|
620
|
+
if single_sub else
|
|
621
|
+
"[bold yellow]No dependencies found.[/bold yellow] Nothing in the scanned "
|
|
622
|
+
"scope references this resource and it references nothing scanned."
|
|
623
|
+
)
|
|
624
|
+
if down or not up:
|
|
625
|
+
console.print(Panel(_build_tree(seed, set(), upward=False),
|
|
626
|
+
title=f"Depends on ({len(down)})", border_style="blue"))
|
|
627
|
+
if up:
|
|
628
|
+
console.print(Panel(_build_tree(seed, set(), upward=True),
|
|
629
|
+
title=f"What depends on it ({len(up)})", border_style="magenta"))
|
|
521
630
|
console.print(f"🔗 [bold]draw.io:[/bold] {out}\n")
|
|
522
631
|
else:
|
|
523
632
|
print(f"Seed: {n.name} ({n.type})")
|