overreach 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- overreach-0.1.0/LICENSE +21 -0
- overreach-0.1.0/PKG-INFO +180 -0
- overreach-0.1.0/README.md +152 -0
- overreach-0.1.0/overreach/__init__.py +8 -0
- overreach-0.1.0/overreach/cli.py +92 -0
- overreach-0.1.0/overreach/collect_dcspm.py +161 -0
- overreach-0.1.0/overreach/collect_graph.py +235 -0
- overreach-0.1.0/overreach/engine.py +61 -0
- overreach-0.1.0/overreach/models.py +53 -0
- overreach-0.1.0/overreach/report.py +55 -0
- overreach-0.1.0/overreach/roles_catalog.py +52 -0
- overreach-0.1.0/overreach/rolesreport.py +98 -0
- overreach-0.1.0/overreach/rules.py +151 -0
- overreach-0.1.0/overreach.egg-info/PKG-INFO +180 -0
- overreach-0.1.0/overreach.egg-info/SOURCES.txt +24 -0
- overreach-0.1.0/overreach.egg-info/dependency_links.txt +1 -0
- overreach-0.1.0/overreach.egg-info/entry_points.txt +2 -0
- overreach-0.1.0/overreach.egg-info/requires.txt +3 -0
- overreach-0.1.0/overreach.egg-info/top_level.txt +1 -0
- overreach-0.1.0/pyproject.toml +44 -0
- overreach-0.1.0/setup.cfg +4 -0
- overreach-0.1.0/tests/test_collect_graph.py +54 -0
- overreach-0.1.0/tests/test_collect_graph_op6.py +45 -0
- overreach-0.1.0/tests/test_dcspm.py +30 -0
- overreach-0.1.0/tests/test_rolesreport.py +42 -0
- overreach-0.1.0/tests/test_rules.py +54 -0
overreach-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Raj Penchala
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
overreach-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: overreach
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Audit Microsoft Entra ID for over-permissioned human identities — deterministic, explainable, read-only.
|
|
5
|
+
Author: Raj Penchala
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/rpmsft9/overreach
|
|
8
|
+
Project-URL: Repository, https://github.com/rpmsft9/overreach
|
|
9
|
+
Project-URL: Issues, https://github.com/rpmsft9/overreach/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/rpmsft9/overreach/blob/main/CHANGELOG.md
|
|
11
|
+
Keywords: entra,azure-ad,least-privilege,iam,pim,identity-governance,security
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Information Technology
|
|
14
|
+
Classifier: Intended Audience :: System Administrators
|
|
15
|
+
Classifier: Topic :: Security
|
|
16
|
+
Classifier: Topic :: System :: Systems Administration :: Authentication/Directory
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Operating System :: OS Independent
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
27
|
+
Dynamic: license-file
|
|
28
|
+
|
|
29
|
+
# overreach
|
|
30
|
+
|
|
31
|
+
**Audit Microsoft Entra ID for over-permissioned human identities — deterministic, explainable, read-only.**
|
|
32
|
+
|
|
33
|
+
`overreach` finds humans who hold more privilege than they need, right-sizes them with
|
|
34
|
+
evidence, and flags the ones whose over-privilege flows into **AI agents**. It is the
|
|
35
|
+
human-identity companion to [nhi-scan](https://github.com/rpmsft9/nhi-scan) (non-human &
|
|
36
|
+
agent identities): nhi-scan owns the machines, `overreach` owns the people, and they
|
|
37
|
+
join at the OBO bridge (OP6).
|
|
38
|
+
|
|
39
|
+
No LLM in the verdict path. Every finding records its evidence and a confidence level.
|
|
40
|
+
|
|
41
|
+
## Install
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
pip install overreach
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Then run `overreach ...` (or `python -m overreach.cli ...` from a checkout).
|
|
48
|
+
|
|
49
|
+
## Quickstart (30 seconds, no tenant needed)
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
overreach scan --input fixtures/sample-tenant.json
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
That runs against a synthetic directory, so you can see the output offline. JSON too:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
python -m overreach.cli scan --input fixtures/sample-tenant.json --format json
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Just want the flat "every role -> everyone assigned to it" report (the one Azure makes you
|
|
62
|
+
assemble yourself), with eligible/active and via-group resolved? Markdown, JSON, or CSV:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
python -m overreach.cli roles-report --input fixtures/sample-tenant.json
|
|
66
|
+
python -m overreach.cli roles-report --input fixtures/sample-tenant.json --format csv > roles.csv
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Against a real tenant (read-only; see **Safety**):
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
az login
|
|
73
|
+
python -m overreach.cli inventory --out inv.json
|
|
74
|
+
python -m overreach.cli scan --input inv.json
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
### Fold in Defender CSPM (usage-based CIEM signals)
|
|
78
|
+
|
|
79
|
+
Microsoft retired Entra Permissions Management (ex-CloudKnox); its CIEM analysis now
|
|
80
|
+
surfaces as **Defender CSPM** recommendations for **stale/unused and over-permissioned
|
|
81
|
+
identities** on Azure (and AWS/GCP) *resource* permissions — the usage-based signal
|
|
82
|
+
`overreach`'s native checks only approximate. If you have Defender CSPM, pull those and
|
|
83
|
+
merge them in (tagged `_Defender CSPM_` so they stay distinct from directory-role findings):
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
python -m overreach.cli dcspm --out dcspm.json # read-only Resource Graph pull
|
|
87
|
+
python -m overreach.cli scan --input inv.json --dcspm dcspm.json # directory roles + resource-plane CIEM
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Offline, the same merge runs against the fixtures:
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
python -m overreach.cli scan --input fixtures/sample-tenant.json --dcspm fixtures/sample-dcspm.json
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Scope split: `overreach`'s native OP1–OP6 cover **Entra directory roles**; DCSPM covers
|
|
97
|
+
**resource / RBAC** permissions. Together they span both planes.
|
|
98
|
+
|
|
99
|
+
## The methodology (why this is more than a checkbox)
|
|
100
|
+
|
|
101
|
+
"Over-privileged" means *holds more than they need* — and **"need" has no ground truth**.
|
|
102
|
+
There is no authoritative record of what each person's job requires. So `overreach` does
|
|
103
|
+
not pretend to a verdict. It **triangulates over-privilege from signals**, each with a
|
|
104
|
+
stated confidence, and presents **candidates for review with evidence**. The tool narrows
|
|
105
|
+
the haystack and shows its work; a human access review confirms "needed or not".
|
|
106
|
+
|
|
107
|
+
The backbone is deterministic signals that need **no model of need**:
|
|
108
|
+
|
|
109
|
+
| ID | Check | Signal | Confidence |
|
|
110
|
+
|----|-------|--------|-----------|
|
|
111
|
+
| **OP1** | Standing privileged role | A privileged role held as a permanent **active** assignment that should be **PIM-eligible** (on-demand). Over-privileged *in time*. | high |
|
|
112
|
+
| **OP2** | Global Admin sprawl | More Global Administrators than the recommended minimum (~5). | high |
|
|
113
|
+
| **OP3** | Dormant privileged account | Holds privilege but hasn't signed in (unused privilege = not needed now). The MVP proxy for usage-based right-sizing. | high |
|
|
114
|
+
| **OP4** | Privileged without MFA | **Amplifier** — raises the *danger* of over-privilege, not the privilege itself. | high |
|
|
115
|
+
| **OP5** | Hidden admin via nesting | Privilege inherited indirectly through a group, easy to miss in a native role view. | high |
|
|
116
|
+
| **OP6** | Privilege → agent bridge | A privileged human who owns/sponsors an **OBO-capable** app or agent: the agent inherits the user's entitlements, so human over-privilege becomes agent blast radius. *The differentiator.* | high |
|
|
117
|
+
|
|
118
|
+
Design principles:
|
|
119
|
+
- **Lean on signals that don't require knowing "need"** (standing, nesting, dormancy, sprawl).
|
|
120
|
+
- **Label amplifiers honestly** (OP4) so findings aren't overstated.
|
|
121
|
+
- **Never claim a verdict you can't evidence.**
|
|
122
|
+
|
|
123
|
+
## How this relates to Defender CSPM
|
|
124
|
+
|
|
125
|
+
A fair question: if Defender CSPM already flags over-permissioned identities, why
|
|
126
|
+
`overreach`? Because DCSPM is **one input**, and `overreach` is the layer around it.
|
|
127
|
+
|
|
128
|
+
- **Different plane.** DCSPM's CIEM analyzes **resource / RBAC** permissions (Azure, AWS,
|
|
129
|
+
GCP). It does **not** assess **Entra directory roles** — Global Administrator, Privileged
|
|
130
|
+
Role Administrator, and the rest — which is where the most dangerous, tenant-takeover
|
|
131
|
+
privilege lives. `overreach`'s native OP1–OP6 target exactly that plane DCSPM leaves blind.
|
|
132
|
+
- **Checks DCSPM doesn't run** on those roles: standing-vs-PIM-eligible (OP1), Global Admin
|
|
133
|
+
sprawl (OP2), dormant privileged accounts (OP3), privileged-without-MFA (OP4), and hidden
|
|
134
|
+
admin via nested groups (OP5).
|
|
135
|
+
- **The agent bridge (OP6)** — connecting a human's privilege to the agents that inherit it
|
|
136
|
+
via OBO / ownership — is something no CIEM product models.
|
|
137
|
+
- **Deterministic, explainable, and no license.** DCSPM is a paid plan and a black-box
|
|
138
|
+
verdict. `overreach` runs against any tenant with read-only Graph scopes, records the
|
|
139
|
+
evidence and confidence for every finding, and works where DCSPM isn't licensed.
|
|
140
|
+
- **It unifies the picture** — directory-role findings (native) + resource-plane findings
|
|
141
|
+
(DCSPM, tagged by source) + non-human identities (via nhi-scan) in one risk model.
|
|
142
|
+
|
|
143
|
+
In short: `overreach` **surrounds** DCSPM — it works without it, covers what it misses, and
|
|
144
|
+
folds it in as the usage-based signal when you have it.
|
|
145
|
+
|
|
146
|
+
## Roadmap
|
|
147
|
+
|
|
148
|
+
- **Usage-based right-sizing** — the strongest signal of all: compare *granted* vs.
|
|
149
|
+
*actually-used* permissions. Needs per-action telemetry. OP3 (dormancy) is the MVP proxy.
|
|
150
|
+
- **Pull Defender CSPM (DCSPM) CIEM signals.** Microsoft retired Entra Permissions
|
|
151
|
+
Management (ex-CloudKnox) as a standalone product; its CIEM analysis now surfaces as
|
|
152
|
+
**Microsoft Defender CSPM recommendations** that flag **stale/unused and over-permissioned
|
|
153
|
+
identities** across Azure (and AWS/GCP) resource permissions. `overreach` can ingest these
|
|
154
|
+
— via Azure Resource Graph over `Microsoft.Security/assessments`, or the Defender for Cloud
|
|
155
|
+
assessments API — and fold them in, upgrading the usage-based signal from a proxy to the
|
|
156
|
+
real thing. Scope split worth stating: DCSPM covers **resource / RBAC** permissions, while
|
|
157
|
+
`overreach`'s native checks cover **Entra directory roles** — the two are complementary, and
|
|
158
|
+
together they cover both planes.
|
|
159
|
+
- ~~Group-nesting resolution in the collector~~ **done** — a role assigned to a group is
|
|
160
|
+
expanded into its transitive user members (tagged `via="group:<name>"`), so OP5 and
|
|
161
|
+
roles-report's via-group populate from a live tenant, not just a supplied inventory.
|
|
162
|
+
- ~~owned_apps / owned_agents in the collector~~ **done** — the collector attaches each
|
|
163
|
+
user's owned app registrations (flagged `obo_capable` when the app requests delegated
|
|
164
|
+
permissions) and their Agent 365 registrations, so OP6 (the agent bridge) fires from a
|
|
165
|
+
live tenant. Agent 365 is preview, so that part is best-effort and skipped where absent.
|
|
166
|
+
- **Peer baselining** — flag users with far more entitlement than functional peers.
|
|
167
|
+
- **Unified report** merging `overreach` (human) and `nhi-scan` (non-human) output.
|
|
168
|
+
|
|
169
|
+
## Safety
|
|
170
|
+
|
|
171
|
+
- **Read-only.** `overreach` never writes to your directory.
|
|
172
|
+
- **Authorized tenants only.** Run it against your own tenant, or one you have explicit
|
|
173
|
+
written permission to audit. Never point it at someone else's directory.
|
|
174
|
+
- **Least-privilege scopes.** The collector needs only `Directory.Read.All`,
|
|
175
|
+
`RoleManagement.Read.Directory`, and `AuditLog.Read.All`, via your own `az login`.
|
|
176
|
+
|
|
177
|
+
## Status
|
|
178
|
+
|
|
179
|
+
Alpha (v0.1.0). The offline `scan` path and all six checks are implemented and tested;
|
|
180
|
+
the Graph collector is a best-effort MVP (see `collect_graph.py` TODOs).
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
# overreach
|
|
2
|
+
|
|
3
|
+
**Audit Microsoft Entra ID for over-permissioned human identities — deterministic, explainable, read-only.**
|
|
4
|
+
|
|
5
|
+
`overreach` finds humans who hold more privilege than they need, right-sizes them with
|
|
6
|
+
evidence, and flags the ones whose over-privilege flows into **AI agents**. It is the
|
|
7
|
+
human-identity companion to [nhi-scan](https://github.com/rpmsft9/nhi-scan) (non-human &
|
|
8
|
+
agent identities): nhi-scan owns the machines, `overreach` owns the people, and they
|
|
9
|
+
join at the OBO bridge (OP6).
|
|
10
|
+
|
|
11
|
+
No LLM in the verdict path. Every finding records its evidence and a confidence level.
|
|
12
|
+
|
|
13
|
+
## Install
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
pip install overreach
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
Then run `overreach ...` (or `python -m overreach.cli ...` from a checkout).
|
|
20
|
+
|
|
21
|
+
## Quickstart (30 seconds, no tenant needed)
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
overreach scan --input fixtures/sample-tenant.json
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
That runs against a synthetic directory, so you can see the output offline. JSON too:
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
python -m overreach.cli scan --input fixtures/sample-tenant.json --format json
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
Just want the flat "every role -> everyone assigned to it" report (the one Azure makes you
|
|
34
|
+
assemble yourself), with eligible/active and via-group resolved? Markdown, JSON, or CSV:
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
python -m overreach.cli roles-report --input fixtures/sample-tenant.json
|
|
38
|
+
python -m overreach.cli roles-report --input fixtures/sample-tenant.json --format csv > roles.csv
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Against a real tenant (read-only; see **Safety**):
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
az login
|
|
45
|
+
python -m overreach.cli inventory --out inv.json
|
|
46
|
+
python -m overreach.cli scan --input inv.json
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
### Fold in Defender CSPM (usage-based CIEM signals)
|
|
50
|
+
|
|
51
|
+
Microsoft retired Entra Permissions Management (ex-CloudKnox); its CIEM analysis now
|
|
52
|
+
surfaces as **Defender CSPM** recommendations for **stale/unused and over-permissioned
|
|
53
|
+
identities** on Azure (and AWS/GCP) *resource* permissions — the usage-based signal
|
|
54
|
+
`overreach`'s native checks only approximate. If you have Defender CSPM, pull those and
|
|
55
|
+
merge them in (tagged `_Defender CSPM_` so they stay distinct from directory-role findings):
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
python -m overreach.cli dcspm --out dcspm.json # read-only Resource Graph pull
|
|
59
|
+
python -m overreach.cli scan --input inv.json --dcspm dcspm.json # directory roles + resource-plane CIEM
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Offline, the same merge runs against the fixtures:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
python -m overreach.cli scan --input fixtures/sample-tenant.json --dcspm fixtures/sample-dcspm.json
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Scope split: `overreach`'s native OP1–OP6 cover **Entra directory roles**; DCSPM covers
|
|
69
|
+
**resource / RBAC** permissions. Together they span both planes.
|
|
70
|
+
|
|
71
|
+
## The methodology (why this is more than a checkbox)
|
|
72
|
+
|
|
73
|
+
"Over-privileged" means *holds more than they need* — and **"need" has no ground truth**.
|
|
74
|
+
There is no authoritative record of what each person's job requires. So `overreach` does
|
|
75
|
+
not pretend to a verdict. It **triangulates over-privilege from signals**, each with a
|
|
76
|
+
stated confidence, and presents **candidates for review with evidence**. The tool narrows
|
|
77
|
+
the haystack and shows its work; a human access review confirms "needed or not".
|
|
78
|
+
|
|
79
|
+
The backbone is deterministic signals that need **no model of need**:
|
|
80
|
+
|
|
81
|
+
| ID | Check | Signal | Confidence |
|
|
82
|
+
|----|-------|--------|-----------|
|
|
83
|
+
| **OP1** | Standing privileged role | A privileged role held as a permanent **active** assignment that should be **PIM-eligible** (on-demand). Over-privileged *in time*. | high |
|
|
84
|
+
| **OP2** | Global Admin sprawl | More Global Administrators than the recommended minimum (~5). | high |
|
|
85
|
+
| **OP3** | Dormant privileged account | Holds privilege but hasn't signed in (unused privilege = not needed now). The MVP proxy for usage-based right-sizing. | high |
|
|
86
|
+
| **OP4** | Privileged without MFA | **Amplifier** — raises the *danger* of over-privilege, not the privilege itself. | high |
|
|
87
|
+
| **OP5** | Hidden admin via nesting | Privilege inherited indirectly through a group, easy to miss in a native role view. | high |
|
|
88
|
+
| **OP6** | Privilege → agent bridge | A privileged human who owns/sponsors an **OBO-capable** app or agent: the agent inherits the user's entitlements, so human over-privilege becomes agent blast radius. *The differentiator.* | high |
|
|
89
|
+
|
|
90
|
+
Design principles:
|
|
91
|
+
- **Lean on signals that don't require knowing "need"** (standing, nesting, dormancy, sprawl).
|
|
92
|
+
- **Label amplifiers honestly** (OP4) so findings aren't overstated.
|
|
93
|
+
- **Never claim a verdict you can't evidence.**
|
|
94
|
+
|
|
95
|
+
## How this relates to Defender CSPM
|
|
96
|
+
|
|
97
|
+
A fair question: if Defender CSPM already flags over-permissioned identities, why
|
|
98
|
+
`overreach`? Because DCSPM is **one input**, and `overreach` is the layer around it.
|
|
99
|
+
|
|
100
|
+
- **Different plane.** DCSPM's CIEM analyzes **resource / RBAC** permissions (Azure, AWS,
|
|
101
|
+
GCP). It does **not** assess **Entra directory roles** — Global Administrator, Privileged
|
|
102
|
+
Role Administrator, and the rest — which is where the most dangerous, tenant-takeover
|
|
103
|
+
privilege lives. `overreach`'s native OP1–OP6 target exactly that plane DCSPM leaves blind.
|
|
104
|
+
- **Checks DCSPM doesn't run** on those roles: standing-vs-PIM-eligible (OP1), Global Admin
|
|
105
|
+
sprawl (OP2), dormant privileged accounts (OP3), privileged-without-MFA (OP4), and hidden
|
|
106
|
+
admin via nested groups (OP5).
|
|
107
|
+
- **The agent bridge (OP6)** — connecting a human's privilege to the agents that inherit it
|
|
108
|
+
via OBO / ownership — is something no CIEM product models.
|
|
109
|
+
- **Deterministic, explainable, and no license.** DCSPM is a paid plan and a black-box
|
|
110
|
+
verdict. `overreach` runs against any tenant with read-only Graph scopes, records the
|
|
111
|
+
evidence and confidence for every finding, and works where DCSPM isn't licensed.
|
|
112
|
+
- **It unifies the picture** — directory-role findings (native) + resource-plane findings
|
|
113
|
+
(DCSPM, tagged by source) + non-human identities (via nhi-scan) in one risk model.
|
|
114
|
+
|
|
115
|
+
In short: `overreach` **surrounds** DCSPM — it works without it, covers what it misses, and
|
|
116
|
+
folds it in as the usage-based signal when you have it.
|
|
117
|
+
|
|
118
|
+
## Roadmap
|
|
119
|
+
|
|
120
|
+
- **Usage-based right-sizing** — the strongest signal of all: compare *granted* vs.
|
|
121
|
+
*actually-used* permissions. Needs per-action telemetry. OP3 (dormancy) is the MVP proxy.
|
|
122
|
+
- **Pull Defender CSPM (DCSPM) CIEM signals.** Microsoft retired Entra Permissions
|
|
123
|
+
Management (ex-CloudKnox) as a standalone product; its CIEM analysis now surfaces as
|
|
124
|
+
**Microsoft Defender CSPM recommendations** that flag **stale/unused and over-permissioned
|
|
125
|
+
identities** across Azure (and AWS/GCP) resource permissions. `overreach` can ingest these
|
|
126
|
+
— via Azure Resource Graph over `Microsoft.Security/assessments`, or the Defender for Cloud
|
|
127
|
+
assessments API — and fold them in, upgrading the usage-based signal from a proxy to the
|
|
128
|
+
real thing. Scope split worth stating: DCSPM covers **resource / RBAC** permissions, while
|
|
129
|
+
`overreach`'s native checks cover **Entra directory roles** — the two are complementary, and
|
|
130
|
+
together they cover both planes.
|
|
131
|
+
- ~~Group-nesting resolution in the collector~~ **done** — a role assigned to a group is
|
|
132
|
+
expanded into its transitive user members (tagged `via="group:<name>"`), so OP5 and
|
|
133
|
+
roles-report's via-group populate from a live tenant, not just a supplied inventory.
|
|
134
|
+
- ~~owned_apps / owned_agents in the collector~~ **done** — the collector attaches each
|
|
135
|
+
user's owned app registrations (flagged `obo_capable` when the app requests delegated
|
|
136
|
+
permissions) and their Agent 365 registrations, so OP6 (the agent bridge) fires from a
|
|
137
|
+
live tenant. Agent 365 is preview, so that part is best-effort and skipped where absent.
|
|
138
|
+
- **Peer baselining** — flag users with far more entitlement than functional peers.
|
|
139
|
+
- **Unified report** merging `overreach` (human) and `nhi-scan` (non-human) output.
|
|
140
|
+
|
|
141
|
+
## Safety
|
|
142
|
+
|
|
143
|
+
- **Read-only.** `overreach` never writes to your directory.
|
|
144
|
+
- **Authorized tenants only.** Run it against your own tenant, or one you have explicit
|
|
145
|
+
written permission to audit. Never point it at someone else's directory.
|
|
146
|
+
- **Least-privilege scopes.** The collector needs only `Directory.Read.All`,
|
|
147
|
+
`RoleManagement.Read.Directory`, and `AuditLog.Read.All`, via your own `az login`.
|
|
148
|
+
|
|
149
|
+
## Status
|
|
150
|
+
|
|
151
|
+
Alpha (v0.1.0). The offline `scan` path and all six checks are implemented and tested;
|
|
152
|
+
the Graph collector is a best-effort MVP (see `collect_graph.py` TODOs).
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""overreach - audit Microsoft Entra ID for over-permissioned human identities.
|
|
2
|
+
|
|
3
|
+
Deterministic and explainable: every finding records the evidence and its
|
|
4
|
+
confidence, and nothing uses an LLM in the verdict path. Read-only by design.
|
|
5
|
+
Companion to nhi-scan (non-human / agent identities); see OP6 for the bridge.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""overreach command-line interface.
|
|
2
|
+
|
|
3
|
+
overreach scan --input fixtures/sample-tenant.json # offline, no tenant needed
|
|
4
|
+
overreach scan --input inv.json --format json
|
|
5
|
+
overreach inventory --out inv.json # read-only gather from Graph (needs az login)
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import argparse
|
|
9
|
+
import sys
|
|
10
|
+
|
|
11
|
+
from . import engine, report
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def main(argv=None):
|
|
15
|
+
parser = argparse.ArgumentParser(
|
|
16
|
+
prog="overreach",
|
|
17
|
+
description="Audit Microsoft Entra ID for over-permissioned human identities (read-only).",
|
|
18
|
+
)
|
|
19
|
+
sub = parser.add_subparsers(dest="cmd", required=True)
|
|
20
|
+
|
|
21
|
+
s = sub.add_parser("scan", help="Scan an inventory JSON and report over-privilege findings.")
|
|
22
|
+
s.add_argument("--input", required=True, help="Path to an inventory JSON (see fixtures/).")
|
|
23
|
+
s.add_argument("--dcspm", help="Optional normalized Defender CSPM JSON to merge in "
|
|
24
|
+
"(see fixtures/sample-dcspm.json).")
|
|
25
|
+
s.add_argument("--format", choices=["md", "json"], default="md")
|
|
26
|
+
s.add_argument("--ga-max", type=int, default=5, help="Max Global Admins before OP2 fires (default 5).")
|
|
27
|
+
|
|
28
|
+
g = sub.add_parser("inventory", help="Gather a read-only inventory from Microsoft Graph.")
|
|
29
|
+
g.add_argument("--out", required=True, help="Where to write the inventory JSON.")
|
|
30
|
+
|
|
31
|
+
d = sub.add_parser("dcspm", help="Gather Defender CSPM CIEM recommendations (read-only, needs az login).")
|
|
32
|
+
d.add_argument("--out", required=True, help="Where to write the normalized DCSPM JSON.")
|
|
33
|
+
|
|
34
|
+
r = sub.add_parser("roles-report",
|
|
35
|
+
help="List every role and everyone assigned to it (eligible/active and via-group resolved).")
|
|
36
|
+
r.add_argument("--input", required=True, help="Path to an inventory JSON (see fixtures/).")
|
|
37
|
+
r.add_argument("--format", choices=["md", "json", "csv"], default="md")
|
|
38
|
+
|
|
39
|
+
args = parser.parse_args(argv)
|
|
40
|
+
|
|
41
|
+
if args.cmd == "scan":
|
|
42
|
+
tenant, identities = engine.load_inventory(args.input)
|
|
43
|
+
findings = engine.scan(identities, {"ga_max": args.ga_max})
|
|
44
|
+
if args.dcspm:
|
|
45
|
+
from . import collect_dcspm
|
|
46
|
+
findings.extend(collect_dcspm.load_findings(args.dcspm))
|
|
47
|
+
out = report.to_markdown(tenant, findings) if args.format == "md" else report.to_json(tenant, findings)
|
|
48
|
+
print(out)
|
|
49
|
+
return 0
|
|
50
|
+
|
|
51
|
+
if args.cmd == "inventory":
|
|
52
|
+
from . import collect_graph
|
|
53
|
+
try:
|
|
54
|
+
data = collect_graph.gather()
|
|
55
|
+
except collect_graph.GraphError as e:
|
|
56
|
+
print(f"error: {e}", file=sys.stderr)
|
|
57
|
+
return 2
|
|
58
|
+
with open(args.out, "w", encoding="utf-8") as f:
|
|
59
|
+
f.write(data)
|
|
60
|
+
print(f"wrote {args.out}")
|
|
61
|
+
return 0
|
|
62
|
+
|
|
63
|
+
if args.cmd == "dcspm":
|
|
64
|
+
from . import collect_dcspm
|
|
65
|
+
try:
|
|
66
|
+
data = collect_dcspm.gather()
|
|
67
|
+
except collect_dcspm.DcspmError as e:
|
|
68
|
+
print(f"error: {e}", file=sys.stderr)
|
|
69
|
+
return 2
|
|
70
|
+
with open(args.out, "w", encoding="utf-8") as f:
|
|
71
|
+
f.write(data)
|
|
72
|
+
print(f"wrote {args.out}")
|
|
73
|
+
return 0
|
|
74
|
+
|
|
75
|
+
if args.cmd == "roles-report":
|
|
76
|
+
from . import rolesreport
|
|
77
|
+
tenant, identities = engine.load_inventory(args.input)
|
|
78
|
+
rows = rolesreport.build(identities)
|
|
79
|
+
if args.format == "md":
|
|
80
|
+
out = rolesreport.to_markdown(tenant, rows)
|
|
81
|
+
elif args.format == "json":
|
|
82
|
+
out = rolesreport.to_json(tenant, rows)
|
|
83
|
+
else:
|
|
84
|
+
out = rolesreport.to_csv(tenant, rows)
|
|
85
|
+
print(out)
|
|
86
|
+
return 0
|
|
87
|
+
|
|
88
|
+
return 1
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
if __name__ == "__main__":
|
|
92
|
+
sys.exit(main())
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
"""Defender CSPM (DCSPM) CIEM collector + mapping.
|
|
2
|
+
|
|
3
|
+
Microsoft retired Entra Permissions Management (ex-CloudKnox) as a standalone
|
|
4
|
+
product. Its CIEM analysis now surfaces as **Microsoft Defender CSPM
|
|
5
|
+
recommendations** that flag stale/unused and over-permissioned identities across
|
|
6
|
+
Azure (and AWS/GCP) *resource* permissions. That analysis is usage-based — it is
|
|
7
|
+
the gold-standard signal overreach's native checks only approximate (OP3 dormancy).
|
|
8
|
+
|
|
9
|
+
This module:
|
|
10
|
+
1. `gather()` — read-only pull of those recommendations from Azure Resource
|
|
11
|
+
Graph (securityresources / Microsoft.Security assessments), normalized to a
|
|
12
|
+
small JSON shape. Needs Defender CSPM enabled and an ARM token (az login).
|
|
13
|
+
2. `load_findings(path)` — read a normalized DCSPM JSON (from gather() or the
|
|
14
|
+
fixture) and map each recommendation to an overreach Finding, tagged
|
|
15
|
+
source="defender-cspm" so the report keeps it distinct from directory-role
|
|
16
|
+
findings.
|
|
17
|
+
|
|
18
|
+
Scope note: DCSPM covers resource / RBAC permissions; overreach's native checks
|
|
19
|
+
cover Entra directory roles. They are complementary — merge both for the full
|
|
20
|
+
picture across the directory plane and the resource plane.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import json
|
|
24
|
+
import subprocess
|
|
25
|
+
import urllib.error
|
|
26
|
+
import urllib.request
|
|
27
|
+
|
|
28
|
+
from .models import Finding
|
|
29
|
+
|
|
30
|
+
ARM = "https://management.azure.com"
|
|
31
|
+
|
|
32
|
+
# The Defender CSPM CIEM recommendations we care about. Match on lowercased display name.
|
|
33
|
+
_ARG_QUERY = """
|
|
34
|
+
securityresources
|
|
35
|
+
| where type == "microsoft.security/assessments"
|
|
36
|
+
| extend displayName = tostring(properties.displayName),
|
|
37
|
+
status = tostring(properties.status.code),
|
|
38
|
+
severity = tostring(properties.metadata.severity),
|
|
39
|
+
resourceId = tostring(properties.resourceDetails.Id)
|
|
40
|
+
| where status == "Unhealthy"
|
|
41
|
+
| where displayName has_any ("permission","permissions","identity","identities",
|
|
42
|
+
"unused","stale","inactive","overprovisioned",
|
|
43
|
+
"over-provisioned","super")
|
|
44
|
+
| project displayName, status, severity, resourceId,
|
|
45
|
+
additionalData = tostring(properties.additionalData)
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
_SEV_MAP = {"critical": "critical", "high": "high", "medium": "medium", "low": "low"}
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class DcspmError(RuntimeError):
|
|
52
|
+
pass
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
# --- gather (live, read-only) ------------------------------------------------
|
|
56
|
+
|
|
57
|
+
def _arm_token():
|
|
58
|
+
try:
|
|
59
|
+
out = subprocess.run(
|
|
60
|
+
["az", "account", "get-access-token", "--resource", ARM,
|
|
61
|
+
"--query", "accessToken", "-o", "tsv"],
|
|
62
|
+
capture_output=True, text=True, check=True,
|
|
63
|
+
)
|
|
64
|
+
token = out.stdout.strip()
|
|
65
|
+
if not token:
|
|
66
|
+
raise DcspmError("empty ARM token from 'az account get-access-token'")
|
|
67
|
+
return token
|
|
68
|
+
except FileNotFoundError:
|
|
69
|
+
raise DcspmError("Azure CLI ('az') not found. Install it and run 'az login' first.")
|
|
70
|
+
except subprocess.CalledProcessError as e:
|
|
71
|
+
raise DcspmError(f"'az account get-access-token' failed: {e.stderr.strip()}")
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def gather():
|
|
75
|
+
"""Return normalized DCSPM CIEM recommendations as a JSON string."""
|
|
76
|
+
token = _arm_token()
|
|
77
|
+
url = f"{ARM}/providers/Microsoft.ResourceGraph/resources?api-version=2022-10-01"
|
|
78
|
+
assessments = []
|
|
79
|
+
skip_token = None
|
|
80
|
+
while True:
|
|
81
|
+
options = {"resultFormat": "objectArray"}
|
|
82
|
+
if skip_token:
|
|
83
|
+
options["$skipToken"] = skip_token
|
|
84
|
+
body = json.dumps({"query": _ARG_QUERY, "options": options}).encode("utf-8")
|
|
85
|
+
req = urllib.request.Request(
|
|
86
|
+
url, data=body, method="POST",
|
|
87
|
+
headers={"Authorization": f"Bearer {token}", "Content-Type": "application/json"},
|
|
88
|
+
)
|
|
89
|
+
try:
|
|
90
|
+
with urllib.request.urlopen(req) as resp:
|
|
91
|
+
payload = json.loads(resp.read().decode("utf-8"))
|
|
92
|
+
except urllib.error.HTTPError as e:
|
|
93
|
+
raise DcspmError(f"Resource Graph query failed: HTTP {e.code}: "
|
|
94
|
+
f"{e.read().decode('utf-8', 'ignore')[:300]}")
|
|
95
|
+
for row in payload.get("data", []):
|
|
96
|
+
assessments.append(_normalize_row(row))
|
|
97
|
+
skip_token = payload.get("$skipToken")
|
|
98
|
+
if not skip_token:
|
|
99
|
+
break
|
|
100
|
+
return json.dumps({"source": "defender-cspm", "assessments": assessments}, indent=2)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _normalize_row(row):
|
|
104
|
+
"""Map a raw Resource Graph assessment row to the normalized shape."""
|
|
105
|
+
# The specific over-permissioned principal often lives in additionalData / sub-assessments;
|
|
106
|
+
# we surface the resource and display name, and best-effort an identity if present. TODO: expand.
|
|
107
|
+
return {
|
|
108
|
+
"display_name": row.get("displayName", ""),
|
|
109
|
+
"status": row.get("status", "Unhealthy"),
|
|
110
|
+
"severity": (row.get("severity") or "Medium"),
|
|
111
|
+
"identity": None,
|
|
112
|
+
"identity_type": None,
|
|
113
|
+
"resource_id": row.get("resourceId", ""),
|
|
114
|
+
"remediation": "",
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
# --- map normalized recommendations to findings (used offline + live) --------
|
|
119
|
+
|
|
120
|
+
def _classify(display_name):
|
|
121
|
+
d = (display_name or "").lower()
|
|
122
|
+
if "super" in d:
|
|
123
|
+
return "DCSPM-SUPER", "Super identity (excessive resource permissions)"
|
|
124
|
+
if "unused" in d:
|
|
125
|
+
return "DCSPM-UNUSED", "Unused identity (resource permissions)"
|
|
126
|
+
if "stale" in d or "inactive" in d:
|
|
127
|
+
return "DCSPM-STALE", "Stale / inactive identity"
|
|
128
|
+
if "overprovisioned" in d or "over-provisioned" in d or "necessary permissions" in d:
|
|
129
|
+
return "DCSPM-OVERPROV", "Over-provisioned identity (resource permissions)"
|
|
130
|
+
return "DCSPM", "Defender CSPM identity recommendation"
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _to_finding(rec):
|
|
134
|
+
rule, title = _classify(rec.get("display_name", ""))
|
|
135
|
+
sev = _SEV_MAP.get(str(rec.get("severity", "medium")).lower(), "medium")
|
|
136
|
+
who = rec.get("identity") or rec.get("resource_id") or "(resource)"
|
|
137
|
+
where = rec.get("resource_id", "")
|
|
138
|
+
scope = f" on {where}" if where and rec.get("identity") else ""
|
|
139
|
+
return Finding(
|
|
140
|
+
rule=rule, title=title,
|
|
141
|
+
identity_id=rec.get("identity") or rec.get("resource_id") or "(resource)",
|
|
142
|
+
identity_upn=who,
|
|
143
|
+
severity=sev, confidence="high", amplifier=False,
|
|
144
|
+
evidence=f"Defender CSPM (usage-based): '{rec.get('display_name', '')}'{scope}.",
|
|
145
|
+
remediation=rec.get("remediation")
|
|
146
|
+
or "Right-size the resource role assignments to the permissions actually used "
|
|
147
|
+
"(per the Defender CSPM recommendation).",
|
|
148
|
+
source="defender-cspm",
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def findings_from_normalized(data):
|
|
153
|
+
"""Map a normalized DCSPM dict (as produced by gather()) to a list of Findings."""
|
|
154
|
+
return [_to_finding(rec) for rec in data.get("assessments", [])]
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def load_findings(path):
|
|
158
|
+
"""Read a normalized DCSPM JSON file and return a list of Findings."""
|
|
159
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
160
|
+
data = json.load(f)
|
|
161
|
+
return findings_from_normalized(data)
|