websec-validator 0.6.3__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {websec_validator-0.6.3/src/websec_validator.egg-info → websec_validator-0.8.0}/PKG-INFO +22 -12
- {websec_validator-0.6.3 → websec_validator-0.8.0}/README.md +21 -11
- {websec_validator-0.6.3 → websec_validator-0.8.0}/pyproject.toml +1 -1
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/briefing.py +13 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/__init__.py +6 -0
- websec_validator-0.8.0/src/websec_validator/extractors/authz.py +391 -0
- websec_validator-0.8.0/src/websec_validator/extractors/authz_dataflow.py +110 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/base.py +37 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/client_exposure.py +54 -7
- websec_validator-0.8.0/src/websec_validator/extractors/crypto_usage.py +107 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/iac_ci.py +106 -4
- websec_validator-0.8.0/src/websec_validator/extractors/llm_security.py +175 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/pii_exposure.py +27 -13
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/routes.py +1 -1
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/surface.py +82 -5
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/transport_security.py +75 -3
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/upload_security.py +15 -6
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/findings.py +148 -4
- {websec_validator-0.6.3 → websec_validator-0.8.0/src/websec_validator.egg-info}/PKG-INFO +22 -12
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator.egg-info/SOURCES.txt +3 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/tests/test_recon.py +449 -0
- websec_validator-0.6.3/src/websec_validator/extractors/authz.py +0 -196
- {websec_validator-0.6.3 → websec_validator-0.8.0}/LICENSE +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/setup.cfg +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/__init__.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/calibration.json +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/calibration.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/cli.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/constitution.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/corpus.json +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/dynamic.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/auth.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/client_integrity.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/graphql.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/integrations.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/policy_consistency.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/schemas.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/stack.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/tenant.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/probes.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/proof.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/recon.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/report.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/rules/error-stack-disclosure.yml +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/rules/insecure-default-secret.yml +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/scanners.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/_lib.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/appsync-cswsh.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/appsync-introspection.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/appsync-subscription-bola.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/bola-cross-tenant.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/bola-write-verbs.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/client-integrity-checklist.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/compare-roles.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/dlp-bypass-offline.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/error-disclosure-probe.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/forged-token.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/hs256-brute-force.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/jwt-attacks.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/mass-assignment.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/password-reuse.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/pii-output-diff.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/race-conditions.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/rate-limit-burst.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/s3-assess.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/ssrf-probes.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/unauth-baseline.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/upload-matrix.sh +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/probes/webhook-forgery.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/reports/FINDINGS-SUMMARY.md.template +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/reports/access-control-matrix.md.template +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/reports/findings-triage.md.template +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/reports/pentest-handover-brief.md.template +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/templates/reports/per-tool-FINDINGS.md.template +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator.egg-info/dependency_links.txt +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator.egg-info/entry_points.txt +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator.egg-info/top_level.txt +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/tests/test_hardening.py +0 -0
- {websec_validator-0.6.3 → websec_validator-0.8.0}/tests/test_pentest_regressions.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: websec-validator
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: Defensive, local-first security recon that briefs your AI coding agent on your own codebase — read-only by default (code in, artifacts out): facts + tailored probe scripts, no LLM, no server, no running app.
|
|
5
5
|
Author: Ricardo Accioly
|
|
6
6
|
License: MIT
|
|
@@ -89,26 +89,29 @@ Then point your agent at the output: **"Read `websec-out/AGENT-BRIEFING.md` and
|
|
|
89
89
|
|
|
90
90
|
> That's the whole user surface: **`run`** (plus the optional, advanced **`dynamic`** live-probing step below). `recon`/`proof`/`calibrate` exist for developing the tool itself and are hidden from `--help` — you never need them.
|
|
91
91
|
|
|
92
|
-
## What it extracts (
|
|
92
|
+
## What it extracts (19 deterministic extractors, no LLM)
|
|
93
93
|
|
|
94
94
|
| | Dimension | Notable output |
|
|
95
95
|
|---|---|---|
|
|
96
96
|
| stack | languages, frameworks, datastores | monorepo-aware (aggregates every manifest) |
|
|
97
97
|
| routes | every endpoint via **OWASP Noir** | method · path · typed params · code path |
|
|
98
98
|
| auth | scheme + login surface + **insecure-default signing secrets** | multi-scheme; flags a hard-coded `JWT_SECRET \|\| 'dev-secret'` fallback (forgeable JWT) |
|
|
99
|
-
| **authz** | access-control map | guard coverage + **write endpoints with no visible guard** + roles |
|
|
99
|
+
| **authz** | access-control map | guard coverage (incl. **router-mount auth**) + **write endpoints with no visible guard** + roles |
|
|
100
|
+
| **authz_dataflow** | authz *correctness* (does the guard trust the right thing?) | **unsigned-cookie authorization** · **claim-keyed authz** (user-influenceable JWT claim) · **transaction-local RLS context** (resets before the query) |
|
|
100
101
|
| tenant | multi-tenancy key candidates | the BOLA boundary, by frequency |
|
|
101
102
|
| **password_policy** | cross-route consistency **+ reuse/history** | complexity drift across routes **+ a set-password path that hashes without a reuse check** |
|
|
102
|
-
| surface | 15 sink classes **+ redirect-SSRF** | user-input-gated sinks (incl. **mass-assignment via object spread**) + var-arg SSRF + error-disclosure
|
|
103
|
+
| surface | 15 sink classes **+ redirect-SSRF** | user-input-gated sinks (incl. **mass-assignment via object spread**) + var-arg SSRF + error-disclosure + follows-redirects-without-per-hop-guard **+ reverse-proxy prefix-escape + host-header open-redirect + SSRF-redirect-hardening** |
|
|
103
104
|
| **upload_security** | unrestricted upload + unsafe serve | deny-list-only, stored-name-from-filename, trust-client-MIME, accept-SVG, **serve without `nosniff`** |
|
|
104
105
|
| schemas | data models + **privileged fields** | Pydantic/SQLAlchemy/Django/Prisma/Mongoose/TypeORM/Zod → `role`/`isAdmin`/`groupId` for mass-assignment targeting |
|
|
105
|
-
| iac_ci | IaC + CI/CD | GHA injection, unpinned actions, tfstate,
|
|
106
|
+
| iac_ci | IaC + CI/CD | GHA injection (**run:-position-aware**), unpinned actions, tfstate, CDK AppSync `API_KEY` anonymous-default-auth, **docker-compose host-takeover (docker.sock / pid:host / privileged) + `.gitleaksignore` secret-suppression audit** |
|
|
106
107
|
| client_exposure | browser leakage | public-var secrets by **name + value-shape (`da2-…`) + CDK build-injection**, server-secret-in-client, source maps |
|
|
107
108
|
| **client_integrity** | tamperable display (client trust boundary) + **WS auth model** | any security-critical sink value (address/IBAN/2FA-seed/API-key/webhook) the user reads or copies, without strict CSP / out-of-band anchor **+ client-tamper-vector, grindable-fingerprint, over-claimed-control, the CSWSH determinant** |
|
|
108
|
-
| **transport_security** | CSP + HSTS
|
|
109
|
+
| **transport_security** | CSP + HSTS + **CORS** + **SRI** baseline | missing/weak CSP, inline event handlers, partial HSTS, **CORS reflect-origin+credentials, external script without SRI, monorepo `next.config` header gap** |
|
|
109
110
|
| **pii_exposure** | unmasked PII at the output boundary | `res.json(rawEntity)` with PII + **a masking control defined but with zero live call sites** (value-shape, not field-name) |
|
|
110
111
|
| graphql | GraphQL surface | introspection (**AppSync `introspectionConfig: DISABLED`-aware**) / playground / depth-limit **+ AppSync subscription-authz (cross-group BOLA)** |
|
|
111
112
|
| integrations | third-party + webhooks **+ outbound-action endpoints** | unsigned webhooks **+ email/SMS/push handlers with no auth or IP-only rate-limit + redundant secret-fetch** |
|
|
113
|
+
| **llm_security** | LLM / AI-agent surface (**OWASP LLM Top 10**) | indirect **prompt injection** (untrusted RAG/tool content → prompt) · **insecure output handling** (model text → tool dispatch) · **excessive agency** · **unbounded generation** (no maxTokens/timeout) · **guardrail fail-open** |
|
|
114
|
+
| **crypto_usage** | crypto-API correctness | **weak password hash** (fast/unsalted SHA-256/MD5) · `jwtVerify` without an `algorithms` allowlist · **predictable principal** (id = hash of email) · non-constant-time secret compare |
|
|
112
115
|
|
|
113
116
|
Plus **derived targeting** — IDOR / SSRF / open-redirect / upload / write / auth-endpoint
|
|
114
117
|
candidates — so probes get pointed at the *exact* endpoints, not fired blindly.
|
|
@@ -216,15 +219,22 @@ publisher** with project `websec-validator`, owner `raccioly`, repo `websec-vali
|
|
|
216
219
|
|
|
217
220
|
## Status / roadmap
|
|
218
221
|
|
|
219
|
-
**Done:**
|
|
222
|
+
**Done:** 19-extractor recon (incl. an **authz-correctness data-flow extractor** — unsigned-cookie /
|
|
223
|
+
claim-keyed authz / transaction-local RLS — plus **CORS-misconfig**, **SRI**, **host-header
|
|
224
|
+
open-redirect** and **SSRF-redirect-hardening** classes, schema/entity → mass-assignment targeting, the **AWS-CDK /
|
|
220
225
|
managed-AppSync / VTL boundary**, **upload-security** + **PII-output-boundary** + **redirect-SSRF**
|
|
221
|
-
+ **password-reuse** classes,
|
|
222
|
-
|
|
223
|
-
|
|
226
|
+
+ **password-reuse** classes, a **man-in-the-browser / tamperable-display** class, an **LLM / AI-agent
|
|
227
|
+
extractor** (OWASP LLM Top 10 — prompt injection / insecure output / excessive agency / unbounded
|
|
228
|
+
generation / guardrail fail-open), a **crypto-usage extractor** (weak password hash / jwtVerify-without-
|
|
229
|
+
algorithms / predictable principal), **docker-compose host-takeover** + **`.gitleaksignore`
|
|
230
|
+
secret-suppression** audits, and a **reverse-proxy prefix-escape** detector), cross-tool de-dup +
|
|
231
|
+
**bundled Semgrep rules**, **router-mount-auth modeling** (cuts the dominant Express-monorepo
|
|
232
|
+
missing-auth false positive), tailored probe staging, agent briefing, traceable findings ledger with
|
|
233
|
+
**calibrated confidence (CJE — Wilson CIs)**, proof harness, test suite (181), **Docker bundle** (all
|
|
224
234
|
scanners + Noir, arch-aware), **dynamic phase v1** (authenticated read-only cross-tenant BOLA —
|
|
225
235
|
validated live, reproduced a hand-pentest's 14/14). Validated against the **REF-PENTEST pen test +
|
|
226
|
-
retest**
|
|
227
|
-
|
|
236
|
+
retest** and re-validated on a large real-world LLM-agent monorepo (HIGH-finding noise 178 → 15, AI +
|
|
237
|
+
crypto surfaces newly covered).
|
|
228
238
|
**Next:** dynamic write-verb BOLA + JWT/auth probes + ZAP/Nuclei two-role diff (gated, they mutate),
|
|
229
239
|
calibration on hand-labeled real repos (more representative base rate), ASVS index lookup, optional
|
|
230
240
|
model-SDK adapters for no-agent fallback.
|
|
@@ -77,26 +77,29 @@ Then point your agent at the output: **"Read `websec-out/AGENT-BRIEFING.md` and
|
|
|
77
77
|
|
|
78
78
|
> That's the whole user surface: **`run`** (plus the optional, advanced **`dynamic`** live-probing step below). `recon`/`proof`/`calibrate` exist for developing the tool itself and are hidden from `--help` — you never need them.
|
|
79
79
|
|
|
80
|
-
## What it extracts (
|
|
80
|
+
## What it extracts (19 deterministic extractors, no LLM)
|
|
81
81
|
|
|
82
82
|
| | Dimension | Notable output |
|
|
83
83
|
|---|---|---|
|
|
84
84
|
| stack | languages, frameworks, datastores | monorepo-aware (aggregates every manifest) |
|
|
85
85
|
| routes | every endpoint via **OWASP Noir** | method · path · typed params · code path |
|
|
86
86
|
| auth | scheme + login surface + **insecure-default signing secrets** | multi-scheme; flags a hard-coded `JWT_SECRET \|\| 'dev-secret'` fallback (forgeable JWT) |
|
|
87
|
-
| **authz** | access-control map | guard coverage + **write endpoints with no visible guard** + roles |
|
|
87
|
+
| **authz** | access-control map | guard coverage (incl. **router-mount auth**) + **write endpoints with no visible guard** + roles |
|
|
88
|
+
| **authz_dataflow** | authz *correctness* (does the guard trust the right thing?) | **unsigned-cookie authorization** · **claim-keyed authz** (user-influenceable JWT claim) · **transaction-local RLS context** (resets before the query) |
|
|
88
89
|
| tenant | multi-tenancy key candidates | the BOLA boundary, by frequency |
|
|
89
90
|
| **password_policy** | cross-route consistency **+ reuse/history** | complexity drift across routes **+ a set-password path that hashes without a reuse check** |
|
|
90
|
-
| surface | 15 sink classes **+ redirect-SSRF** | user-input-gated sinks (incl. **mass-assignment via object spread**) + var-arg SSRF + error-disclosure
|
|
91
|
+
| surface | 15 sink classes **+ redirect-SSRF** | user-input-gated sinks (incl. **mass-assignment via object spread**) + var-arg SSRF + error-disclosure + follows-redirects-without-per-hop-guard **+ reverse-proxy prefix-escape + host-header open-redirect + SSRF-redirect-hardening** |
|
|
91
92
|
| **upload_security** | unrestricted upload + unsafe serve | deny-list-only, stored-name-from-filename, trust-client-MIME, accept-SVG, **serve without `nosniff`** |
|
|
92
93
|
| schemas | data models + **privileged fields** | Pydantic/SQLAlchemy/Django/Prisma/Mongoose/TypeORM/Zod → `role`/`isAdmin`/`groupId` for mass-assignment targeting |
|
|
93
|
-
| iac_ci | IaC + CI/CD | GHA injection, unpinned actions, tfstate,
|
|
94
|
+
| iac_ci | IaC + CI/CD | GHA injection (**run:-position-aware**), unpinned actions, tfstate, CDK AppSync `API_KEY` anonymous-default-auth, **docker-compose host-takeover (docker.sock / pid:host / privileged) + `.gitleaksignore` secret-suppression audit** |
|
|
94
95
|
| client_exposure | browser leakage | public-var secrets by **name + value-shape (`da2-…`) + CDK build-injection**, server-secret-in-client, source maps |
|
|
95
96
|
| **client_integrity** | tamperable display (client trust boundary) + **WS auth model** | any security-critical sink value (address/IBAN/2FA-seed/API-key/webhook) the user reads or copies, without strict CSP / out-of-band anchor **+ client-tamper-vector, grindable-fingerprint, over-claimed-control, the CSWSH determinant** |
|
|
96
|
-
| **transport_security** | CSP + HSTS
|
|
97
|
+
| **transport_security** | CSP + HSTS + **CORS** + **SRI** baseline | missing/weak CSP, inline event handlers, partial HSTS, **CORS reflect-origin+credentials, external script without SRI, monorepo `next.config` header gap** |
|
|
97
98
|
| **pii_exposure** | unmasked PII at the output boundary | `res.json(rawEntity)` with PII + **a masking control defined but with zero live call sites** (value-shape, not field-name) |
|
|
98
99
|
| graphql | GraphQL surface | introspection (**AppSync `introspectionConfig: DISABLED`-aware**) / playground / depth-limit **+ AppSync subscription-authz (cross-group BOLA)** |
|
|
99
100
|
| integrations | third-party + webhooks **+ outbound-action endpoints** | unsigned webhooks **+ email/SMS/push handlers with no auth or IP-only rate-limit + redundant secret-fetch** |
|
|
101
|
+
| **llm_security** | LLM / AI-agent surface (**OWASP LLM Top 10**) | indirect **prompt injection** (untrusted RAG/tool content → prompt) · **insecure output handling** (model text → tool dispatch) · **excessive agency** · **unbounded generation** (no maxTokens/timeout) · **guardrail fail-open** |
|
|
102
|
+
| **crypto_usage** | crypto-API correctness | **weak password hash** (fast/unsalted SHA-256/MD5) · `jwtVerify` without an `algorithms` allowlist · **predictable principal** (id = hash of email) · non-constant-time secret compare |
|
|
100
103
|
|
|
101
104
|
Plus **derived targeting** — IDOR / SSRF / open-redirect / upload / write / auth-endpoint
|
|
102
105
|
candidates — so probes get pointed at the *exact* endpoints, not fired blindly.
|
|
@@ -204,15 +207,22 @@ publisher** with project `websec-validator`, owner `raccioly`, repo `websec-vali
|
|
|
204
207
|
|
|
205
208
|
## Status / roadmap
|
|
206
209
|
|
|
207
|
-
**Done:**
|
|
210
|
+
**Done:** 19-extractor recon (incl. an **authz-correctness data-flow extractor** — unsigned-cookie /
|
|
211
|
+
claim-keyed authz / transaction-local RLS — plus **CORS-misconfig**, **SRI**, **host-header
|
|
212
|
+
open-redirect** and **SSRF-redirect-hardening** classes, schema/entity → mass-assignment targeting, the **AWS-CDK /
|
|
208
213
|
managed-AppSync / VTL boundary**, **upload-security** + **PII-output-boundary** + **redirect-SSRF**
|
|
209
|
-
+ **password-reuse** classes,
|
|
210
|
-
|
|
211
|
-
|
|
214
|
+
+ **password-reuse** classes, a **man-in-the-browser / tamperable-display** class, an **LLM / AI-agent
|
|
215
|
+
extractor** (OWASP LLM Top 10 — prompt injection / insecure output / excessive agency / unbounded
|
|
216
|
+
generation / guardrail fail-open), a **crypto-usage extractor** (weak password hash / jwtVerify-without-
|
|
217
|
+
algorithms / predictable principal), **docker-compose host-takeover** + **`.gitleaksignore`
|
|
218
|
+
secret-suppression** audits, and a **reverse-proxy prefix-escape** detector), cross-tool de-dup +
|
|
219
|
+
**bundled Semgrep rules**, **router-mount-auth modeling** (cuts the dominant Express-monorepo
|
|
220
|
+
missing-auth false positive), tailored probe staging, agent briefing, traceable findings ledger with
|
|
221
|
+
**calibrated confidence (CJE — Wilson CIs)**, proof harness, test suite (181), **Docker bundle** (all
|
|
212
222
|
scanners + Noir, arch-aware), **dynamic phase v1** (authenticated read-only cross-tenant BOLA —
|
|
213
223
|
validated live, reproduced a hand-pentest's 14/14). Validated against the **REF-PENTEST pen test +
|
|
214
|
-
retest**
|
|
215
|
-
|
|
224
|
+
retest** and re-validated on a large real-world LLM-agent monorepo (HIGH-finding noise 178 → 15, AI +
|
|
225
|
+
crypto surfaces newly covered).
|
|
216
226
|
**Next:** dynamic write-verb BOLA + JWT/auth probes + ZAP/Nuclei two-role diff (gated, they mutate),
|
|
217
227
|
calibration on hand-labeled real repos (more representative base rate), ASVS index lookup, optional
|
|
218
228
|
model-SDK adapters for no-agent fallback.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "websec-validator"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.8.0"
|
|
8
8
|
description = "Defensive, local-first security recon that briefs your AI coding agent on your own codebase — read-only by default (code in, artifacts out): facts + tailored probe scripts, no LLM, no server, no running app."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.11"
|
|
@@ -82,6 +82,16 @@ def render(facts: dict, scanners: dict, scan_results: list, probe_manifest: list
|
|
|
82
82
|
pii_findings = pii.get("findings", [])
|
|
83
83
|
pii_section = ("\n".join(f"- **{f.get('severity')}** {f.get('kind')} — `{f.get('file')}`" for f in pii_findings[:20])
|
|
84
84
|
if pii_findings else "_no obvious raw-PII responses / dead masking controls_")
|
|
85
|
+
llm = facts.get("llm_security", {})
|
|
86
|
+
llm_findings = llm.get("findings", [])
|
|
87
|
+
if llm.get("is_ai_app"):
|
|
88
|
+
llm_section = ("\n".join(f"- **{f.get('severity')}** {f.get('kind')} — `{f.get('file')}`" for f in llm_findings[:25])
|
|
89
|
+
if llm_findings else "_LLM call sites present; no obvious prompt-injection / unbounded / "
|
|
90
|
+
"insecure-output tells — verify the agentic surface by hand_")
|
|
91
|
+
if len(llm_findings) > 25:
|
|
92
|
+
llm_section += f"\n- _…and {len(llm_findings) - 25} more (see FACTS.json)_"
|
|
93
|
+
else:
|
|
94
|
+
llm_section = "_no direct LLM SDK call sites detected_"
|
|
85
95
|
ws_line = (facts.get("client_integrity", {}) or {}).get("websocket_auth", "no websocket detected")
|
|
86
96
|
_cs = (facts.get("transport_security", {}) or {}).get("cookie_security")
|
|
87
97
|
if _cs:
|
|
@@ -231,6 +241,9 @@ Production source maps exposed: {client.get("production_source_maps", False)}
|
|
|
231
241
|
**Third-party integrations:** {integ_line}
|
|
232
242
|
{wh_line}
|
|
233
243
|
|
|
244
|
+
**LLM / AI-agent surface (OWASP LLM Top 10 — prompt injection, insecure output, excessive agency, unbounded):**
|
|
245
|
+
{llm_section}
|
|
246
|
+
|
|
234
247
|
## 4. Static findings (no running app needed)
|
|
235
248
|
|
|
236
249
|
Scanners available: {avail}
|
{websec_validator-0.6.3 → websec_validator-0.8.0}/src/websec_validator/extractors/__init__.py
RENAMED
|
@@ -11,12 +11,15 @@ from pathlib import Path
|
|
|
11
11
|
|
|
12
12
|
from .auth import AuthExtractor
|
|
13
13
|
from .authz import AuthzExtractor
|
|
14
|
+
from .authz_dataflow import AuthzDataflowExtractor
|
|
14
15
|
from .base import MAX_FILES, Extractor, RepoContext
|
|
15
16
|
from .client_exposure import ClientExposureExtractor
|
|
16
17
|
from .client_integrity import ClientIntegrityExtractor
|
|
18
|
+
from .crypto_usage import CryptoUsageExtractor
|
|
17
19
|
from .graphql import GraphQLExtractor
|
|
18
20
|
from .iac_ci import IacCiExtractor
|
|
19
21
|
from .integrations import IntegrationsExtractor
|
|
22
|
+
from .llm_security import LlmSecurityExtractor
|
|
20
23
|
from .pii_exposure import PiiExposureExtractor
|
|
21
24
|
from .policy_consistency import PolicyConsistencyExtractor
|
|
22
25
|
from .routes import RoutesExtractor
|
|
@@ -46,6 +49,9 @@ REGISTRY: list[Extractor] = [
|
|
|
46
49
|
PiiExposureExtractor(),
|
|
47
50
|
GraphQLExtractor(),
|
|
48
51
|
IntegrationsExtractor(),
|
|
52
|
+
LlmSecurityExtractor(),
|
|
53
|
+
CryptoUsageExtractor(),
|
|
54
|
+
AuthzDataflowExtractor(),
|
|
49
55
|
]
|
|
50
56
|
|
|
51
57
|
|
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
"""Authorization extractor — the access-control map (who can reach what).
|
|
2
|
+
|
|
3
|
+
Per your methodology this is the highest-value test. For each endpoint we decide
|
|
4
|
+
whether a guard protects it, using several signals:
|
|
5
|
+
1. a guard pattern in the handler's own file (incl. `router.use(authenticate)`)
|
|
6
|
+
or a project-specific auth helper (a `getRequest*Auth`-style getter),
|
|
7
|
+
2. coverage by a Next.js middleware matcher (incl. monorepo `proxy.ts`),
|
|
8
|
+
3. a GLOBAL auth middleware (`app.use(authenticate)`) — when present, routes are
|
|
9
|
+
protected by default and "no visible guard" becomes a *verify* signal,
|
|
10
|
+
4. ROUTER-MOUNT auth: `app.use('/prefix', authMiddleware(...), createXRouter())` —
|
|
11
|
+
resolve the mounted router factory to its file and BFS the local-import graph to
|
|
12
|
+
mark every composed sub-router guarded (this is what inflated the count on the
|
|
13
|
+
Express monorepo: auth lives at the mount, not in the handler file),
|
|
14
|
+
5. a one-hop delegated guard for thin Next route handlers (`route.ts` → `./proxy`).
|
|
15
|
+
|
|
16
|
+
File-level heuristic → results are HINTS the agent confirms. The high-signal
|
|
17
|
+
output is write endpoints with no visible guard that also don't look public.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import posixpath
|
|
23
|
+
import re
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
|
|
26
|
+
from .base import Extractor, RepoContext, is_client_file, is_test_file
|
|
27
|
+
|
|
28
|
+
WRITE_VERBS = {"POST", "PUT", "PATCH", "DELETE"}
|
|
29
|
+
|
|
30
|
+
# endpoint_guards feeds the missing-auth ledger (findings.build_ledger), so capping it low was a
|
|
31
|
+
# silent coverage cliff: a big monorepo's unguarded write #401 never became a finding. Raised to
|
|
32
|
+
# cover realistic monorepos; truncation beyond this is DISCLOSED (endpoint_guards_truncated), never
|
|
33
|
+
# silent — mirrors constitution.py's "…and N more" pattern.
|
|
34
|
+
_MAX_ENDPOINT_GUARDS = 5000
|
|
35
|
+
|
|
36
|
+
GUARD = re.compile(
|
|
37
|
+
r"requireAuth|requirePermission|requireRole|requireGroupAccess|isAuthenticated|"
|
|
38
|
+
r"@login_required|@jwt_required|@permission_required|@roles_required|ensureAuth|"
|
|
39
|
+
r"withAuth|getServerSession|getToken\s*\(|verifyToken|authMiddleware|@UseGuards|"
|
|
40
|
+
r"@Roles\b|Depends\s*\(\s*(?:get_current_user|oauth2_scheme|require_)|Security\s*\(|"
|
|
41
|
+
r"PermissionRequired|LoginRequired|passport\.authenticate|"
|
|
42
|
+
r"\.use\s*\(\s*[\w.]*(?:[Aa]uth|[Vv]erifyToken|[Rr]equire|[Gg]uard|jwt)\w*", re.I)
|
|
43
|
+
|
|
44
|
+
# a global, path-less auth middleware → everything downstream is protected by default
|
|
45
|
+
GLOBAL_AUTH = re.compile(
|
|
46
|
+
r"app\.use\s*\(\s*[\w.]*(?:authenticate|requireAuth|authMiddleware|verifyToken|"
|
|
47
|
+
r"isAuthenticated|jwtMiddleware|ensureAuth)\w*\s*\)", re.I)
|
|
48
|
+
|
|
49
|
+
# Does a Next.js middleware/proxy file actually enforce AUTH (vs. i18n/headers only)?
|
|
50
|
+
# `auth((req)=>…)` / `withAuth` / `req.auth` / getToken / getServerSession / redirect-to-login /
|
|
51
|
+
# a 401 / Clerk / Supabase updateSession all signal a global auth gate.
|
|
52
|
+
MW_AUTH = re.compile(
|
|
53
|
+
r"\bauth\s*\(|withAuth\b|req\.auth\b|getToken\s*\(|getServerSession\s*\(|clerkMiddleware|"
|
|
54
|
+
r"updateSession\s*\(|NextResponse\.redirect\([^)]*(?:login|signin)|status:\s*401|"
|
|
55
|
+
r"['\"]Authentication required['\"]", re.I)
|
|
56
|
+
|
|
57
|
+
PUBLIC_HINT = re.compile(
|
|
58
|
+
r"/(login|logout|register|signup|signin|health|healthz|ping|status|webhooks?|"
|
|
59
|
+
r"public|\.well-known|robots|favicon|sitemap|callback|refresh|csrf|metrics)\b", re.I)
|
|
60
|
+
|
|
61
|
+
ROLE = re.compile(
|
|
62
|
+
r"@Roles\s*\(([^)]*)\)|allowedRoles\s*=\s*\[([^\]]*)\]|"
|
|
63
|
+
r"\b(?:role|roles)\b\s*[!=]==?\s*['\"]([\w:.-]+)['\"]|"
|
|
64
|
+
r"has_?[Rr]ole\s*\(\s*['\"]([\w:.-]+)['\"]|"
|
|
65
|
+
r"authorizeRoles\s*\(([^)]*)\)|permission_required\s*\(\s*['\"]([\w:.-]+)['\"]")
|
|
66
|
+
|
|
67
|
+
# F5: a call to a decoder/parser named "unsafe"/"unverified"/"noVerify"/"skipVerify"
|
|
68
|
+
# (e.g. decodeJwtPayloadUnsafe) — dangerous when its result feeds an auth decision.
|
|
69
|
+
UNSAFE_DECODER = re.compile(r"\b([A-Za-z_]\w*(?:[Uu]nsafe|[Uu]nverified|[Nn]o[Vv]erif\w*|[Ss]kip[Vv]erif\w*)\w*)\s*\(")
|
|
70
|
+
# does this file actually make an auth/identity decision? (so the unsafe decode matters)
|
|
71
|
+
AUTH_CONTEXT = re.compile(
|
|
72
|
+
r"require(?:Auth|Admin|Role|Permission)|isAdmin|authoriz|getToken\s*\(|getServerSession|"
|
|
73
|
+
r"req\.auth\b|currentUser|jwt\.(?:decode|verify)|decodeJwt", re.I)
|
|
74
|
+
|
|
75
|
+
# --- Router-mount auth (the dominant Express monorepo false-positive, validated on a real LLM-agent monorepo)
|
|
76
|
+
# Auth is frequently applied at the MOUNT, not in the handler: `app.use('/api/x', apiAuth, ...,
|
|
77
|
+
# createApiRouter(db))`. The handler files (routes/feature/*.ts) then carry no in-file guard, so the
|
|
78
|
+
# old GLOBAL_AUTH (which only matched a PATH-LESS `app.use(authMiddleware)`) saw nothing and every
|
|
79
|
+
# endpoint was flagged. We instead: (1) find each `.use(...)` mount, (2) decide if its arg list
|
|
80
|
+
# carries an auth middleware, (3) collect the router FACTORIES it mounts, (4) resolve each factory to
|
|
81
|
+
# its defining file, and (5) BFS the local-import graph from there to mark every route-bearing file
|
|
82
|
+
# composed under an authed mount as guarded — without crossing into a separately-UNAUTHed mount.
|
|
83
|
+
USE_CALL = re.compile(r"(\b[\w.]+)\.use\s*\(")
|
|
84
|
+
# Only a TOP-LEVEL app mount (`app.use`/`server.use`) without auth establishes an UNAUTHED router.
|
|
85
|
+
# An inner `router.use('/sub', createSubRouter())` legitimately omits auth because the parent prefix
|
|
86
|
+
# already applied it (Express runs parent middleware before the sub-router) — counting those as
|
|
87
|
+
# unauthed wrongly excluded inherited sub-routers (routes/feature/sub.ts) from coverage.
|
|
88
|
+
APP_RECEIVER = re.compile(r"(?:^|\.)(?:app|server|application|httpServer|expressApp|api)$", re.I)
|
|
89
|
+
# an auth middleware appearing in a mount arg list (a local guard const like `apiAuth`, a factory
|
|
90
|
+
# like `authMiddleware({...})`, or a named guard). Deliberately NOT matching `authenticatedWriteRateLimit`.
|
|
91
|
+
MOUNT_AUTH = re.compile(
|
|
92
|
+
r"\bauthMiddleware\b|\brequireAuth\b|\brequireAdmin\b|\brequireRole\b|\brequirePermission\b|"
|
|
93
|
+
r"\brequireTenantContext\b|\bensureAuth\w*|\bisAuthenticated\b|\bverifyToken\b|\bjwtMiddleware\b|"
|
|
94
|
+
r"\bauthenticate\b|passport\.authenticate|\b[A-Za-z_]\w*Auth\b|\bauth\b\s*[,)]", re.I)
|
|
95
|
+
FACTORY_CALL = re.compile(r"\b((?:create|make|build|register|mount|init|setup|use)\w*(?:Router|Routes)|\w+Router)\s*\(")
|
|
96
|
+
FACTORY_DEF = re.compile(
|
|
97
|
+
r"(?:export\s+)?(?:async\s+)?function\s+((?:create|make|build|register|mount|init|setup)\w*(?:Router|Routes)|\w+Router)\b"
|
|
98
|
+
r"|(?:export\s+)?(?:const|let|var)\s+((?:create|make|build|register|mount|init|setup)\w*(?:Router|Routes)|\w+Router)\s*[:=]")
|
|
99
|
+
IMPORT_REL = re.compile(r"""(?:from\s*|require\s*\(\s*|import\s*\(\s*)['"](\.[^'"]+)['"]""")
|
|
100
|
+
ROUTE_MARK = re.compile(
|
|
101
|
+
r"\.(?:get|post|put|patch|delete|all|options|head)\s*\(|\.route\s*\(|\brouter\b|FastifyInstance"
|
|
102
|
+
r"|@(?:Get|Post|Put|Patch|Delete|Controller)\(", re.I)
|
|
103
|
+
# Project-specific auth helpers the in-file GUARD list misses — a getter that returns an auth state
|
|
104
|
+
# (e.g. Next's `getRequestSessionAuth()` → fail-closed) or a named auth-failure response. Required
|
|
105
|
+
# to clear the packages/web proxy handlers that ARE authenticated. Conservative: name must read as an
|
|
106
|
+
# auth/token/session getter or an explicit auth-failure helper, so a benign util isn't mistaken for one.
|
|
107
|
+
CUSTOM_GUARD = re.compile(
|
|
108
|
+
r"\bget(?:Request)?\w*(?:Auth|Token|Session)\w*\s*\(|\brequire\w*(?:Auth|Token|Session)\w*\s*\("
|
|
109
|
+
r"|\bensure\w*(?:Auth|Session)\w*\s*\(|\bassert\w*(?:Auth|Session)\w*\s*\("
|
|
110
|
+
r"|[A-Za-z]\w*Auth(?:Failure|Required)Response\b", re.I)
|
|
111
|
+
_IMPORT_EXTS = (".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs")
|
|
112
|
+
NEXT_ROUTE = re.compile(r"(?:^|/)route\.[cm]?[jt]sx?$|(?:^|/)api/.*\.[cm]?[jt]sx?$", re.I)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _call_args(text: str, paren_idx: int) -> str:
|
|
116
|
+
"""Return the substring inside the balanced parens of a call whose '(' is at paren_idx."""
|
|
117
|
+
depth, out, i, n = 0, [], paren_idx, len(text)
|
|
118
|
+
while i < n and len(out) < 4000:
|
|
119
|
+
ch = text[i]
|
|
120
|
+
if ch == "(":
|
|
121
|
+
depth += 1
|
|
122
|
+
if depth == 1:
|
|
123
|
+
i += 1
|
|
124
|
+
continue
|
|
125
|
+
elif ch == ")":
|
|
126
|
+
depth -= 1
|
|
127
|
+
if depth == 0:
|
|
128
|
+
return "".join(out)
|
|
129
|
+
if depth >= 1:
|
|
130
|
+
out.append(ch)
|
|
131
|
+
i += 1
|
|
132
|
+
return "".join(out)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _resolve_import(importer_rel: str, spec: str, rel_set: set) -> "str | None":
|
|
136
|
+
base = posixpath.normpath(posixpath.join(posixpath.dirname(importer_rel), spec))
|
|
137
|
+
cands = [base]
|
|
138
|
+
# TS ESM convention: the import writes a `.js` extension but the file on disk is `.ts`/`.tsx`
|
|
139
|
+
# (`import './sub/feature.js'` → `routes/sub/feature.ts`). Without this the import graph
|
|
140
|
+
# dead-ends at the first hop and mount-auth coverage never reaches the sub-routers.
|
|
141
|
+
m = re.match(r"(.*)\.(?:js|jsx|mjs|cjs)$", base)
|
|
142
|
+
if m:
|
|
143
|
+
cands += [m.group(1) + e for e in (".ts", ".tsx", ".mts", ".cts")] + [m.group(1)]
|
|
144
|
+
for c in cands:
|
|
145
|
+
if c in rel_set:
|
|
146
|
+
return c
|
|
147
|
+
for ext in _IMPORT_EXTS:
|
|
148
|
+
if base + ext in rel_set:
|
|
149
|
+
return base + ext
|
|
150
|
+
for idx in ("/index.ts", "/index.tsx", "/index.js", "/index.jsx", "/index.mjs"):
|
|
151
|
+
if base + idx in rel_set:
|
|
152
|
+
return base + idx
|
|
153
|
+
return None
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _mount_auth_coverage(ctx: RepoContext) -> dict:
|
|
157
|
+
"""Find Express router-mount auth and return the set of route files it covers.
|
|
158
|
+
|
|
159
|
+
Returns {covered:set[rel], detected:bool, authed_factories:[...], rel_map, rel_set}.
|
|
160
|
+
"""
|
|
161
|
+
rel_map = {ctx.rel(p): p for p in ctx.code_files}
|
|
162
|
+
rel_set = set(rel_map)
|
|
163
|
+
|
|
164
|
+
def txt(rel: str) -> str:
|
|
165
|
+
p = rel_map.get(rel)
|
|
166
|
+
return ctx.text(p) if p else ""
|
|
167
|
+
|
|
168
|
+
authed_factories: set = set()
|
|
169
|
+
unauthed_factories: set = set()
|
|
170
|
+
for rel, p in rel_map.items():
|
|
171
|
+
# Test harnesses wire `app.use(createXRouter())` WITHOUT auth to exercise a router in
|
|
172
|
+
# isolation — that is test scaffolding, not the production mount, and counting it wrongly
|
|
173
|
+
# marked real sub-routers (memory/marketplace-content) as unauthed. Skip test files.
|
|
174
|
+
if is_test_file(rel):
|
|
175
|
+
continue
|
|
176
|
+
text = ctx.text(p)
|
|
177
|
+
if ".use(" not in text:
|
|
178
|
+
continue
|
|
179
|
+
for m in USE_CALL.finditer(text):
|
|
180
|
+
args = _call_args(text, m.end() - 1)
|
|
181
|
+
facs = set(FACTORY_CALL.findall(args))
|
|
182
|
+
if not facs:
|
|
183
|
+
continue
|
|
184
|
+
if MOUNT_AUTH.search(args):
|
|
185
|
+
authed_factories.update(facs)
|
|
186
|
+
elif APP_RECEIVER.search(m.group(1)):
|
|
187
|
+
unauthed_factories.update(facs)
|
|
188
|
+
# else: inner `router.use(...)` with no auth → inherits the parent mount's auth; ignore
|
|
189
|
+
unauthed_factories -= authed_factories # an authed mount anywhere wins
|
|
190
|
+
|
|
191
|
+
factory_file: dict = {} # factory name -> defining file
|
|
192
|
+
for rel, p in rel_map.items():
|
|
193
|
+
if is_test_file(rel):
|
|
194
|
+
continue
|
|
195
|
+
text = ctx.text(p)
|
|
196
|
+
if "Router" not in text and "Routes" not in text:
|
|
197
|
+
continue
|
|
198
|
+
for a, b in FACTORY_DEF.findall(text):
|
|
199
|
+
nm = a or b
|
|
200
|
+
if nm and nm not in factory_file:
|
|
201
|
+
factory_file[nm] = rel
|
|
202
|
+
|
|
203
|
+
authed_files = {factory_file[n] for n in authed_factories if n in factory_file}
|
|
204
|
+
unauthed_files = {factory_file[n] for n in unauthed_factories if n in factory_file} - authed_files
|
|
205
|
+
|
|
206
|
+
# BFS the local-import graph from each authed factory file; mark every route-bearing file
|
|
207
|
+
# composed under it as covered, but never traverse INTO a separately-unauthed mount's file.
|
|
208
|
+
covered: set = set()
|
|
209
|
+
visited: set = set(authed_files)
|
|
210
|
+
frontier: list = list(authed_files)
|
|
211
|
+
# Cap bounds the walk to the repo's own files (the import graph is finite); a big assembler file
|
|
212
|
+
# can have 100+ relative imports, so the per-file cap must be generous or sub-routers past it
|
|
213
|
+
# (e.g. routes/feature/sub.ts) are silently dropped.
|
|
214
|
+
while frontier and len(visited) <= 6000:
|
|
215
|
+
cur = frontier.pop()
|
|
216
|
+
t = txt(cur)
|
|
217
|
+
if not t:
|
|
218
|
+
continue
|
|
219
|
+
if cur in authed_files or ROUTE_MARK.search(t):
|
|
220
|
+
covered.add(cur)
|
|
221
|
+
for spec in IMPORT_REL.findall(t)[:400]:
|
|
222
|
+
tgt = _resolve_import(cur, spec, rel_set)
|
|
223
|
+
if (tgt and tgt not in visited and tgt not in unauthed_files
|
|
224
|
+
and not is_test_file(tgt) and not is_client_file(tgt)):
|
|
225
|
+
visited.add(tgt)
|
|
226
|
+
frontier.append(tgt)
|
|
227
|
+
|
|
228
|
+
return {"covered": covered, "detected": bool(authed_files),
|
|
229
|
+
"authed_factories": sorted(authed_factories)[:40],
|
|
230
|
+
"rel_map": rel_map, "rel_set": rel_set}
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _parse_next_middleware(ctx: RepoContext) -> dict:
|
|
234
|
+
# Next 15.5+/16 renamed `middleware.ts` → `proxy.ts` (both filenames are valid; the
|
|
235
|
+
# framework recognizes either). Missing this made the tool report "no global auth" on
|
|
236
|
+
# Next 16 apps and flag every handler — the single biggest false-positive cluster.
|
|
237
|
+
for cand in ("middleware.ts", "middleware.js", "src/middleware.ts", "src/middleware.js",
|
|
238
|
+
"proxy.ts", "proxy.js", "src/proxy.ts", "src/proxy.js"):
|
|
239
|
+
txt = ctx.manifest(cand)
|
|
240
|
+
if not txt:
|
|
241
|
+
continue
|
|
242
|
+
matchers = re.findall(r"matcher\s*:\s*\[([^\]]*)\]", txt)
|
|
243
|
+
patterns = re.findall(r"['\"]([^'\"]+)['\"]", matchers[0]) if matchers else []
|
|
244
|
+
roles = [m for grp in ROLE.findall(txt) for m in grp if m]
|
|
245
|
+
return {"present": True, "file": cand, "matchers": patterns,
|
|
246
|
+
"is_auth": bool(MW_AUTH.search(txt)), "role_checks": roles}
|
|
247
|
+
return {"present": False, "matchers": [], "is_auth": False}
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _matcher_covers(path: str, matchers: list) -> bool:
|
|
251
|
+
for m in matchers:
|
|
252
|
+
base = m.split(":")[0].split("(")[0].rstrip("/*")
|
|
253
|
+
if base and path.startswith(base):
|
|
254
|
+
return True
|
|
255
|
+
if m.startswith("/(") or m == "/:path*":
|
|
256
|
+
return True
|
|
257
|
+
return False
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _collect_roles(text: str, roles: set) -> None:
|
|
261
|
+
for grp in ROLE.findall(text or ""):
|
|
262
|
+
for m in grp:
|
|
263
|
+
if not m:
|
|
264
|
+
continue
|
|
265
|
+
for part in m.split(","):
|
|
266
|
+
v = part.strip().strip("'\" ")
|
|
267
|
+
if v and len(v) < 40:
|
|
268
|
+
roles.add(v)
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
class AuthzExtractor(Extractor):
|
|
272
|
+
name = "authz"
|
|
273
|
+
category = "authz"
|
|
274
|
+
|
|
275
|
+
def extract(self, ctx: RepoContext, facts: dict) -> dict:
|
|
276
|
+
endpoints = (facts.get("routes") or {}).get("endpoints", [])
|
|
277
|
+
mw = _parse_next_middleware(ctx)
|
|
278
|
+
mw_auth = mw.get("is_auth", False)
|
|
279
|
+
|
|
280
|
+
# Router-mount auth coverage (Express `app.use('/x', authMiddleware, createXRouter())`).
|
|
281
|
+
mount = _mount_auth_coverage(ctx)
|
|
282
|
+
mount_covered, mount_detected = mount["covered"], mount["detected"]
|
|
283
|
+
rel_map, rel_set = mount["rel_map"], mount["rel_set"]
|
|
284
|
+
|
|
285
|
+
# A THIN Next.js route handler (route.ts / api/*) can delegate its auth one hop to a relative
|
|
286
|
+
# module (e.g. `route.ts` → `./proxy.ts` where getRequestSessionAuth lives). Follow that one
|
|
287
|
+
# hop — bounded to short handler files so the FN risk (marking a real gap as guarded) stays low.
|
|
288
|
+
def _imported_guard(rel: str, text: str) -> bool:
|
|
289
|
+
if not rel or not NEXT_ROUTE.search(rel) or len(text.splitlines()) > 80:
|
|
290
|
+
return False
|
|
291
|
+
for spec in IMPORT_REL.findall(text)[:20]:
|
|
292
|
+
tgt = _resolve_import(rel, spec, rel_set)
|
|
293
|
+
if tgt:
|
|
294
|
+
t2 = ctx.text(rel_map[tgt]) if tgt in rel_map else ""
|
|
295
|
+
if t2 and (GUARD.search(t2) or CUSTOM_GUARD.search(t2)):
|
|
296
|
+
return True
|
|
297
|
+
return False
|
|
298
|
+
|
|
299
|
+
# global auth = an Express path-less auth middleware OR a Next auth middleware/proxy OR a
|
|
300
|
+
# detected router-mount-auth pattern (routes are protected centrally, at the mount).
|
|
301
|
+
global_auth = (mw_auth or mount_detected
|
|
302
|
+
or any(GLOBAL_AUTH.search(t) for _p, _r, t in ctx.iter_code()))
|
|
303
|
+
roles: set = set(mw.get("role_checks", []))
|
|
304
|
+
protected = no_guard = unknown = 0
|
|
305
|
+
no_guard_writes, egs = [], []
|
|
306
|
+
|
|
307
|
+
for e in endpoints:
|
|
308
|
+
cp = e.get("code_path", "")
|
|
309
|
+
text = ctx.text(Path(cp)) if cp else ""
|
|
310
|
+
_collect_roles(text, roles)
|
|
311
|
+
relcp = ctx.rel(Path(cp)) if cp else ""
|
|
312
|
+
# a matcher only counts as a guard when the middleware actually does auth — a
|
|
313
|
+
# non-auth middleware.ts (i18n/headers) must NOT mark routes protected. Mount coverage,
|
|
314
|
+
# the project's custom auth helper, and a one-hop delegated guard also count.
|
|
315
|
+
guarded = (bool(text and (GUARD.search(text) or CUSTOM_GUARD.search(text)))
|
|
316
|
+
or (relcp and relcp in mount_covered)
|
|
317
|
+
or (mw_auth and _matcher_covers(e.get("path", ""), mw.get("matchers", [])))
|
|
318
|
+
or _imported_guard(relcp, text))
|
|
319
|
+
egs.append({"method": e.get("method"), "path": e.get("path"), "code_path": relcp,
|
|
320
|
+
"guarded": bool(guarded), "analyzed": bool(text),
|
|
321
|
+
"public_hint": bool(PUBLIC_HINT.search(e.get("path", "")))})
|
|
322
|
+
if guarded:
|
|
323
|
+
protected += 1
|
|
324
|
+
elif not text:
|
|
325
|
+
unknown += 1
|
|
326
|
+
else:
|
|
327
|
+
no_guard += 1
|
|
328
|
+
if e.get("method") in WRITE_VERBS and not PUBLIC_HINT.search(e.get("path", "")):
|
|
329
|
+
no_guard_writes.append(f"{e['method']} {e['path']} ({relcp or '?'})")
|
|
330
|
+
|
|
331
|
+
# F5: files that make an auth decision AND call an unsafe/unverified decoder
|
|
332
|
+
unsafe_decoders = []
|
|
333
|
+
for _p, rel, text in ctx.iter_code():
|
|
334
|
+
if AUTH_CONTEXT.search(text):
|
|
335
|
+
for dec in sorted(set(UNSAFE_DECODER.findall(text))):
|
|
336
|
+
unsafe_decoders.append({"file": rel, "decoder": dec})
|
|
337
|
+
|
|
338
|
+
# A guard DEFINED in a file that also calls an unsafe/unverified decoder authenticates via
|
|
339
|
+
# an unverified decode. Routes that call such a guard are the static "at-risk" set for the
|
|
340
|
+
# forged-token bypass class — the dynamic probe confirms which actually fall, but this points
|
|
341
|
+
# at them even with NO live target (turns the F5 hypothesis into named routes).
|
|
342
|
+
unverified_routes: list = []
|
|
343
|
+
unsafe_files = {ud["file"] for ud in unsafe_decoders}
|
|
344
|
+
if unsafe_files:
|
|
345
|
+
guard_def = re.compile(r"(?:export\s+)?(?:async\s+)?(?:function|const)\s+"
|
|
346
|
+
r"(require\w+|ensure\w+|\w*[Aa]uth\w*|verify\w+)\b")
|
|
347
|
+
unsafe_guards = set()
|
|
348
|
+
for _p, rel, text in ctx.iter_code():
|
|
349
|
+
if rel in unsafe_files:
|
|
350
|
+
unsafe_guards.update(g for g in guard_def.findall(text) if len(g) >= 5)
|
|
351
|
+
if unsafe_guards:
|
|
352
|
+
call = re.compile(r"\b(?:" + "|".join(re.escape(g) for g in sorted(unsafe_guards)) + r")\s*\(")
|
|
353
|
+
for e in endpoints:
|
|
354
|
+
cp = e.get("code_path", "")
|
|
355
|
+
t = ctx.text(Path(cp)) if cp else ""
|
|
356
|
+
if t and call.search(t):
|
|
357
|
+
unverified_routes.append(f"{e.get('method')} {e.get('path')}")
|
|
358
|
+
unverified_routes = sorted(set(unverified_routes))[:60]
|
|
359
|
+
|
|
360
|
+
if global_auth:
|
|
361
|
+
if mount_detected:
|
|
362
|
+
where = f"router-mount auth (`app.use('/prefix', <auth>, createXRouter())`) on {len(mount_covered)} route file(s)"
|
|
363
|
+
elif mw_auth:
|
|
364
|
+
where = f"`{mw['file']}` (matcher {mw.get('matchers') or '—'})"
|
|
365
|
+
else:
|
|
366
|
+
where = "`app.use(<auth>)`"
|
|
367
|
+
note = (f"A GLOBAL/mount auth pattern ({where}) was detected — most routes are protected centrally, "
|
|
368
|
+
"not in each handler file. Endpoints under an authed mount are reported as guarded. "
|
|
369
|
+
"Any list below is write endpoints with NO guard via the handler file, an authed mount, the "
|
|
370
|
+
"Next matcher, or a one-hop delegated guard; verify each is either covered or an intentional "
|
|
371
|
+
"public exemption (e.g. a signed-token download) — don't assume they're vulnerable.")
|
|
372
|
+
else:
|
|
373
|
+
note = ("No global/mount auth middleware detected. Write endpoints with no visible guard are "
|
|
374
|
+
"high-signal missing-authz leads — verify each.")
|
|
375
|
+
|
|
376
|
+
return {
|
|
377
|
+
"global_auth_middleware": global_auth,
|
|
378
|
+
"mount_auth_detected": mount_detected,
|
|
379
|
+
"mount_authed_factories": mount["authed_factories"],
|
|
380
|
+
"mount_covered_files": len(mount_covered),
|
|
381
|
+
"next_middleware": mw,
|
|
382
|
+
"roles_detected": sorted(r for r in roles if r),
|
|
383
|
+
"guard_summary": {"with_visible_guard": protected,
|
|
384
|
+
"no_visible_guard": no_guard, "unknown": unknown},
|
|
385
|
+
"endpoint_guards": egs[:_MAX_ENDPOINT_GUARDS],
|
|
386
|
+
"endpoint_guards_truncated": max(0, len(egs) - _MAX_ENDPOINT_GUARDS),
|
|
387
|
+
"write_endpoints_without_visible_guard": sorted(set(no_guard_writes))[:60],
|
|
388
|
+
"unsafe_auth_decoders": unsafe_decoders[:30],
|
|
389
|
+
"unverified_signature_routes": unverified_routes,
|
|
390
|
+
"note": note,
|
|
391
|
+
}
|