websec-validator 0.7.0__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. {websec_validator-0.7.0/src/websec_validator.egg-info → websec_validator-0.8.0}/PKG-INFO +9 -6
  2. {websec_validator-0.7.0 → websec_validator-0.8.0}/README.md +8 -5
  3. {websec_validator-0.7.0 → websec_validator-0.8.0}/pyproject.toml +1 -1
  4. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/__init__.py +2 -0
  5. websec_validator-0.8.0/src/websec_validator/extractors/authz_dataflow.py +110 -0
  6. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/surface.py +33 -0
  7. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/transport_security.py +68 -2
  8. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/findings.py +60 -3
  9. {websec_validator-0.7.0 → websec_validator-0.8.0/src/websec_validator.egg-info}/PKG-INFO +9 -6
  10. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator.egg-info/SOURCES.txt +1 -0
  11. {websec_validator-0.7.0 → websec_validator-0.8.0}/tests/test_recon.py +114 -0
  12. {websec_validator-0.7.0 → websec_validator-0.8.0}/LICENSE +0 -0
  13. {websec_validator-0.7.0 → websec_validator-0.8.0}/setup.cfg +0 -0
  14. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/__init__.py +0 -0
  15. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/briefing.py +0 -0
  16. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/calibration.json +0 -0
  17. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/calibration.py +0 -0
  18. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/cli.py +0 -0
  19. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/constitution.py +0 -0
  20. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/corpus.json +0 -0
  21. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/dynamic.py +0 -0
  22. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/auth.py +0 -0
  23. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/authz.py +0 -0
  24. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/base.py +0 -0
  25. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/client_exposure.py +0 -0
  26. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/client_integrity.py +0 -0
  27. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/crypto_usage.py +0 -0
  28. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/graphql.py +0 -0
  29. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/iac_ci.py +0 -0
  30. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/integrations.py +0 -0
  31. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/llm_security.py +0 -0
  32. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/pii_exposure.py +0 -0
  33. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/policy_consistency.py +0 -0
  34. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/routes.py +0 -0
  35. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/schemas.py +0 -0
  36. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/stack.py +0 -0
  37. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/tenant.py +0 -0
  38. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/extractors/upload_security.py +0 -0
  39. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/probes.py +0 -0
  40. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/proof.py +0 -0
  41. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/recon.py +0 -0
  42. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/report.py +0 -0
  43. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/rules/error-stack-disclosure.yml +0 -0
  44. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/rules/insecure-default-secret.yml +0 -0
  45. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/scanners.py +0 -0
  46. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/_lib.py +0 -0
  47. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/appsync-cswsh.sh +0 -0
  48. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/appsync-introspection.sh +0 -0
  49. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/appsync-subscription-bola.sh +0 -0
  50. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/bola-cross-tenant.sh +0 -0
  51. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/bola-write-verbs.py +0 -0
  52. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/client-integrity-checklist.sh +0 -0
  53. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/compare-roles.sh +0 -0
  54. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/dlp-bypass-offline.py +0 -0
  55. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/error-disclosure-probe.sh +0 -0
  56. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/forged-token.sh +0 -0
  57. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/hs256-brute-force.py +0 -0
  58. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/jwt-attacks.sh +0 -0
  59. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/mass-assignment.py +0 -0
  60. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/password-reuse.sh +0 -0
  61. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/pii-output-diff.sh +0 -0
  62. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/race-conditions.py +0 -0
  63. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/rate-limit-burst.sh +0 -0
  64. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/s3-assess.sh +0 -0
  65. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/ssrf-probes.sh +0 -0
  66. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/unauth-baseline.sh +0 -0
  67. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/upload-matrix.sh +0 -0
  68. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/probes/webhook-forgery.py +0 -0
  69. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/reports/FINDINGS-SUMMARY.md.template +0 -0
  70. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/reports/access-control-matrix.md.template +0 -0
  71. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/reports/findings-triage.md.template +0 -0
  72. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/reports/pentest-handover-brief.md.template +0 -0
  73. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator/templates/reports/per-tool-FINDINGS.md.template +0 -0
  74. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator.egg-info/dependency_links.txt +0 -0
  75. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator.egg-info/entry_points.txt +0 -0
  76. {websec_validator-0.7.0 → websec_validator-0.8.0}/src/websec_validator.egg-info/top_level.txt +0 -0
  77. {websec_validator-0.7.0 → websec_validator-0.8.0}/tests/test_hardening.py +0 -0
  78. {websec_validator-0.7.0 → websec_validator-0.8.0}/tests/test_pentest_regressions.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: websec-validator
3
- Version: 0.7.0
3
+ Version: 0.8.0
4
4
  Summary: Defensive, local-first security recon that briefs your AI coding agent on your own codebase — read-only by default (code in, artifacts out): facts + tailored probe scripts, no LLM, no server, no running app.
5
5
  Author: Ricardo Accioly
6
6
  License: MIT
@@ -89,23 +89,24 @@ Then point your agent at the output: **"Read `websec-out/AGENT-BRIEFING.md` and
89
89
 
90
90
  > That's the whole user surface: **`run`** (plus the optional, advanced **`dynamic`** live-probing step below). `recon`/`proof`/`calibrate` exist for developing the tool itself and are hidden from `--help` — you never need them.
91
91
 
92
- ## What it extracts (18 deterministic extractors, no LLM)
92
+ ## What it extracts (19 deterministic extractors, no LLM)
93
93
 
94
94
  | | Dimension | Notable output |
95
95
  |---|---|---|
96
96
  | stack | languages, frameworks, datastores | monorepo-aware (aggregates every manifest) |
97
97
  | routes | every endpoint via **OWASP Noir** | method · path · typed params · code path |
98
98
  | auth | scheme + login surface + **insecure-default signing secrets** | multi-scheme; flags a hard-coded `JWT_SECRET \|\| 'dev-secret'` fallback (forgeable JWT) |
99
- | **authz** | access-control map | guard coverage + **write endpoints with no visible guard** + roles |
99
+ | **authz** | access-control map | guard coverage (incl. **router-mount auth**) + **write endpoints with no visible guard** + roles |
100
+ | **authz_dataflow** | authz *correctness* (does the guard trust the right thing?) | **unsigned-cookie authorization** · **claim-keyed authz** (user-influenceable JWT claim) · **transaction-local RLS context** (resets before the query) |
100
101
  | tenant | multi-tenancy key candidates | the BOLA boundary, by frequency |
101
102
  | **password_policy** | cross-route consistency **+ reuse/history** | complexity drift across routes **+ a set-password path that hashes without a reuse check** |
102
- | surface | 15 sink classes **+ redirect-SSRF** | user-input-gated sinks (incl. **mass-assignment via object spread**) + var-arg SSRF + error-disclosure **+ follows-redirects-without-per-hop-guard** |
103
+ | surface | 15 sink classes **+ redirect-SSRF** | user-input-gated sinks (incl. **mass-assignment via object spread**) + var-arg SSRF + error-disclosure + follows-redirects-without-per-hop-guard **+ reverse-proxy prefix-escape + host-header open-redirect + SSRF-redirect-hardening** |
103
104
  | **upload_security** | unrestricted upload + unsafe serve | deny-list-only, stored-name-from-filename, trust-client-MIME, accept-SVG, **serve without `nosniff`** |
104
105
  | schemas | data models + **privileged fields** | Pydantic/SQLAlchemy/Django/Prisma/Mongoose/TypeORM/Zod → `role`/`isAdmin`/`groupId` for mass-assignment targeting |
105
106
  | iac_ci | IaC + CI/CD | GHA injection (**run:-position-aware**), unpinned actions, tfstate, CDK AppSync `API_KEY` anonymous-default-auth, **docker-compose host-takeover (docker.sock / pid:host / privileged) + `.gitleaksignore` secret-suppression audit** |
106
107
  | client_exposure | browser leakage | public-var secrets by **name + value-shape (`da2-…`) + CDK build-injection**, server-secret-in-client, source maps |
107
108
  | **client_integrity** | tamperable display (client trust boundary) + **WS auth model** | any security-critical sink value (address/IBAN/2FA-seed/API-key/webhook) the user reads or copies, without strict CSP / out-of-band anchor **+ client-tamper-vector, grindable-fingerprint, over-claimed-control, the CSWSH determinant** |
108
- | **transport_security** | CSP + HSTS header baseline | missing/weak CSP, inline event handlers, **partial HSTS (set on /api but not the HTML page)** |
109
+ | **transport_security** | CSP + HSTS + **CORS** + **SRI** baseline | missing/weak CSP, inline event handlers, partial HSTS, **CORS reflect-origin+credentials, external script without SRI, monorepo `next.config` header gap** |
109
110
  | **pii_exposure** | unmasked PII at the output boundary | `res.json(rawEntity)` with PII + **a masking control defined but with zero live call sites** (value-shape, not field-name) |
110
111
  | graphql | GraphQL surface | introspection (**AppSync `introspectionConfig: DISABLED`-aware**) / playground / depth-limit **+ AppSync subscription-authz (cross-group BOLA)** |
111
112
  | integrations | third-party + webhooks **+ outbound-action endpoints** | unsigned webhooks **+ email/SMS/push handlers with no auth or IP-only rate-limit + redundant secret-fetch** |
@@ -218,7 +219,9 @@ publisher** with project `websec-validator`, owner `raccioly`, repo `websec-vali
218
219
 
219
220
  ## Status / roadmap
220
221
 
221
- **Done:** 18-extractor recon (incl. schema/entity → mass-assignment targeting, the **AWS-CDK /
222
+ **Done:** 19-extractor recon (incl. an **authz-correctness data-flow extractor** — unsigned-cookie /
223
+ claim-keyed authz / transaction-local RLS — plus **CORS-misconfig**, **SRI**, **host-header
224
+ open-redirect** and **SSRF-redirect-hardening** classes, schema/entity → mass-assignment targeting, the **AWS-CDK /
222
225
  managed-AppSync / VTL boundary**, **upload-security** + **PII-output-boundary** + **redirect-SSRF**
223
226
  + **password-reuse** classes, a **man-in-the-browser / tamperable-display** class, an **LLM / AI-agent
224
227
  extractor** (OWASP LLM Top 10 — prompt injection / insecure output / excessive agency / unbounded
@@ -77,23 +77,24 @@ Then point your agent at the output: **"Read `websec-out/AGENT-BRIEFING.md` and
77
77
 
78
78
  > That's the whole user surface: **`run`** (plus the optional, advanced **`dynamic`** live-probing step below). `recon`/`proof`/`calibrate` exist for developing the tool itself and are hidden from `--help` — you never need them.
79
79
 
80
- ## What it extracts (18 deterministic extractors, no LLM)
80
+ ## What it extracts (19 deterministic extractors, no LLM)
81
81
 
82
82
  | | Dimension | Notable output |
83
83
  |---|---|---|
84
84
  | stack | languages, frameworks, datastores | monorepo-aware (aggregates every manifest) |
85
85
  | routes | every endpoint via **OWASP Noir** | method · path · typed params · code path |
86
86
  | auth | scheme + login surface + **insecure-default signing secrets** | multi-scheme; flags a hard-coded `JWT_SECRET \|\| 'dev-secret'` fallback (forgeable JWT) |
87
- | **authz** | access-control map | guard coverage + **write endpoints with no visible guard** + roles |
87
+ | **authz** | access-control map | guard coverage (incl. **router-mount auth**) + **write endpoints with no visible guard** + roles |
88
+ | **authz_dataflow** | authz *correctness* (does the guard trust the right thing?) | **unsigned-cookie authorization** · **claim-keyed authz** (user-influenceable JWT claim) · **transaction-local RLS context** (resets before the query) |
88
89
  | tenant | multi-tenancy key candidates | the BOLA boundary, by frequency |
89
90
  | **password_policy** | cross-route consistency **+ reuse/history** | complexity drift across routes **+ a set-password path that hashes without a reuse check** |
90
- | surface | 15 sink classes **+ redirect-SSRF** | user-input-gated sinks (incl. **mass-assignment via object spread**) + var-arg SSRF + error-disclosure **+ follows-redirects-without-per-hop-guard** |
91
+ | surface | 15 sink classes **+ redirect-SSRF** | user-input-gated sinks (incl. **mass-assignment via object spread**) + var-arg SSRF + error-disclosure + follows-redirects-without-per-hop-guard **+ reverse-proxy prefix-escape + host-header open-redirect + SSRF-redirect-hardening** |
91
92
  | **upload_security** | unrestricted upload + unsafe serve | deny-list-only, stored-name-from-filename, trust-client-MIME, accept-SVG, **serve without `nosniff`** |
92
93
  | schemas | data models + **privileged fields** | Pydantic/SQLAlchemy/Django/Prisma/Mongoose/TypeORM/Zod → `role`/`isAdmin`/`groupId` for mass-assignment targeting |
93
94
  | iac_ci | IaC + CI/CD | GHA injection (**run:-position-aware**), unpinned actions, tfstate, CDK AppSync `API_KEY` anonymous-default-auth, **docker-compose host-takeover (docker.sock / pid:host / privileged) + `.gitleaksignore` secret-suppression audit** |
94
95
  | client_exposure | browser leakage | public-var secrets by **name + value-shape (`da2-…`) + CDK build-injection**, server-secret-in-client, source maps |
95
96
  | **client_integrity** | tamperable display (client trust boundary) + **WS auth model** | any security-critical sink value (address/IBAN/2FA-seed/API-key/webhook) the user reads or copies, without strict CSP / out-of-band anchor **+ client-tamper-vector, grindable-fingerprint, over-claimed-control, the CSWSH determinant** |
96
- | **transport_security** | CSP + HSTS header baseline | missing/weak CSP, inline event handlers, **partial HSTS (set on /api but not the HTML page)** |
97
+ | **transport_security** | CSP + HSTS + **CORS** + **SRI** baseline | missing/weak CSP, inline event handlers, partial HSTS, **CORS reflect-origin+credentials, external script without SRI, monorepo `next.config` header gap** |
97
98
  | **pii_exposure** | unmasked PII at the output boundary | `res.json(rawEntity)` with PII + **a masking control defined but with zero live call sites** (value-shape, not field-name) |
98
99
  | graphql | GraphQL surface | introspection (**AppSync `introspectionConfig: DISABLED`-aware**) / playground / depth-limit **+ AppSync subscription-authz (cross-group BOLA)** |
99
100
  | integrations | third-party + webhooks **+ outbound-action endpoints** | unsigned webhooks **+ email/SMS/push handlers with no auth or IP-only rate-limit + redundant secret-fetch** |
@@ -206,7 +207,9 @@ publisher** with project `websec-validator`, owner `raccioly`, repo `websec-vali
206
207
 
207
208
  ## Status / roadmap
208
209
 
209
- **Done:** 18-extractor recon (incl. schema/entity → mass-assignment targeting, the **AWS-CDK /
210
+ **Done:** 19-extractor recon (incl. an **authz-correctness data-flow extractor** — unsigned-cookie /
211
+ claim-keyed authz / transaction-local RLS — plus **CORS-misconfig**, **SRI**, **host-header
212
+ open-redirect** and **SSRF-redirect-hardening** classes, schema/entity → mass-assignment targeting, the **AWS-CDK /
210
213
  managed-AppSync / VTL boundary**, **upload-security** + **PII-output-boundary** + **redirect-SSRF**
211
214
  + **password-reuse** classes, a **man-in-the-browser / tamperable-display** class, an **LLM / AI-agent
212
215
  extractor** (OWASP LLM Top 10 — prompt injection / insecure output / excessive agency / unbounded
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "websec-validator"
7
- version = "0.7.0"
7
+ version = "0.8.0"
8
8
  description = "Defensive, local-first security recon that briefs your AI coding agent on your own codebase — read-only by default (code in, artifacts out): facts + tailored probe scripts, no LLM, no server, no running app."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11"
@@ -11,6 +11,7 @@ from pathlib import Path
11
11
 
12
12
  from .auth import AuthExtractor
13
13
  from .authz import AuthzExtractor
14
+ from .authz_dataflow import AuthzDataflowExtractor
14
15
  from .base import MAX_FILES, Extractor, RepoContext
15
16
  from .client_exposure import ClientExposureExtractor
16
17
  from .client_integrity import ClientIntegrityExtractor
@@ -50,6 +51,7 @@ REGISTRY: list[Extractor] = [
50
51
  IntegrationsExtractor(),
51
52
  LlmSecurityExtractor(),
52
53
  CryptoUsageExtractor(),
54
+ AuthzDataflowExtractor(),
53
55
  ]
54
56
 
55
57
 
@@ -0,0 +1,110 @@
1
+ """Authorization data-flow extractor — authz CORRECTNESS, not just presence.
2
+
3
+ `authz.py` answers "is there a guard on this route?". This answers the next question — "does the guard
4
+ trust the right thing?" — for three real broken-access-control patterns the route-level model can't see:
5
+
6
+ - **Unsigned-cookie authorization (CWE-565/602)** — an access decision keyed on a client-settable,
7
+ unsigned cookie (`twin-access-level`, `*role*`, `*allowed*`). httpOnly stops JS reads but the user
8
+ still controls their own cookie jar (devtools/curl), so the gate is forgeable.
9
+ - **Claim → authz decision (CWE-639/807)** — an authorization check that compares a profile-ish JWT
10
+ BODY claim (office/role/group/tenant) the user may influence, instead of a server-resolved record.
11
+ - **Transaction-local RLS context (CWE-1188)** — `set_config('app.*', x, true)` (is_local=true)
12
+ emitted at autocommit (no surrounding transaction): the RLS principal resets before the handler's
13
+ query runs, so row-level security evaluates with an EMPTY context — defense-in-depth theater.
14
+
15
+ All are file-level co-occurrence heuristics, server-side + test-excluded, framed as leads to verify —
16
+ a static scan can't prove the cookie is the one that gates, only point the agent at the file.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import re
22
+
23
+ from .base import Extractor, RepoContext, is_client_file, is_test_file
24
+
25
+ # --- unsigned-cookie authorization ---
26
+ COOKIE_AUTHZ_READ = re.compile(
27
+ r"(?:cookies?\.get|req\.cookies|getCookie|cookieStore\.get)\s*\(?\s*['\"]"
28
+ r"([\w-]*(?:access|role|admin|allow|auth|tier|perm|priv|level|scope|tenant|clerk-user)[\w-]*)['\"]", re.I)
29
+ AUTHZ_GATE = re.compile(
30
+ r"\b403\b|status\s*\(\s*40[13]|\.redirect\s*\([^)]*(?:login|signin|default|unauthorized|forbidden)"
31
+ r"|[!=]==\s*['\"](?:full|admin|allowed|true|owner)|normalizeAccess|accessLevel|hasAccess|requireFull|denyAccess", re.I)
32
+ COOKIE_VERIFY = re.compile(
33
+ r"jwtVerify|jwt\.verify|verifyToken|verifyCookie|\bhmac\b|timingSafeEqual|unseal|\bdecrypt\b"
34
+ r"|signed\s*:\s*true|cookie-signature|verifySignature|getServerSession", re.I)
35
+
36
+ # --- claim → authz decision ---
37
+ AUTHZ_FN = re.compile(
38
+ r"function\s+(?:can|authorize|check|assert)\w*|(?:const|let)\s+(?:can|authorize|check)\w*\s*=\s*"
39
+ r"|\bcan[A-Z]\w*\s*\(|\bauthoriz\w*\s*\(|visibilit|buildVisibility|accessControl|ACL\b", re.I)
40
+ # a profile-ish claim taken from the JWT BODY / request identity (vs a server-resolved record)
41
+ CLAIM_SRC = re.compile(
42
+ r"jwtPayload\s*\[\s*['\"](?:office|role|roles|group|org|department|tenant|plan|tier|scope)"
43
+ r"|payload\.(?:office|role|group|org|department|tenant)\b"
44
+ r"|claims?\s*[\[.]\s*['\"]?(?:office|role|group|org|department|tenant)"
45
+ r"|\.tenant(?:Office|Role|Group|Plan)\b|req\.(?:user|auth)\.(?:office|role|group|org|department)\b", re.I)
46
+ CLAIM_COMPARE = re.compile(r"[!=]==|\.(?:includes|has|some)\s*\(|\bin\s+\[", re.I)
47
+
48
+ # --- transaction-local RLS context ---
49
+ SET_CONFIG_LOCAL = re.compile(r"set_config\s*\(\s*['\"]app\.[\w.]+['\"]\s*,[^,()]+,\s*true\s*\)", re.I)
50
+ IN_TRANSACTION = re.compile(
51
+ r"\bBEGIN\b|\bSTART\s+TRANSACTION\b|\.transaction\s*\(|withTransaction|db\.transaction|\btx\b\.|\btrx\b\.|\bBEGIN;|"
52
+ r"transaction\s*\(\s*async", re.I)
53
+
54
+
55
+ class AuthzDataflowExtractor(Extractor):
56
+ name = "authz_dataflow"
57
+ category = "authz"
58
+
59
+ def extract(self, ctx: RepoContext, facts: dict) -> dict:
60
+ findings: list = []
61
+ seen: set = set()
62
+
63
+ def add(sev, kind, attack, rel, detail):
64
+ if (kind, rel) in seen:
65
+ return
66
+ seen.add((kind, rel))
67
+ findings.append({"severity": sev, "kind": kind, "attack_class": attack,
68
+ "file": rel, "detail": detail})
69
+
70
+ for _p, rel, text in ctx.iter_code():
71
+ if is_test_file(rel) or is_client_file(rel, text):
72
+ continue
73
+
74
+ # 1. authorization decision keyed on an unsigned cookie
75
+ m = COOKIE_AUTHZ_READ.search(text)
76
+ if m and AUTHZ_GATE.search(text) and not COOKIE_VERIFY.search(text):
77
+ add("MEDIUM", "unsigned-cookie-authz", "cookie-authz", rel,
78
+ f"An access decision appears to be keyed on a client-settable cookie "
79
+ f"(`{m.group(1)}`) with no signature/JWT/HMAC verification in this file. httpOnly only "
80
+ "blocks JS reads — the user still controls their own cookie jar, so they can forge the "
81
+ "value and pass the gate (CWE-565/602). Bind access to a signed value (read it from the "
82
+ "verified session JWT, or HMAC the cookie) and re-derive server-side.")
83
+
84
+ # 2. authz decision comparing a user-influenceable JWT claim
85
+ if AUTHZ_FN.search(text) and CLAIM_SRC.search(text) and CLAIM_COMPARE.search(text):
86
+ add("LOW", "claim-based-authz", "claim-authz", rel,
87
+ "An authorization check appears to compare a profile-ish JWT BODY claim "
88
+ "(office/role/group/tenant) that the user may be able to influence, rather than a "
89
+ "value resolved from the authenticated record (CWE-639/807). VERIFY the claim is "
90
+ "server-resolved and non-user-editable; otherwise resolve it from the tenant/user row "
91
+ "inside the request, and never write a client-asserted claim back as the record of truth.")
92
+
93
+ # 3. transaction-local RLS context set outside a transaction (resets before the query)
94
+ if SET_CONFIG_LOCAL.search(text) and not IN_TRANSACTION.search(text):
95
+ add("MEDIUM", "rls-context-no-transaction", "rls-context", rel,
96
+ "`set_config('app.*', …, true)` sets the RLS principal TRANSACTION-LOCALLY "
97
+ "(is_local=true) but there's no surrounding transaction in this file — at autocommit "
98
+ "the setting resets the instant that one statement commits, before any handler query "
99
+ "runs, so RLS evaluates with an EMPTY context (CWE-1188, defense-in-depth theater). "
100
+ "Set it INSIDE the transaction that runs the tenant-scoped query, on the SAME "
101
+ "connection the handler uses (or use is_local=false on a dedicated per-request connection).")
102
+
103
+ by_sev: dict = {}
104
+ for f in findings:
105
+ by_sev[f["severity"]] = by_sev.get(f["severity"], 0) + 1
106
+ return {"findings": findings, "by_severity": by_sev,
107
+ "note": (f"{len(findings)} authz-correctness lead(s) — these check whether a guard trusts the "
108
+ "RIGHT thing (signed cookie / server-resolved claim / live RLS context), not just whether "
109
+ "a guard exists. Verify each against the data flow.") if findings
110
+ else "No unsigned-cookie / claim-keyed-authz / transaction-local-RLS tells found."}
@@ -108,6 +108,24 @@ DOTSEG_GUARD = re.compile(
108
108
  r"includes\s*\(\s*['\"]\.\.|===\s*['\"]\.\.['\"]|['\"]\.\.['\"]\s*\)|%2e|%2f|decodeURIComponent"
109
109
  r"|\bnormalize\b|startsWith\s*\(\s*['\"]/[\w]|sanitiz|assertPath|safeJoin|\bresolve\b[^;]{0,40}startsWith", re.I)
110
110
 
111
+ # Host-header → redirect (open redirect / cache poisoning, CWE-601): a redirect Location/origin built
112
+ # from the attacker-controllable Host / X-Forwarded-Host header with no host allow-list comparison.
113
+ HOST_HEADER_READ = re.compile(
114
+ r"(?:req|request|ctx)\.(?:headers?\s*[\[.]|get\s*\(\s*)['\"]?(?:x-forwarded-host|forwarded|host)\b"
115
+ r"|headers\.get\s*\(\s*['\"](?:x-forwarded-host|host|forwarded)", re.I)
116
+ REDIRECT_SINK = re.compile(
117
+ r"\.redirect\s*\(|NextResponse\.redirect|res\.setHeader\s*\(\s*['\"]Location|['\"]Location['\"]\s*[:,]|sendRedirect", re.I)
118
+ HOST_ALLOWLIST = re.compile(
119
+ r"allowedHosts?|allow[_-]?list|publicOrigin|publicWebOrigin|trustedHosts?|isAllowedHost|whitelist|brandfolderHostnames", re.I)
120
+ # SSRF-hardening: an outbound client DELIBERATELY following redirects with no host allow-list /
121
+ # private-range deny — a 30x to an internal/metadata host is then fetched server-side (CWE-918),
122
+ # independent of whether the INITIAL url is user-tainted. Reaches worker/job scripts the route scan misses.
123
+ FOLLOW_REDIR = re.compile(r"follow_redirects\s*=\s*True|allow_redirects\s*=\s*True|maxRedirects\s*:\s*[1-9]", re.I)
124
+ OUTBOUND_CLIENT = re.compile(r"\b(?:requests\.\w+|httpx\.\w+|\bfetch\s*\(|axios|got\s*\(|node-fetch|urllib\.request)", re.I)
125
+ SSRF_PRIVATE_GUARD = re.compile(
126
+ r"allow_redirects\s*=\s*False|follow_redirects\s*=\s*False|maxRedirects\s*:\s*0|169\.254|RFC1918|is_private|"
127
+ r"ip_address|private_?range|block.?(?:internal|private)|allowedHosts?|allow[_-]?list", re.I)
128
+
111
129
 
112
130
  class SurfaceExtractor(Extractor):
113
131
  name = "surface"
@@ -122,6 +140,8 @@ class SurfaceExtractor(Extractor):
122
140
  counts: dict = {k: 0 for k in SINKS}
123
141
  ssrf_redirect: list = [] # SSRF sink in a file with NO per-hop redirect guard (#1)
124
142
  proxy_escape: list = [] # reverse-proxy prefix-escape (confined-deputy)
143
+ host_redirect: list = [] # redirect built from the Host/X-Forwarded-Host header (open redirect)
144
+ follows_redirect: list = [] # outbound client follows redirects with no allow-list (SSRF-hardening)
125
145
  for _p, rel, text in ctx.iter_code():
126
146
  # Test fixtures are never a runtime sink; SSRF + outbound-HTTP are SERVER-ONLY classes,
127
147
  # so client (.tsx/'use client') and build/CLI scripts can't host them (validated: these
@@ -153,6 +173,17 @@ class SurfaceExtractor(Extractor):
153
173
  if (len(proxy_escape) < 30 and not nonserver and PROXY_JOIN.search(text)
154
174
  and PROXY_FETCH.search(text) and not DOTSEG_GUARD.search(text)):
155
175
  proxy_escape.append(rel)
176
+ # host-header → redirect (open redirect via Host/X-Forwarded-Host). Server-only; needs a
177
+ # redirect sink + a host-header read + no host allow-list in the file.
178
+ if (len(host_redirect) < 30 and not nonserver and HOST_HEADER_READ.search(text)
179
+ and REDIRECT_SINK.search(text) and not HOST_ALLOWLIST.search(text)):
180
+ host_redirect.append(rel)
181
+ # SSRF-hardening: follows redirects with no allow-list / private-range deny. Worker/job
182
+ # SCRIPTS are in scope here (they fetch server-side), so only client + tests are excluded.
183
+ if (len(follows_redirect) < 30 and not is_client_file(rel, text)
184
+ and FOLLOW_REDIR.search(text) and OUTBOUND_CLIENT.search(text)
185
+ and not SSRF_PRIVATE_GUARD.search(text)):
186
+ follows_redirect.append(rel)
156
187
 
157
188
  sinks = {k: {"probe": SINKS[k][0], "count": counts[k], "files": found[k]}
158
189
  for k in SINKS if counts[k]}
@@ -161,6 +192,8 @@ class SurfaceExtractor(Extractor):
161
192
  "sink_counts": {k: counts[k] for k in SINKS if counts[k]},
162
193
  "ssrf_redirect_unguarded": ssrf_redirect, # validate EVERY hop, not just the input URL (#1)
163
194
  "proxy_prefix_escape": proxy_escape, # confined-deputy via `..` in catch-all path
195
+ "host_header_redirect": host_redirect, # open redirect via Host/X-Forwarded-Host
196
+ "follows_redirect_no_allowlist": follows_redirect, # SSRF-hardening (redirect-to-internal)
164
197
  "datastore_class": ("sql" if has_sql else ("nosql" if has_nosql else "unknown")),
165
198
  "note": "Each sink hit is user-input-gated (req./request./concat/interp), so these are "
166
199
  "higher-confidence leads. Cross-reference the files with routes.targeting to pick "
@@ -19,7 +19,7 @@ from __future__ import annotations
19
19
 
20
20
  import re
21
21
 
22
- from .base import Extractor, RepoContext
22
+ from .base import Extractor, RepoContext, is_script_file, is_test_file
23
23
 
24
24
  CSP_ANY = re.compile(r"Content-Security-Policy|contentSecurityPolicy|helmet[\s\S]{0,40}?\bcsp\b"
25
25
  r"|useCspNonce|cspDirectives", re.I)
@@ -54,6 +54,22 @@ CK_SECURE = re.compile(
54
54
  r"[A-Za-z_$][\w$.]*\s*[=!]==|[A-Za-z_$][\w$.]*\s*\?|[A-Za-z_$][\w$.]*\([^)]*\))", re.I)
55
55
  CK_SAMESITE = re.compile(r"samesite", re.I)
56
56
 
57
+ # CORS misconfiguration — the high-impact form is an Allow-Origin that REFLECTS the request Origin (or
58
+ # `*`) TOGETHER with Allow-Credentials:true, which lets any site read authenticated responses.
59
+ CORS_REFLECT = re.compile(
60
+ r"Access-Control-Allow-Origin['\"]?\s*[,:][^,\n)]{0,60}(?:req\.|request\.|headers?\.origin|get\s*\(\s*['\"]origin|\borigin\b)"
61
+ r"|cors\s*\(\s*\{[^}]*origin\s*:\s*true|origin\s*:\s*(?:true|function|\(origin)|reflectOrigin|originReflect", re.I)
62
+ CORS_WILDCARD = re.compile(r"Access-Control-Allow-Origin['\"]?\s*[,:]\s*['\"]\*['\"]|\borigin\s*:\s*['\"]\*['\"]", re.I)
63
+ CORS_CREDS = re.compile(r"Access-Control-Allow-Credentials['\"]?\s*[,:]\s*['\"]?true|credentials\s*:\s*true", re.I)
64
+ # an external <script src="https://…"> with no Subresource-Integrity (supply-chain: a CDN compromise
65
+ # runs arbitrary JS in your origin). Only meaningful in code that emits HTML.
66
+ EXT_SCRIPT = re.compile(r"<script\b[^>]*\ssrc\s*=\s*['\"]https?://[^'\"]+['\"][^>]*>", re.I)
67
+ SRI_OK = re.compile(r"\bintegrity\s*=", re.I)
68
+ # a Next.js config that defines security headers via headers()
69
+ NEXT_HEADERS_FN = re.compile(r"async\s+headers\s*\(|\bheaders\s*\(\s*\)\s*\{|key\s*:\s*['\"](?:Content-Security-Policy|X-Frame-Options|Strict-Transport-Security|X-Content-Type-Options)['\"]", re.I)
70
+ NEXT_CSP = re.compile(r"Content-Security-Policy", re.I)
71
+ NEXT_XFO = re.compile(r"X-Frame-Options|frame-ancestors", re.I)
72
+
57
73
 
58
74
  class TransportSecurityExtractor(Extractor):
59
75
  name = "transport_security"
@@ -69,6 +85,7 @@ class TransportSecurityExtractor(Extractor):
69
85
  inline_handlers = []
70
86
  sets_cookie = ck_httponly = ck_secure = ck_samesite = False
71
87
  hsts_files, hsts_api_only, hsts_html = [], True, False
88
+ extra_findings: list = [] # CORS / SRI / next-config — emitted alongside the CSP/HSTS set
72
89
 
73
90
  # config manifests carry headers too (next.config, vercel.json, netlify.toml, _headers)
74
91
  manifests = "\n".join(ctx.manifest(n) for n in
@@ -76,9 +93,37 @@ class TransportSecurityExtractor(Extractor):
76
93
  "netlify.toml", "public/_headers", "static/_headers", "nginx.conf"))
77
94
 
78
95
  for _p, rel, text in ctx.iter_code():
96
+ if is_test_file(rel) or is_script_file(rel):
97
+ continue
79
98
  if HTML_SURFACE.search(rel) or HTML_CONTENT.search(text):
80
99
  html_surface = True
81
100
  blob = text
101
+ # CORS misconfig — reflected/wildcard Allow-Origin together with credentials = any site
102
+ # reads authed responses (CWE-942). Server-side only.
103
+ if CORS_CREDS.search(blob) and (CORS_REFLECT.search(blob) or CORS_WILDCARD.search(blob)):
104
+ extra_findings.append({"severity": "HIGH", "kind": "cors-credentials-any-origin",
105
+ "attack_class": "cors-misconfig", "file": rel,
106
+ "detail": "CORS reflects the request Origin (or uses `*`) AND sets "
107
+ "Allow-Credentials:true — any website can make credentialed cross-origin "
108
+ "requests and READ the authenticated responses (CWE-942). Allow-list exact "
109
+ "trusted origins; never reflect Origin or use `*` when credentials are on."})
110
+ elif CORS_REFLECT.search(blob):
111
+ extra_findings.append({"severity": "MEDIUM", "kind": "cors-reflects-origin",
112
+ "attack_class": "cors-misconfig", "file": rel,
113
+ "detail": "CORS appears to reflect the request Origin (echo-back / `origin:true`) "
114
+ "rather than allow-listing exact origins. Safe only without credentials and with a "
115
+ "strict allow-list — verify it can't be turned into a credentialed cross-origin read."})
116
+ # external script with no SRI, in code that emits HTML
117
+ if (HTML_CONTENT.search(blob) or HTML_SURFACE.search(rel)):
118
+ for m in EXT_SCRIPT.finditer(blob):
119
+ if not SRI_OK.search(m.group(0)):
120
+ extra_findings.append({"severity": "MEDIUM", "kind": "external-script-no-sri",
121
+ "attack_class": "subresource-integrity", "file": rel,
122
+ "detail": "An external <script src=\"https://…\"> is loaded with no "
123
+ "Subresource-Integrity (`integrity=`) hash / version pin — a CDN or "
124
+ "package compromise runs arbitrary JS in this origin (CWE-829). Pin the "
125
+ "version + add an SRI hash + `crossorigin`, or self-host the bundle."})
126
+ break
82
127
  if CSP_ANY.search(blob):
83
128
  csp_present = True
84
129
  if CSP_SCRIPT_SELF.search(blob):
@@ -121,9 +166,30 @@ class TransportSecurityExtractor(Extractor):
121
166
  if HSTS_PRELOAD.search(manifests):
122
167
  hsts_preload = True
123
168
 
169
+ # Monorepo-aware Next.js config header gap — `transport_security` previously only read the
170
+ # ROOT next.config via manifests, so a `packages/web/next.config.ts` was invisible. Glob every
171
+ # next.config.* and flag one with no security-header block (CSP + X-Frame-Options).
172
+ for nc in (ctx.glob("**/next.config.js") + ctx.glob("**/next.config.mjs")
173
+ + ctx.glob("**/next.config.ts")):
174
+ rel, txt = ctx.rel(nc), ctx.text(nc)
175
+ if not NEXT_HEADERS_FN.search(txt):
176
+ extra_findings.append({"severity": "MEDIUM", "kind": "nextjs-no-security-headers",
177
+ "attack_class": "missing-csp", "file": rel,
178
+ "detail": "This Next.js config defines no security-header block (`headers()`) — "
179
+ "so no app-wide Content-Security-Policy, X-Frame-Options/frame-ancestors, HSTS, "
180
+ "X-Content-Type-Options, or Referrer-Policy unless an upstream edge sets them. Add "
181
+ "a `headers()` matcher (start CSP report-only) or document that nginx/CDN owns them."})
182
+ elif not (NEXT_CSP.search(txt) and NEXT_XFO.search(txt)):
183
+ miss = ", ".join(n for n, ok in (("CSP", NEXT_CSP.search(txt)),
184
+ ("X-Frame-Options/frame-ancestors", NEXT_XFO.search(txt))) if not ok)
185
+ extra_findings.append({"severity": "LOW", "kind": "nextjs-partial-security-headers",
186
+ "attack_class": "missing-csp", "file": rel,
187
+ "detail": f"Next.js config has a headers() block but is missing {miss}. Add the "
188
+ "clickjacking/XSS-defense headers (verify against the live response if the edge sets some)."})
189
+
124
190
  strict_csp = bool(csp_present and csp_self and csp_nonce and not csp_unsafe)
125
191
  web_surface = html_surface or has_routes
126
- findings = []
192
+ findings = list(extra_findings)
127
193
 
128
194
  if html_surface:
129
195
  if not csp_present:
@@ -111,6 +111,16 @@ STANDARDS = {
111
111
  "predictable-principal": (["CWE-330 Use of Insufficiently Random Values", "CWE-340 Predictable from Observable State"],
112
112
  "ASVS V6.3.1", ["API1:2023 BOLA"]),
113
113
  "timing-unsafe-compare": (["CWE-208 Observable Timing Discrepancy"], "ASVS V6.2.3", ["API2:2023 Broken Authentication"]),
114
+ # --- transport / access-control-dataflow classes (0.8.0) ---
115
+ "cors-misconfig": (["CWE-942 Permissive Cross-domain Policy with Untrusted Domains"], "ASVS V14.5.3",
116
+ ["API8:2023 Misconfiguration"]),
117
+ "subresource-integrity": (["CWE-829 Inclusion of Functionality from Untrusted Control Sphere"], "ASVS V14.2.3",
118
+ ["API8:2023 Misconfiguration"]),
119
+ "cookie-authz": (["CWE-565 Reliance on Cookies Without Validation", "CWE-602 Client-Side Enforcement of Server-Side Security"],
120
+ "ASVS V3.4", ["API1:2023 BOLA", "API5:2023 BFLA"]),
121
+ "claim-authz": (["CWE-639 Authorization Bypass Through User-Controlled Key", "CWE-807 Reliance on Untrusted Inputs in a Security Decision"],
122
+ "ASVS V4.2.1", ["API1:2023 BOLA"]),
123
+ "rls-context": (["CWE-1188 Insecure Default Initialization of Resource"], "ASVS V4.1.3", ["API1:2023 BOLA"]),
114
124
  }
115
125
  REMEDIATION = {
116
126
  "missing-auth": "Add an auth guard to the handler (e.g. requireAuth()/getServerSession()), or a "
@@ -208,6 +218,21 @@ REMEDIATION = {
208
218
  "timing-unsafe-compare": "Compare request-supplied secrets/tokens/signatures with `crypto.timingSafeEqual` "
209
219
  "(equal-length buffers) or `compare_digest`, never `===`/`!==`, to remove the timing "
210
220
  "side-channel.",
221
+ "cors-misconfig": "Allow-list exact trusted origins; never reflect the request Origin or use `*` when "
222
+ "`Allow-Credentials: true`. If credentials aren't needed, drop them so a strict allow-list "
223
+ "isn't load-bearing.",
224
+ "subresource-integrity": "Pin the external resource to an exact version and add a Subresource-Integrity "
225
+ "`integrity=` hash + `crossorigin`, or self-host it, and constrain it with a CSP so a "
226
+ "CDN/package compromise can't run arbitrary JS in your origin.",
227
+ "cookie-authz": "Never make an authorization decision on an unsigned/unverified cookie — bind access to a "
228
+ "signed value (read it from the verified session JWT, or HMAC the cookie) and re-derive on the "
229
+ "server; treat client cookies as untrusted hints only.",
230
+ "claim-authz": "Resolve the fields used for authorization (office/role/tenant) from the authenticated record "
231
+ "server-side, not from the JWT body claim; never write a client-asserted claim back as the record "
232
+ "of truth. Verify the resolved row owns the resource.",
233
+ "rls-context": "Set the RLS context (`set_config('app.*', x, true)`) INSIDE the transaction that runs the "
234
+ "tenant-scoped query, on the SAME connection the handler uses — a transaction-local setting "
235
+ "emitted at autocommit resets before the query, so RLS evaluates with an empty context.",
211
236
  }
212
237
  _DEFAULT_REM = "Review and remediate per the cited standard."
213
238
 
@@ -429,6 +454,25 @@ def build_ledger(facts: dict, unified: dict | None, dynamic: dict | None = None,
429
454
  "caller reaches any upstream route with valid creds (confused deputy). Reject `.`/`..`/`%2e` "
430
455
  "segments or assert the normalized pathname still starts with the prefix."}]))
431
456
 
457
+ # ---- 3d. Host-header → redirect (open redirect / cache poisoning) ----
458
+ for rel in (facts.get("surface", {}).get("host_header_redirect", []) or []):
459
+ out.append(_f(f"Host-header open-redirect: {rel}", "attack-surface", "open-redirect",
460
+ "MEDIUM", "LOW", rel,
461
+ [{"layer": "recon", "detail": "a redirect Location/origin is built from the attacker-controllable "
462
+ "Host / X-Forwarded-Host header with no host allow-list — an on-path or cache-poisoning attacker "
463
+ "can send the user to an arbitrary host (CWE-601). Pin the redirect base to a server-configured "
464
+ "origin allow-list, never request headers."}]))
465
+
466
+ # ---- 3e. SSRF-hardening: outbound client follows redirects with no allow-list ----
467
+ for rel in (facts.get("surface", {}).get("follows_redirect_no_allowlist", []) or []):
468
+ out.append(_f(f"Follows redirects with no allow-list: {rel}", "attack-surface", "ssrf",
469
+ "MEDIUM", "LOW", rel,
470
+ [{"layer": "recon", "detail": "an outbound HTTP client deliberately follows redirects "
471
+ "(follow_redirects/allow_redirects=True, maxRedirects>0) with no host allow-list or private-range "
472
+ "deny — a 30x to 169.254.169.254 / an RFC1918 host is fetched server-side regardless of whether "
473
+ "the initial URL is user-tainted (CWE-918). Re-validate every hop against an allow-list; deny "
474
+ "loopback/link-local/private ranges."}]))
475
+
432
476
  # ---- 4. Client-side secret exposure (HIGH — ships to browser) ----
433
477
  # Name-based + value-shape (rename-proof) + CDK build-injection (#3) all land here.
434
478
  _cx = facts.get("client_exposure", {})
@@ -495,9 +539,14 @@ def build_ledger(facts: dict, unified: dict | None, dynamic: dict | None = None,
495
539
 
496
540
  # ---- 8b. Transport / browser-hardening header baseline (CSP #3, HSTS #4) ----
497
541
  for fnd in (facts.get("transport_security", {}) or {}).get("findings", []):
498
- out.append(_f(f"{fnd.get('kind')}: browser/transport hardening header", "transport",
499
- fnd.get("attack_class", "missing-csp"), fnd.get("severity", "LOW"), "LOW",
500
- "(response headers)", [{"layer": "recon", "detail": fnd.get("detail", "")}]))
542
+ # header-baseline findings have no file (they're about response headers); CORS/SRI/next-config
543
+ # findings carry a real file — use it so the location is actionable.
544
+ loc = fnd.get("file") or "(response headers)"
545
+ title = f"{fnd.get('kind')}: {fnd.get('file')}" if fnd.get("file") else \
546
+ f"{fnd.get('kind')}: browser/transport hardening header"
547
+ out.append(_f(title, "transport", fnd.get("attack_class", "missing-csp"),
548
+ fnd.get("severity", "LOW"), "LOW", loc,
549
+ [{"layer": "recon", "detail": fnd.get("detail", "")}]))
501
550
 
502
551
  # ---- 9. Inbound webhooks with no signature verification (forgery / replay) ----
503
552
  # Recon found webhook handlers with no HMAC/signature check. This was surfaced in the briefing
@@ -548,6 +597,14 @@ def build_ledger(facts: dict, unified: dict | None, dynamic: dict | None = None,
548
597
  fnd.get("severity", "MEDIUM"), "MEDIUM", fnd.get("file", ""),
549
598
  [{"layer": "recon", "detail": fnd.get("detail", "")}]))
550
599
 
600
+ # ---- 14. Authz data-flow — does the guard trust the right thing? (unsigned-cookie authz,
601
+ # claim-keyed authz, transaction-local RLS context). ----
602
+ for fnd in (facts.get("authz_dataflow", {}) or {}).get("findings", []):
603
+ out.append(_f(f"{fnd.get('kind')}: {fnd.get('file')}", "access-control",
604
+ fnd.get("attack_class", "cookie-authz"),
605
+ fnd.get("severity", "MEDIUM"), "LOW", fnd.get("file", ""),
606
+ [{"layer": "recon", "detail": fnd.get("detail", "")}]))
607
+
551
608
  # ---- suppress + rank ----
552
609
  kept = [f for f in out if not _suppressed(f, suppressions)]
553
610
  suppressed_n = len(out) - len(kept)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: websec-validator
3
- Version: 0.7.0
3
+ Version: 0.8.0
4
4
  Summary: Defensive, local-first security recon that briefs your AI coding agent on your own codebase — read-only by default (code in, artifacts out): facts + tailored probe scripts, no LLM, no server, no running app.
5
5
  Author: Ricardo Accioly
6
6
  License: MIT
@@ -89,23 +89,24 @@ Then point your agent at the output: **"Read `websec-out/AGENT-BRIEFING.md` and
89
89
 
90
90
  > That's the whole user surface: **`run`** (plus the optional, advanced **`dynamic`** live-probing step below). `recon`/`proof`/`calibrate` exist for developing the tool itself and are hidden from `--help` — you never need them.
91
91
 
92
- ## What it extracts (18 deterministic extractors, no LLM)
92
+ ## What it extracts (19 deterministic extractors, no LLM)
93
93
 
94
94
  | | Dimension | Notable output |
95
95
  |---|---|---|
96
96
  | stack | languages, frameworks, datastores | monorepo-aware (aggregates every manifest) |
97
97
  | routes | every endpoint via **OWASP Noir** | method · path · typed params · code path |
98
98
  | auth | scheme + login surface + **insecure-default signing secrets** | multi-scheme; flags a hard-coded `JWT_SECRET \|\| 'dev-secret'` fallback (forgeable JWT) |
99
- | **authz** | access-control map | guard coverage + **write endpoints with no visible guard** + roles |
99
+ | **authz** | access-control map | guard coverage (incl. **router-mount auth**) + **write endpoints with no visible guard** + roles |
100
+ | **authz_dataflow** | authz *correctness* (does the guard trust the right thing?) | **unsigned-cookie authorization** · **claim-keyed authz** (user-influenceable JWT claim) · **transaction-local RLS context** (resets before the query) |
100
101
  | tenant | multi-tenancy key candidates | the BOLA boundary, by frequency |
101
102
  | **password_policy** | cross-route consistency **+ reuse/history** | complexity drift across routes **+ a set-password path that hashes without a reuse check** |
102
- | surface | 15 sink classes **+ redirect-SSRF** | user-input-gated sinks (incl. **mass-assignment via object spread**) + var-arg SSRF + error-disclosure **+ follows-redirects-without-per-hop-guard** |
103
+ | surface | 15 sink classes **+ redirect-SSRF** | user-input-gated sinks (incl. **mass-assignment via object spread**) + var-arg SSRF + error-disclosure + follows-redirects-without-per-hop-guard **+ reverse-proxy prefix-escape + host-header open-redirect + SSRF-redirect-hardening** |
103
104
  | **upload_security** | unrestricted upload + unsafe serve | deny-list-only, stored-name-from-filename, trust-client-MIME, accept-SVG, **serve without `nosniff`** |
104
105
  | schemas | data models + **privileged fields** | Pydantic/SQLAlchemy/Django/Prisma/Mongoose/TypeORM/Zod → `role`/`isAdmin`/`groupId` for mass-assignment targeting |
105
106
  | iac_ci | IaC + CI/CD | GHA injection (**run:-position-aware**), unpinned actions, tfstate, CDK AppSync `API_KEY` anonymous-default-auth, **docker-compose host-takeover (docker.sock / pid:host / privileged) + `.gitleaksignore` secret-suppression audit** |
106
107
  | client_exposure | browser leakage | public-var secrets by **name + value-shape (`da2-…`) + CDK build-injection**, server-secret-in-client, source maps |
107
108
  | **client_integrity** | tamperable display (client trust boundary) + **WS auth model** | any security-critical sink value (address/IBAN/2FA-seed/API-key/webhook) the user reads or copies, without strict CSP / out-of-band anchor **+ client-tamper-vector, grindable-fingerprint, over-claimed-control, the CSWSH determinant** |
108
- | **transport_security** | CSP + HSTS header baseline | missing/weak CSP, inline event handlers, **partial HSTS (set on /api but not the HTML page)** |
109
+ | **transport_security** | CSP + HSTS + **CORS** + **SRI** baseline | missing/weak CSP, inline event handlers, partial HSTS, **CORS reflect-origin+credentials, external script without SRI, monorepo `next.config` header gap** |
109
110
  | **pii_exposure** | unmasked PII at the output boundary | `res.json(rawEntity)` with PII + **a masking control defined but with zero live call sites** (value-shape, not field-name) |
110
111
  | graphql | GraphQL surface | introspection (**AppSync `introspectionConfig: DISABLED`-aware**) / playground / depth-limit **+ AppSync subscription-authz (cross-group BOLA)** |
111
112
  | integrations | third-party + webhooks **+ outbound-action endpoints** | unsigned webhooks **+ email/SMS/push handlers with no auth or IP-only rate-limit + redundant secret-fetch** |
@@ -218,7 +219,9 @@ publisher** with project `websec-validator`, owner `raccioly`, repo `websec-vali
218
219
 
219
220
  ## Status / roadmap
220
221
 
221
- **Done:** 18-extractor recon (incl. schema/entity → mass-assignment targeting, the **AWS-CDK /
222
+ **Done:** 19-extractor recon (incl. an **authz-correctness data-flow extractor** — unsigned-cookie /
223
+ claim-keyed authz / transaction-local RLS — plus **CORS-misconfig**, **SRI**, **host-header
224
+ open-redirect** and **SSRF-redirect-hardening** classes, schema/entity → mass-assignment targeting, the **AWS-CDK /
222
225
  managed-AppSync / VTL boundary**, **upload-security** + **PII-output-boundary** + **redirect-SSRF**
223
226
  + **password-reuse** classes, a **man-in-the-browser / tamperable-display** class, an **LLM / AI-agent
224
227
  extractor** (OWASP LLM Top 10 — prompt injection / insecure output / excessive agency / unbounded
@@ -23,6 +23,7 @@ src/websec_validator.egg-info/top_level.txt
23
23
  src/websec_validator/extractors/__init__.py
24
24
  src/websec_validator/extractors/auth.py
25
25
  src/websec_validator/extractors/authz.py
26
+ src/websec_validator/extractors/authz_dataflow.py
26
27
  src/websec_validator/extractors/base.py
27
28
  src/websec_validator/extractors/client_exposure.py
28
29
  src/websec_validator/extractors/client_integrity.py
@@ -32,6 +32,7 @@ from websec_validator.extractors.iac_ci import IacCiExtractor # noqa: E
32
32
  from websec_validator.extractors.transport_security import TransportSecurityExtractor # noqa: E402
33
33
  from websec_validator.extractors.llm_security import LlmSecurityExtractor # noqa: E402
34
34
  from websec_validator.extractors.crypto_usage import CryptoUsageExtractor # noqa: E402
35
+ from websec_validator.extractors.authz_dataflow import AuthzDataflowExtractor # noqa: E402
35
36
 
36
37
  FIX = Path(__file__).resolve().parent / "fixtures"
37
38
 
@@ -848,5 +849,118 @@ class Wave2bDetectorTests(unittest.TestCase):
848
849
  self.assertNotIn("timing-unsafe-compare", self._kinds(out))
849
850
 
850
851
 
852
+ class Wave3DeferredDetectorTests(unittest.TestCase):
853
+ """0.8.0 deferred-backlog detectors: CORS, Next-config headers, SRI, host-redirect,
854
+ follows-redirect, unsigned-cookie authz, claim-authz, transaction-local RLS."""
855
+
856
+ # --- transport: CORS / SRI / next-config ---
857
+ def _transport(self, files, frameworks=("next", "express")):
858
+ d = Path(tempfile.mkdtemp())
859
+ for name, body in files.items():
860
+ p = d / name
861
+ p.parent.mkdir(parents=True, exist_ok=True)
862
+ p.write_text(body)
863
+ return TransportSecurityExtractor().extract(
864
+ RepoContext(d), {"stack": {"frameworks": list(frameworks)}, "routes": {"endpoints": [{"method": "GET", "path": "/x"}]}})
865
+
866
+ def _tkinds(self, out):
867
+ return {f["kind"] for f in out["findings"]}
868
+
869
+ def test_cors_reflect_with_credentials_high(self):
870
+ out = self._transport({"srv.ts": "res.setHeader('Access-Control-Allow-Origin', req.headers.origin);\n"
871
+ "res.setHeader('Access-Control-Allow-Credentials', 'true');"})
872
+ hi = [f for f in out["findings"] if f["kind"] == "cors-credentials-any-origin"]
873
+ self.assertTrue(hi)
874
+ self.assertEqual(hi[0]["severity"], "HIGH")
875
+
876
+ def test_cors_allowlist_not_flagged(self):
877
+ out = self._transport({"srv.ts": "app.use(cors({ origin: ['https://app.example.com'], credentials: true }));"})
878
+ self.assertNotIn("cors-credentials-any-origin", self._tkinds(out))
879
+
880
+ def test_external_script_without_sri_flagged(self):
881
+ out = self._transport({"docs.ts": "res.send(`<html><script src=\"https://cdn.jsdelivr.net/npm/x/y.js\"></script></html>`);"})
882
+ self.assertIn("external-script-no-sri", self._tkinds(out))
883
+
884
+ def test_external_script_with_sri_not_flagged(self):
885
+ out = self._transport({"docs.ts": "res.send(`<html><script src=\"https://cdn/x.js\" integrity=\"sha384-abc\"></script></html>`);"})
886
+ self.assertNotIn("external-script-no-sri", self._tkinds(out))
887
+
888
+ def test_nextjs_config_no_headers_flagged(self):
889
+ out = self._transport({"packages/web/next.config.ts": "export default { reactStrictMode: true };"})
890
+ self.assertIn("nextjs-no-security-headers", self._tkinds(out))
891
+
892
+ def test_nextjs_config_with_headers_not_flagged(self):
893
+ out = self._transport({"packages/web/next.config.ts":
894
+ "export default { async headers(){ return [{ source:'/(.*)', headers:["
895
+ "{key:'Content-Security-Policy',value:\"default-src 'self'\"},"
896
+ "{key:'X-Frame-Options',value:'DENY'}]}]; } };"})
897
+ ks = self._tkinds(out)
898
+ self.assertNotIn("nextjs-no-security-headers", ks)
899
+ self.assertNotIn("nextjs-partial-security-headers", ks)
900
+
901
+ # --- surface: host-redirect / follows-redirect ---
902
+ def _surface(self, files, ds=("postgres",)):
903
+ d = Path(tempfile.mkdtemp())
904
+ for name, body in files.items():
905
+ p = d / name
906
+ p.parent.mkdir(parents=True, exist_ok=True)
907
+ p.write_text(body)
908
+ return SurfaceExtractor().extract(RepoContext(d), {"stack": {"datastores": list(ds)}})
909
+
910
+ def test_host_header_redirect_flagged(self):
911
+ out = self._surface({"h.ts": "const host = req.headers['x-forwarded-host'];\n"
912
+ "return res.redirect(`https://${host}/next`);"})
913
+ self.assertEqual(len(out["host_header_redirect"]), 1)
914
+
915
+ def test_host_header_redirect_with_allowlist_not_flagged(self):
916
+ out = self._surface({"h.ts": "const host = req.headers['x-forwarded-host'];\n"
917
+ "const base = allowedHosts.includes(host) ? host : publicWebOrigin;\n"
918
+ "return res.redirect(`https://${base}/next`);"})
919
+ self.assertEqual(out["host_header_redirect"], [])
920
+
921
+ def test_follows_redirects_no_allowlist_flagged(self):
922
+ out = self._surface({"job.py": "r = httpx.get(url, follow_redirects=True)"})
923
+ self.assertEqual(len(out["follows_redirect_no_allowlist"]), 1)
924
+
925
+ def test_follows_redirects_disabled_not_flagged(self):
926
+ out = self._surface({"job.py": "r = httpx.get(url, follow_redirects=False)"})
927
+ self.assertEqual(out["follows_redirect_no_allowlist"], [])
928
+
929
+ # --- authz_dataflow: cookie-authz / claim-authz / rls-context ---
930
+ def _adf(self, files):
931
+ d = Path(tempfile.mkdtemp())
932
+ for name, body in files.items():
933
+ d.joinpath(name).write_text(body)
934
+ return AuthzDataflowExtractor().extract(RepoContext(d), {})
935
+
936
+ def _akinds(self, out):
937
+ return {f["kind"] for f in out["findings"]}
938
+
939
+ def test_unsigned_cookie_authz_flagged(self):
940
+ out = self._adf({"mw.ts": "const lvl = req.cookies.get('twin-access-level');\n"
941
+ "if (lvl !== 'full') return new Response('forbidden', { status: 403 });"})
942
+ self.assertIn("unsigned-cookie-authz", self._akinds(out))
943
+
944
+ def test_cookie_authz_with_jwt_verify_not_flagged(self):
945
+ out = self._adf({"mw.ts": "const lvl = req.cookies.get('twin-access-level');\n"
946
+ "const ok = await jwtVerify(req.cookies.get('twin-token'), key);\n"
947
+ "if (lvl !== 'full') return new Response('forbidden', { status: 403 });"})
948
+ self.assertNotIn("unsigned-cookie-authz", self._akinds(out))
949
+
950
+ def test_claim_based_authz_flagged(self):
951
+ out = self._adf({"acl.ts": "export function canAccessDocument(ctx, doc){ "
952
+ "return ctx.tenantOffice === doc.authorOffice; }"})
953
+ self.assertIn("claim-based-authz", self._akinds(out))
954
+
955
+ def test_rls_context_no_transaction_flagged(self):
956
+ out = self._adf({"rls.ts": "await db.execute(sql`SELECT set_config('app.user_id', ${userId}, true)`);"})
957
+ self.assertIn("rls-context-no-transaction", self._akinds(out))
958
+
959
+ def test_rls_context_in_transaction_not_flagged(self):
960
+ out = self._adf({"rls.ts": "await db.transaction(async (tx) => { "
961
+ "await tx.execute(sql`SELECT set_config('app.user_id', ${userId}, true)`); });"})
962
+ self.assertNotIn("rls-context-no-transaction", self._akinds(out))
963
+
964
+
851
965
  if __name__ == "__main__":
852
966
  unittest.main()