websec-validator 0.6.3__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. {websec_validator-0.6.3/src/websec_validator.egg-info → websec_validator-0.7.0}/PKG-INFO +16 -9
  2. {websec_validator-0.6.3 → websec_validator-0.7.0}/README.md +15 -8
  3. {websec_validator-0.6.3 → websec_validator-0.7.0}/pyproject.toml +1 -1
  4. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/briefing.py +13 -0
  5. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/__init__.py +4 -0
  6. websec_validator-0.7.0/src/websec_validator/extractors/authz.py +391 -0
  7. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/base.py +37 -0
  8. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/client_exposure.py +54 -7
  9. websec_validator-0.7.0/src/websec_validator/extractors/crypto_usage.py +107 -0
  10. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/iac_ci.py +106 -4
  11. websec_validator-0.7.0/src/websec_validator/extractors/llm_security.py +175 -0
  12. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/pii_exposure.py +27 -13
  13. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/routes.py +1 -1
  14. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/surface.py +49 -5
  15. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/transport_security.py +7 -1
  16. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/upload_security.py +15 -6
  17. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/findings.py +88 -1
  18. {websec_validator-0.6.3 → websec_validator-0.7.0/src/websec_validator.egg-info}/PKG-INFO +16 -9
  19. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator.egg-info/SOURCES.txt +2 -0
  20. {websec_validator-0.6.3 → websec_validator-0.7.0}/tests/test_recon.py +335 -0
  21. websec_validator-0.6.3/src/websec_validator/extractors/authz.py +0 -196
  22. {websec_validator-0.6.3 → websec_validator-0.7.0}/LICENSE +0 -0
  23. {websec_validator-0.6.3 → websec_validator-0.7.0}/setup.cfg +0 -0
  24. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/__init__.py +0 -0
  25. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/calibration.json +0 -0
  26. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/calibration.py +0 -0
  27. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/cli.py +0 -0
  28. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/constitution.py +0 -0
  29. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/corpus.json +0 -0
  30. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/dynamic.py +0 -0
  31. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/auth.py +0 -0
  32. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/client_integrity.py +0 -0
  33. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/graphql.py +0 -0
  34. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/integrations.py +0 -0
  35. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/policy_consistency.py +0 -0
  36. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/schemas.py +0 -0
  37. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/stack.py +0 -0
  38. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/extractors/tenant.py +0 -0
  39. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/probes.py +0 -0
  40. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/proof.py +0 -0
  41. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/recon.py +0 -0
  42. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/report.py +0 -0
  43. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/rules/error-stack-disclosure.yml +0 -0
  44. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/rules/insecure-default-secret.yml +0 -0
  45. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/scanners.py +0 -0
  46. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/_lib.py +0 -0
  47. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/appsync-cswsh.sh +0 -0
  48. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/appsync-introspection.sh +0 -0
  49. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/appsync-subscription-bola.sh +0 -0
  50. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/bola-cross-tenant.sh +0 -0
  51. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/bola-write-verbs.py +0 -0
  52. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/client-integrity-checklist.sh +0 -0
  53. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/compare-roles.sh +0 -0
  54. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/dlp-bypass-offline.py +0 -0
  55. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/error-disclosure-probe.sh +0 -0
  56. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/forged-token.sh +0 -0
  57. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/hs256-brute-force.py +0 -0
  58. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/jwt-attacks.sh +0 -0
  59. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/mass-assignment.py +0 -0
  60. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/password-reuse.sh +0 -0
  61. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/pii-output-diff.sh +0 -0
  62. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/race-conditions.py +0 -0
  63. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/rate-limit-burst.sh +0 -0
  64. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/s3-assess.sh +0 -0
  65. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/ssrf-probes.sh +0 -0
  66. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/unauth-baseline.sh +0 -0
  67. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/upload-matrix.sh +0 -0
  68. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/probes/webhook-forgery.py +0 -0
  69. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/reports/FINDINGS-SUMMARY.md.template +0 -0
  70. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/reports/access-control-matrix.md.template +0 -0
  71. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/reports/findings-triage.md.template +0 -0
  72. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/reports/pentest-handover-brief.md.template +0 -0
  73. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator/templates/reports/per-tool-FINDINGS.md.template +0 -0
  74. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator.egg-info/dependency_links.txt +0 -0
  75. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator.egg-info/entry_points.txt +0 -0
  76. {websec_validator-0.6.3 → websec_validator-0.7.0}/src/websec_validator.egg-info/top_level.txt +0 -0
  77. {websec_validator-0.6.3 → websec_validator-0.7.0}/tests/test_hardening.py +0 -0
  78. {websec_validator-0.6.3 → websec_validator-0.7.0}/tests/test_pentest_regressions.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: websec-validator
3
- Version: 0.6.3
3
+ Version: 0.7.0
4
4
  Summary: Defensive, local-first security recon that briefs your AI coding agent on your own codebase — read-only by default (code in, artifacts out): facts + tailored probe scripts, no LLM, no server, no running app.
5
5
  Author: Ricardo Accioly
6
6
  License: MIT
@@ -89,7 +89,7 @@ Then point your agent at the output: **"Read `websec-out/AGENT-BRIEFING.md` and
89
89
 
90
90
  > That's the whole user surface: **`run`** (plus the optional, advanced **`dynamic`** live-probing step below). `recon`/`proof`/`calibrate` exist for developing the tool itself and are hidden from `--help` — you never need them.
91
91
 
92
- ## What it extracts (16 deterministic extractors, no LLM)
92
+ ## What it extracts (18 deterministic extractors, no LLM)
93
93
 
94
94
  | | Dimension | Notable output |
95
95
  |---|---|---|
@@ -102,13 +102,15 @@ Then point your agent at the output: **"Read `websec-out/AGENT-BRIEFING.md` and
102
102
  | surface | 15 sink classes **+ redirect-SSRF** | user-input-gated sinks (incl. **mass-assignment via object spread**) + var-arg SSRF + error-disclosure **+ follows-redirects-without-per-hop-guard** |
103
103
  | **upload_security** | unrestricted upload + unsafe serve | deny-list-only, stored-name-from-filename, trust-client-MIME, accept-SVG, **serve without `nosniff`** |
104
104
  | schemas | data models + **privileged fields** | Pydantic/SQLAlchemy/Django/Prisma/Mongoose/TypeORM/Zod → `role`/`isAdmin`/`groupId` for mass-assignment targeting |
105
- | iac_ci | IaC + CI/CD | GHA injection, unpinned actions, tfstate, **CDK AppSync `API_KEY` anonymous-default-auth + WAF-as-control smell** |
105
+ | iac_ci | IaC + CI/CD | GHA injection (**run:-position-aware**), unpinned actions, tfstate, CDK AppSync `API_KEY` anonymous-default-auth, **docker-compose host-takeover (docker.sock / pid:host / privileged) + `.gitleaksignore` secret-suppression audit** |
106
106
  | client_exposure | browser leakage | public-var secrets by **name + value-shape (`da2-…`) + CDK build-injection**, server-secret-in-client, source maps |
107
107
  | **client_integrity** | tamperable display (client trust boundary) + **WS auth model** | any security-critical sink value (address/IBAN/2FA-seed/API-key/webhook) the user reads or copies, without strict CSP / out-of-band anchor **+ client-tamper-vector, grindable-fingerprint, over-claimed-control, the CSWSH determinant** |
108
108
  | **transport_security** | CSP + HSTS header baseline | missing/weak CSP, inline event handlers, **partial HSTS (set on /api but not the HTML page)** |
109
109
  | **pii_exposure** | unmasked PII at the output boundary | `res.json(rawEntity)` with PII + **a masking control defined but with zero live call sites** (value-shape, not field-name) |
110
110
  | graphql | GraphQL surface | introspection (**AppSync `introspectionConfig: DISABLED`-aware**) / playground / depth-limit **+ AppSync subscription-authz (cross-group BOLA)** |
111
111
  | integrations | third-party + webhooks **+ outbound-action endpoints** | unsigned webhooks **+ email/SMS/push handlers with no auth or IP-only rate-limit + redundant secret-fetch** |
112
+ | **llm_security** | LLM / AI-agent surface (**OWASP LLM Top 10**) | indirect **prompt injection** (untrusted RAG/tool content → prompt) · **insecure output handling** (model text → tool dispatch) · **excessive agency** · **unbounded generation** (no maxTokens/timeout) · **guardrail fail-open** |
113
+ | **crypto_usage** | crypto-API correctness | **weak password hash** (fast/unsalted SHA-256/MD5) · `jwtVerify` without an `algorithms` allowlist · **predictable principal** (id = hash of email) · non-constant-time secret compare |
112
114
 
113
115
  Plus **derived targeting** — IDOR / SSRF / open-redirect / upload / write / auth-endpoint
114
116
  candidates — so probes get pointed at the *exact* endpoints, not fired blindly.
@@ -216,15 +218,20 @@ publisher** with project `websec-validator`, owner `raccioly`, repo `websec-vali
216
218
 
217
219
  ## Status / roadmap
218
220
 
219
- **Done:** 15-extractor recon (incl. schema/entity → mass-assignment targeting, the **AWS-CDK /
221
+ **Done:** 18-extractor recon (incl. schema/entity → mass-assignment targeting, the **AWS-CDK /
220
222
  managed-AppSync / VTL boundary**, **upload-security** + **PII-output-boundary** + **redirect-SSRF**
221
- + **password-reuse** classes, and a **man-in-the-browser / tamperable-display** class), cross-tool
222
- de-dup + **bundled Semgrep rules**, tailored probe staging, agent briefing, traceable findings ledger
223
- with **calibrated confidence (CJE — Wilson CIs)**, proof harness, test suite, **Docker bundle** (all
223
+ + **password-reuse** classes, a **man-in-the-browser / tamperable-display** class, an **LLM / AI-agent
224
+ extractor** (OWASP LLM Top 10 — prompt injection / insecure output / excessive agency / unbounded
225
+ generation / guardrail fail-open), a **crypto-usage extractor** (weak password hash / jwtVerify-without-
226
+ algorithms / predictable principal), **docker-compose host-takeover** + **`.gitleaksignore`
227
+ secret-suppression** audits, and a **reverse-proxy prefix-escape** detector), cross-tool de-dup +
228
+ **bundled Semgrep rules**, **router-mount-auth modeling** (cuts the dominant Express-monorepo
229
+ missing-auth false positive), tailored probe staging, agent briefing, traceable findings ledger with
230
+ **calibrated confidence (CJE — Wilson CIs)**, proof harness, test suite (181), **Docker bundle** (all
224
231
  scanners + Noir, arch-aware), **dynamic phase v1** (authenticated read-only cross-tenant BOLA —
225
232
  validated live, reproduced a hand-pentest's 14/14). Validated against the **REF-PENTEST pen test +
226
- retest** (incl. correcting two findings the retest disproved: AppSync introspection *is* disablable
227
- engine-level, and API_KEY-default is anonymous-auth, not CSWSH).
233
+ retest** and re-validated on a large real-world LLM-agent monorepo (HIGH-finding noise 178 → 15, AI +
234
+ crypto surfaces newly covered).
228
235
  **Next:** dynamic write-verb BOLA + JWT/auth probes + ZAP/Nuclei two-role diff (gated, they mutate),
229
236
  calibration on hand-labeled real repos (more representative base rate), ASVS index lookup, optional
230
237
  model-SDK adapters for no-agent fallback.
@@ -77,7 +77,7 @@ Then point your agent at the output: **"Read `websec-out/AGENT-BRIEFING.md` and
77
77
 
78
78
  > That's the whole user surface: **`run`** (plus the optional, advanced **`dynamic`** live-probing step below). `recon`/`proof`/`calibrate` exist for developing the tool itself and are hidden from `--help` — you never need them.
79
79
 
80
- ## What it extracts (16 deterministic extractors, no LLM)
80
+ ## What it extracts (18 deterministic extractors, no LLM)
81
81
 
82
82
  | | Dimension | Notable output |
83
83
  |---|---|---|
@@ -90,13 +90,15 @@ Then point your agent at the output: **"Read `websec-out/AGENT-BRIEFING.md` and
90
90
  | surface | 15 sink classes **+ redirect-SSRF** | user-input-gated sinks (incl. **mass-assignment via object spread**) + var-arg SSRF + error-disclosure **+ follows-redirects-without-per-hop-guard** |
91
91
  | **upload_security** | unrestricted upload + unsafe serve | deny-list-only, stored-name-from-filename, trust-client-MIME, accept-SVG, **serve without `nosniff`** |
92
92
  | schemas | data models + **privileged fields** | Pydantic/SQLAlchemy/Django/Prisma/Mongoose/TypeORM/Zod → `role`/`isAdmin`/`groupId` for mass-assignment targeting |
93
- | iac_ci | IaC + CI/CD | GHA injection, unpinned actions, tfstate, **CDK AppSync `API_KEY` anonymous-default-auth + WAF-as-control smell** |
93
+ | iac_ci | IaC + CI/CD | GHA injection (**run:-position-aware**), unpinned actions, tfstate, CDK AppSync `API_KEY` anonymous-default-auth, **docker-compose host-takeover (docker.sock / pid:host / privileged) + `.gitleaksignore` secret-suppression audit** |
94
94
  | client_exposure | browser leakage | public-var secrets by **name + value-shape (`da2-…`) + CDK build-injection**, server-secret-in-client, source maps |
95
95
  | **client_integrity** | tamperable display (client trust boundary) + **WS auth model** | any security-critical sink value (address/IBAN/2FA-seed/API-key/webhook) the user reads or copies, without strict CSP / out-of-band anchor **+ client-tamper-vector, grindable-fingerprint, over-claimed-control, the CSWSH determinant** |
96
96
  | **transport_security** | CSP + HSTS header baseline | missing/weak CSP, inline event handlers, **partial HSTS (set on /api but not the HTML page)** |
97
97
  | **pii_exposure** | unmasked PII at the output boundary | `res.json(rawEntity)` with PII + **a masking control defined but with zero live call sites** (value-shape, not field-name) |
98
98
  | graphql | GraphQL surface | introspection (**AppSync `introspectionConfig: DISABLED`-aware**) / playground / depth-limit **+ AppSync subscription-authz (cross-group BOLA)** |
99
99
  | integrations | third-party + webhooks **+ outbound-action endpoints** | unsigned webhooks **+ email/SMS/push handlers with no auth or IP-only rate-limit + redundant secret-fetch** |
100
+ | **llm_security** | LLM / AI-agent surface (**OWASP LLM Top 10**) | indirect **prompt injection** (untrusted RAG/tool content → prompt) · **insecure output handling** (model text → tool dispatch) · **excessive agency** · **unbounded generation** (no maxTokens/timeout) · **guardrail fail-open** |
101
+ | **crypto_usage** | crypto-API correctness | **weak password hash** (fast/unsalted SHA-256/MD5) · `jwtVerify` without an `algorithms` allowlist · **predictable principal** (id = hash of email) · non-constant-time secret compare |
100
102
 
101
103
  Plus **derived targeting** — IDOR / SSRF / open-redirect / upload / write / auth-endpoint
102
104
  candidates — so probes get pointed at the *exact* endpoints, not fired blindly.
@@ -204,15 +206,20 @@ publisher** with project `websec-validator`, owner `raccioly`, repo `websec-vali
204
206
 
205
207
  ## Status / roadmap
206
208
 
207
- **Done:** 15-extractor recon (incl. schema/entity → mass-assignment targeting, the **AWS-CDK /
209
+ **Done:** 18-extractor recon (incl. schema/entity → mass-assignment targeting, the **AWS-CDK /
208
210
  managed-AppSync / VTL boundary**, **upload-security** + **PII-output-boundary** + **redirect-SSRF**
209
- + **password-reuse** classes, and a **man-in-the-browser / tamperable-display** class), cross-tool
210
- de-dup + **bundled Semgrep rules**, tailored probe staging, agent briefing, traceable findings ledger
211
- with **calibrated confidence (CJE — Wilson CIs)**, proof harness, test suite, **Docker bundle** (all
211
+ + **password-reuse** classes, a **man-in-the-browser / tamperable-display** class, an **LLM / AI-agent
212
+ extractor** (OWASP LLM Top 10 — prompt injection / insecure output / excessive agency / unbounded
213
+ generation / guardrail fail-open), a **crypto-usage extractor** (weak password hash / jwtVerify-without-
214
+ algorithms / predictable principal), **docker-compose host-takeover** + **`.gitleaksignore`
215
+ secret-suppression** audits, and a **reverse-proxy prefix-escape** detector), cross-tool de-dup +
216
+ **bundled Semgrep rules**, **router-mount-auth modeling** (cuts the dominant Express-monorepo
217
+ missing-auth false positive), tailored probe staging, agent briefing, traceable findings ledger with
218
+ **calibrated confidence (CJE — Wilson CIs)**, proof harness, test suite (181), **Docker bundle** (all
212
219
  scanners + Noir, arch-aware), **dynamic phase v1** (authenticated read-only cross-tenant BOLA —
213
220
  validated live, reproduced a hand-pentest's 14/14). Validated against the **REF-PENTEST pen test +
214
- retest** (incl. correcting two findings the retest disproved: AppSync introspection *is* disablable
215
- engine-level, and API_KEY-default is anonymous-auth, not CSWSH).
221
+ retest** and re-validated on a large real-world LLM-agent monorepo (HIGH-finding noise 178 → 15, AI +
222
+ crypto surfaces newly covered).
216
223
  **Next:** dynamic write-verb BOLA + JWT/auth probes + ZAP/Nuclei two-role diff (gated, they mutate),
217
224
  calibration on hand-labeled real repos (more representative base rate), ASVS index lookup, optional
218
225
  model-SDK adapters for no-agent fallback.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "websec-validator"
7
- version = "0.6.3"
7
+ version = "0.7.0"
8
8
  description = "Defensive, local-first security recon that briefs your AI coding agent on your own codebase — read-only by default (code in, artifacts out): facts + tailored probe scripts, no LLM, no server, no running app."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11"
@@ -82,6 +82,16 @@ def render(facts: dict, scanners: dict, scan_results: list, probe_manifest: list
82
82
  pii_findings = pii.get("findings", [])
83
83
  pii_section = ("\n".join(f"- **{f.get('severity')}** {f.get('kind')} — `{f.get('file')}`" for f in pii_findings[:20])
84
84
  if pii_findings else "_no obvious raw-PII responses / dead masking controls_")
85
+ llm = facts.get("llm_security", {})
86
+ llm_findings = llm.get("findings", [])
87
+ if llm.get("is_ai_app"):
88
+ llm_section = ("\n".join(f"- **{f.get('severity')}** {f.get('kind')} — `{f.get('file')}`" for f in llm_findings[:25])
89
+ if llm_findings else "_LLM call sites present; no obvious prompt-injection / unbounded / "
90
+ "insecure-output tells — verify the agentic surface by hand_")
91
+ if len(llm_findings) > 25:
92
+ llm_section += f"\n- _…and {len(llm_findings) - 25} more (see FACTS.json)_"
93
+ else:
94
+ llm_section = "_no direct LLM SDK call sites detected_"
85
95
  ws_line = (facts.get("client_integrity", {}) or {}).get("websocket_auth", "no websocket detected")
86
96
  _cs = (facts.get("transport_security", {}) or {}).get("cookie_security")
87
97
  if _cs:
@@ -231,6 +241,9 @@ Production source maps exposed: {client.get("production_source_maps", False)}
231
241
  **Third-party integrations:** {integ_line}
232
242
  {wh_line}
233
243
 
244
+ **LLM / AI-agent surface (OWASP LLM Top 10 — prompt injection, insecure output, excessive agency, unbounded):**
245
+ {llm_section}
246
+
234
247
  ## 4. Static findings (no running app needed)
235
248
 
236
249
  Scanners available: {avail}
@@ -14,9 +14,11 @@ from .authz import AuthzExtractor
14
14
  from .base import MAX_FILES, Extractor, RepoContext
15
15
  from .client_exposure import ClientExposureExtractor
16
16
  from .client_integrity import ClientIntegrityExtractor
17
+ from .crypto_usage import CryptoUsageExtractor
17
18
  from .graphql import GraphQLExtractor
18
19
  from .iac_ci import IacCiExtractor
19
20
  from .integrations import IntegrationsExtractor
21
+ from .llm_security import LlmSecurityExtractor
20
22
  from .pii_exposure import PiiExposureExtractor
21
23
  from .policy_consistency import PolicyConsistencyExtractor
22
24
  from .routes import RoutesExtractor
@@ -46,6 +48,8 @@ REGISTRY: list[Extractor] = [
46
48
  PiiExposureExtractor(),
47
49
  GraphQLExtractor(),
48
50
  IntegrationsExtractor(),
51
+ LlmSecurityExtractor(),
52
+ CryptoUsageExtractor(),
49
53
  ]
50
54
 
51
55
 
@@ -0,0 +1,391 @@
1
+ """Authorization extractor — the access-control map (who can reach what).
2
+
3
+ Per your methodology this is the highest-value test. For each endpoint we decide
4
+ whether a guard protects it, using several signals:
5
+ 1. a guard pattern in the handler's own file (incl. `router.use(authenticate)`)
6
+ or a project-specific auth helper (a `getRequest*Auth`-style getter),
7
+ 2. coverage by a Next.js middleware matcher (incl. monorepo `proxy.ts`),
8
+ 3. a GLOBAL auth middleware (`app.use(authenticate)`) — when present, routes are
9
+ protected by default and "no visible guard" becomes a *verify* signal,
10
+ 4. ROUTER-MOUNT auth: `app.use('/prefix', authMiddleware(...), createXRouter())` —
11
+ resolve the mounted router factory to its file and BFS the local-import graph to
12
+ mark every composed sub-router guarded (this is what inflated the count on the
13
+ Express monorepo: auth lives at the mount, not in the handler file),
14
+ 5. a one-hop delegated guard for thin Next route handlers (`route.ts` → `./proxy`).
15
+
16
+ File-level heuristic → results are HINTS the agent confirms. The high-signal
17
+ output is write endpoints with no visible guard that also don't look public.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import posixpath
23
+ import re
24
+ from pathlib import Path
25
+
26
+ from .base import Extractor, RepoContext, is_client_file, is_test_file
27
+
28
+ WRITE_VERBS = {"POST", "PUT", "PATCH", "DELETE"}
29
+
30
+ # endpoint_guards feeds the missing-auth ledger (findings.build_ledger), so capping it low was a
31
+ # silent coverage cliff: a big monorepo's unguarded write #401 never became a finding. Raised to
32
+ # cover realistic monorepos; truncation beyond this is DISCLOSED (endpoint_guards_truncated), never
33
+ # silent — mirrors constitution.py's "…and N more" pattern.
34
+ _MAX_ENDPOINT_GUARDS = 5000
35
+
36
+ GUARD = re.compile(
37
+ r"requireAuth|requirePermission|requireRole|requireGroupAccess|isAuthenticated|"
38
+ r"@login_required|@jwt_required|@permission_required|@roles_required|ensureAuth|"
39
+ r"withAuth|getServerSession|getToken\s*\(|verifyToken|authMiddleware|@UseGuards|"
40
+ r"@Roles\b|Depends\s*\(\s*(?:get_current_user|oauth2_scheme|require_)|Security\s*\(|"
41
+ r"PermissionRequired|LoginRequired|passport\.authenticate|"
42
+ r"\.use\s*\(\s*[\w.]*(?:[Aa]uth|[Vv]erifyToken|[Rr]equire|[Gg]uard|jwt)\w*", re.I)
43
+
44
+ # a global, path-less auth middleware → everything downstream is protected by default
45
+ GLOBAL_AUTH = re.compile(
46
+ r"app\.use\s*\(\s*[\w.]*(?:authenticate|requireAuth|authMiddleware|verifyToken|"
47
+ r"isAuthenticated|jwtMiddleware|ensureAuth)\w*\s*\)", re.I)
48
+
49
+ # Does a Next.js middleware/proxy file actually enforce AUTH (vs. i18n/headers only)?
50
+ # `auth((req)=>…)` / `withAuth` / `req.auth` / getToken / getServerSession / redirect-to-login /
51
+ # a 401 / Clerk / Supabase updateSession all signal a global auth gate.
52
+ MW_AUTH = re.compile(
53
+ r"\bauth\s*\(|withAuth\b|req\.auth\b|getToken\s*\(|getServerSession\s*\(|clerkMiddleware|"
54
+ r"updateSession\s*\(|NextResponse\.redirect\([^)]*(?:login|signin)|status:\s*401|"
55
+ r"['\"]Authentication required['\"]", re.I)
56
+
57
+ PUBLIC_HINT = re.compile(
58
+ r"/(login|logout|register|signup|signin|health|healthz|ping|status|webhooks?|"
59
+ r"public|\.well-known|robots|favicon|sitemap|callback|refresh|csrf|metrics)\b", re.I)
60
+
61
+ ROLE = re.compile(
62
+ r"@Roles\s*\(([^)]*)\)|allowedRoles\s*=\s*\[([^\]]*)\]|"
63
+ r"\b(?:role|roles)\b\s*[!=]==?\s*['\"]([\w:.-]+)['\"]|"
64
+ r"has_?[Rr]ole\s*\(\s*['\"]([\w:.-]+)['\"]|"
65
+ r"authorizeRoles\s*\(([^)]*)\)|permission_required\s*\(\s*['\"]([\w:.-]+)['\"]")
66
+
67
+ # F5: a call to a decoder/parser named "unsafe"/"unverified"/"noVerify"/"skipVerify"
68
+ # (e.g. decodeJwtPayloadUnsafe) — dangerous when its result feeds an auth decision.
69
+ UNSAFE_DECODER = re.compile(r"\b([A-Za-z_]\w*(?:[Uu]nsafe|[Uu]nverified|[Nn]o[Vv]erif\w*|[Ss]kip[Vv]erif\w*)\w*)\s*\(")
70
+ # does this file actually make an auth/identity decision? (so the unsafe decode matters)
71
+ AUTH_CONTEXT = re.compile(
72
+ r"require(?:Auth|Admin|Role|Permission)|isAdmin|authoriz|getToken\s*\(|getServerSession|"
73
+ r"req\.auth\b|currentUser|jwt\.(?:decode|verify)|decodeJwt", re.I)
74
+
75
+ # --- Router-mount auth (the dominant Express monorepo false-positive, validated on a real LLM-agent monorepo)
76
+ # Auth is frequently applied at the MOUNT, not in the handler: `app.use('/api/x', apiAuth, ...,
77
+ # createApiRouter(db))`. The handler files (routes/feature/*.ts) then carry no in-file guard, so the
78
+ # old GLOBAL_AUTH (which only matched a PATH-LESS `app.use(authMiddleware)`) saw nothing and every
79
+ # endpoint was flagged. We instead: (1) find each `.use(...)` mount, (2) decide if its arg list
80
+ # carries an auth middleware, (3) collect the router FACTORIES it mounts, (4) resolve each factory to
81
+ # its defining file, and (5) BFS the local-import graph from there to mark every route-bearing file
82
+ # composed under an authed mount as guarded — without crossing into a separately-UNAUTHed mount.
83
+ USE_CALL = re.compile(r"(\b[\w.]+)\.use\s*\(")
84
+ # Only a TOP-LEVEL app mount (`app.use`/`server.use`) without auth establishes an UNAUTHED router.
85
+ # An inner `router.use('/sub', createSubRouter())` legitimately omits auth because the parent prefix
86
+ # already applied it (Express runs parent middleware before the sub-router) — counting those as
87
+ # unauthed wrongly excluded inherited sub-routers (routes/feature/sub.ts) from coverage.
88
+ APP_RECEIVER = re.compile(r"(?:^|\.)(?:app|server|application|httpServer|expressApp|api)$", re.I)
89
+ # an auth middleware appearing in a mount arg list (a local guard const like `apiAuth`, a factory
90
+ # like `authMiddleware({...})`, or a named guard). Deliberately NOT matching `authenticatedWriteRateLimit`.
91
+ MOUNT_AUTH = re.compile(
92
+ r"\bauthMiddleware\b|\brequireAuth\b|\brequireAdmin\b|\brequireRole\b|\brequirePermission\b|"
93
+ r"\brequireTenantContext\b|\bensureAuth\w*|\bisAuthenticated\b|\bverifyToken\b|\bjwtMiddleware\b|"
94
+ r"\bauthenticate\b|passport\.authenticate|\b[A-Za-z_]\w*Auth\b|\bauth\b\s*[,)]", re.I)
95
+ FACTORY_CALL = re.compile(r"\b((?:create|make|build|register|mount|init|setup|use)\w*(?:Router|Routes)|\w+Router)\s*\(")
96
+ FACTORY_DEF = re.compile(
97
+ r"(?:export\s+)?(?:async\s+)?function\s+((?:create|make|build|register|mount|init|setup)\w*(?:Router|Routes)|\w+Router)\b"
98
+ r"|(?:export\s+)?(?:const|let|var)\s+((?:create|make|build|register|mount|init|setup)\w*(?:Router|Routes)|\w+Router)\s*[:=]")
99
+ IMPORT_REL = re.compile(r"""(?:from\s*|require\s*\(\s*|import\s*\(\s*)['"](\.[^'"]+)['"]""")
100
+ ROUTE_MARK = re.compile(
101
+ r"\.(?:get|post|put|patch|delete|all|options|head)\s*\(|\.route\s*\(|\brouter\b|FastifyInstance"
102
+ r"|@(?:Get|Post|Put|Patch|Delete|Controller)\(", re.I)
103
+ # Project-specific auth helpers the in-file GUARD list misses — a getter that returns an auth state
104
+ # (e.g. Next's `getRequestSessionAuth()` → fail-closed) or a named auth-failure response. Required
105
+ # to clear the packages/web proxy handlers that ARE authenticated. Conservative: name must read as an
106
+ # auth/token/session getter or an explicit auth-failure helper, so a benign util isn't mistaken for one.
107
+ CUSTOM_GUARD = re.compile(
108
+ r"\bget(?:Request)?\w*(?:Auth|Token|Session)\w*\s*\(|\brequire\w*(?:Auth|Token|Session)\w*\s*\("
109
+ r"|\bensure\w*(?:Auth|Session)\w*\s*\(|\bassert\w*(?:Auth|Session)\w*\s*\("
110
+ r"|[A-Za-z]\w*Auth(?:Failure|Required)Response\b", re.I)
111
+ _IMPORT_EXTS = (".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs")
112
+ NEXT_ROUTE = re.compile(r"(?:^|/)route\.[cm]?[jt]sx?$|(?:^|/)api/.*\.[cm]?[jt]sx?$", re.I)
113
+
114
+
115
+ def _call_args(text: str, paren_idx: int) -> str:
116
+ """Return the substring inside the balanced parens of a call whose '(' is at paren_idx."""
117
+ depth, out, i, n = 0, [], paren_idx, len(text)
118
+ while i < n and len(out) < 4000:
119
+ ch = text[i]
120
+ if ch == "(":
121
+ depth += 1
122
+ if depth == 1:
123
+ i += 1
124
+ continue
125
+ elif ch == ")":
126
+ depth -= 1
127
+ if depth == 0:
128
+ return "".join(out)
129
+ if depth >= 1:
130
+ out.append(ch)
131
+ i += 1
132
+ return "".join(out)
133
+
134
+
135
+ def _resolve_import(importer_rel: str, spec: str, rel_set: set) -> "str | None":
136
+ base = posixpath.normpath(posixpath.join(posixpath.dirname(importer_rel), spec))
137
+ cands = [base]
138
+ # TS ESM convention: the import writes a `.js` extension but the file on disk is `.ts`/`.tsx`
139
+ # (`import './sub/feature.js'` → `routes/sub/feature.ts`). Without this the import graph
140
+ # dead-ends at the first hop and mount-auth coverage never reaches the sub-routers.
141
+ m = re.match(r"(.*)\.(?:js|jsx|mjs|cjs)$", base)
142
+ if m:
143
+ cands += [m.group(1) + e for e in (".ts", ".tsx", ".mts", ".cts")] + [m.group(1)]
144
+ for c in cands:
145
+ if c in rel_set:
146
+ return c
147
+ for ext in _IMPORT_EXTS:
148
+ if base + ext in rel_set:
149
+ return base + ext
150
+ for idx in ("/index.ts", "/index.tsx", "/index.js", "/index.jsx", "/index.mjs"):
151
+ if base + idx in rel_set:
152
+ return base + idx
153
+ return None
154
+
155
+
156
+ def _mount_auth_coverage(ctx: RepoContext) -> dict:
157
+ """Find Express router-mount auth and return the set of route files it covers.
158
+
159
+ Returns {covered:set[rel], detected:bool, authed_factories:[...], rel_map, rel_set}.
160
+ """
161
+ rel_map = {ctx.rel(p): p for p in ctx.code_files}
162
+ rel_set = set(rel_map)
163
+
164
+ def txt(rel: str) -> str:
165
+ p = rel_map.get(rel)
166
+ return ctx.text(p) if p else ""
167
+
168
+ authed_factories: set = set()
169
+ unauthed_factories: set = set()
170
+ for rel, p in rel_map.items():
171
+ # Test harnesses wire `app.use(createXRouter())` WITHOUT auth to exercise a router in
172
+ # isolation — that is test scaffolding, not the production mount, and counting it wrongly
173
+ # marked real sub-routers (memory/marketplace-content) as unauthed. Skip test files.
174
+ if is_test_file(rel):
175
+ continue
176
+ text = ctx.text(p)
177
+ if ".use(" not in text:
178
+ continue
179
+ for m in USE_CALL.finditer(text):
180
+ args = _call_args(text, m.end() - 1)
181
+ facs = set(FACTORY_CALL.findall(args))
182
+ if not facs:
183
+ continue
184
+ if MOUNT_AUTH.search(args):
185
+ authed_factories.update(facs)
186
+ elif APP_RECEIVER.search(m.group(1)):
187
+ unauthed_factories.update(facs)
188
+ # else: inner `router.use(...)` with no auth → inherits the parent mount's auth; ignore
189
+ unauthed_factories -= authed_factories # an authed mount anywhere wins
190
+
191
+ factory_file: dict = {} # factory name -> defining file
192
+ for rel, p in rel_map.items():
193
+ if is_test_file(rel):
194
+ continue
195
+ text = ctx.text(p)
196
+ if "Router" not in text and "Routes" not in text:
197
+ continue
198
+ for a, b in FACTORY_DEF.findall(text):
199
+ nm = a or b
200
+ if nm and nm not in factory_file:
201
+ factory_file[nm] = rel
202
+
203
+ authed_files = {factory_file[n] for n in authed_factories if n in factory_file}
204
+ unauthed_files = {factory_file[n] for n in unauthed_factories if n in factory_file} - authed_files
205
+
206
+ # BFS the local-import graph from each authed factory file; mark every route-bearing file
207
+ # composed under it as covered, but never traverse INTO a separately-unauthed mount's file.
208
+ covered: set = set()
209
+ visited: set = set(authed_files)
210
+ frontier: list = list(authed_files)
211
+ # Cap bounds the walk to the repo's own files (the import graph is finite); a big assembler file
212
+ # can have 100+ relative imports, so the per-file cap must be generous or sub-routers past it
213
+ # (e.g. routes/feature/sub.ts) are silently dropped.
214
+ while frontier and len(visited) <= 6000:
215
+ cur = frontier.pop()
216
+ t = txt(cur)
217
+ if not t:
218
+ continue
219
+ if cur in authed_files or ROUTE_MARK.search(t):
220
+ covered.add(cur)
221
+ for spec in IMPORT_REL.findall(t)[:400]:
222
+ tgt = _resolve_import(cur, spec, rel_set)
223
+ if (tgt and tgt not in visited and tgt not in unauthed_files
224
+ and not is_test_file(tgt) and not is_client_file(tgt)):
225
+ visited.add(tgt)
226
+ frontier.append(tgt)
227
+
228
+ return {"covered": covered, "detected": bool(authed_files),
229
+ "authed_factories": sorted(authed_factories)[:40],
230
+ "rel_map": rel_map, "rel_set": rel_set}
231
+
232
+
233
+ def _parse_next_middleware(ctx: RepoContext) -> dict:
234
+ # Next 15.5+/16 renamed `middleware.ts` → `proxy.ts` (both filenames are valid; the
235
+ # framework recognizes either). Missing this made the tool report "no global auth" on
236
+ # Next 16 apps and flag every handler — the single biggest false-positive cluster.
237
+ for cand in ("middleware.ts", "middleware.js", "src/middleware.ts", "src/middleware.js",
238
+ "proxy.ts", "proxy.js", "src/proxy.ts", "src/proxy.js"):
239
+ txt = ctx.manifest(cand)
240
+ if not txt:
241
+ continue
242
+ matchers = re.findall(r"matcher\s*:\s*\[([^\]]*)\]", txt)
243
+ patterns = re.findall(r"['\"]([^'\"]+)['\"]", matchers[0]) if matchers else []
244
+ roles = [m for grp in ROLE.findall(txt) for m in grp if m]
245
+ return {"present": True, "file": cand, "matchers": patterns,
246
+ "is_auth": bool(MW_AUTH.search(txt)), "role_checks": roles}
247
+ return {"present": False, "matchers": [], "is_auth": False}
248
+
249
+
250
+ def _matcher_covers(path: str, matchers: list) -> bool:
251
+ for m in matchers:
252
+ base = m.split(":")[0].split("(")[0].rstrip("/*")
253
+ if base and path.startswith(base):
254
+ return True
255
+ if m.startswith("/(") or m == "/:path*":
256
+ return True
257
+ return False
258
+
259
+
260
+ def _collect_roles(text: str, roles: set) -> None:
261
+ for grp in ROLE.findall(text or ""):
262
+ for m in grp:
263
+ if not m:
264
+ continue
265
+ for part in m.split(","):
266
+ v = part.strip().strip("'\" ")
267
+ if v and len(v) < 40:
268
+ roles.add(v)
269
+
270
+
271
+ class AuthzExtractor(Extractor):
272
+ name = "authz"
273
+ category = "authz"
274
+
275
+ def extract(self, ctx: RepoContext, facts: dict) -> dict:
276
+ endpoints = (facts.get("routes") or {}).get("endpoints", [])
277
+ mw = _parse_next_middleware(ctx)
278
+ mw_auth = mw.get("is_auth", False)
279
+
280
+ # Router-mount auth coverage (Express `app.use('/x', authMiddleware, createXRouter())`).
281
+ mount = _mount_auth_coverage(ctx)
282
+ mount_covered, mount_detected = mount["covered"], mount["detected"]
283
+ rel_map, rel_set = mount["rel_map"], mount["rel_set"]
284
+
285
+ # A THIN Next.js route handler (route.ts / api/*) can delegate its auth one hop to a relative
286
+ # module (e.g. `route.ts` → `./proxy.ts` where getRequestSessionAuth lives). Follow that one
287
+ # hop — bounded to short handler files so the FN risk (marking a real gap as guarded) stays low.
288
+ def _imported_guard(rel: str, text: str) -> bool:
289
+ if not rel or not NEXT_ROUTE.search(rel) or len(text.splitlines()) > 80:
290
+ return False
291
+ for spec in IMPORT_REL.findall(text)[:20]:
292
+ tgt = _resolve_import(rel, spec, rel_set)
293
+ if tgt:
294
+ t2 = ctx.text(rel_map[tgt]) if tgt in rel_map else ""
295
+ if t2 and (GUARD.search(t2) or CUSTOM_GUARD.search(t2)):
296
+ return True
297
+ return False
298
+
299
+ # global auth = an Express path-less auth middleware OR a Next auth middleware/proxy OR a
300
+ # detected router-mount-auth pattern (routes are protected centrally, at the mount).
301
+ global_auth = (mw_auth or mount_detected
302
+ or any(GLOBAL_AUTH.search(t) for _p, _r, t in ctx.iter_code()))
303
+ roles: set = set(mw.get("role_checks", []))
304
+ protected = no_guard = unknown = 0
305
+ no_guard_writes, egs = [], []
306
+
307
+ for e in endpoints:
308
+ cp = e.get("code_path", "")
309
+ text = ctx.text(Path(cp)) if cp else ""
310
+ _collect_roles(text, roles)
311
+ relcp = ctx.rel(Path(cp)) if cp else ""
312
+ # a matcher only counts as a guard when the middleware actually does auth — a
313
+ # non-auth middleware.ts (i18n/headers) must NOT mark routes protected. Mount coverage,
314
+ # the project's custom auth helper, and a one-hop delegated guard also count.
315
+ guarded = (bool(text and (GUARD.search(text) or CUSTOM_GUARD.search(text)))
316
+ or (relcp and relcp in mount_covered)
317
+ or (mw_auth and _matcher_covers(e.get("path", ""), mw.get("matchers", [])))
318
+ or _imported_guard(relcp, text))
319
+ egs.append({"method": e.get("method"), "path": e.get("path"), "code_path": relcp,
320
+ "guarded": bool(guarded), "analyzed": bool(text),
321
+ "public_hint": bool(PUBLIC_HINT.search(e.get("path", "")))})
322
+ if guarded:
323
+ protected += 1
324
+ elif not text:
325
+ unknown += 1
326
+ else:
327
+ no_guard += 1
328
+ if e.get("method") in WRITE_VERBS and not PUBLIC_HINT.search(e.get("path", "")):
329
+ no_guard_writes.append(f"{e['method']} {e['path']} ({relcp or '?'})")
330
+
331
+ # F5: files that make an auth decision AND call an unsafe/unverified decoder
332
+ unsafe_decoders = []
333
+ for _p, rel, text in ctx.iter_code():
334
+ if AUTH_CONTEXT.search(text):
335
+ for dec in sorted(set(UNSAFE_DECODER.findall(text))):
336
+ unsafe_decoders.append({"file": rel, "decoder": dec})
337
+
338
+ # A guard DEFINED in a file that also calls an unsafe/unverified decoder authenticates via
339
+ # an unverified decode. Routes that call such a guard are the static "at-risk" set for the
340
+ # forged-token bypass class — the dynamic probe confirms which actually fall, but this points
341
+ # at them even with NO live target (turns the F5 hypothesis into named routes).
342
+ unverified_routes: list = []
343
+ unsafe_files = {ud["file"] for ud in unsafe_decoders}
344
+ if unsafe_files:
345
+ guard_def = re.compile(r"(?:export\s+)?(?:async\s+)?(?:function|const)\s+"
346
+ r"(require\w+|ensure\w+|\w*[Aa]uth\w*|verify\w+)\b")
347
+ unsafe_guards = set()
348
+ for _p, rel, text in ctx.iter_code():
349
+ if rel in unsafe_files:
350
+ unsafe_guards.update(g for g in guard_def.findall(text) if len(g) >= 5)
351
+ if unsafe_guards:
352
+ call = re.compile(r"\b(?:" + "|".join(re.escape(g) for g in sorted(unsafe_guards)) + r")\s*\(")
353
+ for e in endpoints:
354
+ cp = e.get("code_path", "")
355
+ t = ctx.text(Path(cp)) if cp else ""
356
+ if t and call.search(t):
357
+ unverified_routes.append(f"{e.get('method')} {e.get('path')}")
358
+ unverified_routes = sorted(set(unverified_routes))[:60]
359
+
360
+ if global_auth:
361
+ if mount_detected:
362
+ where = f"router-mount auth (`app.use('/prefix', <auth>, createXRouter())`) on {len(mount_covered)} route file(s)"
363
+ elif mw_auth:
364
+ where = f"`{mw['file']}` (matcher {mw.get('matchers') or '—'})"
365
+ else:
366
+ where = "`app.use(<auth>)`"
367
+ note = (f"A GLOBAL/mount auth pattern ({where}) was detected — most routes are protected centrally, "
368
+ "not in each handler file. Endpoints under an authed mount are reported as guarded. "
369
+ "Any list below is write endpoints with NO guard via the handler file, an authed mount, the "
370
+ "Next matcher, or a one-hop delegated guard; verify each is either covered or an intentional "
371
+ "public exemption (e.g. a signed-token download) — don't assume they're vulnerable.")
372
+ else:
373
+ note = ("No global/mount auth middleware detected. Write endpoints with no visible guard are "
374
+ "high-signal missing-authz leads — verify each.")
375
+
376
+ return {
377
+ "global_auth_middleware": global_auth,
378
+ "mount_auth_detected": mount_detected,
379
+ "mount_authed_factories": mount["authed_factories"],
380
+ "mount_covered_files": len(mount_covered),
381
+ "next_middleware": mw,
382
+ "roles_detected": sorted(r for r in roles if r),
383
+ "guard_summary": {"with_visible_guard": protected,
384
+ "no_visible_guard": no_guard, "unknown": unknown},
385
+ "endpoint_guards": egs[:_MAX_ENDPOINT_GUARDS],
386
+ "endpoint_guards_truncated": max(0, len(egs) - _MAX_ENDPOINT_GUARDS),
387
+ "write_endpoints_without_visible_guard": sorted(set(no_guard_writes))[:60],
388
+ "unsafe_auth_decoders": unsafe_decoders[:30],
389
+ "unverified_signature_routes": unverified_routes,
390
+ "note": note,
391
+ }
@@ -10,6 +10,7 @@ still say something useful.
10
10
  from __future__ import annotations
11
11
 
12
12
  import fnmatch
13
+ import re
13
14
  from pathlib import Path
14
15
 
15
16
  SKIP_DIRS = {".git", "node_modules", "dist", "build", ".next", ".nuxt", "venv",
@@ -57,6 +58,42 @@ def path_in_skip_dir(path: str, root: "Path | str | None" = None) -> bool:
57
58
  return any(part in SKIP_DIRS for part in p.split("/"))
58
59
 
59
60
 
61
+ # --- file-class helpers -------------------------------------------------------------------------
62
+ # Many sink/exposure extractors over-report because iter_code() walks the WHOLE tree — tests,
63
+ # build/CI scripts, and browser code get scanned as if they were deployed server request handlers.
64
+ # A test fixture's fake secret, an `e2e/*.spec.ts` relative fetch, a `scripts/deploy.mjs` outbound
65
+ # call — none are a runtime attack surface. These centralize the classification so every extractor
66
+ # decides the same way (validated against a real LLM-agent monorepo: the dominant client-exposure / ssrf /
67
+ # pii false-positive driver). Each extractor opts in to whichever classes it should skip.
68
+ _TEST_FILE = re.compile(
69
+ r"(?:^|/)(?:tests?|__tests__|__mocks__|spec|specs|e2e|cypress|fixtures?|mocks?|stories|testdata|testing)/"
70
+ r"|\.(?:test|spec|stories|e2e|cy)\.[cm]?[jt]sx?$"
71
+ r"|(?:^|/)[\w.-]*\.config\.[cm]?[jt]sx?$" # vite/vitest/jest/playwright/next/... .config.*
72
+ r"|(?:^|/)(?:playwright|vitest|jest|cypress)\.[\w.]*$", re.I)
73
+ # build / ops / CLI scripts run by an operator or CI, not reachable from an inbound HTTP request.
74
+ _SCRIPT_FILE = re.compile(r"(?:^|/)(?:scripts?|bin|\.bin|ops|operations|migrations?|seeds?)/", re.I)
75
+ # browser / client-side code. SSRF and server-secret-exposure are server-only classes; a `.tsx`
76
+ # React component, a hook, or a `'use client'` module runs in the visitor's browser to the app's OWN
77
+ # origin, so an outbound fetch there is same-origin, not an SSRF/exfil sink.
78
+ _CLIENT_FILE = re.compile(r"\.(?:tsx|jsx)$|(?:^|/)(?:components?|hooks?|contexts?|widgets?|ui)/", re.I)
79
+
80
+
81
+ def is_test_file(rel: str) -> bool:
82
+ return bool(_TEST_FILE.search((rel or "").replace("\\", "/")))
83
+
84
+
85
+ def is_script_file(rel: str) -> bool:
86
+ return bool(_SCRIPT_FILE.search((rel or "").replace("\\", "/")))
87
+
88
+
89
+ def is_client_file(rel: str, text: str = "") -> bool:
90
+ rel = (rel or "").replace("\\", "/")
91
+ if _CLIENT_FILE.search(rel):
92
+ return True
93
+ head = text[:300]
94
+ return "'use client'" in head or '"use client"' in head
95
+
96
+
60
97
  class RepoContext:
61
98
  """Walk the tree once; cache file text; serve cheap queries to every extractor."""
62
99