sourcecode 3.8.0__py3-none-any.whl → 4.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sourcecode/__init__.py +1 -1
- sourcecode/cache.py +48 -0
- sourcecode/cache_model.py +7 -0
- sourcecode/cli.py +111 -9
- sourcecode/confidence_analyzer.py +7 -2
- sourcecode/defect_identity.py +24 -0
- sourcecode/environment_resolution.py +1 -1
- sourcecode/filter_surface.py +243 -12
- sourcecode/identity_fallback.py +244 -0
- sourcecode/migrate_check.py +27 -5
- sourcecode/parse_cache.py +23 -0
- sourcecode/posture.py +57 -4
- sourcecode/remedies.py +87 -0
- sourcecode/risk.py +345 -0
- sourcecode/security_config_scan.py +262 -0
- sourcecode/security_posture.py +216 -23
- sourcecode/spring_findings.py +13 -3
- sourcecode/spring_profiles.py +6 -2
- sourcecode/spring_properties.py +1 -1
- sourcecode/spring_security_audit.py +88 -0
- sourcecode/spring_tx_analyzer.py +42 -0
- sourcecode/summarizer.py +7 -2
- sourcecode/verify_repo.py +8 -0
- {sourcecode-3.8.0.dist-info → sourcecode-4.0.0.dist-info}/METADATA +3 -2
- {sourcecode-3.8.0.dist-info → sourcecode-4.0.0.dist-info}/RECORD +28 -24
- {sourcecode-3.8.0.dist-info → sourcecode-4.0.0.dist-info}/WHEEL +0 -0
- {sourcecode-3.8.0.dist-info → sourcecode-4.0.0.dist-info}/entry_points.txt +0 -0
- {sourcecode-3.8.0.dist-info → sourcecode-4.0.0.dist-info}/licenses/LICENSE +0 -0
sourcecode/__init__.py
CHANGED
sourcecode/cache.py
CHANGED
|
@@ -354,6 +354,7 @@ def status(repo_root: Path) -> dict[str, Any]:
|
|
|
354
354
|
"cores": 0, "snapshots": 0, "views": 0, "cas_blobs": 0,
|
|
355
355
|
"total_size_bytes": 0, "total_size_mb": 0.0,
|
|
356
356
|
"current_git_head": current_head,
|
|
357
|
+
"stores": _store_breakdown(repo_root, 0),
|
|
357
358
|
**ris_fields,
|
|
358
359
|
}
|
|
359
360
|
cores = list(cache_d.glob("core-*.json.gz"))
|
|
@@ -371,10 +372,57 @@ def status(repo_root: Path) -> dict[str, Any]:
|
|
|
371
372
|
"total_size_bytes": total_bytes,
|
|
372
373
|
"total_size_mb": round(total_bytes / (1024 * 1024), 2),
|
|
373
374
|
"current_git_head": current_head,
|
|
375
|
+
# C3-36: `CAS blobs: 0`, `Total size: 0.1 MB` immediately after an 89-second
|
|
376
|
+
# warm of 3 342 files. Every figure above was true and described one store
|
|
377
|
+
# of three — the warm's output mostly lands in the shared CIR and the parse
|
|
378
|
+
# cache, which this command never counted. A status that reports a third of
|
|
379
|
+
# the state reads as a warm that did nothing.
|
|
380
|
+
"stores": _store_breakdown(repo_root, total_bytes),
|
|
374
381
|
**ris_fields,
|
|
375
382
|
}
|
|
376
383
|
|
|
377
384
|
|
|
385
|
+
def _store_breakdown(repo_root: Path, core_bytes: int) -> "dict[str, Any]":
|
|
386
|
+
"""What a warm actually populated, by store. Best-effort per store: a store
|
|
387
|
+
that cannot be inspected is reported as unavailable, never as empty."""
|
|
388
|
+
out: "dict[str, Any]" = {
|
|
389
|
+
"core": {
|
|
390
|
+
"cache_dir": str(cache_dir(repo_root)),
|
|
391
|
+
"bytes": core_bytes,
|
|
392
|
+
"scope": "this repository",
|
|
393
|
+
"holds": "core snapshots, rendered views and their CAS blobs",
|
|
394
|
+
},
|
|
395
|
+
}
|
|
396
|
+
try:
|
|
397
|
+
from sourcecode.context_cache import ContextCache # noqa: PLC0415
|
|
398
|
+
|
|
399
|
+
ctx = ContextCache.for_repo(repo_root).stats()
|
|
400
|
+
out["shared_cir"] = {
|
|
401
|
+
"cache_dir": ctx["cache_dir"],
|
|
402
|
+
"entries": ctx["contexts"],
|
|
403
|
+
"bytes": ctx["bytes_stored"],
|
|
404
|
+
"scope": "this repository",
|
|
405
|
+
"holds": "the shared Canonical IR that explain/impact/posture reuse",
|
|
406
|
+
}
|
|
407
|
+
except Exception:
|
|
408
|
+
out["shared_cir"] = {"available": False}
|
|
409
|
+
try:
|
|
410
|
+
from sourcecode import parse_cache as _pc # noqa: PLC0415
|
|
411
|
+
|
|
412
|
+
out["parse"] = {
|
|
413
|
+
**_pc.store_stats(),
|
|
414
|
+
"holds": "per-file parses, content-addressed across every repository",
|
|
415
|
+
}
|
|
416
|
+
except Exception:
|
|
417
|
+
out["parse"] = {"available": False}
|
|
418
|
+
out["note"] = (
|
|
419
|
+
"`cache warm` populates all of these; the core store alone is a third of "
|
|
420
|
+
"the answer, and reading it as the whole is how a completed warm looks like "
|
|
421
|
+
"an empty cache."
|
|
422
|
+
)
|
|
423
|
+
return out
|
|
424
|
+
|
|
425
|
+
|
|
378
426
|
def clear(repo_root: Path, *, clear_ris: bool = False) -> int:
|
|
379
427
|
"""Delete cache files for *repo_root*. Returns the number of files removed.
|
|
380
428
|
|
sourcecode/cache_model.py
CHANGED
|
@@ -131,6 +131,13 @@ COMMANDS: tuple[CommandCache, ...] = (
|
|
|
131
131
|
"builds — the parse it used to repeat for itself. `--diff` compares two profile "
|
|
132
132
|
"sets over that one IR, so the second side costs the resolution only.",
|
|
133
133
|
"10.1 s → 1.6 s"),
|
|
134
|
+
CommandCache("risk", ("cir", "parse"), "shared", False,
|
|
135
|
+
"Composes what the audit, impact-chain and the posture already answer, so it "
|
|
136
|
+
"pays each of their costs once over the shared CIR a warm builds — one parse "
|
|
137
|
+
"for the whole composition, and the reachability query is cached per symbol "
|
|
138
|
+
"within the run.",
|
|
139
|
+
"not measured on the battery yet — the composition is bounded by the "
|
|
140
|
+
"`spring-audit` + `impact-chain` costs listed here, not by new analysis"),
|
|
134
141
|
CommandCache("endpoints", ("ris", "parse"), "shared", False,
|
|
135
142
|
"Recomputes the endpoint surface on every run, over a parse a warm has already "
|
|
136
143
|
"paid for. Until 3.7.0 the extractor parsed every file itself instead of reading "
|
sourcecode/cli.py
CHANGED
|
@@ -183,7 +183,7 @@ COMMAND_TIERS: "tuple[tuple[str, str, tuple[str, ...]], ...]" = (
|
|
|
183
183
|
"cache", "auth", "mcp", "telemetry",
|
|
184
184
|
)),
|
|
185
185
|
("experimental", "shape may change in a minor — do not gate CI on it", (
|
|
186
|
-
"posture", "archetype",
|
|
186
|
+
"risk", "posture", "archetype",
|
|
187
187
|
)),
|
|
188
188
|
# `retrieve` publishes 15+ intents whose answers the other commands already
|
|
189
189
|
# give better: measured on the battery, `security-surface` merely re-states
|
|
@@ -5475,11 +5475,21 @@ def validation_cmd(
|
|
|
5475
5475
|
# The payload carries the same note, so a pipeline loses nothing.
|
|
5476
5476
|
if _note:
|
|
5477
5477
|
_notice(f"Note: {_note}")
|
|
5478
|
+
# C2-15: the console said "0 body endpoints, 1179 gaps" over a payload carrying
|
|
5479
|
+
# `body_endpoints_in_code: 1201` — it printed the *declared-constraint* count
|
|
5480
|
+
# under the *code-surface* name, so the one line most readers see stated the
|
|
5481
|
+
# opposite of the answer. Both axes are named, in their own units.
|
|
5482
|
+
_declared = _summary.get("endpoints_with_body", 0)
|
|
5483
|
+
_in_code = _summary.get("body_endpoints_in_code")
|
|
5484
|
+
_bodies = (
|
|
5485
|
+
f"{_in_code} body endpoints in code, {_declared} with a declared constraint "
|
|
5486
|
+
f"surface" if _in_code is not None
|
|
5487
|
+
else f"{_declared} routes with a declared constraint surface"
|
|
5488
|
+
)
|
|
5478
5489
|
_emit_command_output(
|
|
5479
5490
|
output, output_path, copy,
|
|
5480
5491
|
success_msg=f"Validation surface written to {output_path} "
|
|
5481
|
-
f"({_summary.get('
|
|
5482
|
-
f"{_summary.get('gaps', 0)} gaps)",
|
|
5492
|
+
f"({_bodies}, {_summary.get('gaps', 0)} gaps)",
|
|
5483
5493
|
)
|
|
5484
5494
|
|
|
5485
5495
|
from sourcecode.mcp_nudge import nudge_mcp_if_needed as _nudge
|
|
@@ -6308,11 +6318,13 @@ def _render_gate_coverage_section(result: "SpringAuditResult") -> list[str]: #
|
|
|
6308
6318
|
gc = (result.security_posture or {}).get("gate_coverage")
|
|
6309
6319
|
if not gc:
|
|
6310
6320
|
return []
|
|
6311
|
-
not_covered = gc.get("endpoints_not_carrying_gate",
|
|
6321
|
+
not_covered = gc.get("endpoints_not_carrying_gate", 0)
|
|
6312
6322
|
gates = ", ".join(f"`{g}`" for g in gc.get("gate_annotations", [])) or "the detected gate"
|
|
6313
6323
|
# One unit, named: these are endpoints, not handler methods. The two counts differ
|
|
6314
6324
|
# (one method can serve several mappings) and must never be summed together.
|
|
6315
|
-
|
|
6325
|
+
# C2-14: read only the explicit keys — falling back to the retired aliases is how
|
|
6326
|
+
# a name that lies about its unit survives its own removal.
|
|
6327
|
+
total = gc.get("endpoints_total", 0)
|
|
6316
6328
|
lines: list[str] = ["", "---", ""]
|
|
6317
6329
|
if not_covered == 0:
|
|
6318
6330
|
_gated = gc.get("endpoints_carrying_gate", total)
|
|
@@ -6326,7 +6338,6 @@ def _render_gate_coverage_section(result: "SpringAuditResult") -> list[str]: #
|
|
|
6326
6338
|
lines.append(f"✅ **Gate coverage** — all {total} endpoints carry {gates}.")
|
|
6327
6339
|
return lines
|
|
6328
6340
|
|
|
6329
|
-
covered = gc.get("possibly_filter_covered", 0)
|
|
6330
6341
|
_standard = gc.get("endpoints_standard_guarded", 0)
|
|
6331
6342
|
lines.append(
|
|
6332
6343
|
f"🔓 **Gate coverage** — {not_covered} of {total} endpoints do not carry "
|
|
@@ -6338,9 +6349,18 @@ def _render_gate_coverage_section(result: "SpringAuditResult") -> list[str]: #
|
|
|
6338
6349
|
f"standard guard, {not_covered} on neither — the three sum to {total}._"
|
|
6339
6350
|
)
|
|
6340
6351
|
if gc.get("reconstructed_filter_patterns"):
|
|
6352
|
+
# C1-20: the split, not the total. "N match a filter pattern" is the sentence
|
|
6353
|
+
# a reader turns into "N are covered", and in the field the filters doing the
|
|
6354
|
+
# matching were CORS, headers and logging.
|
|
6355
|
+
_auth = gc.get("filter_covered_authenticating", 0)
|
|
6356
|
+
_gap = gc.get("filter_gap_non_authenticating", 0)
|
|
6357
|
+
_unknown = gc.get("filter_authentication_unknown", 0)
|
|
6341
6358
|
lines.append(
|
|
6342
|
-
f"_{
|
|
6343
|
-
f"
|
|
6359
|
+
f"_{_auth} are covered by a servlet filter that authenticates; {_gap} match "
|
|
6360
|
+
f"only filters that do not check the caller (CORS, headers, logging and the "
|
|
6361
|
+
f"like); {_unknown} match a filter whose implementation is not in this "
|
|
6362
|
+
f"repository; {gc.get('no_matching_filter_pattern', 0)} match no pattern at "
|
|
6363
|
+
f"all. Filter-chain order is not reconstructed._"
|
|
6344
6364
|
)
|
|
6345
6365
|
lines += ["", "<details>", "<summary>Handlers without the gate</summary>", ""]
|
|
6346
6366
|
lines += [
|
|
@@ -6840,6 +6860,74 @@ def verify_cmd(
|
|
|
6840
6860
|
raise typer.Exit(code=report.exit_code)
|
|
6841
6861
|
|
|
6842
6862
|
|
|
6863
|
+
@app.command("risk")
|
|
6864
|
+
def risk_cmd(
|
|
6865
|
+
path: Path = typer.Argument(
|
|
6866
|
+
Path("."),
|
|
6867
|
+
help="Repository path (default: current directory).",
|
|
6868
|
+
),
|
|
6869
|
+
limit: int = typer.Option(
|
|
6870
|
+
50, "--limit", help="How many composed risks to publish (highest first)."
|
|
6871
|
+
),
|
|
6872
|
+
min_band: str = typer.Option(
|
|
6873
|
+
"low",
|
|
6874
|
+
"--min-band",
|
|
6875
|
+
help="Floor on the composed band: critical | high | medium | low.",
|
|
6876
|
+
),
|
|
6877
|
+
output_path: Optional[Path] = typer.Option(
|
|
6878
|
+
None, "--output", "-o", help="Write the report to a file instead of stdout."
|
|
6879
|
+
),
|
|
6880
|
+
format: str = typer.Option("json", "--format", "-f", help="Output format: json or yaml."),
|
|
6881
|
+
) -> None:
|
|
6882
|
+
"""[EXPERIMENTAL] What each defect actually costs, once reach and access are in it.
|
|
6883
|
+
|
|
6884
|
+
\b
|
|
6885
|
+
The other commands answer one axis each, correctly, and leave the composition
|
|
6886
|
+
to the reader. In the field, one class was `medium` in `spring-audit`,
|
|
6887
|
+
`medium/5.0` in `impact-chain`, `coverage_unknown → permit_all` in `posture`,
|
|
6888
|
+
and executed a stored procedure that mutates the database. Composed, that is
|
|
6889
|
+
unauthenticated write access under the release build; separately, it was two
|
|
6890
|
+
commands saying "medium".
|
|
6891
|
+
|
|
6892
|
+
\b
|
|
6893
|
+
severity_effective = defect_severity × reachability × auth_verdict × write_effect
|
|
6894
|
+
|
|
6895
|
+
\b
|
|
6896
|
+
No new analysis: every factor is read from the command that already publishes
|
|
6897
|
+
it, and every row publishes its four factors with the authority each came
|
|
6898
|
+
from, so a reader can disagree with one and keep the rest. An axis that could
|
|
6899
|
+
not be measured is `unknown`, multiplies by 1.0, and is named in `blind_axes`.
|
|
6900
|
+
|
|
6901
|
+
\b
|
|
6902
|
+
Examples:
|
|
6903
|
+
ask risk .
|
|
6904
|
+
ask risk . --min-band high
|
|
6905
|
+
ask risk . --limit 10 -o risk.json
|
|
6906
|
+
"""
|
|
6907
|
+
from sourcecode.risk import build_risk
|
|
6908
|
+
|
|
6909
|
+
path = _admit_path(path)
|
|
6910
|
+
if min_band not in ("critical", "high", "medium", "low"):
|
|
6911
|
+
_emit_error_json(
|
|
6912
|
+
INVALID_INPUT_CODE,
|
|
6913
|
+
f"--min-band expects critical | high | medium | low (got {min_band!r}).",
|
|
6914
|
+
hint="Example: --min-band high",
|
|
6915
|
+
expected="critical|high|medium|low",
|
|
6916
|
+
)
|
|
6917
|
+
raise typer.Exit(code=1)
|
|
6918
|
+
|
|
6919
|
+
data = build_risk(path, limit=limit, min_band=min_band)
|
|
6920
|
+
_emit_command_output(
|
|
6921
|
+
_serialize_dict(data, format),
|
|
6922
|
+
output_path,
|
|
6923
|
+
False,
|
|
6924
|
+
success_msg=(
|
|
6925
|
+
f"risk written to {output_path} ({data['shown']} of "
|
|
6926
|
+
f"{data['total_defects']} defects composed)"
|
|
6927
|
+
),
|
|
6928
|
+
)
|
|
6929
|
+
|
|
6930
|
+
|
|
6843
6931
|
@app.command("posture")
|
|
6844
6932
|
def posture_cmd(
|
|
6845
6933
|
path: Path = typer.Argument(
|
|
@@ -9969,6 +10057,20 @@ def cache_status_cmd(
|
|
|
9969
10057
|
typer.echo(f"Views: {stats['views']}")
|
|
9970
10058
|
typer.echo(f"CAS blobs: {stats['cas_blobs']}")
|
|
9971
10059
|
typer.echo(f"Total size: {stats['total_size_mb']} MB")
|
|
10060
|
+
# C3-36: the three lines above describe ONE of the three stores a warm
|
|
10061
|
+
# fills. Printed alone after an 89 s warm they read as "nothing was
|
|
10062
|
+
# cached", which is the opposite of what happened.
|
|
10063
|
+
_stores = stats.get("stores") or {}
|
|
10064
|
+
for _label, _key in (("Shared CIR", "shared_cir"), ("Parse cache", "parse")):
|
|
10065
|
+
_store = _stores.get(_key) or {}
|
|
10066
|
+
if not _store or _store.get("available") is False:
|
|
10067
|
+
typer.echo(f"{_label + ':':<13}unavailable")
|
|
10068
|
+
continue
|
|
10069
|
+
_mb = round(_store.get("bytes", 0) / (1024 * 1024), 2)
|
|
10070
|
+
_scope = " (shared across repositories)" if _store.get("scope") == "shared" else ""
|
|
10071
|
+
typer.echo(
|
|
10072
|
+
f"{_label + ':':<13}{_store.get('entries', 0)} entries, {_mb} MB{_scope}"
|
|
10073
|
+
)
|
|
9972
10074
|
# RIS section
|
|
9973
10075
|
if stats.get("ris_exists"):
|
|
9974
10076
|
_stale_tag = " [STALE]" if stats.get("ris_is_stale") else ""
|
|
@@ -10326,7 +10428,7 @@ HELP_PANELS: "tuple[tuple[str, tuple[str, ...]], ...]" = (
|
|
|
10326
10428
|
"cache", "auth", "mcp", "telemetry", "baseline",
|
|
10327
10429
|
)),
|
|
10328
10430
|
("Experimental — shape may change", (
|
|
10329
|
-
"archetype", "retrieve",
|
|
10431
|
+
"risk", "archetype", "retrieve",
|
|
10330
10432
|
)),
|
|
10331
10433
|
)
|
|
10332
10434
|
|
|
@@ -343,9 +343,14 @@ class ConfidenceAnalyzer:
|
|
|
343
343
|
gaps.append(AnalysisGap(
|
|
344
344
|
area="testing",
|
|
345
345
|
reason=(
|
|
346
|
+
# C1-19: "Java files" here has always meant the non-test ones
|
|
347
|
+
# — the denominator of a test ratio cannot include the tests.
|
|
348
|
+
# Unqualified, it read as a third file count contradicting
|
|
349
|
+
# `migrate-check.java_files_scanned` (which counts them all).
|
|
346
350
|
f"Backend test coverage critical: {len(_java_tests)} test files "
|
|
347
|
-
f"for {len(_java_prod)} Java files "
|
|
348
|
-
f"({_ratio:.1%}) —
|
|
351
|
+
f"for {len(_java_prod)} non-test Java files "
|
|
352
|
+
f"(of {len(_java_all)} Java files, {_ratio:.1%}) — "
|
|
353
|
+
f"{_java_test_facts.basis}"
|
|
349
354
|
),
|
|
350
355
|
impact="high",
|
|
351
356
|
))
|
sourcecode/defect_identity.py
CHANGED
|
@@ -93,6 +93,25 @@ def assign_defect_ids(findings: "Iterable[SpringFinding]") -> None:
|
|
|
93
93
|
finding.defect_id = make_defect_id(category, kind, symbol)
|
|
94
94
|
|
|
95
95
|
|
|
96
|
+
def _witness_sites(witnesses: "list[SpringFinding]") -> "dict[str, Any]":
|
|
97
|
+
"""The distinct places this defect was observed, when the witnesses record one."""
|
|
98
|
+
sites: "list[dict[str, Any]]" = []
|
|
99
|
+
seen: "set[tuple[str, Any]]" = set()
|
|
100
|
+
for finding in witnesses:
|
|
101
|
+
site = (finding.evidence or {}).get("call_site")
|
|
102
|
+
if not isinstance(site, dict):
|
|
103
|
+
continue
|
|
104
|
+
key = (str(site.get("source_file") or ""), site.get("line"))
|
|
105
|
+
if key in seen:
|
|
106
|
+
continue
|
|
107
|
+
seen.add(key)
|
|
108
|
+
sites.append(site)
|
|
109
|
+
if not sites:
|
|
110
|
+
return {}
|
|
111
|
+
sites.sort(key=lambda s: (str(s.get("source_file") or ""), s.get("line") or 0))
|
|
112
|
+
return {"witness_sites": sites, "distinct_site_count": len(sites)}
|
|
113
|
+
|
|
114
|
+
|
|
96
115
|
def group_by_defect(findings: "list[SpringFinding]") -> list[dict[str, Any]]:
|
|
97
116
|
"""One row per defect, most severe first, each naming its witnesses.
|
|
98
117
|
|
|
@@ -137,6 +156,11 @@ def group_by_defect(findings: "list[SpringFinding]") -> list[dict[str, Any]]:
|
|
|
137
156
|
"rule_ids": sorted({f.pattern_id for f in witnesses}),
|
|
138
157
|
"witness_count": len(witnesses),
|
|
139
158
|
"witnesses": [f.id for f in witnesses],
|
|
159
|
+
# C2-16: witnesses of one defect are all located at the symbol that
|
|
160
|
+
# carries the remedy, so several of them print the same line and read
|
|
161
|
+
# as duplicates. Where each was actually observed is listed here, and
|
|
162
|
+
# a count of distinct sites so "N witnesses" can be checked against it.
|
|
163
|
+
**_witness_sites(witnesses),
|
|
140
164
|
})
|
|
141
165
|
rows.sort(key=lambda r: (SEVERITY_RANK.get(r["severity"], 9), r["symbol"]))
|
|
142
166
|
return rows
|
|
@@ -209,7 +209,7 @@ def collect_signals(root: Path) -> list[Signal]:
|
|
|
209
209
|
if _PROPERTY_KEY not in text and _ENV_KEY not in text:
|
|
210
210
|
continue
|
|
211
211
|
try:
|
|
212
|
-
rel =
|
|
212
|
+
rel = path.relative_to(root).as_posix()
|
|
213
213
|
except ValueError:
|
|
214
214
|
rel = str(path)
|
|
215
215
|
lines = text.splitlines()
|
sourcecode/filter_surface.py
CHANGED
|
@@ -38,6 +38,117 @@ _QUOTED_RE = re.compile(r'"([^"]*)"')
|
|
|
38
38
|
_MEMBER_RE = re.compile(r"(urlPatterns|value)\s*=", re.DOTALL)
|
|
39
39
|
|
|
40
40
|
|
|
41
|
+
#: Published Servlet / Spring Security / JAAS vocabulary by which a filter
|
|
42
|
+
#: *establishes* a caller identity. Recorded as evidence for a verdict; never a
|
|
43
|
+
#: predicate over a client's own class name (VAI) — `AuthFilter` proves nothing and
|
|
44
|
+
#: `M3FiltroSeguridad` is not less of an authenticator for being called that.
|
|
45
|
+
_ESTABLISHES_IDENTITY = (
|
|
46
|
+
"SecurityContextHolder",
|
|
47
|
+
"setAuthentication",
|
|
48
|
+
"AuthenticationManager",
|
|
49
|
+
"AuthenticationProvider",
|
|
50
|
+
"UsernamePasswordAuthenticationToken",
|
|
51
|
+
"PreAuthenticatedAuthenticationToken",
|
|
52
|
+
"AbstractAuthenticationProcessingFilter",
|
|
53
|
+
"AuthenticationEntryPoint",
|
|
54
|
+
".authenticate(",
|
|
55
|
+
"request.login(",
|
|
56
|
+
"getUserPrincipal(",
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
#: Vocabulary by which a filter *reads a credential* off the request.
|
|
60
|
+
_READS_CREDENTIAL = (
|
|
61
|
+
'"Authorization"',
|
|
62
|
+
"HttpHeaders.AUTHORIZATION",
|
|
63
|
+
"AUTHORIZATION",
|
|
64
|
+
"Bearer ",
|
|
65
|
+
"parseClaimsJws",
|
|
66
|
+
"parseSignedClaims",
|
|
67
|
+
"JWTVerifier",
|
|
68
|
+
"verifyToken",
|
|
69
|
+
"validateToken",
|
|
70
|
+
"getSession(false)",
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
#: Vocabulary by which a filter *refuses* a request it did not authenticate. Alone
|
|
74
|
+
#: this is not authentication — a rate limiter refuses too — so it only counts
|
|
75
|
+
#: alongside a credential read.
|
|
76
|
+
_REFUSES = (
|
|
77
|
+
"SC_UNAUTHORIZED",
|
|
78
|
+
"SC_FORBIDDEN",
|
|
79
|
+
"sendError(401",
|
|
80
|
+
"sendError(403",
|
|
81
|
+
"setStatus(401",
|
|
82
|
+
"setStatus(403",
|
|
83
|
+
"AccessDeniedException",
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
#: The three verdicts. Named after what was established, not after a confidence
|
|
87
|
+
#: level, because "unknown" here is a specific thing: the implementation was never
|
|
88
|
+
#: read, so nothing was ruled in or out.
|
|
89
|
+
AUTHENTICATES = "authenticates"
|
|
90
|
+
DOES_NOT_AUTHENTICATE = "does_not_authenticate"
|
|
91
|
+
AUTHENTICATION_UNKNOWN = "unknown"
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def authentication_verdict(source: str) -> "tuple[str, list[str]]":
|
|
95
|
+
"""Does this filter implementation authenticate the caller, and on what evidence?
|
|
96
|
+
|
|
97
|
+
C1-20. A pattern match tells you a filter *runs* on a path; it says nothing about
|
|
98
|
+
what the filter does there. In the field, three filters matched 2 635 endpoints
|
|
99
|
+
with `/*`, `/api/*` and `/api/v1/*` — and they were CORS, header and logging
|
|
100
|
+
filters. The reassurance was entirely in the reader's head.
|
|
101
|
+
|
|
102
|
+
Two independent ways to answer yes, both structural:
|
|
103
|
+
* the filter establishes an identity (it touches the security context, an
|
|
104
|
+
authentication manager, or a principal); or
|
|
105
|
+
* it reads a credential off the request **and** refuses requests over it.
|
|
106
|
+
|
|
107
|
+
A body we could not read is `unknown`, never `does_not_authenticate` — the whole
|
|
108
|
+
point is to stop absence of evidence from being published as evidence.
|
|
109
|
+
"""
|
|
110
|
+
if not source:
|
|
111
|
+
return AUTHENTICATION_UNKNOWN, []
|
|
112
|
+
seen_identity = [tok for tok in _ESTABLISHES_IDENTITY if tok in source]
|
|
113
|
+
seen_credential = [tok for tok in _READS_CREDENTIAL if tok in source]
|
|
114
|
+
seen_refusal = [tok for tok in _REFUSES if tok in source]
|
|
115
|
+
if seen_identity:
|
|
116
|
+
return AUTHENTICATES, sorted(set(seen_identity + seen_credential))
|
|
117
|
+
if seen_credential and seen_refusal:
|
|
118
|
+
return AUTHENTICATES, sorted(set(seen_credential + seen_refusal))
|
|
119
|
+
return DOES_NOT_AUTHENTICATE, sorted(set(seen_credential + seen_refusal))
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
@dataclass(frozen=True)
|
|
123
|
+
class FilterDeclaration:
|
|
124
|
+
"""One declared servlet filter: what it is mapped to, and what it does there."""
|
|
125
|
+
|
|
126
|
+
name: str
|
|
127
|
+
source: str
|
|
128
|
+
patterns: "tuple[str, ...]"
|
|
129
|
+
authenticates: str = AUTHENTICATION_UNKNOWN
|
|
130
|
+
evidence: "tuple[str, ...]" = ()
|
|
131
|
+
implementation_file: Optional[str] = None
|
|
132
|
+
|
|
133
|
+
def to_dict(self) -> dict:
|
|
134
|
+
out: dict = {
|
|
135
|
+
"filter": self.name,
|
|
136
|
+
"declared_in": self.source,
|
|
137
|
+
"patterns": sorted(self.patterns),
|
|
138
|
+
"authenticates": self.authenticates,
|
|
139
|
+
}
|
|
140
|
+
if self.evidence:
|
|
141
|
+
out["evidence"] = list(self.evidence)
|
|
142
|
+
if self.implementation_file:
|
|
143
|
+
out["implementation_file"] = self.implementation_file
|
|
144
|
+
if self.authenticates == AUTHENTICATION_UNKNOWN:
|
|
145
|
+
out["reason"] = (
|
|
146
|
+
"the implementation class was not found in this repository, so nothing "
|
|
147
|
+
"was ruled in or out"
|
|
148
|
+
)
|
|
149
|
+
return out
|
|
150
|
+
|
|
151
|
+
|
|
41
152
|
@dataclass
|
|
42
153
|
class FilterSurface:
|
|
43
154
|
"""Reconstructed servlet filter URL patterns + their provenance."""
|
|
@@ -45,6 +156,10 @@ class FilterSurface:
|
|
|
45
156
|
patterns: list[str] = field(default_factory=list)
|
|
46
157
|
# pattern → list of human-readable sources (filter class / web.xml) — evidence only
|
|
47
158
|
provenance: dict[str, list[str]] = field(default_factory=dict)
|
|
159
|
+
# Per-filter records. `patterns` above stays the flat union it always was, so
|
|
160
|
+
# every existing consumer keeps working; the split C1-20 needs is per filter,
|
|
161
|
+
# because the question is not "is a filter here" but "does *that* filter check".
|
|
162
|
+
filters: list[FilterDeclaration] = field(default_factory=list)
|
|
48
163
|
|
|
49
164
|
def is_empty(self) -> bool:
|
|
50
165
|
return not self.patterns
|
|
@@ -60,10 +175,45 @@ class FilterSurface:
|
|
|
60
175
|
return pat
|
|
61
176
|
return None
|
|
62
177
|
|
|
178
|
+
def matching_filters(self, path: Optional[str]) -> "list[FilterDeclaration]":
|
|
179
|
+
"""Every declared filter whose patterns cover ``path``, in a stable order."""
|
|
180
|
+
if not path:
|
|
181
|
+
return []
|
|
182
|
+
return [
|
|
183
|
+
f for f in sorted(self.filters, key=lambda d: (d.name, d.source))
|
|
184
|
+
if any(_servlet_pattern_matches(p, path) for p in f.patterns)
|
|
185
|
+
]
|
|
186
|
+
|
|
187
|
+
def authenticating_pattern(self, path: Optional[str]) -> Optional[str]:
|
|
188
|
+
"""The pattern of the first filter covering ``path`` that authenticates."""
|
|
189
|
+
for f in self.matching_filters(path):
|
|
190
|
+
if f.authenticates != AUTHENTICATES:
|
|
191
|
+
continue
|
|
192
|
+
for pat in sorted(f.patterns):
|
|
193
|
+
if _servlet_pattern_matches(pat, path):
|
|
194
|
+
return pat
|
|
195
|
+
return None
|
|
196
|
+
|
|
197
|
+
def coverage_verdict(self, path: Optional[str]) -> str:
|
|
198
|
+
"""How this path stands relative to the *authenticating* filter surface:
|
|
199
|
+
`authenticates`, `unknown` (a filter covers it whose body we never read), or
|
|
200
|
+
`does_not_authenticate` (covered only by filters that demonstrably do not).
|
|
201
|
+
A path no pattern covers returns `does_not_authenticate` — the caller keeps
|
|
202
|
+
that case separate as "no matching pattern"."""
|
|
203
|
+
matched = self.matching_filters(path)
|
|
204
|
+
if any(f.authenticates == AUTHENTICATES for f in matched):
|
|
205
|
+
return AUTHENTICATES
|
|
206
|
+
if any(f.authenticates == AUTHENTICATION_UNKNOWN for f in matched):
|
|
207
|
+
return AUTHENTICATION_UNKNOWN
|
|
208
|
+
return DOES_NOT_AUTHENTICATE
|
|
209
|
+
|
|
63
210
|
def to_dict(self) -> dict:
|
|
64
211
|
return {
|
|
65
212
|
"patterns": sorted(self.patterns),
|
|
66
213
|
"provenance": {k: sorted(v) for k, v in sorted(self.provenance.items())},
|
|
214
|
+
"filters": [
|
|
215
|
+
f.to_dict() for f in sorted(self.filters, key=lambda d: (d.name, d.source))
|
|
216
|
+
],
|
|
67
217
|
}
|
|
68
218
|
|
|
69
219
|
|
|
@@ -114,18 +264,56 @@ def _patterns_from_webfilter(source: str) -> list[str]:
|
|
|
114
264
|
def _patterns_from_web_xml(xml_path: Path) -> list[str]:
|
|
115
265
|
"""Servlet URL patterns from ``<filter-mapping><url-pattern>`` in one web.xml.
|
|
116
266
|
Namespace-agnostic; never raises on malformed XML (returns what it parsed)."""
|
|
117
|
-
|
|
267
|
+
return [pat for _name, pats in _mappings_from_web_xml(xml_path) for pat in pats]
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def _local(tag: str) -> str:
|
|
271
|
+
return tag.rsplit("}", 1)[-1]
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def _child_text(elem: "ET.Element", name: str) -> str:
|
|
275
|
+
for child in elem.iter():
|
|
276
|
+
if _local(child.tag) == name and (child.text or "").strip():
|
|
277
|
+
return child.text.strip()
|
|
278
|
+
return ""
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _mappings_from_web_xml(xml_path: Path) -> "list[tuple[str, list[str]]]":
|
|
282
|
+
"""``(filter-name, url-patterns)`` per ``<filter-mapping>``. The name is what
|
|
283
|
+
correlates a mapping with the class that implements it (C1-20): without it, a
|
|
284
|
+
pattern is anonymous and every filter looks like every other filter."""
|
|
285
|
+
out: "list[tuple[str, list[str]]]" = []
|
|
118
286
|
try:
|
|
119
287
|
root = ET.parse(str(xml_path)).getroot()
|
|
120
288
|
except Exception:
|
|
121
289
|
return out
|
|
122
290
|
for elem in root.iter():
|
|
123
|
-
|
|
124
|
-
if tag != "filter-mapping":
|
|
291
|
+
if _local(elem.tag) != "filter-mapping":
|
|
125
292
|
continue
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
293
|
+
patterns = [
|
|
294
|
+
child.text.strip()
|
|
295
|
+
for child in elem.iter()
|
|
296
|
+
if _local(child.tag) == "url-pattern" and (child.text or "").strip()
|
|
297
|
+
]
|
|
298
|
+
if patterns:
|
|
299
|
+
out.append((_child_text(elem, "filter-name"), patterns))
|
|
300
|
+
return out
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _filter_classes_from_web_xml(xml_path: Path) -> "dict[str, str]":
|
|
304
|
+
"""``<filter>`` declarations: filter-name → filter-class."""
|
|
305
|
+
out: "dict[str, str]" = {}
|
|
306
|
+
try:
|
|
307
|
+
root = ET.parse(str(xml_path)).getroot()
|
|
308
|
+
except Exception:
|
|
309
|
+
return out
|
|
310
|
+
for elem in root.iter():
|
|
311
|
+
if _local(elem.tag) != "filter":
|
|
312
|
+
continue
|
|
313
|
+
name = _child_text(elem, "filter-name")
|
|
314
|
+
klass = _child_text(elem, "filter-class")
|
|
315
|
+
if name and klass:
|
|
316
|
+
out[name] = klass
|
|
129
317
|
return out
|
|
130
318
|
|
|
131
319
|
|
|
@@ -154,20 +342,63 @@ def build_filter_surface(
|
|
|
154
342
|
if source not in surface.provenance[pattern]:
|
|
155
343
|
surface.provenance[pattern].append(source)
|
|
156
344
|
|
|
345
|
+
# Simple name → repo-relative path, so a `<filter-class>` in web.xml can be read
|
|
346
|
+
# for what it actually does. Ambiguous simple names (two classes, one name) are
|
|
347
|
+
# dropped rather than guessed: the wrong body would produce a confident verdict
|
|
348
|
+
# about the wrong filter, which is the failure this row exists to remove.
|
|
349
|
+
by_simple_name: "dict[str, Optional[str]]" = {}
|
|
157
350
|
for rel in java_files:
|
|
158
|
-
|
|
351
|
+
stem = rel.rsplit("/", 1)[-1][: -len(".java")] if rel.endswith(".java") else ""
|
|
352
|
+
if not stem:
|
|
353
|
+
continue
|
|
354
|
+
by_simple_name[stem] = None if stem in by_simple_name else rel
|
|
355
|
+
|
|
356
|
+
def _read(rel: Optional[str]) -> str:
|
|
357
|
+
if not rel:
|
|
358
|
+
return ""
|
|
159
359
|
try:
|
|
160
|
-
|
|
360
|
+
return (root / rel).read_text(encoding="utf-8", errors="replace")
|
|
161
361
|
except OSError:
|
|
162
|
-
|
|
362
|
+
return ""
|
|
363
|
+
|
|
364
|
+
for rel in java_files:
|
|
365
|
+
src = _read(rel)
|
|
163
366
|
if "@WebFilter" not in src:
|
|
164
367
|
continue
|
|
165
|
-
|
|
368
|
+
patterns = _patterns_from_webfilter(src)
|
|
369
|
+
if not patterns:
|
|
370
|
+
continue
|
|
371
|
+
for pat in patterns:
|
|
166
372
|
_add(pat, f"@WebFilter:{rel}")
|
|
373
|
+
verdict, evidence = authentication_verdict(src)
|
|
374
|
+
surface.filters.append(FilterDeclaration(
|
|
375
|
+
name=rel.rsplit("/", 1)[-1][: -len(".java")],
|
|
376
|
+
source=f"@WebFilter:{rel}",
|
|
377
|
+
patterns=tuple(patterns),
|
|
378
|
+
authenticates=verdict,
|
|
379
|
+
evidence=tuple(evidence),
|
|
380
|
+
implementation_file=rel,
|
|
381
|
+
))
|
|
167
382
|
|
|
168
383
|
for xml_path in sorted(root.rglob("web.xml")):
|
|
169
384
|
rel = xml_path.relative_to(root).as_posix()
|
|
170
|
-
|
|
171
|
-
|
|
385
|
+
classes = _filter_classes_from_web_xml(xml_path)
|
|
386
|
+
for name, patterns in _mappings_from_web_xml(xml_path):
|
|
387
|
+
for pat in patterns:
|
|
388
|
+
_add(pat, f"web.xml:{rel}")
|
|
389
|
+
klass = classes.get(name, "")
|
|
390
|
+
impl = by_simple_name.get(klass.rsplit(".", 1)[-1]) if klass else None
|
|
391
|
+
body = _read(impl)
|
|
392
|
+
verdict, evidence = (
|
|
393
|
+
authentication_verdict(body) if body else (AUTHENTICATION_UNKNOWN, [])
|
|
394
|
+
)
|
|
395
|
+
surface.filters.append(FilterDeclaration(
|
|
396
|
+
name=klass or name or "(unnamed filter-mapping)",
|
|
397
|
+
source=f"web.xml:{rel}",
|
|
398
|
+
patterns=tuple(patterns),
|
|
399
|
+
authenticates=verdict,
|
|
400
|
+
evidence=tuple(evidence),
|
|
401
|
+
implementation_file=impl,
|
|
402
|
+
))
|
|
172
403
|
|
|
173
404
|
return surface
|