sourcecode 3.8.0__py3-none-any.whl → 4.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sourcecode/__init__.py CHANGED
@@ -4,4 +4,4 @@ ASK Engine is the product. ``ask`` is the canonical CLI command; ``sourcecode``
4
4
  the legacy compatibility alias and the Python/PyPI package name. See
5
5
  docs/PRODUCT_IDENTITY.md (normative)."""
6
6
 
7
- __version__ = "3.8.0"
7
+ __version__ = "4.0.0"
sourcecode/cache.py CHANGED
@@ -354,6 +354,7 @@ def status(repo_root: Path) -> dict[str, Any]:
354
354
  "cores": 0, "snapshots": 0, "views": 0, "cas_blobs": 0,
355
355
  "total_size_bytes": 0, "total_size_mb": 0.0,
356
356
  "current_git_head": current_head,
357
+ "stores": _store_breakdown(repo_root, 0),
357
358
  **ris_fields,
358
359
  }
359
360
  cores = list(cache_d.glob("core-*.json.gz"))
@@ -371,10 +372,57 @@ def status(repo_root: Path) -> dict[str, Any]:
371
372
  "total_size_bytes": total_bytes,
372
373
  "total_size_mb": round(total_bytes / (1024 * 1024), 2),
373
374
  "current_git_head": current_head,
375
+ # C3-36: `CAS blobs: 0`, `Total size: 0.1 MB` immediately after an 89-second
376
+ # warm of 3 342 files. Every figure above was true and described one store
377
+ # of three — the warm's output mostly lands in the shared CIR and the parse
378
+ # cache, which this command never counted. A status that reports a third of
379
+ # the state reads as a warm that did nothing.
380
+ "stores": _store_breakdown(repo_root, total_bytes),
374
381
  **ris_fields,
375
382
  }
376
383
 
377
384
 
385
+ def _store_breakdown(repo_root: Path, core_bytes: int) -> "dict[str, Any]":
386
+ """What a warm actually populated, by store. Best-effort per store: a store
387
+ that cannot be inspected is reported as unavailable, never as empty."""
388
+ out: "dict[str, Any]" = {
389
+ "core": {
390
+ "cache_dir": str(cache_dir(repo_root)),
391
+ "bytes": core_bytes,
392
+ "scope": "this repository",
393
+ "holds": "core snapshots, rendered views and their CAS blobs",
394
+ },
395
+ }
396
+ try:
397
+ from sourcecode.context_cache import ContextCache # noqa: PLC0415
398
+
399
+ ctx = ContextCache.for_repo(repo_root).stats()
400
+ out["shared_cir"] = {
401
+ "cache_dir": ctx["cache_dir"],
402
+ "entries": ctx["contexts"],
403
+ "bytes": ctx["bytes_stored"],
404
+ "scope": "this repository",
405
+ "holds": "the shared Canonical IR that explain/impact/posture reuse",
406
+ }
407
+ except Exception:
408
+ out["shared_cir"] = {"available": False}
409
+ try:
410
+ from sourcecode import parse_cache as _pc # noqa: PLC0415
411
+
412
+ out["parse"] = {
413
+ **_pc.store_stats(),
414
+ "holds": "per-file parses, content-addressed across every repository",
415
+ }
416
+ except Exception:
417
+ out["parse"] = {"available": False}
418
+ out["note"] = (
419
+ "`cache warm` populates all of these; the core store alone is a third of "
420
+ "the answer, and reading it as the whole is how a completed warm looks like "
421
+ "an empty cache."
422
+ )
423
+ return out
424
+
425
+
378
426
  def clear(repo_root: Path, *, clear_ris: bool = False) -> int:
379
427
  """Delete cache files for *repo_root*. Returns the number of files removed.
380
428
 
sourcecode/cache_model.py CHANGED
@@ -131,6 +131,13 @@ COMMANDS: tuple[CommandCache, ...] = (
131
131
  "builds — the parse it used to repeat for itself. `--diff` compares two profile "
132
132
  "sets over that one IR, so the second side costs the resolution only.",
133
133
  "10.1 s → 1.6 s"),
134
+ CommandCache("risk", ("cir", "parse"), "shared", False,
135
+ "Composes what the audit, impact-chain and the posture already answer, so it "
136
+ "pays each of their costs once over the shared CIR a warm builds — one parse "
137
+ "for the whole composition, and the reachability query is cached per symbol "
138
+ "within the run.",
139
+ "not measured on the battery yet — the composition is bounded by the "
140
+ "`spring-audit` + `impact-chain` costs listed here, not by new analysis"),
134
141
  CommandCache("endpoints", ("ris", "parse"), "shared", False,
135
142
  "Recomputes the endpoint surface on every run, over a parse a warm has already "
136
143
  "paid for. Until 3.7.0 the extractor parsed every file itself instead of reading "
sourcecode/cli.py CHANGED
@@ -183,7 +183,7 @@ COMMAND_TIERS: "tuple[tuple[str, str, tuple[str, ...]], ...]" = (
183
183
  "cache", "auth", "mcp", "telemetry",
184
184
  )),
185
185
  ("experimental", "shape may change in a minor — do not gate CI on it", (
186
- "posture", "archetype",
186
+ "risk", "posture", "archetype",
187
187
  )),
188
188
  # `retrieve` publishes 15+ intents whose answers the other commands already
189
189
  # give better: measured on the battery, `security-surface` merely re-states
@@ -5475,11 +5475,21 @@ def validation_cmd(
5475
5475
  # The payload carries the same note, so a pipeline loses nothing.
5476
5476
  if _note:
5477
5477
  _notice(f"Note: {_note}")
5478
+ # C2-15: the console said "0 body endpoints, 1179 gaps" over a payload carrying
5479
+ # `body_endpoints_in_code: 1201` — it printed the *declared-constraint* count
5480
+ # under the *code-surface* name, so the one line most readers see stated the
5481
+ # opposite of the answer. Both axes are named, in their own units.
5482
+ _declared = _summary.get("endpoints_with_body", 0)
5483
+ _in_code = _summary.get("body_endpoints_in_code")
5484
+ _bodies = (
5485
+ f"{_in_code} body endpoints in code, {_declared} with a declared constraint "
5486
+ f"surface" if _in_code is not None
5487
+ else f"{_declared} routes with a declared constraint surface"
5488
+ )
5478
5489
  _emit_command_output(
5479
5490
  output, output_path, copy,
5480
5491
  success_msg=f"Validation surface written to {output_path} "
5481
- f"({_summary.get('endpoints_with_body', 0)} body endpoints, "
5482
- f"{_summary.get('gaps', 0)} gaps)",
5492
+ f"({_bodies}, {_summary.get('gaps', 0)} gaps)",
5483
5493
  )
5484
5494
 
5485
5495
  from sourcecode.mcp_nudge import nudge_mcp_if_needed as _nudge
@@ -6308,11 +6318,13 @@ def _render_gate_coverage_section(result: "SpringAuditResult") -> list[str]: #
6308
6318
  gc = (result.security_posture or {}).get("gate_coverage")
6309
6319
  if not gc:
6310
6320
  return []
6311
- not_covered = gc.get("endpoints_not_carrying_gate", gc.get("not_carrying_gate", 0))
6321
+ not_covered = gc.get("endpoints_not_carrying_gate", 0)
6312
6322
  gates = ", ".join(f"`{g}`" for g in gc.get("gate_annotations", [])) or "the detected gate"
6313
6323
  # One unit, named: these are endpoints, not handler methods. The two counts differ
6314
6324
  # (one method can serve several mappings) and must never be summed together.
6315
- total = gc.get("endpoints_total", gc.get("total_controller_handlers", 0))
6325
+ # C2-14: read only the explicit keys — falling back to the retired aliases is how
6326
+ # a name that lies about its unit survives its own removal.
6327
+ total = gc.get("endpoints_total", 0)
6316
6328
  lines: list[str] = ["", "---", ""]
6317
6329
  if not_covered == 0:
6318
6330
  _gated = gc.get("endpoints_carrying_gate", total)
@@ -6326,7 +6338,6 @@ def _render_gate_coverage_section(result: "SpringAuditResult") -> list[str]: #
6326
6338
  lines.append(f"✅ **Gate coverage** — all {total} endpoints carry {gates}.")
6327
6339
  return lines
6328
6340
 
6329
- covered = gc.get("possibly_filter_covered", 0)
6330
6341
  _standard = gc.get("endpoints_standard_guarded", 0)
6331
6342
  lines.append(
6332
6343
  f"🔓 **Gate coverage** — {not_covered} of {total} endpoints do not carry "
@@ -6338,9 +6349,18 @@ def _render_gate_coverage_section(result: "SpringAuditResult") -> list[str]: #
6338
6349
  f"standard guard, {not_covered} on neither — the three sum to {total}._"
6339
6350
  )
6340
6351
  if gc.get("reconstructed_filter_patterns"):
6352
+ # C1-20: the split, not the total. "N match a filter pattern" is the sentence
6353
+ # a reader turns into "N are covered", and in the field the filters doing the
6354
+ # matching were CORS, headers and logging.
6355
+ _auth = gc.get("filter_covered_authenticating", 0)
6356
+ _gap = gc.get("filter_gap_non_authenticating", 0)
6357
+ _unknown = gc.get("filter_authentication_unknown", 0)
6341
6358
  lines.append(
6342
- f"_{covered} of those match a reconstructed servlet filter pattern "
6343
- f"(possibly filter-covered); {gc.get('no_matching_filter_pattern', 0)} match none._"
6359
+ f"_{_auth} are covered by a servlet filter that authenticates; {_gap} match "
6360
+ f"only filters that do not check the caller (CORS, headers, logging and the "
6361
+ f"like); {_unknown} match a filter whose implementation is not in this "
6362
+ f"repository; {gc.get('no_matching_filter_pattern', 0)} match no pattern at "
6363
+ f"all. Filter-chain order is not reconstructed._"
6344
6364
  )
6345
6365
  lines += ["", "<details>", "<summary>Handlers without the gate</summary>", ""]
6346
6366
  lines += [
@@ -6840,6 +6860,74 @@ def verify_cmd(
6840
6860
  raise typer.Exit(code=report.exit_code)
6841
6861
 
6842
6862
 
6863
+ @app.command("risk")
6864
+ def risk_cmd(
6865
+ path: Path = typer.Argument(
6866
+ Path("."),
6867
+ help="Repository path (default: current directory).",
6868
+ ),
6869
+ limit: int = typer.Option(
6870
+ 50, "--limit", help="How many composed risks to publish (highest first)."
6871
+ ),
6872
+ min_band: str = typer.Option(
6873
+ "low",
6874
+ "--min-band",
6875
+ help="Floor on the composed band: critical | high | medium | low.",
6876
+ ),
6877
+ output_path: Optional[Path] = typer.Option(
6878
+ None, "--output", "-o", help="Write the report to a file instead of stdout."
6879
+ ),
6880
+ format: str = typer.Option("json", "--format", "-f", help="Output format: json or yaml."),
6881
+ ) -> None:
6882
+ """[EXPERIMENTAL] What each defect actually costs, once reach and access are in it.
6883
+
6884
+ \b
6885
+ The other commands answer one axis each, correctly, and leave the composition
6886
+ to the reader. In the field, one class was `medium` in `spring-audit`,
6887
+ `medium/5.0` in `impact-chain`, `coverage_unknown → permit_all` in `posture`,
6888
+ and executed a stored procedure that mutates the database. Composed, that is
6889
+ unauthenticated write access under the release build; separately, it was two
6890
+ commands saying "medium".
6891
+
6892
+ \b
6893
+ severity_effective = defect_severity × reachability × auth_verdict × write_effect
6894
+
6895
+ \b
6896
+ No new analysis: every factor is read from the command that already publishes
6897
+ it, and every row publishes its four factors with the authority each came
6898
+ from, so a reader can disagree with one and keep the rest. An axis that could
6899
+ not be measured is `unknown`, multiplies by 1.0, and is named in `blind_axes`.
6900
+
6901
+ \b
6902
+ Examples:
6903
+ ask risk .
6904
+ ask risk . --min-band high
6905
+ ask risk . --limit 10 -o risk.json
6906
+ """
6907
+ from sourcecode.risk import build_risk
6908
+
6909
+ path = _admit_path(path)
6910
+ if min_band not in ("critical", "high", "medium", "low"):
6911
+ _emit_error_json(
6912
+ INVALID_INPUT_CODE,
6913
+ f"--min-band expects critical | high | medium | low (got {min_band!r}).",
6914
+ hint="Example: --min-band high",
6915
+ expected="critical|high|medium|low",
6916
+ )
6917
+ raise typer.Exit(code=1)
6918
+
6919
+ data = build_risk(path, limit=limit, min_band=min_band)
6920
+ _emit_command_output(
6921
+ _serialize_dict(data, format),
6922
+ output_path,
6923
+ False,
6924
+ success_msg=(
6925
+ f"risk written to {output_path} ({data['shown']} of "
6926
+ f"{data['total_defects']} defects composed)"
6927
+ ),
6928
+ )
6929
+
6930
+
6843
6931
  @app.command("posture")
6844
6932
  def posture_cmd(
6845
6933
  path: Path = typer.Argument(
@@ -9969,6 +10057,20 @@ def cache_status_cmd(
9969
10057
  typer.echo(f"Views: {stats['views']}")
9970
10058
  typer.echo(f"CAS blobs: {stats['cas_blobs']}")
9971
10059
  typer.echo(f"Total size: {stats['total_size_mb']} MB")
10060
+ # C3-36: the three lines above describe ONE of the three stores a warm
10061
+ # fills. Printed alone after an 89 s warm they read as "nothing was
10062
+ # cached", which is the opposite of what happened.
10063
+ _stores = stats.get("stores") or {}
10064
+ for _label, _key in (("Shared CIR", "shared_cir"), ("Parse cache", "parse")):
10065
+ _store = _stores.get(_key) or {}
10066
+ if not _store or _store.get("available") is False:
10067
+ typer.echo(f"{_label + ':':<13}unavailable")
10068
+ continue
10069
+ _mb = round(_store.get("bytes", 0) / (1024 * 1024), 2)
10070
+ _scope = " (shared across repositories)" if _store.get("scope") == "shared" else ""
10071
+ typer.echo(
10072
+ f"{_label + ':':<13}{_store.get('entries', 0)} entries, {_mb} MB{_scope}"
10073
+ )
9972
10074
  # RIS section
9973
10075
  if stats.get("ris_exists"):
9974
10076
  _stale_tag = " [STALE]" if stats.get("ris_is_stale") else ""
@@ -10326,7 +10428,7 @@ HELP_PANELS: "tuple[tuple[str, tuple[str, ...]], ...]" = (
10326
10428
  "cache", "auth", "mcp", "telemetry", "baseline",
10327
10429
  )),
10328
10430
  ("Experimental — shape may change", (
10329
- "archetype", "retrieve",
10431
+ "risk", "archetype", "retrieve",
10330
10432
  )),
10331
10433
  )
10332
10434
 
@@ -343,9 +343,14 @@ class ConfidenceAnalyzer:
343
343
  gaps.append(AnalysisGap(
344
344
  area="testing",
345
345
  reason=(
346
+ # C1-19: "Java files" here has always meant the non-test ones
347
+ # — the denominator of a test ratio cannot include the tests.
348
+ # Unqualified, it read as a third file count contradicting
349
+ # `migrate-check.java_files_scanned` (which counts them all).
346
350
  f"Backend test coverage critical: {len(_java_tests)} test files "
347
- f"for {len(_java_prod)} Java files "
348
- f"({_ratio:.1%}) — {_java_test_facts.basis}"
351
+ f"for {len(_java_prod)} non-test Java files "
352
+ f"(of {len(_java_all)} Java files, {_ratio:.1%}) — "
353
+ f"{_java_test_facts.basis}"
349
354
  ),
350
355
  impact="high",
351
356
  ))
@@ -93,6 +93,25 @@ def assign_defect_ids(findings: "Iterable[SpringFinding]") -> None:
93
93
  finding.defect_id = make_defect_id(category, kind, symbol)
94
94
 
95
95
 
96
+ def _witness_sites(witnesses: "list[SpringFinding]") -> "dict[str, Any]":
97
+ """The distinct places this defect was observed, when the witnesses record one."""
98
+ sites: "list[dict[str, Any]]" = []
99
+ seen: "set[tuple[str, Any]]" = set()
100
+ for finding in witnesses:
101
+ site = (finding.evidence or {}).get("call_site")
102
+ if not isinstance(site, dict):
103
+ continue
104
+ key = (str(site.get("source_file") or ""), site.get("line"))
105
+ if key in seen:
106
+ continue
107
+ seen.add(key)
108
+ sites.append(site)
109
+ if not sites:
110
+ return {}
111
+ sites.sort(key=lambda s: (str(s.get("source_file") or ""), s.get("line") or 0))
112
+ return {"witness_sites": sites, "distinct_site_count": len(sites)}
113
+
114
+
96
115
  def group_by_defect(findings: "list[SpringFinding]") -> list[dict[str, Any]]:
97
116
  """One row per defect, most severe first, each naming its witnesses.
98
117
 
@@ -137,6 +156,11 @@ def group_by_defect(findings: "list[SpringFinding]") -> list[dict[str, Any]]:
137
156
  "rule_ids": sorted({f.pattern_id for f in witnesses}),
138
157
  "witness_count": len(witnesses),
139
158
  "witnesses": [f.id for f in witnesses],
159
+ # C2-16: witnesses of one defect are all located at the symbol that
160
+ # carries the remedy, so several of them print the same line and read
161
+ # as duplicates. Where each was actually observed is listed here, and
162
+ # a count of distinct sites so "N witnesses" can be checked against it.
163
+ **_witness_sites(witnesses),
140
164
  })
141
165
  rows.sort(key=lambda r: (SEVERITY_RANK.get(r["severity"], 9), r["symbol"]))
142
166
  return rows
@@ -209,7 +209,7 @@ def collect_signals(root: Path) -> list[Signal]:
209
209
  if _PROPERTY_KEY not in text and _ENV_KEY not in text:
210
210
  continue
211
211
  try:
212
- rel = str(path.relative_to(root))
212
+ rel = path.relative_to(root).as_posix()
213
213
  except ValueError:
214
214
  rel = str(path)
215
215
  lines = text.splitlines()
@@ -38,6 +38,117 @@ _QUOTED_RE = re.compile(r'"([^"]*)"')
38
38
  _MEMBER_RE = re.compile(r"(urlPatterns|value)\s*=", re.DOTALL)
39
39
 
40
40
 
41
+ #: Published Servlet / Spring Security / JAAS vocabulary by which a filter
42
+ #: *establishes* a caller identity. Recorded as evidence for a verdict; never a
43
+ #: predicate over a client's own class name (VAI) — `AuthFilter` proves nothing and
44
+ #: `M3FiltroSeguridad` is not less of an authenticator for being called that.
45
+ _ESTABLISHES_IDENTITY = (
46
+ "SecurityContextHolder",
47
+ "setAuthentication",
48
+ "AuthenticationManager",
49
+ "AuthenticationProvider",
50
+ "UsernamePasswordAuthenticationToken",
51
+ "PreAuthenticatedAuthenticationToken",
52
+ "AbstractAuthenticationProcessingFilter",
53
+ "AuthenticationEntryPoint",
54
+ ".authenticate(",
55
+ "request.login(",
56
+ "getUserPrincipal(",
57
+ )
58
+
59
+ #: Vocabulary by which a filter *reads a credential* off the request.
60
+ _READS_CREDENTIAL = (
61
+ '"Authorization"',
62
+ "HttpHeaders.AUTHORIZATION",
63
+ "AUTHORIZATION",
64
+ "Bearer ",
65
+ "parseClaimsJws",
66
+ "parseSignedClaims",
67
+ "JWTVerifier",
68
+ "verifyToken",
69
+ "validateToken",
70
+ "getSession(false)",
71
+ )
72
+
73
+ #: Vocabulary by which a filter *refuses* a request it did not authenticate. Alone
74
+ #: this is not authentication — a rate limiter refuses too — so it only counts
75
+ #: alongside a credential read.
76
+ _REFUSES = (
77
+ "SC_UNAUTHORIZED",
78
+ "SC_FORBIDDEN",
79
+ "sendError(401",
80
+ "sendError(403",
81
+ "setStatus(401",
82
+ "setStatus(403",
83
+ "AccessDeniedException",
84
+ )
85
+
86
+ #: The three verdicts. Named after what was established, not after a confidence
87
+ #: level, because "unknown" here is a specific thing: the implementation was never
88
+ #: read, so nothing was ruled in or out.
89
+ AUTHENTICATES = "authenticates"
90
+ DOES_NOT_AUTHENTICATE = "does_not_authenticate"
91
+ AUTHENTICATION_UNKNOWN = "unknown"
92
+
93
+
94
+ def authentication_verdict(source: str) -> "tuple[str, list[str]]":
95
+ """Does this filter implementation authenticate the caller, and on what evidence?
96
+
97
+ C1-20. A pattern match tells you a filter *runs* on a path; it says nothing about
98
+ what the filter does there. In the field, three filters matched 2 635 endpoints
99
+ with `/*`, `/api/*` and `/api/v1/*` — and they were CORS, header and logging
100
+ filters. The reassurance was entirely in the reader's head.
101
+
102
+ Two independent ways to answer yes, both structural:
103
+ * the filter establishes an identity (it touches the security context, an
104
+ authentication manager, or a principal); or
105
+ * it reads a credential off the request **and** refuses requests over it.
106
+
107
+ A body we could not read is `unknown`, never `does_not_authenticate` — the whole
108
+ point is to stop absence of evidence from being published as evidence.
109
+ """
110
+ if not source:
111
+ return AUTHENTICATION_UNKNOWN, []
112
+ seen_identity = [tok for tok in _ESTABLISHES_IDENTITY if tok in source]
113
+ seen_credential = [tok for tok in _READS_CREDENTIAL if tok in source]
114
+ seen_refusal = [tok for tok in _REFUSES if tok in source]
115
+ if seen_identity:
116
+ return AUTHENTICATES, sorted(set(seen_identity + seen_credential))
117
+ if seen_credential and seen_refusal:
118
+ return AUTHENTICATES, sorted(set(seen_credential + seen_refusal))
119
+ return DOES_NOT_AUTHENTICATE, sorted(set(seen_credential + seen_refusal))
120
+
121
+
122
+ @dataclass(frozen=True)
123
+ class FilterDeclaration:
124
+ """One declared servlet filter: what it is mapped to, and what it does there."""
125
+
126
+ name: str
127
+ source: str
128
+ patterns: "tuple[str, ...]"
129
+ authenticates: str = AUTHENTICATION_UNKNOWN
130
+ evidence: "tuple[str, ...]" = ()
131
+ implementation_file: Optional[str] = None
132
+
133
+ def to_dict(self) -> dict:
134
+ out: dict = {
135
+ "filter": self.name,
136
+ "declared_in": self.source,
137
+ "patterns": sorted(self.patterns),
138
+ "authenticates": self.authenticates,
139
+ }
140
+ if self.evidence:
141
+ out["evidence"] = list(self.evidence)
142
+ if self.implementation_file:
143
+ out["implementation_file"] = self.implementation_file
144
+ if self.authenticates == AUTHENTICATION_UNKNOWN:
145
+ out["reason"] = (
146
+ "the implementation class was not found in this repository, so nothing "
147
+ "was ruled in or out"
148
+ )
149
+ return out
150
+
151
+
41
152
  @dataclass
42
153
  class FilterSurface:
43
154
  """Reconstructed servlet filter URL patterns + their provenance."""
@@ -45,6 +156,10 @@ class FilterSurface:
45
156
  patterns: list[str] = field(default_factory=list)
46
157
  # pattern → list of human-readable sources (filter class / web.xml) — evidence only
47
158
  provenance: dict[str, list[str]] = field(default_factory=dict)
159
+ # Per-filter records. `patterns` above stays the flat union it always was, so
160
+ # every existing consumer keeps working; the split C1-20 needs is per filter,
161
+ # because the question is not "is a filter here" but "does *that* filter check".
162
+ filters: list[FilterDeclaration] = field(default_factory=list)
48
163
 
49
164
  def is_empty(self) -> bool:
50
165
  return not self.patterns
@@ -60,10 +175,45 @@ class FilterSurface:
60
175
  return pat
61
176
  return None
62
177
 
178
+ def matching_filters(self, path: Optional[str]) -> "list[FilterDeclaration]":
179
+ """Every declared filter whose patterns cover ``path``, in a stable order."""
180
+ if not path:
181
+ return []
182
+ return [
183
+ f for f in sorted(self.filters, key=lambda d: (d.name, d.source))
184
+ if any(_servlet_pattern_matches(p, path) for p in f.patterns)
185
+ ]
186
+
187
+ def authenticating_pattern(self, path: Optional[str]) -> Optional[str]:
188
+ """The pattern of the first filter covering ``path`` that authenticates."""
189
+ for f in self.matching_filters(path):
190
+ if f.authenticates != AUTHENTICATES:
191
+ continue
192
+ for pat in sorted(f.patterns):
193
+ if _servlet_pattern_matches(pat, path):
194
+ return pat
195
+ return None
196
+
197
+ def coverage_verdict(self, path: Optional[str]) -> str:
198
+ """How this path stands relative to the *authenticating* filter surface:
199
+ `authenticates`, `unknown` (a filter covers it whose body we never read), or
200
+ `does_not_authenticate` (covered only by filters that demonstrably do not).
201
+ A path no pattern covers returns `does_not_authenticate` — the caller keeps
202
+ that case separate as "no matching pattern"."""
203
+ matched = self.matching_filters(path)
204
+ if any(f.authenticates == AUTHENTICATES for f in matched):
205
+ return AUTHENTICATES
206
+ if any(f.authenticates == AUTHENTICATION_UNKNOWN for f in matched):
207
+ return AUTHENTICATION_UNKNOWN
208
+ return DOES_NOT_AUTHENTICATE
209
+
63
210
  def to_dict(self) -> dict:
64
211
  return {
65
212
  "patterns": sorted(self.patterns),
66
213
  "provenance": {k: sorted(v) for k, v in sorted(self.provenance.items())},
214
+ "filters": [
215
+ f.to_dict() for f in sorted(self.filters, key=lambda d: (d.name, d.source))
216
+ ],
67
217
  }
68
218
 
69
219
 
@@ -114,18 +264,56 @@ def _patterns_from_webfilter(source: str) -> list[str]:
114
264
  def _patterns_from_web_xml(xml_path: Path) -> list[str]:
115
265
  """Servlet URL patterns from ``<filter-mapping><url-pattern>`` in one web.xml.
116
266
  Namespace-agnostic; never raises on malformed XML (returns what it parsed)."""
117
- out: list[str] = []
267
+ return [pat for _name, pats in _mappings_from_web_xml(xml_path) for pat in pats]
268
+
269
+
270
+ def _local(tag: str) -> str:
271
+ return tag.rsplit("}", 1)[-1]
272
+
273
+
274
+ def _child_text(elem: "ET.Element", name: str) -> str:
275
+ for child in elem.iter():
276
+ if _local(child.tag) == name and (child.text or "").strip():
277
+ return child.text.strip()
278
+ return ""
279
+
280
+
281
+ def _mappings_from_web_xml(xml_path: Path) -> "list[tuple[str, list[str]]]":
282
+ """``(filter-name, url-patterns)`` per ``<filter-mapping>``. The name is what
283
+ correlates a mapping with the class that implements it (C1-20): without it, a
284
+ pattern is anonymous and every filter looks like every other filter."""
285
+ out: "list[tuple[str, list[str]]]" = []
118
286
  try:
119
287
  root = ET.parse(str(xml_path)).getroot()
120
288
  except Exception:
121
289
  return out
122
290
  for elem in root.iter():
123
- tag = elem.tag.rsplit("}", 1)[-1] # strip any namespace
124
- if tag != "filter-mapping":
291
+ if _local(elem.tag) != "filter-mapping":
125
292
  continue
126
- for child in elem.iter():
127
- if child.tag.rsplit("}", 1)[-1] == "url-pattern" and (child.text or "").strip():
128
- out.append(child.text.strip())
293
+ patterns = [
294
+ child.text.strip()
295
+ for child in elem.iter()
296
+ if _local(child.tag) == "url-pattern" and (child.text or "").strip()
297
+ ]
298
+ if patterns:
299
+ out.append((_child_text(elem, "filter-name"), patterns))
300
+ return out
301
+
302
+
303
+ def _filter_classes_from_web_xml(xml_path: Path) -> "dict[str, str]":
304
+ """``<filter>`` declarations: filter-name → filter-class."""
305
+ out: "dict[str, str]" = {}
306
+ try:
307
+ root = ET.parse(str(xml_path)).getroot()
308
+ except Exception:
309
+ return out
310
+ for elem in root.iter():
311
+ if _local(elem.tag) != "filter":
312
+ continue
313
+ name = _child_text(elem, "filter-name")
314
+ klass = _child_text(elem, "filter-class")
315
+ if name and klass:
316
+ out[name] = klass
129
317
  return out
130
318
 
131
319
 
@@ -154,20 +342,63 @@ def build_filter_surface(
154
342
  if source not in surface.provenance[pattern]:
155
343
  surface.provenance[pattern].append(source)
156
344
 
345
+ # Simple name → repo-relative path, so a `<filter-class>` in web.xml can be read
346
+ # for what it actually does. Ambiguous simple names (two classes, one name) are
347
+ # dropped rather than guessed: the wrong body would produce a confident verdict
348
+ # about the wrong filter, which is the failure this row exists to remove.
349
+ by_simple_name: "dict[str, Optional[str]]" = {}
157
350
  for rel in java_files:
158
- abs_path = root / rel
351
+ stem = rel.rsplit("/", 1)[-1][: -len(".java")] if rel.endswith(".java") else ""
352
+ if not stem:
353
+ continue
354
+ by_simple_name[stem] = None if stem in by_simple_name else rel
355
+
356
+ def _read(rel: Optional[str]) -> str:
357
+ if not rel:
358
+ return ""
159
359
  try:
160
- src = abs_path.read_text(encoding="utf-8", errors="replace")
360
+ return (root / rel).read_text(encoding="utf-8", errors="replace")
161
361
  except OSError:
162
- continue
362
+ return ""
363
+
364
+ for rel in java_files:
365
+ src = _read(rel)
163
366
  if "@WebFilter" not in src:
164
367
  continue
165
- for pat in _patterns_from_webfilter(src):
368
+ patterns = _patterns_from_webfilter(src)
369
+ if not patterns:
370
+ continue
371
+ for pat in patterns:
166
372
  _add(pat, f"@WebFilter:{rel}")
373
+ verdict, evidence = authentication_verdict(src)
374
+ surface.filters.append(FilterDeclaration(
375
+ name=rel.rsplit("/", 1)[-1][: -len(".java")],
376
+ source=f"@WebFilter:{rel}",
377
+ patterns=tuple(patterns),
378
+ authenticates=verdict,
379
+ evidence=tuple(evidence),
380
+ implementation_file=rel,
381
+ ))
167
382
 
168
383
  for xml_path in sorted(root.rglob("web.xml")):
169
384
  rel = xml_path.relative_to(root).as_posix()
170
- for pat in _patterns_from_web_xml(xml_path):
171
- _add(pat, f"web.xml:{rel}")
385
+ classes = _filter_classes_from_web_xml(xml_path)
386
+ for name, patterns in _mappings_from_web_xml(xml_path):
387
+ for pat in patterns:
388
+ _add(pat, f"web.xml:{rel}")
389
+ klass = classes.get(name, "")
390
+ impl = by_simple_name.get(klass.rsplit(".", 1)[-1]) if klass else None
391
+ body = _read(impl)
392
+ verdict, evidence = (
393
+ authentication_verdict(body) if body else (AUTHENTICATION_UNKNOWN, [])
394
+ )
395
+ surface.filters.append(FilterDeclaration(
396
+ name=klass or name or "(unnamed filter-mapping)",
397
+ source=f"web.xml:{rel}",
398
+ patterns=tuple(patterns),
399
+ authenticates=verdict,
400
+ evidence=tuple(evidence),
401
+ implementation_file=impl,
402
+ ))
172
403
 
173
404
  return surface