sad-py 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sad/__init__.py +24 -0
- sad/__main__.py +3 -0
- sad/analyzer.py +255 -0
- sad/cli.py +74 -0
- sad/crawler/__init__.py +192 -0
- sad/html/__init__.py +94 -0
- sad/ingest/__init__.py +205 -0
- sad/javascript/__init__.py +3 -0
- sad/javascript/engine.py +1264 -0
- sad/javascript/parser.py +45 -0
- sad/javascript/resources.py +75 -0
- sad/javascript/sourcemaps.py +114 -0
- sad/javascript/values.py +249 -0
- sad/models.py +125 -0
- sad/normalization/__init__.py +114 -0
- sad/output/__init__.py +46 -0
- sad/protocols/__init__.py +50 -0
- sad/py.typed +0 -0
- sad_py-0.1.0.dist-info/METADATA +122 -0
- sad_py-0.1.0.dist-info/RECORD +23 -0
- sad_py-0.1.0.dist-info/WHEEL +4 -0
- sad_py-0.1.0.dist-info/entry_points.txt +2 -0
- sad_py-0.1.0.dist-info/licenses/LICENSE +21 -0
sad/__init__.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""SAD: Statistic API Discovery. Public API version 0.1."""
|
|
2
|
+
|
|
3
|
+
from .analyzer import Analyzer
|
|
4
|
+
from .models import (
|
|
5
|
+
AnalysisIssue,
|
|
6
|
+
AnalysisResult,
|
|
7
|
+
Dataset,
|
|
8
|
+
Endpoint,
|
|
9
|
+
Evidence,
|
|
10
|
+
HTTPExchange,
|
|
11
|
+
Resource,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
__version__ = "0.1.0"
|
|
15
|
+
__all__ = [
|
|
16
|
+
"Analyzer",
|
|
17
|
+
"AnalysisIssue",
|
|
18
|
+
"AnalysisResult",
|
|
19
|
+
"Dataset",
|
|
20
|
+
"Endpoint",
|
|
21
|
+
"Evidence",
|
|
22
|
+
"HTTPExchange",
|
|
23
|
+
"Resource",
|
|
24
|
+
]
|
sad/__main__.py
ADDED
sad/analyzer.py
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
"""Library-first analysis orchestration; all CLI and integrations use this API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from urllib.parse import urlsplit
|
|
7
|
+
|
|
8
|
+
from sad.crawler import CrawlConfig, Crawler
|
|
9
|
+
from sad.html import HTMLCollector
|
|
10
|
+
from sad.ingest import BurpImporter, read_bounded
|
|
11
|
+
from sad.javascript.engine import Engine
|
|
12
|
+
from sad.javascript.parser import ASTCache
|
|
13
|
+
from sad.javascript.resources import inline_source_map, references, source_map
|
|
14
|
+
from sad.javascript.sourcemaps import SourceMapIndex
|
|
15
|
+
from sad.models import AnalysisResult, Dataset, Endpoint, Evidence, HTTPExchange, Resource
|
|
16
|
+
from sad.normalization import correlate, normalize
|
|
17
|
+
from sad.protocols import SinkRegistry
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class Analyzer:
|
|
21
|
+
"""Reusable analyzer. AST cache is shared; each analysis has isolated value state.
|
|
22
|
+
|
|
23
|
+
Input files and datasets are offline. Only analyze_url/analyze_urls make requests.
|
|
24
|
+
Environment substitutions are explicit; the host environment is never imported.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
def __init__(
|
|
28
|
+
self,
|
|
29
|
+
*,
|
|
30
|
+
base_url: str | None = None,
|
|
31
|
+
environment: dict[str, str] | None = None,
|
|
32
|
+
registry: SinkRegistry | None = None,
|
|
33
|
+
max_steps: int = 200_000,
|
|
34
|
+
max_depth: int = 20,
|
|
35
|
+
max_resource_bytes: int = 8_000_000,
|
|
36
|
+
path_mappings: dict[str, str] | None = None,
|
|
37
|
+
) -> None:
|
|
38
|
+
if max_steps < 1:
|
|
39
|
+
raise ValueError("max_steps must be positive")
|
|
40
|
+
self.path_mappings = path_mappings or {}
|
|
41
|
+
self.base_url = base_url
|
|
42
|
+
self.environment = environment
|
|
43
|
+
self.registry = registry or SinkRegistry()
|
|
44
|
+
self.max_steps = max_steps
|
|
45
|
+
self.max_depth = max_depth
|
|
46
|
+
self.cache = ASTCache(max_bytes=max_resource_bytes)
|
|
47
|
+
|
|
48
|
+
def analyze_js(self, source: str, *, filename: str = "input.js") -> AnalysisResult:
|
|
49
|
+
return self.analyze_dataset(Dataset(resources=[Resource(filename, source)]))
|
|
50
|
+
|
|
51
|
+
def analyze_url(self, url: str, *, config: CrawlConfig | None = None) -> AnalysisResult:
|
|
52
|
+
return self.analyze_urls([url], config=config)
|
|
53
|
+
|
|
54
|
+
def analyze_urls(self, urls: list[str], *, config: CrawlConfig | None = None) -> AnalysisResult:
|
|
55
|
+
return self.analyze_dataset(Crawler(config, cache=self.cache).collect(urls))
|
|
56
|
+
|
|
57
|
+
def analyze_exchanges(self, exchanges: list[HTTPExchange]) -> AnalysisResult:
|
|
58
|
+
return self.analyze_dataset(Dataset(exchanges=exchanges))
|
|
59
|
+
|
|
60
|
+
def analyze_path(self, path: str | Path) -> AnalysisResult:
|
|
61
|
+
root = Path(path)
|
|
62
|
+
if not root.exists():
|
|
63
|
+
raise FileNotFoundError(path)
|
|
64
|
+
dataset = Dataset()
|
|
65
|
+
files = sorted(root.rglob("*")) if root.is_dir() else [root]
|
|
66
|
+
for file in files:
|
|
67
|
+
if not file.is_file() or file.is_symlink():
|
|
68
|
+
continue
|
|
69
|
+
if any(part in ("node_modules", ".git", ".venv", "__pycache__") for part in file.parts):
|
|
70
|
+
continue
|
|
71
|
+
suffix = file.suffix.lower()
|
|
72
|
+
if suffix in (".js", ".mjs", ".cjs", ".jsx", ".ts", ".tsx", ".html", ".htm", ".map"):
|
|
73
|
+
try:
|
|
74
|
+
content = read_bounded(file, self.cache.max_bytes).decode(
|
|
75
|
+
"utf-8", errors="replace"
|
|
76
|
+
)
|
|
77
|
+
media = (
|
|
78
|
+
"text/html"
|
|
79
|
+
if suffix in (".html", ".htm")
|
|
80
|
+
else "application/json"
|
|
81
|
+
if suffix == ".map"
|
|
82
|
+
else "application/javascript"
|
|
83
|
+
)
|
|
84
|
+
dataset.resources.append(Resource(file.as_posix(), content, media))
|
|
85
|
+
except (OSError, ValueError) as exc:
|
|
86
|
+
dataset.diagnostics.append(str(exc))
|
|
87
|
+
elif suffix in (".xml", ".har", ".http", ".json"):
|
|
88
|
+
try:
|
|
89
|
+
imported = BurpImporter.load(file, base_url=self.base_url)
|
|
90
|
+
dataset.resources.extend(imported.resources)
|
|
91
|
+
dataset.exchanges.extend(imported.exchanges)
|
|
92
|
+
dataset.diagnostics.extend(imported.diagnostics)
|
|
93
|
+
except (ValueError, OSError) as exc:
|
|
94
|
+
dataset.diagnostics.append(f"{file}: {exc}")
|
|
95
|
+
return self.analyze_dataset(dataset)
|
|
96
|
+
|
|
97
|
+
def analyze_dataset(self, dataset: Dataset) -> AnalysisResult:
|
|
98
|
+
result = AnalysisResult(diagnostics=list(dataset.diagnostics))
|
|
99
|
+
resources = list(dataset.resources)
|
|
100
|
+
for exchange in dataset.exchanges:
|
|
101
|
+
result.endpoints.append(
|
|
102
|
+
Endpoint(
|
|
103
|
+
method=exchange.method,
|
|
104
|
+
url=exchange.url,
|
|
105
|
+
protocol="websocket"
|
|
106
|
+
if exchange.request_headers.get("upgrade", "").lower() == "websocket"
|
|
107
|
+
else "http",
|
|
108
|
+
headers=exchange.request_headers or None,
|
|
109
|
+
body=exchange.request_body,
|
|
110
|
+
source_file=exchange.source_file,
|
|
111
|
+
content_type=exchange.request_headers.get("content-type"),
|
|
112
|
+
origin="observed",
|
|
113
|
+
evidence=[
|
|
114
|
+
Evidence(
|
|
115
|
+
"captured-request",
|
|
116
|
+
f"{exchange.method} {exchange.url}",
|
|
117
|
+
exchange.source_file,
|
|
118
|
+
)
|
|
119
|
+
],
|
|
120
|
+
observations=[
|
|
121
|
+
{
|
|
122
|
+
"url": exchange.url,
|
|
123
|
+
"method": exchange.method,
|
|
124
|
+
"status": exchange.status,
|
|
125
|
+
"headers": exchange.request_headers,
|
|
126
|
+
"body": exchange.request_body,
|
|
127
|
+
}
|
|
128
|
+
],
|
|
129
|
+
)
|
|
130
|
+
)
|
|
131
|
+
if exchange.response_body is not None:
|
|
132
|
+
resources.append(
|
|
133
|
+
Resource(
|
|
134
|
+
exchange.url,
|
|
135
|
+
exchange.response_body,
|
|
136
|
+
exchange.response_headers.get("content-type", ""),
|
|
137
|
+
exchange.source_file,
|
|
138
|
+
)
|
|
139
|
+
)
|
|
140
|
+
javascript = []
|
|
141
|
+
maps: dict[str, SourceMapIndex] = {}
|
|
142
|
+
map_aliases: dict[str, list[str]] = {}
|
|
143
|
+
origins: dict[str, str | None] = {}
|
|
144
|
+
seen: set[tuple[str, str]] = set()
|
|
145
|
+
for resource in resources:
|
|
146
|
+
key = (resource.url, resource.content)
|
|
147
|
+
if key in seen:
|
|
148
|
+
continue
|
|
149
|
+
seen.add(key)
|
|
150
|
+
if len(resource.content.encode("utf-8")) > self.cache.max_bytes:
|
|
151
|
+
result.diagnostics.append(f"resource too large: {resource.url}")
|
|
152
|
+
continue
|
|
153
|
+
source = resource.source_file or resource.url
|
|
154
|
+
base = (
|
|
155
|
+
self.base_url
|
|
156
|
+
or resource.document_url
|
|
157
|
+
or (resource.url if urlsplit(resource.url).scheme in ("http", "https") else None)
|
|
158
|
+
)
|
|
159
|
+
origins[source] = base
|
|
160
|
+
path = urlsplit(resource.url).path
|
|
161
|
+
try:
|
|
162
|
+
if "html" in resource.media_type or path.endswith((".html", ".htm")):
|
|
163
|
+
parser = HTMLCollector(resource)
|
|
164
|
+
parser.feed(resource.content)
|
|
165
|
+
javascript.extend(parser.inline)
|
|
166
|
+
result.endpoints.extend(parser.forms)
|
|
167
|
+
elif path.endswith(".map"):
|
|
168
|
+
maps[resource.url] = SourceMapIndex(resource)
|
|
169
|
+
maps[resource.url.removesuffix(".map")] = maps[resource.url]
|
|
170
|
+
expanded = source_map(resource, self.cache.max_bytes)
|
|
171
|
+
javascript.extend(expanded)
|
|
172
|
+
for item in expanded:
|
|
173
|
+
origins[item.source_file or item.url] = base
|
|
174
|
+
elif (
|
|
175
|
+
"javascript" in resource.media_type
|
|
176
|
+
or "ecmascript" in resource.media_type
|
|
177
|
+
or path.endswith((".js", ".mjs", ".cjs", ".ts", ".jsx", ".tsx"))
|
|
178
|
+
):
|
|
179
|
+
javascript.append(resource)
|
|
180
|
+
for reference in references(resource, self.cache):
|
|
181
|
+
if reference.startswith("data:"):
|
|
182
|
+
inline = inline_source_map(reference, self.cache.max_bytes)
|
|
183
|
+
inline = Resource(
|
|
184
|
+
resource.url + ".map", inline.content, inline.media_type
|
|
185
|
+
)
|
|
186
|
+
maps[resource.source_file or resource.url] = SourceMapIndex(inline)
|
|
187
|
+
expanded = source_map(inline, self.cache.max_bytes)
|
|
188
|
+
javascript.extend(expanded)
|
|
189
|
+
for item in expanded:
|
|
190
|
+
origins[item.source_file or item.url] = base
|
|
191
|
+
else:
|
|
192
|
+
map_aliases.setdefault(resource.source_file or resource.url, []).append(
|
|
193
|
+
reference
|
|
194
|
+
)
|
|
195
|
+
except (ValueError, RecursionError) as exc:
|
|
196
|
+
result.diagnostics.append(f"{resource.url}: {exc}")
|
|
197
|
+
for source, candidates in map_aliases.items():
|
|
198
|
+
for reference in candidates:
|
|
199
|
+
if reference in maps:
|
|
200
|
+
maps[source] = maps[reference]
|
|
201
|
+
engine = Engine(
|
|
202
|
+
self.cache,
|
|
203
|
+
self.registry,
|
|
204
|
+
max_steps=self.max_steps,
|
|
205
|
+
max_depth=self.max_depth,
|
|
206
|
+
environment=self.environment,
|
|
207
|
+
path_mappings=self.path_mappings,
|
|
208
|
+
)
|
|
209
|
+
static = engine.analyze(javascript)
|
|
210
|
+
result.endpoints.extend(static.endpoints)
|
|
211
|
+
result.diagnostics.extend(static.diagnostics)
|
|
212
|
+
result.call_graph = static.call_graph
|
|
213
|
+
result.issues.extend(static.issues)
|
|
214
|
+
result.analysis_incomplete = static.analysis_incomplete
|
|
215
|
+
valid = []
|
|
216
|
+
for endpoint in result.endpoints:
|
|
217
|
+
try:
|
|
218
|
+
valid.append(normalize(endpoint, origins.get(endpoint.source_file, self.base_url)))
|
|
219
|
+
mapping = maps.get(endpoint.source_file)
|
|
220
|
+
location = (
|
|
221
|
+
mapping.lookup(endpoint.source_line or 1, endpoint.source_column or 1)
|
|
222
|
+
if mapping
|
|
223
|
+
else None
|
|
224
|
+
)
|
|
225
|
+
if location:
|
|
226
|
+
endpoint.evidence.append(
|
|
227
|
+
Evidence(
|
|
228
|
+
"source-map",
|
|
229
|
+
f"Mapped from {endpoint.source_file}:{endpoint.source_line}",
|
|
230
|
+
location.source_file,
|
|
231
|
+
location.line,
|
|
232
|
+
location.column,
|
|
233
|
+
)
|
|
234
|
+
)
|
|
235
|
+
endpoint.source_file, endpoint.source_line, endpoint.source_column = (
|
|
236
|
+
location.source_file,
|
|
237
|
+
location.line,
|
|
238
|
+
location.column,
|
|
239
|
+
)
|
|
240
|
+
except ValueError as exc:
|
|
241
|
+
result.diagnostics.append(f"Invalid endpoint {endpoint.url!r}: {exc}")
|
|
242
|
+
result.endpoints = correlate(valid)
|
|
243
|
+
for diagnostic in list(result.diagnostics):
|
|
244
|
+
if not any(issue.reason == diagnostic for issue in result.issues):
|
|
245
|
+
result.add_issue("input-incomplete", diagnostic)
|
|
246
|
+
result.diagnostics = sorted(set(result.diagnostics))
|
|
247
|
+
result.issues.sort(
|
|
248
|
+
key=lambda issue: (
|
|
249
|
+
issue.source_file or "",
|
|
250
|
+
issue.source_line or 0,
|
|
251
|
+
issue.code,
|
|
252
|
+
issue.reason,
|
|
253
|
+
)
|
|
254
|
+
)
|
|
255
|
+
return result
|
sad/cli.py
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Thin command-line frontend; exit 2 for input errors, 1 with --strict diagnostics."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from sad import Analyzer, __version__
|
|
10
|
+
from sad.crawler import CrawlConfig
|
|
11
|
+
from sad.ingest import BurpImporter, read_bounded
|
|
12
|
+
from sad.output import render
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def main(argv: list[str] | None = None) -> int:
|
|
16
|
+
parser = argparse.ArgumentParser(prog="sad", description="Static frontend API discovery")
|
|
17
|
+
parser.add_argument("--version", action="version", version=__version__)
|
|
18
|
+
commands = parser.add_subparsers(dest="command", required=True)
|
|
19
|
+
for command in ("analyze", "analyze-js", "import-burp", "crawl"):
|
|
20
|
+
sub = commands.add_parser(command)
|
|
21
|
+
sub.add_argument("input", nargs="+" if command == "crawl" else None)
|
|
22
|
+
sub.add_argument("--base-url")
|
|
23
|
+
sub.add_argument("--format", choices=("json", "jsonl", "csv", "table"), default="table")
|
|
24
|
+
sub.add_argument("-o", "--output", type=Path)
|
|
25
|
+
sub.add_argument(
|
|
26
|
+
"--max-steps",
|
|
27
|
+
type=int,
|
|
28
|
+
default=200_000,
|
|
29
|
+
help="global static-analysis step budget, shared fairly among supplied sources",
|
|
30
|
+
)
|
|
31
|
+
sub.add_argument(
|
|
32
|
+
"--strict", action="store_true", help="exit 1 if any diagnostic is emitted"
|
|
33
|
+
)
|
|
34
|
+
if command == "crawl":
|
|
35
|
+
sub.add_argument("--depth", type=int, default=2)
|
|
36
|
+
sub.add_argument("--max-pages", type=int, default=30)
|
|
37
|
+
sub.add_argument("--max-resources", type=int, default=300)
|
|
38
|
+
sub.add_argument("--allow-origin", action="append", default=[])
|
|
39
|
+
sub.add_argument("--exclude", action="append", default=[])
|
|
40
|
+
args = parser.parse_args(argv)
|
|
41
|
+
try:
|
|
42
|
+
analyzer = Analyzer(base_url=args.base_url, max_steps=args.max_steps)
|
|
43
|
+
if args.command == "crawl":
|
|
44
|
+
result = analyzer.analyze_urls(
|
|
45
|
+
args.input,
|
|
46
|
+
config=CrawlConfig(
|
|
47
|
+
depth=args.depth,
|
|
48
|
+
max_pages=args.max_pages,
|
|
49
|
+
max_resources=args.max_resources,
|
|
50
|
+
allowed_origins=tuple(args.allow_origin),
|
|
51
|
+
exclusions=tuple(args.exclude),
|
|
52
|
+
),
|
|
53
|
+
)
|
|
54
|
+
elif args.command == "import-burp":
|
|
55
|
+
result = analyzer.analyze_dataset(BurpImporter.load(args.input, base_url=args.base_url))
|
|
56
|
+
elif args.command == "analyze-js":
|
|
57
|
+
result = analyzer.analyze_js(
|
|
58
|
+
read_bounded(args.input).decode("utf-8", errors="replace"), filename=args.input
|
|
59
|
+
)
|
|
60
|
+
else:
|
|
61
|
+
result = analyzer.analyze_path(args.input)
|
|
62
|
+
output = render(result, args.format)
|
|
63
|
+
if args.output:
|
|
64
|
+
args.output.write_text(output, encoding="utf-8")
|
|
65
|
+
else:
|
|
66
|
+
sys.stdout.write(output)
|
|
67
|
+
if result.analysis_incomplete:
|
|
68
|
+
print("sad: analysis incomplete; results are partial", file=sys.stderr)
|
|
69
|
+
for diagnostic in result.diagnostics:
|
|
70
|
+
print(f"sad: {diagnostic}", file=sys.stderr)
|
|
71
|
+
return 1 if args.strict and result.diagnostics else 0
|
|
72
|
+
except (OSError, ValueError) as exc:
|
|
73
|
+
print(f"sad: {exc}", file=sys.stderr)
|
|
74
|
+
return 2
|
sad/crawler/__init__.py
ADDED
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
"""Bounded resource collection with scope-checked canonical seed redirects."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections import deque
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from fnmatch import fnmatch
|
|
8
|
+
from urllib.parse import urljoin, urlsplit
|
|
9
|
+
|
|
10
|
+
import httpx
|
|
11
|
+
|
|
12
|
+
from sad.html import HTMLCollector
|
|
13
|
+
from sad.javascript.parser import ASTCache
|
|
14
|
+
from sad.javascript.resources import references
|
|
15
|
+
from sad.models import Dataset, Resource
|
|
16
|
+
from sad.normalization import normalize_url
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def origin(url: str) -> tuple[str, str | None, int | None]:
|
|
20
|
+
parts = urlsplit(url)
|
|
21
|
+
return parts.scheme, parts.hostname, parts.port or (443 if parts.scheme == "https" else 80)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class CrawlConfig:
|
|
26
|
+
depth: int = 2
|
|
27
|
+
max_pages: int = 30
|
|
28
|
+
max_resources: int = 300
|
|
29
|
+
max_bytes: int = 8_000_000
|
|
30
|
+
max_total_bytes: int = 64_000_000
|
|
31
|
+
timeout: float = 10.0
|
|
32
|
+
allowed_origins: tuple[str, ...] = ()
|
|
33
|
+
exclusions: tuple[str, ...] = ()
|
|
34
|
+
max_redirects: int = 5
|
|
35
|
+
|
|
36
|
+
def __post_init__(self) -> None:
|
|
37
|
+
if (
|
|
38
|
+
self.depth < 0
|
|
39
|
+
or min(self.max_pages, self.max_resources, self.max_bytes, self.max_total_bytes) < 1
|
|
40
|
+
):
|
|
41
|
+
raise ValueError("crawl limits must be positive; depth may be zero")
|
|
42
|
+
if self.timeout <= 0 or self.max_redirects < 0:
|
|
43
|
+
raise ValueError("invalid timeout or redirect limit")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class Crawler:
|
|
47
|
+
def __init__(
|
|
48
|
+
self,
|
|
49
|
+
config: CrawlConfig | None = None,
|
|
50
|
+
*,
|
|
51
|
+
client: httpx.Client | None = None,
|
|
52
|
+
cache: ASTCache | None = None,
|
|
53
|
+
) -> None:
|
|
54
|
+
self.config = config or CrawlConfig()
|
|
55
|
+
self.client = client
|
|
56
|
+
self.cache = cache or ASTCache()
|
|
57
|
+
|
|
58
|
+
def collect(self, seeds: list[str]) -> Dataset:
|
|
59
|
+
cfg = self.config
|
|
60
|
+
for seed in seeds:
|
|
61
|
+
if urlsplit(seed).scheme not in ("http", "https"):
|
|
62
|
+
raise ValueError("crawl seeds must be HTTP(S) URLs")
|
|
63
|
+
allowed = {origin(s) for s in [*seeds, *cfg.allowed_origins]}
|
|
64
|
+
seed_urls = {normalize_url(s) for s in seeds}
|
|
65
|
+
|
|
66
|
+
def canonical_redirect(source: str, target: str) -> bool:
|
|
67
|
+
source_scheme, source_host, source_port = origin(source)
|
|
68
|
+
target_scheme, target_host, target_port = origin(target)
|
|
69
|
+
return (
|
|
70
|
+
bool(source_host and target_host)
|
|
71
|
+
and (
|
|
72
|
+
source_host == target_host
|
|
73
|
+
or target_host == f"www.{source_host}"
|
|
74
|
+
or source_host == f"www.{target_host}"
|
|
75
|
+
)
|
|
76
|
+
and (
|
|
77
|
+
source_scheme == target_scheme
|
|
78
|
+
or (source_scheme, target_scheme) == ("http", "https")
|
|
79
|
+
)
|
|
80
|
+
and source_port == (443 if source_scheme == "https" else 80)
|
|
81
|
+
and target_port == (443 if target_scheme == "https" else 80)
|
|
82
|
+
and target_scheme in ("http", "https")
|
|
83
|
+
and not urlsplit(target).username
|
|
84
|
+
and not any(fnmatch(target, x) for x in cfg.exclusions)
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
def in_scope(url: str) -> bool:
|
|
88
|
+
parts = urlsplit(url)
|
|
89
|
+
return (
|
|
90
|
+
parts.scheme in ("http", "https")
|
|
91
|
+
and not parts.username
|
|
92
|
+
and origin(url) in allowed
|
|
93
|
+
and not any(fnmatch(url, x) for x in cfg.exclusions)
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
queue = deque((normalize_url(s), 0, True, s) for s in seeds)
|
|
97
|
+
seen: set[str] = set()
|
|
98
|
+
dataset = Dataset()
|
|
99
|
+
pages = total_bytes = 0
|
|
100
|
+
client = self.client or httpx.Client(
|
|
101
|
+
timeout=cfg.timeout, follow_redirects=False, trust_env=False
|
|
102
|
+
)
|
|
103
|
+
try:
|
|
104
|
+
while queue and len(seen) < cfg.max_resources:
|
|
105
|
+
url, depth, page, document = queue.popleft()
|
|
106
|
+
try:
|
|
107
|
+
url = normalize_url(url)
|
|
108
|
+
except ValueError as exc:
|
|
109
|
+
dataset.diagnostics.append(f"Invalid resource URL: {exc}")
|
|
110
|
+
continue
|
|
111
|
+
if url in seen or not in_scope(url) or (page and pages >= cfg.max_pages):
|
|
112
|
+
continue
|
|
113
|
+
seen.add(url)
|
|
114
|
+
if page:
|
|
115
|
+
pages += 1
|
|
116
|
+
try:
|
|
117
|
+
current = url
|
|
118
|
+
for redirect in range(cfg.max_redirects + 1):
|
|
119
|
+
with client.stream(
|
|
120
|
+
"GET",
|
|
121
|
+
current,
|
|
122
|
+
follow_redirects=False,
|
|
123
|
+
timeout=cfg.timeout,
|
|
124
|
+
headers={"User-Agent": "SAD/0.1 static-resource-collector"},
|
|
125
|
+
) as response:
|
|
126
|
+
if response.is_redirect:
|
|
127
|
+
target = normalize_url(
|
|
128
|
+
urljoin(current, response.headers.get("location", ""))
|
|
129
|
+
)
|
|
130
|
+
if redirect == cfg.max_redirects:
|
|
131
|
+
raise ValueError(f"blocked redirect to {target}")
|
|
132
|
+
if not in_scope(target):
|
|
133
|
+
if (
|
|
134
|
+
url in seed_urls
|
|
135
|
+
and canonical_redirect(url, target)
|
|
136
|
+
and canonical_redirect(current, target)
|
|
137
|
+
):
|
|
138
|
+
allowed.add(origin(target))
|
|
139
|
+
else:
|
|
140
|
+
raise ValueError(f"blocked redirect to {target}")
|
|
141
|
+
current = target
|
|
142
|
+
continue
|
|
143
|
+
response.raise_for_status()
|
|
144
|
+
chunks = bytearray()
|
|
145
|
+
for chunk in response.iter_bytes():
|
|
146
|
+
total_bytes += len(chunk)
|
|
147
|
+
if (
|
|
148
|
+
len(chunks) + len(chunk) > cfg.max_bytes
|
|
149
|
+
or total_bytes > cfg.max_total_bytes
|
|
150
|
+
):
|
|
151
|
+
raise ValueError("download byte budget exceeded")
|
|
152
|
+
chunks.extend(chunk)
|
|
153
|
+
media = response.headers.get("content-type", "").split(";", 1)[0]
|
|
154
|
+
resource = Resource(
|
|
155
|
+
current,
|
|
156
|
+
chunks.decode("utf-8", errors="replace"),
|
|
157
|
+
media,
|
|
158
|
+
document_url=current if "html" in media else document,
|
|
159
|
+
)
|
|
160
|
+
dataset.resources.append(resource)
|
|
161
|
+
seen.add(current)
|
|
162
|
+
break
|
|
163
|
+
if total_bytes > cfg.max_total_bytes:
|
|
164
|
+
break
|
|
165
|
+
if "html" in media:
|
|
166
|
+
parser = HTMLCollector(resource)
|
|
167
|
+
parser.feed(resource.content)
|
|
168
|
+
queue.extend((s, depth, False, parser.base) for s in parser.scripts)
|
|
169
|
+
if depth < cfg.depth:
|
|
170
|
+
queue.extend((s, depth + 1, True, s) for s in parser.links)
|
|
171
|
+
elif (
|
|
172
|
+
"javascript" in media
|
|
173
|
+
or "ecmascript" in media
|
|
174
|
+
or urlsplit(current).path.endswith((".js", ".mjs", ".ts"))
|
|
175
|
+
):
|
|
176
|
+
queue.extend(
|
|
177
|
+
(s, depth, False, document)
|
|
178
|
+
for s in references(resource, self.cache)
|
|
179
|
+
if not s.startswith("data:")
|
|
180
|
+
)
|
|
181
|
+
except (httpx.HTTPError, ValueError) as exc:
|
|
182
|
+
dataset.diagnostics.append(f"{url}: {exc}")
|
|
183
|
+
if total_bytes > cfg.max_total_bytes:
|
|
184
|
+
break
|
|
185
|
+
if queue:
|
|
186
|
+
dataset.diagnostics.append(
|
|
187
|
+
"Collection stopped with queued resources; a configured limit was reached"
|
|
188
|
+
)
|
|
189
|
+
finally:
|
|
190
|
+
if self.client is None:
|
|
191
|
+
client.close()
|
|
192
|
+
return dataset
|
sad/html/__init__.py
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""HTML resource and declarative form extraction using the standard HTML parser."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from html.parser import HTMLParser
|
|
6
|
+
from urllib.parse import urljoin
|
|
7
|
+
|
|
8
|
+
from sad.models import Endpoint, Evidence, Resource
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class HTMLCollector(HTMLParser):
|
|
12
|
+
def __init__(self, resource: Resource) -> None:
|
|
13
|
+
super().__init__(convert_charrefs=True)
|
|
14
|
+
self.resource = resource
|
|
15
|
+
self.base = resource.url
|
|
16
|
+
self.links: list[str] = []
|
|
17
|
+
self.scripts: list[str] = []
|
|
18
|
+
self.inline: list[Resource] = []
|
|
19
|
+
self.forms: list[Endpoint] = []
|
|
20
|
+
self._script: list[str] | None = None
|
|
21
|
+
self._line = 0
|
|
22
|
+
self._form: Endpoint | None = None
|
|
23
|
+
self._base_seen = False
|
|
24
|
+
|
|
25
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
26
|
+
values = dict(attrs)
|
|
27
|
+
line, col = self.getpos()
|
|
28
|
+
if tag == "base" and values.get("href") and not self._base_seen:
|
|
29
|
+
self.base = urljoin(self.resource.url, values["href"] or "")
|
|
30
|
+
self._base_seen = True
|
|
31
|
+
if tag == "a" and values.get("href"):
|
|
32
|
+
self.links.append(urljoin(self.base, values["href"] or ""))
|
|
33
|
+
if tag == "link" and values.get("rel") in ("modulepreload", "preload"):
|
|
34
|
+
if values.get("as") == "script" or values.get("rel") == "modulepreload":
|
|
35
|
+
self.scripts.append(urljoin(self.base, values.get("href") or ""))
|
|
36
|
+
if tag == "script":
|
|
37
|
+
script_type = (values.get("type") or "").lower()
|
|
38
|
+
if script_type not in ("", "module", "text/javascript", "application/javascript"):
|
|
39
|
+
return
|
|
40
|
+
if values.get("src"):
|
|
41
|
+
self.scripts.append(urljoin(self.base, values["src"] or ""))
|
|
42
|
+
else:
|
|
43
|
+
self._script, self._line = [], line - 1
|
|
44
|
+
if tag == "form":
|
|
45
|
+
method = (values.get("method") or "GET").upper()
|
|
46
|
+
if method == "DIALOG":
|
|
47
|
+
return
|
|
48
|
+
self._form = Endpoint(
|
|
49
|
+
method=method,
|
|
50
|
+
url=urljoin(self.base, values.get("action") or self.resource.url),
|
|
51
|
+
source_file=self.resource.source_file or self.resource.url,
|
|
52
|
+
source_line=line,
|
|
53
|
+
source_column=col + 1,
|
|
54
|
+
framework="html-form",
|
|
55
|
+
content_type=values.get("enctype") or "application/x-www-form-urlencoded",
|
|
56
|
+
body={} if method != "GET" else None,
|
|
57
|
+
evidence=[
|
|
58
|
+
Evidence(
|
|
59
|
+
"html-form",
|
|
60
|
+
self.get_starttag_text() or "",
|
|
61
|
+
self.resource.url,
|
|
62
|
+
line,
|
|
63
|
+
col + 1,
|
|
64
|
+
)
|
|
65
|
+
],
|
|
66
|
+
)
|
|
67
|
+
self.forms.append(self._form)
|
|
68
|
+
if tag in ("input", "textarea", "select", "button") and self._form and values.get("name"):
|
|
69
|
+
if "disabled" in values:
|
|
70
|
+
return
|
|
71
|
+
name = values["name"] or ""
|
|
72
|
+
value = values.get("value")
|
|
73
|
+
if self._form.method == "GET":
|
|
74
|
+
self._form.query_parameters.append({"name": name, "value": value})
|
|
75
|
+
else:
|
|
76
|
+
self._form.body[name] = value
|
|
77
|
+
|
|
78
|
+
def handle_data(self, data: str) -> None:
|
|
79
|
+
if self._script is not None:
|
|
80
|
+
self._script.append(data)
|
|
81
|
+
|
|
82
|
+
def handle_endtag(self, tag: str) -> None:
|
|
83
|
+
if tag == "script" and self._script is not None:
|
|
84
|
+
self.inline.append(
|
|
85
|
+
Resource(
|
|
86
|
+
self.resource.url + f"#inline-{len(self.inline)}.js",
|
|
87
|
+
"".join(self._script),
|
|
88
|
+
source_file=self.resource.source_file or self.resource.url,
|
|
89
|
+
line_offset=self._line,
|
|
90
|
+
)
|
|
91
|
+
)
|
|
92
|
+
self._script = None
|
|
93
|
+
if tag == "form":
|
|
94
|
+
self._form = None
|