sad-py 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sad/__init__.py ADDED
@@ -0,0 +1,24 @@
1
+ """SAD: Statistic API Discovery. Public API version 0.1."""
2
+
3
+ from .analyzer import Analyzer
4
+ from .models import (
5
+ AnalysisIssue,
6
+ AnalysisResult,
7
+ Dataset,
8
+ Endpoint,
9
+ Evidence,
10
+ HTTPExchange,
11
+ Resource,
12
+ )
13
+
14
+ __version__ = "0.1.0"
15
+ __all__ = [
16
+ "Analyzer",
17
+ "AnalysisIssue",
18
+ "AnalysisResult",
19
+ "Dataset",
20
+ "Endpoint",
21
+ "Evidence",
22
+ "HTTPExchange",
23
+ "Resource",
24
+ ]
sad/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ raise SystemExit(main())
sad/analyzer.py ADDED
@@ -0,0 +1,255 @@
1
+ """Library-first analysis orchestration; all CLI and integrations use this API."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from urllib.parse import urlsplit
7
+
8
+ from sad.crawler import CrawlConfig, Crawler
9
+ from sad.html import HTMLCollector
10
+ from sad.ingest import BurpImporter, read_bounded
11
+ from sad.javascript.engine import Engine
12
+ from sad.javascript.parser import ASTCache
13
+ from sad.javascript.resources import inline_source_map, references, source_map
14
+ from sad.javascript.sourcemaps import SourceMapIndex
15
+ from sad.models import AnalysisResult, Dataset, Endpoint, Evidence, HTTPExchange, Resource
16
+ from sad.normalization import correlate, normalize
17
+ from sad.protocols import SinkRegistry
18
+
19
+
20
+ class Analyzer:
21
+ """Reusable analyzer. AST cache is shared; each analysis has isolated value state.
22
+
23
+ Input files and datasets are offline. Only analyze_url/analyze_urls make requests.
24
+ Environment substitutions are explicit; the host environment is never imported.
25
+ """
26
+
27
+ def __init__(
28
+ self,
29
+ *,
30
+ base_url: str | None = None,
31
+ environment: dict[str, str] | None = None,
32
+ registry: SinkRegistry | None = None,
33
+ max_steps: int = 200_000,
34
+ max_depth: int = 20,
35
+ max_resource_bytes: int = 8_000_000,
36
+ path_mappings: dict[str, str] | None = None,
37
+ ) -> None:
38
+ if max_steps < 1:
39
+ raise ValueError("max_steps must be positive")
40
+ self.path_mappings = path_mappings or {}
41
+ self.base_url = base_url
42
+ self.environment = environment
43
+ self.registry = registry or SinkRegistry()
44
+ self.max_steps = max_steps
45
+ self.max_depth = max_depth
46
+ self.cache = ASTCache(max_bytes=max_resource_bytes)
47
+
48
+ def analyze_js(self, source: str, *, filename: str = "input.js") -> AnalysisResult:
49
+ return self.analyze_dataset(Dataset(resources=[Resource(filename, source)]))
50
+
51
+ def analyze_url(self, url: str, *, config: CrawlConfig | None = None) -> AnalysisResult:
52
+ return self.analyze_urls([url], config=config)
53
+
54
+ def analyze_urls(self, urls: list[str], *, config: CrawlConfig | None = None) -> AnalysisResult:
55
+ return self.analyze_dataset(Crawler(config, cache=self.cache).collect(urls))
56
+
57
+ def analyze_exchanges(self, exchanges: list[HTTPExchange]) -> AnalysisResult:
58
+ return self.analyze_dataset(Dataset(exchanges=exchanges))
59
+
60
+ def analyze_path(self, path: str | Path) -> AnalysisResult:
61
+ root = Path(path)
62
+ if not root.exists():
63
+ raise FileNotFoundError(path)
64
+ dataset = Dataset()
65
+ files = sorted(root.rglob("*")) if root.is_dir() else [root]
66
+ for file in files:
67
+ if not file.is_file() or file.is_symlink():
68
+ continue
69
+ if any(part in ("node_modules", ".git", ".venv", "__pycache__") for part in file.parts):
70
+ continue
71
+ suffix = file.suffix.lower()
72
+ if suffix in (".js", ".mjs", ".cjs", ".jsx", ".ts", ".tsx", ".html", ".htm", ".map"):
73
+ try:
74
+ content = read_bounded(file, self.cache.max_bytes).decode(
75
+ "utf-8", errors="replace"
76
+ )
77
+ media = (
78
+ "text/html"
79
+ if suffix in (".html", ".htm")
80
+ else "application/json"
81
+ if suffix == ".map"
82
+ else "application/javascript"
83
+ )
84
+ dataset.resources.append(Resource(file.as_posix(), content, media))
85
+ except (OSError, ValueError) as exc:
86
+ dataset.diagnostics.append(str(exc))
87
+ elif suffix in (".xml", ".har", ".http", ".json"):
88
+ try:
89
+ imported = BurpImporter.load(file, base_url=self.base_url)
90
+ dataset.resources.extend(imported.resources)
91
+ dataset.exchanges.extend(imported.exchanges)
92
+ dataset.diagnostics.extend(imported.diagnostics)
93
+ except (ValueError, OSError) as exc:
94
+ dataset.diagnostics.append(f"{file}: {exc}")
95
+ return self.analyze_dataset(dataset)
96
+
97
+ def analyze_dataset(self, dataset: Dataset) -> AnalysisResult:
98
+ result = AnalysisResult(diagnostics=list(dataset.diagnostics))
99
+ resources = list(dataset.resources)
100
+ for exchange in dataset.exchanges:
101
+ result.endpoints.append(
102
+ Endpoint(
103
+ method=exchange.method,
104
+ url=exchange.url,
105
+ protocol="websocket"
106
+ if exchange.request_headers.get("upgrade", "").lower() == "websocket"
107
+ else "http",
108
+ headers=exchange.request_headers or None,
109
+ body=exchange.request_body,
110
+ source_file=exchange.source_file,
111
+ content_type=exchange.request_headers.get("content-type"),
112
+ origin="observed",
113
+ evidence=[
114
+ Evidence(
115
+ "captured-request",
116
+ f"{exchange.method} {exchange.url}",
117
+ exchange.source_file,
118
+ )
119
+ ],
120
+ observations=[
121
+ {
122
+ "url": exchange.url,
123
+ "method": exchange.method,
124
+ "status": exchange.status,
125
+ "headers": exchange.request_headers,
126
+ "body": exchange.request_body,
127
+ }
128
+ ],
129
+ )
130
+ )
131
+ if exchange.response_body is not None:
132
+ resources.append(
133
+ Resource(
134
+ exchange.url,
135
+ exchange.response_body,
136
+ exchange.response_headers.get("content-type", ""),
137
+ exchange.source_file,
138
+ )
139
+ )
140
+ javascript = []
141
+ maps: dict[str, SourceMapIndex] = {}
142
+ map_aliases: dict[str, list[str]] = {}
143
+ origins: dict[str, str | None] = {}
144
+ seen: set[tuple[str, str]] = set()
145
+ for resource in resources:
146
+ key = (resource.url, resource.content)
147
+ if key in seen:
148
+ continue
149
+ seen.add(key)
150
+ if len(resource.content.encode("utf-8")) > self.cache.max_bytes:
151
+ result.diagnostics.append(f"resource too large: {resource.url}")
152
+ continue
153
+ source = resource.source_file or resource.url
154
+ base = (
155
+ self.base_url
156
+ or resource.document_url
157
+ or (resource.url if urlsplit(resource.url).scheme in ("http", "https") else None)
158
+ )
159
+ origins[source] = base
160
+ path = urlsplit(resource.url).path
161
+ try:
162
+ if "html" in resource.media_type or path.endswith((".html", ".htm")):
163
+ parser = HTMLCollector(resource)
164
+ parser.feed(resource.content)
165
+ javascript.extend(parser.inline)
166
+ result.endpoints.extend(parser.forms)
167
+ elif path.endswith(".map"):
168
+ maps[resource.url] = SourceMapIndex(resource)
169
+ maps[resource.url.removesuffix(".map")] = maps[resource.url]
170
+ expanded = source_map(resource, self.cache.max_bytes)
171
+ javascript.extend(expanded)
172
+ for item in expanded:
173
+ origins[item.source_file or item.url] = base
174
+ elif (
175
+ "javascript" in resource.media_type
176
+ or "ecmascript" in resource.media_type
177
+ or path.endswith((".js", ".mjs", ".cjs", ".ts", ".jsx", ".tsx"))
178
+ ):
179
+ javascript.append(resource)
180
+ for reference in references(resource, self.cache):
181
+ if reference.startswith("data:"):
182
+ inline = inline_source_map(reference, self.cache.max_bytes)
183
+ inline = Resource(
184
+ resource.url + ".map", inline.content, inline.media_type
185
+ )
186
+ maps[resource.source_file or resource.url] = SourceMapIndex(inline)
187
+ expanded = source_map(inline, self.cache.max_bytes)
188
+ javascript.extend(expanded)
189
+ for item in expanded:
190
+ origins[item.source_file or item.url] = base
191
+ else:
192
+ map_aliases.setdefault(resource.source_file or resource.url, []).append(
193
+ reference
194
+ )
195
+ except (ValueError, RecursionError) as exc:
196
+ result.diagnostics.append(f"{resource.url}: {exc}")
197
+ for source, candidates in map_aliases.items():
198
+ for reference in candidates:
199
+ if reference in maps:
200
+ maps[source] = maps[reference]
201
+ engine = Engine(
202
+ self.cache,
203
+ self.registry,
204
+ max_steps=self.max_steps,
205
+ max_depth=self.max_depth,
206
+ environment=self.environment,
207
+ path_mappings=self.path_mappings,
208
+ )
209
+ static = engine.analyze(javascript)
210
+ result.endpoints.extend(static.endpoints)
211
+ result.diagnostics.extend(static.diagnostics)
212
+ result.call_graph = static.call_graph
213
+ result.issues.extend(static.issues)
214
+ result.analysis_incomplete = static.analysis_incomplete
215
+ valid = []
216
+ for endpoint in result.endpoints:
217
+ try:
218
+ valid.append(normalize(endpoint, origins.get(endpoint.source_file, self.base_url)))
219
+ mapping = maps.get(endpoint.source_file)
220
+ location = (
221
+ mapping.lookup(endpoint.source_line or 1, endpoint.source_column or 1)
222
+ if mapping
223
+ else None
224
+ )
225
+ if location:
226
+ endpoint.evidence.append(
227
+ Evidence(
228
+ "source-map",
229
+ f"Mapped from {endpoint.source_file}:{endpoint.source_line}",
230
+ location.source_file,
231
+ location.line,
232
+ location.column,
233
+ )
234
+ )
235
+ endpoint.source_file, endpoint.source_line, endpoint.source_column = (
236
+ location.source_file,
237
+ location.line,
238
+ location.column,
239
+ )
240
+ except ValueError as exc:
241
+ result.diagnostics.append(f"Invalid endpoint {endpoint.url!r}: {exc}")
242
+ result.endpoints = correlate(valid)
243
+ for diagnostic in list(result.diagnostics):
244
+ if not any(issue.reason == diagnostic for issue in result.issues):
245
+ result.add_issue("input-incomplete", diagnostic)
246
+ result.diagnostics = sorted(set(result.diagnostics))
247
+ result.issues.sort(
248
+ key=lambda issue: (
249
+ issue.source_file or "",
250
+ issue.source_line or 0,
251
+ issue.code,
252
+ issue.reason,
253
+ )
254
+ )
255
+ return result
sad/cli.py ADDED
@@ -0,0 +1,74 @@
1
+ """Thin command-line frontend; exit 2 for input errors, 1 with --strict diagnostics."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import sys
7
+ from pathlib import Path
8
+
9
+ from sad import Analyzer, __version__
10
+ from sad.crawler import CrawlConfig
11
+ from sad.ingest import BurpImporter, read_bounded
12
+ from sad.output import render
13
+
14
+
15
+ def main(argv: list[str] | None = None) -> int:
16
+ parser = argparse.ArgumentParser(prog="sad", description="Static frontend API discovery")
17
+ parser.add_argument("--version", action="version", version=__version__)
18
+ commands = parser.add_subparsers(dest="command", required=True)
19
+ for command in ("analyze", "analyze-js", "import-burp", "crawl"):
20
+ sub = commands.add_parser(command)
21
+ sub.add_argument("input", nargs="+" if command == "crawl" else None)
22
+ sub.add_argument("--base-url")
23
+ sub.add_argument("--format", choices=("json", "jsonl", "csv", "table"), default="table")
24
+ sub.add_argument("-o", "--output", type=Path)
25
+ sub.add_argument(
26
+ "--max-steps",
27
+ type=int,
28
+ default=200_000,
29
+ help="global static-analysis step budget, shared fairly among supplied sources",
30
+ )
31
+ sub.add_argument(
32
+ "--strict", action="store_true", help="exit 1 if any diagnostic is emitted"
33
+ )
34
+ if command == "crawl":
35
+ sub.add_argument("--depth", type=int, default=2)
36
+ sub.add_argument("--max-pages", type=int, default=30)
37
+ sub.add_argument("--max-resources", type=int, default=300)
38
+ sub.add_argument("--allow-origin", action="append", default=[])
39
+ sub.add_argument("--exclude", action="append", default=[])
40
+ args = parser.parse_args(argv)
41
+ try:
42
+ analyzer = Analyzer(base_url=args.base_url, max_steps=args.max_steps)
43
+ if args.command == "crawl":
44
+ result = analyzer.analyze_urls(
45
+ args.input,
46
+ config=CrawlConfig(
47
+ depth=args.depth,
48
+ max_pages=args.max_pages,
49
+ max_resources=args.max_resources,
50
+ allowed_origins=tuple(args.allow_origin),
51
+ exclusions=tuple(args.exclude),
52
+ ),
53
+ )
54
+ elif args.command == "import-burp":
55
+ result = analyzer.analyze_dataset(BurpImporter.load(args.input, base_url=args.base_url))
56
+ elif args.command == "analyze-js":
57
+ result = analyzer.analyze_js(
58
+ read_bounded(args.input).decode("utf-8", errors="replace"), filename=args.input
59
+ )
60
+ else:
61
+ result = analyzer.analyze_path(args.input)
62
+ output = render(result, args.format)
63
+ if args.output:
64
+ args.output.write_text(output, encoding="utf-8")
65
+ else:
66
+ sys.stdout.write(output)
67
+ if result.analysis_incomplete:
68
+ print("sad: analysis incomplete; results are partial", file=sys.stderr)
69
+ for diagnostic in result.diagnostics:
70
+ print(f"sad: {diagnostic}", file=sys.stderr)
71
+ return 1 if args.strict and result.diagnostics else 0
72
+ except (OSError, ValueError) as exc:
73
+ print(f"sad: {exc}", file=sys.stderr)
74
+ return 2
@@ -0,0 +1,192 @@
1
+ """Bounded resource collection with scope-checked canonical seed redirects."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections import deque
6
+ from dataclasses import dataclass
7
+ from fnmatch import fnmatch
8
+ from urllib.parse import urljoin, urlsplit
9
+
10
+ import httpx
11
+
12
+ from sad.html import HTMLCollector
13
+ from sad.javascript.parser import ASTCache
14
+ from sad.javascript.resources import references
15
+ from sad.models import Dataset, Resource
16
+ from sad.normalization import normalize_url
17
+
18
+
19
+ def origin(url: str) -> tuple[str, str | None, int | None]:
20
+ parts = urlsplit(url)
21
+ return parts.scheme, parts.hostname, parts.port or (443 if parts.scheme == "https" else 80)
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class CrawlConfig:
26
+ depth: int = 2
27
+ max_pages: int = 30
28
+ max_resources: int = 300
29
+ max_bytes: int = 8_000_000
30
+ max_total_bytes: int = 64_000_000
31
+ timeout: float = 10.0
32
+ allowed_origins: tuple[str, ...] = ()
33
+ exclusions: tuple[str, ...] = ()
34
+ max_redirects: int = 5
35
+
36
+ def __post_init__(self) -> None:
37
+ if (
38
+ self.depth < 0
39
+ or min(self.max_pages, self.max_resources, self.max_bytes, self.max_total_bytes) < 1
40
+ ):
41
+ raise ValueError("crawl limits must be positive; depth may be zero")
42
+ if self.timeout <= 0 or self.max_redirects < 0:
43
+ raise ValueError("invalid timeout or redirect limit")
44
+
45
+
46
+ class Crawler:
47
+ def __init__(
48
+ self,
49
+ config: CrawlConfig | None = None,
50
+ *,
51
+ client: httpx.Client | None = None,
52
+ cache: ASTCache | None = None,
53
+ ) -> None:
54
+ self.config = config or CrawlConfig()
55
+ self.client = client
56
+ self.cache = cache or ASTCache()
57
+
58
+ def collect(self, seeds: list[str]) -> Dataset:
59
+ cfg = self.config
60
+ for seed in seeds:
61
+ if urlsplit(seed).scheme not in ("http", "https"):
62
+ raise ValueError("crawl seeds must be HTTP(S) URLs")
63
+ allowed = {origin(s) for s in [*seeds, *cfg.allowed_origins]}
64
+ seed_urls = {normalize_url(s) for s in seeds}
65
+
66
+ def canonical_redirect(source: str, target: str) -> bool:
67
+ source_scheme, source_host, source_port = origin(source)
68
+ target_scheme, target_host, target_port = origin(target)
69
+ return (
70
+ bool(source_host and target_host)
71
+ and (
72
+ source_host == target_host
73
+ or target_host == f"www.{source_host}"
74
+ or source_host == f"www.{target_host}"
75
+ )
76
+ and (
77
+ source_scheme == target_scheme
78
+ or (source_scheme, target_scheme) == ("http", "https")
79
+ )
80
+ and source_port == (443 if source_scheme == "https" else 80)
81
+ and target_port == (443 if target_scheme == "https" else 80)
82
+ and target_scheme in ("http", "https")
83
+ and not urlsplit(target).username
84
+ and not any(fnmatch(target, x) for x in cfg.exclusions)
85
+ )
86
+
87
+ def in_scope(url: str) -> bool:
88
+ parts = urlsplit(url)
89
+ return (
90
+ parts.scheme in ("http", "https")
91
+ and not parts.username
92
+ and origin(url) in allowed
93
+ and not any(fnmatch(url, x) for x in cfg.exclusions)
94
+ )
95
+
96
+ queue = deque((normalize_url(s), 0, True, s) for s in seeds)
97
+ seen: set[str] = set()
98
+ dataset = Dataset()
99
+ pages = total_bytes = 0
100
+ client = self.client or httpx.Client(
101
+ timeout=cfg.timeout, follow_redirects=False, trust_env=False
102
+ )
103
+ try:
104
+ while queue and len(seen) < cfg.max_resources:
105
+ url, depth, page, document = queue.popleft()
106
+ try:
107
+ url = normalize_url(url)
108
+ except ValueError as exc:
109
+ dataset.diagnostics.append(f"Invalid resource URL: {exc}")
110
+ continue
111
+ if url in seen or not in_scope(url) or (page and pages >= cfg.max_pages):
112
+ continue
113
+ seen.add(url)
114
+ if page:
115
+ pages += 1
116
+ try:
117
+ current = url
118
+ for redirect in range(cfg.max_redirects + 1):
119
+ with client.stream(
120
+ "GET",
121
+ current,
122
+ follow_redirects=False,
123
+ timeout=cfg.timeout,
124
+ headers={"User-Agent": "SAD/0.1 static-resource-collector"},
125
+ ) as response:
126
+ if response.is_redirect:
127
+ target = normalize_url(
128
+ urljoin(current, response.headers.get("location", ""))
129
+ )
130
+ if redirect == cfg.max_redirects:
131
+ raise ValueError(f"blocked redirect to {target}")
132
+ if not in_scope(target):
133
+ if (
134
+ url in seed_urls
135
+ and canonical_redirect(url, target)
136
+ and canonical_redirect(current, target)
137
+ ):
138
+ allowed.add(origin(target))
139
+ else:
140
+ raise ValueError(f"blocked redirect to {target}")
141
+ current = target
142
+ continue
143
+ response.raise_for_status()
144
+ chunks = bytearray()
145
+ for chunk in response.iter_bytes():
146
+ total_bytes += len(chunk)
147
+ if (
148
+ len(chunks) + len(chunk) > cfg.max_bytes
149
+ or total_bytes > cfg.max_total_bytes
150
+ ):
151
+ raise ValueError("download byte budget exceeded")
152
+ chunks.extend(chunk)
153
+ media = response.headers.get("content-type", "").split(";", 1)[0]
154
+ resource = Resource(
155
+ current,
156
+ chunks.decode("utf-8", errors="replace"),
157
+ media,
158
+ document_url=current if "html" in media else document,
159
+ )
160
+ dataset.resources.append(resource)
161
+ seen.add(current)
162
+ break
163
+ if total_bytes > cfg.max_total_bytes:
164
+ break
165
+ if "html" in media:
166
+ parser = HTMLCollector(resource)
167
+ parser.feed(resource.content)
168
+ queue.extend((s, depth, False, parser.base) for s in parser.scripts)
169
+ if depth < cfg.depth:
170
+ queue.extend((s, depth + 1, True, s) for s in parser.links)
171
+ elif (
172
+ "javascript" in media
173
+ or "ecmascript" in media
174
+ or urlsplit(current).path.endswith((".js", ".mjs", ".ts"))
175
+ ):
176
+ queue.extend(
177
+ (s, depth, False, document)
178
+ for s in references(resource, self.cache)
179
+ if not s.startswith("data:")
180
+ )
181
+ except (httpx.HTTPError, ValueError) as exc:
182
+ dataset.diagnostics.append(f"{url}: {exc}")
183
+ if total_bytes > cfg.max_total_bytes:
184
+ break
185
+ if queue:
186
+ dataset.diagnostics.append(
187
+ "Collection stopped with queued resources; a configured limit was reached"
188
+ )
189
+ finally:
190
+ if self.client is None:
191
+ client.close()
192
+ return dataset
sad/html/__init__.py ADDED
@@ -0,0 +1,94 @@
1
+ """HTML resource and declarative form extraction using the standard HTML parser."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from html.parser import HTMLParser
6
+ from urllib.parse import urljoin
7
+
8
+ from sad.models import Endpoint, Evidence, Resource
9
+
10
+
11
+ class HTMLCollector(HTMLParser):
12
+ def __init__(self, resource: Resource) -> None:
13
+ super().__init__(convert_charrefs=True)
14
+ self.resource = resource
15
+ self.base = resource.url
16
+ self.links: list[str] = []
17
+ self.scripts: list[str] = []
18
+ self.inline: list[Resource] = []
19
+ self.forms: list[Endpoint] = []
20
+ self._script: list[str] | None = None
21
+ self._line = 0
22
+ self._form: Endpoint | None = None
23
+ self._base_seen = False
24
+
25
+ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
26
+ values = dict(attrs)
27
+ line, col = self.getpos()
28
+ if tag == "base" and values.get("href") and not self._base_seen:
29
+ self.base = urljoin(self.resource.url, values["href"] or "")
30
+ self._base_seen = True
31
+ if tag == "a" and values.get("href"):
32
+ self.links.append(urljoin(self.base, values["href"] or ""))
33
+ if tag == "link" and values.get("rel") in ("modulepreload", "preload"):
34
+ if values.get("as") == "script" or values.get("rel") == "modulepreload":
35
+ self.scripts.append(urljoin(self.base, values.get("href") or ""))
36
+ if tag == "script":
37
+ script_type = (values.get("type") or "").lower()
38
+ if script_type not in ("", "module", "text/javascript", "application/javascript"):
39
+ return
40
+ if values.get("src"):
41
+ self.scripts.append(urljoin(self.base, values["src"] or ""))
42
+ else:
43
+ self._script, self._line = [], line - 1
44
+ if tag == "form":
45
+ method = (values.get("method") or "GET").upper()
46
+ if method == "DIALOG":
47
+ return
48
+ self._form = Endpoint(
49
+ method=method,
50
+ url=urljoin(self.base, values.get("action") or self.resource.url),
51
+ source_file=self.resource.source_file or self.resource.url,
52
+ source_line=line,
53
+ source_column=col + 1,
54
+ framework="html-form",
55
+ content_type=values.get("enctype") or "application/x-www-form-urlencoded",
56
+ body={} if method != "GET" else None,
57
+ evidence=[
58
+ Evidence(
59
+ "html-form",
60
+ self.get_starttag_text() or "",
61
+ self.resource.url,
62
+ line,
63
+ col + 1,
64
+ )
65
+ ],
66
+ )
67
+ self.forms.append(self._form)
68
+ if tag in ("input", "textarea", "select", "button") and self._form and values.get("name"):
69
+ if "disabled" in values:
70
+ return
71
+ name = values["name"] or ""
72
+ value = values.get("value")
73
+ if self._form.method == "GET":
74
+ self._form.query_parameters.append({"name": name, "value": value})
75
+ else:
76
+ self._form.body[name] = value
77
+
78
+ def handle_data(self, data: str) -> None:
79
+ if self._script is not None:
80
+ self._script.append(data)
81
+
82
+ def handle_endtag(self, tag: str) -> None:
83
+ if tag == "script" and self._script is not None:
84
+ self.inline.append(
85
+ Resource(
86
+ self.resource.url + f"#inline-{len(self.inline)}.js",
87
+ "".join(self._script),
88
+ source_file=self.resource.source_file or self.resource.url,
89
+ line_offset=self._line,
90
+ )
91
+ )
92
+ self._script = None
93
+ if tag == "form":
94
+ self._form = None