bmad-plus 0.19.0 → 0.21.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,7 +3,8 @@
3
3
  SEO Fetch — Secure HTTP page fetcher for SEO analysis.
4
4
 
5
5
  Features:
6
- - SSRF protection (blocks private/loopback/reserved IPs)
6
+ - SSRF protection (blocks private/loopback/reserved IPs, re-checked on every
7
+ redirect hop, connection pinned to the validated address)
7
8
  - Multi-UA support (standard, Googlebot, GPTBot, ClaudeBot)
8
9
  - Redirect chain tracking
9
10
  - Cookie handling
@@ -16,12 +17,16 @@ License: MIT
16
17
  import argparse
17
18
  import ipaddress
18
19
  import json
20
+ import os
19
21
  import socket
20
22
  import sys
23
+ import time
21
24
  from urllib.parse import urljoin, urlparse
22
25
 
23
26
  try:
24
27
  import requests
28
+ from requests.adapters import DEFAULT_POOLBLOCK, HTTPAdapter
29
+ from urllib3 import PoolManager
25
30
  except ImportError:
26
31
  print("Error: requests library required. Install: pip install requests", file=sys.stderr)
27
32
  sys.exit(1)
@@ -61,10 +66,37 @@ DEFAULT_HEADERS = {
61
66
 
62
67
  # ── Security: SSRF Prevention ──────────────────────────────────────
63
68
 
69
+ ALLOWED_SCHEMES = frozenset({"http", "https"})
70
+
71
+ # RFC 6052 well-known NAT64 prefix: the last 32 bits are the IPv4 host the
72
+ # translator will reach, so the address is only as safe as that IPv4 host.
73
+ _NAT64_PREFIX = ipaddress.ip_network("64:ff9b::/96")
74
+
75
+
76
+ class UnsafeURLError(requests.exceptions.InvalidURL):
77
+ """A URL, or the address it resolves to, must never be fetched."""
78
+
79
+
64
80
  def _ip_is_blocked(ip: "ipaddress._BaseAddress") -> bool:
65
- """Return True if an IP falls in any range that must never be reached."""
81
+ """Return True if an IP falls in any range that must never be reached.
82
+
83
+ Anything that is not globally routable is refused: private, loopback,
84
+ link-local (cloud metadata at 169.254.169.254), carrier-grade NAT
85
+ (100.64.0.0/10, e.g. Alibaba metadata at 100.100.100.200), documentation
86
+ and benchmarking ranges, multicast and reserved space. IPv6 forms that
87
+ embed an IPv4 destination are judged by that destination.
88
+ """
89
+ if ip.version == 6:
90
+ embedded = ip.ipv4_mapped
91
+ if embedded is None and ip in _NAT64_PREFIX:
92
+ embedded = ipaddress.IPv4Address(int(ip) & 0xFFFFFFFF)
93
+ if embedded is not None:
94
+ return _ip_is_blocked(embedded)
95
+ if ip.is_site_local:
96
+ return True
66
97
  return bool(
67
- ip.is_private
98
+ not ip.is_global
99
+ or ip.is_private
68
100
  or ip.is_loopback
69
101
  or ip.is_reserved
70
102
  or ip.is_link_local
@@ -73,47 +105,152 @@ def _ip_is_blocked(ip: "ipaddress._BaseAddress") -> bool:
73
105
  )
74
106
 
75
107
 
76
- def is_safe_url(url: str) -> bool:
77
- """Block requests to private, loopback, and reserved IP addresses.
108
+ def check_url(url: str) -> str:
109
+ """Apply the URL-level rules and return the hostname, without resolving it.
78
110
 
79
- Fails CLOSED: a missing host, a non-HTTP(S) scheme, a DNS resolution
80
- error, or an unparseable/blocked address all cause the URL to be
81
- rejected. Every resolved address (IPv4 and IPv6) must be public.
111
+ Fails CLOSED with UnsafeURLError: a scheme outside http/https, embedded
112
+ credentials, a missing host or an invalid port rejects the URL.
82
113
  """
83
114
  parsed = urlparse(url)
115
+ if parsed.scheme not in ALLOWED_SCHEMES:
116
+ raise UnsafeURLError(f"scheme not allowed: {parsed.scheme or '(none)'}")
117
+ # "user:pass@" would be sent as an Authorization header and makes
118
+ # "https://trusted.example@evil.example/" style URLs misleading.
119
+ if parsed.username is not None or parsed.password is not None:
120
+ raise UnsafeURLError("credentials in URL are not allowed")
84
121
  hostname = parsed.hostname
85
-
86
122
  if not hostname:
87
- return False
123
+ raise UnsafeURLError("URL has no host")
124
+ try:
125
+ parsed.port
126
+ except ValueError:
127
+ raise UnsafeURLError("URL has an invalid port") from None
128
+ return hostname
88
129
 
89
- if parsed.scheme not in ("http", "https"):
90
- return False
91
130
 
131
+ def resolve_public_address(hostname: str) -> str:
132
+ """Resolve a hostname and return the public address to connect to.
133
+
134
+ Fails CLOSED with UnsafeURLError on a DNS error or when any resolved
135
+ address (IPv4 or IPv6) is not public: every address the name resolves to
136
+ must be public, so the returned one is safe whichever the resolver would
137
+ have preferred.
138
+ """
92
139
  try:
93
- # Resolve ALL IP addresses (IPv4 and IPv6) via getaddrinfo
94
140
  addrinfo = socket.getaddrinfo(hostname, None)
95
- except socket.gaierror:
96
- return False # Fail closed: unresolvable host is treated as unsafe
97
-
98
- if not addrinfo:
99
- return False # Fail closed: no addresses resolved
141
+ except (socket.gaierror, UnicodeError):
142
+ raise UnsafeURLError(f"cannot resolve host {hostname}") from None
100
143
 
144
+ addresses = []
101
145
  for entry in addrinfo:
102
- ip_str = entry[4][0] # sockaddr[0] contains the IP string
103
146
  try:
104
- ip = ipaddress.ip_address(ip_str)
147
+ ip = ipaddress.ip_address(entry[4][0])
105
148
  except ValueError:
106
- return False # Fail closed: unparseable address
107
- # IPv4-mapped IPv6 (::ffff:a.b.c.d) must be checked as its IPv4 form
108
- mapped = getattr(ip, "ipv4_mapped", None)
109
- if mapped is not None and _ip_is_blocked(mapped):
110
- return False
149
+ raise UnsafeURLError(f"{hostname} resolved to an unparseable address") from None
111
150
  if _ip_is_blocked(ip):
112
- return False
151
+ raise UnsafeURLError(f"{hostname} resolves to a private/internal address ({ip})")
152
+ addresses.append(ip)
153
+ if not addresses:
154
+ raise UnsafeURLError(f"cannot resolve host {hostname}")
155
+
156
+ # A pinned connection cannot fall back to another address, and IPv6 routes
157
+ # are often missing in containers and CI: prefer IPv4 on dual-stack hosts.
158
+ preferred = next((ip for ip in addresses if ip.version == 4), addresses[0])
159
+ return str(preferred)
160
+
113
161
 
162
+ def resolve_target(url: str) -> "tuple[str, str]":
163
+ """Validate a URL and return ``(hostname, address)`` to connect to.
164
+
165
+ Combines check_url() and resolve_public_address(); see their rules.
166
+ """
167
+ hostname = check_url(url)
168
+ return hostname, resolve_public_address(hostname)
169
+
170
+
171
+ def is_safe_url(url: str) -> bool:
172
+ """Return True when resolve_target() accepts the URL (see its rules)."""
173
+ try:
174
+ resolve_target(url)
175
+ except UnsafeURLError:
176
+ return False
114
177
  return True
115
178
 
116
179
 
180
+ class _PinnedPoolManager(PoolManager):
181
+ """urllib3 pool manager that only opens pools on validated IP literals.
182
+
183
+ Every requests version reaches the network through
184
+ PoolManager.connection_from_host (directly, or via connection_from_url),
185
+ whatever adapter hook it calls first. Validating here, instead of in a
186
+ requests hook that has been renamed across releases, keeps the guard
187
+ closed on old and future requests versions alike.
188
+ """
189
+
190
+ def connection_from_host(self, host, port=None, scheme="http", pool_kwargs=None):
191
+ hostname = (host or "").strip("[]")
192
+ if not hostname:
193
+ raise UnsafeURLError("URL has no host")
194
+ address = resolve_public_address(hostname)
195
+ if scheme == "https":
196
+ pool_kwargs = dict(pool_kwargs or {})
197
+ pool_kwargs["server_hostname"] = hostname
198
+ pool_kwargs["assert_hostname"] = hostname
199
+ return super().connection_from_host(
200
+ address, port=port, scheme=scheme, pool_kwargs=pool_kwargs
201
+ )
202
+
203
+
204
+ class PinnedAddressAdapter(HTTPAdapter):
205
+ """Transport adapter that connects to the exact address it validated.
206
+
207
+ A plain requests call resolves the host again when urllib3 opens the
208
+ socket, so a hostile DNS server can answer a public address to the SSRF
209
+ check and a private one to the connection (DNS rebinding). This adapter's
210
+ pool manager resolves and validates each host itself, then opens the
211
+ connection pool on that IP literal. The Host header, TLS SNI and the
212
+ certificate hostname check still use the original name, so HTTPS
213
+ verification is unchanged. Proxies are refused: a proxy would resolve the
214
+ name itself and bypass the pinned address.
215
+ """
216
+
217
+ def init_poolmanager(self, connections, maxsize, block=DEFAULT_POOLBLOCK, **pool_kwargs):
218
+ super().init_poolmanager(connections, maxsize, block, **pool_kwargs)
219
+ # Rebuild with the exact settings requests chose for this version.
220
+ self.poolmanager = _PinnedPoolManager(
221
+ num_pools=connections, **self.poolmanager.connection_pool_kw
222
+ )
223
+
224
+ def proxy_manager_for(self, proxy, **proxy_kwargs):
225
+ raise UnsafeURLError("proxies are not supported by the pinned transport")
226
+
227
+ def send(self, request, **kwargs):
228
+ check_url(request.url)
229
+ # urllib3 would derive Host from the pool's IP literal; keep the name.
230
+ if "Host" not in request.headers:
231
+ request.headers["Host"] = urlparse(request.url).netloc.rpartition("@")[2]
232
+ return super().send(request, **kwargs)
233
+
234
+
235
+ def create_session() -> requests.Session:
236
+ """Build a requests session whose every connection is SSRF-checked and pinned.
237
+
238
+ Environment proxies and ~/.netrc credentials are ignored: a proxy defeats
239
+ address pinning and netrc would attach credentials to attacker-chosen hosts.
240
+ A custom CA bundle named by REQUESTS_CA_BUNDLE or CURL_CA_BUNDLE (corporate
241
+ TLS inspection) is still honoured, since it does not weaken pinning.
242
+ """
243
+ session = requests.Session()
244
+ session.trust_env = False
245
+ ca_bundle = os.environ.get("REQUESTS_CA_BUNDLE") or os.environ.get("CURL_CA_BUNDLE")
246
+ if ca_bundle:
247
+ session.verify = ca_bundle
248
+ adapter = PinnedAddressAdapter()
249
+ session.mount("http://", adapter)
250
+ session.mount("https://", adapter)
251
+ return session
252
+
253
+
117
254
  # ── Core Fetcher ───────────────────────────────────────────────────
118
255
 
119
256
  def fetch_page(
@@ -147,37 +284,24 @@ def fetch_page(
147
284
  url = f"https://{url}"
148
285
  parsed = urlparse(url)
149
286
 
150
- if parsed.scheme not in ("http", "https"):
287
+ if parsed.scheme not in ALLOWED_SCHEMES:
151
288
  result["error"] = f"Invalid URL scheme: {parsed.scheme}"
152
289
  return result
153
290
 
154
- # SSRF check
155
- if not is_safe_url(url):
156
- resolved = "unknown"
157
- try:
158
- # Use getaddrinfo for consistent multi-address resolution
159
- addrinfo = socket.getaddrinfo(parsed.hostname, None)
160
- resolved = ", ".join(set(entry[4][0] for entry in addrinfo))
161
- except Exception:
162
- pass
163
- result["error"] = f"Blocked: URL resolves to private/internal IP ({resolved})"
164
- return result
165
-
291
+ current_url = url
166
292
  try:
167
- session = requests.Session()
293
+ session = create_session()
168
294
 
169
295
  headers = dict(DEFAULT_HEADERS)
170
296
  ua_string = USER_AGENTS.get(user_agent, user_agent)
171
297
  headers["User-Agent"] = ua_string
172
298
 
173
- import time
174
299
  start = time.monotonic()
175
300
 
176
- # Follow redirects manually so is_safe_url() runs on EVERY hop.
177
- # Letting requests follow redirects internally would allow a
178
- # public URL to redirect (302) to an internal/metadata endpoint
179
- # (redirect-based SSRF), bypassing the initial check.
180
- current_url = url
301
+ # Follow redirects manually so every hop goes back through the pinned
302
+ # adapter, which validates the target before connecting. Letting
303
+ # requests follow redirects internally would hide the hop count and
304
+ # the chain; the adapter still guards each connection either way.
181
305
  redirect_chain = []
182
306
  hops = 0
183
307
  response = None
@@ -198,19 +322,9 @@ def fetch_page(
198
322
  break
199
323
 
200
324
  next_url = urljoin(current_url, location)
201
- next_parsed = urlparse(next_url)
202
-
203
- if next_parsed.scheme not in ("http", "https"):
204
- result["error"] = (
205
- f"Blocked redirect to non-HTTP(S) scheme: {next_parsed.scheme}"
206
- )
207
- return result
208
-
209
- # Re-validate the redirect target (blocks redirect-based SSRF)
210
- if not is_safe_url(next_url):
211
- result["error"] = (
212
- f"Blocked: redirect to private/internal URL ({next_url})"
213
- )
325
+ next_scheme = urlparse(next_url).scheme
326
+ if next_scheme not in ALLOWED_SCHEMES:
327
+ result["error"] = f"Blocked redirect to non-HTTP(S) scheme: {next_scheme}"
214
328
  return result
215
329
 
216
330
  hops += 1
@@ -221,6 +335,7 @@ def fetch_page(
221
335
  redirect_chain.append(
222
336
  {"url": current_url, "status": response.status_code}
223
337
  )
338
+ response.close()
224
339
  current_url = next_url
225
340
 
226
341
  elapsed_ms = round((time.monotonic() - start) * 1000)
@@ -233,6 +348,11 @@ def fetch_page(
233
348
  result["response_time_ms"] = elapsed_ms
234
349
  result["redirect_chain"] = redirect_chain
235
350
 
351
+ except UnsafeURLError as e:
352
+ if current_url == url:
353
+ result["error"] = f"Blocked: {e}"
354
+ else:
355
+ result["error"] = f"Blocked: redirect to private/internal URL ({current_url}): {e}"
236
356
  except requests.exceptions.Timeout:
237
357
  result["error"] = f"Request timed out after {timeout}s"
238
358
  except requests.exceptions.TooManyRedirects:
@@ -20,6 +20,7 @@
20
20
  "updatedAt": { "type": "string", "format": "date-time" },
21
21
  "finishedAt": { "type": ["string", "null"], "format": "date-time" },
22
22
  "overallNote": { "type": "string" },
23
+ "redactions": { "type": "integer", "minimum": 1, "description": "Credential-like values replaced by [REDACTED] in the notes before the run was written. Absent when nothing was replaced." },
23
24
  "carriedFrom": {
24
25
  "type": "object",
25
26
  "description": "Present when the page continued a run that answered another revision of the recipe. The rule is fixed: an answer travelled only when its line was identical (same step, same letter, same text); every other line was asked again. The earlier run is left untouched.",
@@ -128,8 +128,20 @@ for (const modulePath of [
128
128
  configured.action(command.action);
129
129
  }
130
130
 
131
+ // Stdin is resumed above so prompts work on Windows. Once the command has settled
132
+ // nothing reads it, and a resumed terminal would keep the process alive: the shell
133
+ // never got its prompt back until Ctrl+C (0.20.0 acceptance run).
134
+ function releaseStdin() {
135
+ if (!process.stdin.isTTY) return;
136
+ process.stdin.pause();
137
+ process.stdin.unref();
138
+ }
139
+
131
140
  // Await asynchronous registry checks and updates before completing the CLI.
132
- program.parseAsync(process.argv).catch((error) => {
133
- console.error(`BMAD+: ${error.message}`);
134
- process.exitCode = 1;
135
- });
141
+ program
142
+ .parseAsync(process.argv)
143
+ .catch((error) => {
144
+ console.error(`BMAD+: ${error.message}`);
145
+ process.exitCode = 1;
146
+ })
147
+ .finally(releaseStdin);