browser-agent-server 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,591 @@
1
+ """Real headful Chrome controller with zero-leak CDP (no Runtime.enable) and X11 Turnstile solver."""
2
+ from __future__ import annotations
3
+
4
+ import base64
5
+ import hashlib
6
+ import json
7
+ import logging
8
+ import os
9
+ import re
10
+ import shutil
11
+ import socket
12
+ import struct
13
+ import subprocess
14
+ import time
15
+ import urllib.request
16
+ from pathlib import Path
17
+ from urllib.parse import urlparse
18
+
19
+ from .x11_input import X11Display
20
+
21
+ log = logging.getLogger("browser_agent.chrome")
22
+
23
+ CF_CHALLENGE_MARKERS = (
24
+ "just a moment...",
25
+ "attention required! | cloudflare",
26
+ "verify you are human",
27
+ "checking your browser before accessing",
28
+ "cf-turnstile",
29
+ "challenges.cloudflare.com/turnstile",
30
+ "_cf_chl_opt",
31
+ )
32
+
33
+ # Forbidden CDP domains that trigger Cloudflare / DataDome / PerimeterX detection
34
+ FORBIDDEN_CDP_METHODS = frozenset({"Runtime.enable", "Console.enable", "Debugger.enable"})
35
+
36
+
37
+ class StealthCDPConnection:
38
+ """Minimal RFC-6455 WebSocket client for Chrome DevTools Protocol without Runtime.enable leaks."""
39
+
40
+ def __init__(self, ws_url: str, timeout: float = 20.0):
41
+ parsed = urlparse(ws_url)
42
+ host = parsed.hostname or "127.0.0.1"
43
+ port = parsed.port or 9222
44
+ path = parsed.path or "/"
45
+ if parsed.query:
46
+ path = f"{path}?{parsed.query}"
47
+
48
+ self.sock = socket.create_connection((host, port), timeout=timeout)
49
+ self._handshake(host, port, path)
50
+ self._msg_id = 0
51
+
52
+ def _handshake(self, host: str, port: int, path: str) -> None:
53
+ raw_key = os.urandom(16)
54
+ sec_key = base64.b64encode(raw_key).decode("ascii")
55
+ req = (
56
+ f"GET {path} HTTP/1.1\r\n"
57
+ f"Host: {host}:{port}\r\n"
58
+ "Upgrade: websocket\r\n"
59
+ "Connection: Upgrade\r\n"
60
+ f"Sec-WebSocket-Key: {sec_key}\r\n"
61
+ "Sec-WebSocket-Version: 13\r\n\r\n"
62
+ )
63
+ self.sock.sendall(req.encode("ascii"))
64
+ resp = b""
65
+ while b"\r\n\r\n" not in resp:
66
+ chunk = self.sock.recv(4096)
67
+ if not chunk:
68
+ raise ConnectionError("WebSocket handshake closed prematurely")
69
+ resp += chunk
70
+ header_block, self._rx_buf = resp.split(b"\r\n\r\n", 1)
71
+ if b"101" not in header_block.splitlines()[0]:
72
+ raise ConnectionError(f"WebSocket handshake failed: {header_block[:120]!r}")
73
+ expected = base64.b64encode(
74
+ hashlib.sha1((sec_key + "258EAFA5-E914-47DA-95CA-C5AB0DC85B11").encode("ascii")).digest()
75
+ )
76
+ if expected.lower() not in header_block.lower():
77
+ raise ConnectionError("Invalid Sec-WebSocket-Accept header")
78
+
79
+ def send_frame(self, payload: bytes, opcode: int = 0x1) -> None:
80
+ mask_key = os.urandom(4)
81
+ length = len(payload)
82
+ header = bytearray([0x80 | opcode])
83
+ if length < 126:
84
+ header.append(0x80 | length)
85
+ elif length < 65536:
86
+ header.append(0x80 | 126)
87
+ header.extend(struct.pack("!H", length))
88
+ else:
89
+ header.append(0x80 | 127)
90
+ header.extend(struct.pack("!Q", length))
91
+ header.extend(mask_key)
92
+ masked = bytes(b ^ mask_key[i % 4] for i, b in enumerate(payload))
93
+ self.sock.sendall(bytes(header) + masked)
94
+
95
+ def _read_exact(self, n: int) -> bytes:
96
+ while len(self._rx_buf) < n:
97
+ chunk = self.sock.recv(65536)
98
+ if not chunk:
99
+ raise ConnectionError("WebSocket closed while reading frame")
100
+ self._rx_buf += chunk
101
+ out, self._rx_buf = self._rx_buf[:n], self._rx_buf[n:]
102
+ return out
103
+
104
+ def recv_frame(self) -> bytes:
105
+ while True:
106
+ b1, b2 = self._read_exact(2)
107
+ opcode = b1 & 0x0F
108
+ masked = bool(b2 & 0x80)
109
+ length = b2 & 0x7F
110
+ if length == 126:
111
+ length = struct.unpack("!H", self._read_exact(2))[0]
112
+ elif length == 127:
113
+ length = struct.unpack("!Q", self._read_exact(8))[0]
114
+ mask_key = self._read_exact(4) if masked else b""
115
+ data = self._read_exact(length)
116
+ if masked:
117
+ data = bytes(b ^ mask_key[i % 4] for i, b in enumerate(data))
118
+ if opcode == 0x8: # Close
119
+ raise ConnectionError("WebSocket closed by peer")
120
+ if opcode == 0x9: # Ping -> Pong
121
+ self.send_frame(data, opcode=0xA)
122
+ continue
123
+ if opcode in (0x1, 0x2):
124
+ return data
125
+
126
+ def call(self, method: str, params: dict | None = None, timeout: float = 15.0) -> dict:
127
+ if method in FORBIDDEN_CDP_METHODS:
128
+ raise ValueError(f"Forbidden CDP method {method} blocked to prevent bot detection")
129
+ self._msg_id += 1
130
+ req_id = self._msg_id
131
+ msg = json.dumps({"id": req_id, "method": method, "params": params or {}}).encode("utf-8")
132
+ self.sock.settimeout(timeout)
133
+ self.send_frame(msg)
134
+ deadline = time.time() + timeout
135
+ while time.time() < deadline:
136
+ raw = self.recv_frame()
137
+ data = json.loads(raw.decode("utf-8", errors="replace"))
138
+ if data.get("id") == req_id:
139
+ if "error" in data:
140
+ raise RuntimeError(f"CDP error in {method}: {data['error']}")
141
+ return data.get("result") or {}
142
+ raise TimeoutError(f"CDP call {method} timed out after {timeout}s")
143
+
144
+ def close(self) -> None:
145
+ try:
146
+ self.sock.close()
147
+ except OSError:
148
+ pass
149
+
150
+
151
+ def find_chrome_binary() -> str:
152
+ for candidate in (
153
+ os.environ.get("CHROME_BIN", ""),
154
+ "google-chrome-stable",
155
+ "google-chrome",
156
+ "chromium-browser",
157
+ "chromium",
158
+ "/usr/bin/google-chrome-stable",
159
+ "/usr/bin/chromium-browser",
160
+ "/usr/bin/chromium",
161
+ ):
162
+ if candidate and shutil.which(candidate):
163
+ return shutil.which(candidate) or candidate
164
+ raise FileNotFoundError(
165
+ "Chrome/Chromium binary not found. Run scripts/setup_browser_agent.sh to install google-chrome-stable."
166
+ )
167
+
168
+
169
+ def is_warp_proxy_ready(host: str = "127.0.0.1", port: int = 40000) -> bool:
170
+ """Check if local Cloudflare WARP SOCKS5 proxy is listening on 127.0.0.1:40000."""
171
+ try:
172
+ with socket.create_connection((host, port), timeout=0.4):
173
+ return True
174
+ except OSError:
175
+ return False
176
+
177
+
178
+ def is_cloudflare_challenge(html: str, title: str = "") -> bool:
179
+ hay = f"{title}\n{html[:12000]}".lower()
180
+ return any(m in hay for m in CF_CHALLENGE_MARKERS)
181
+
182
+
183
+ def _default_state_dir() -> Path:
184
+ """XDG state directory for the agent (profile, logs). Override with BROWSER_AGENT_STATE_DIR."""
185
+ if os.environ.get("BROWSER_AGENT_STATE_DIR"):
186
+ return Path(os.environ["BROWSER_AGENT_STATE_DIR"])
187
+ xdg = os.environ.get("XDG_STATE_HOME") or str(Path.home() / ".local" / "state")
188
+ return Path(xdg) / "browser-agent"
189
+
190
+
191
+ class RealChromeSession:
192
+ """Manages a real headful Chrome window on Xvfb with hardware-level X11 mouse/keyboard control."""
193
+
194
+ def __init__(
195
+ self,
196
+ display: str = ":99",
197
+ width: int = 1280,
198
+ height: int = 720,
199
+ cdp_port: int = 9222,
200
+ user_data_dir: str | None = None,
201
+ proxy_url: str | None = None,
202
+ ):
203
+ self.x11 = X11Display(display=display, width=width, height=height)
204
+ self.cdp_port = cdp_port
205
+ if user_data_dir is None:
206
+ # Chrome profile with persisted logins. Set BROWSER_AGENT_PROFILE_DIR to keep
207
+ # a pre-existing profile (e.g. /tmp/browser-agent-profile on legacy installs).
208
+ user_data_dir = os.environ.get("BROWSER_AGENT_PROFILE_DIR") or str(_default_state_dir() / "profile")
209
+ self.user_data_dir = Path(user_data_dir)
210
+ self.proxy_url = proxy_url
211
+ self._active_proxy: str | None = None
212
+ self._chrome_proc: subprocess.Popen | None = None
213
+ self._stderr_path = _default_state_dir() / "chrome.err"
214
+ self._stderr_file = None
215
+ self.last_used: float = time.time()
216
+ # Typical Chrome top UI bar height (tabs + address bar) in openbox
217
+ self.chrome_top_offset: int = 82
218
+
219
+ def is_running(self) -> bool:
220
+ return self._chrome_proc is not None and self._chrome_proc.poll() is None
221
+
222
+ def _read_chrome_stderr(self) -> str:
223
+ try:
224
+ if self._stderr_path.exists():
225
+ lines = self._stderr_path.read_text(encoding="utf-8", errors="replace").strip().splitlines()
226
+ return " | ".join(lines[-8:])[:500]
227
+ except OSError:
228
+ pass
229
+ return ""
230
+
231
+ def start(self, use_proxy: bool | None = None) -> None:
232
+ self.last_used = time.time()
233
+ self.x11.start()
234
+ want_proxy = self.proxy_url or ("socks5://127.0.0.1:40000" if (use_proxy and is_warp_proxy_ready()) else None)
235
+ if self.is_running():
236
+ if want_proxy == self._active_proxy:
237
+ return
238
+ log.info("Restarting Chrome to switch proxy (%s -> %s)", self._active_proxy or "direct", want_proxy or "direct")
239
+ self._stop_chrome_only()
240
+
241
+ self.user_data_dir.mkdir(parents=True, exist_ok=True)
242
+ for lock_name in ("SingletonLock", "SingletonCookie", "SingletonSocket"):
243
+ lock_path = self.user_data_dir / lock_name
244
+ if lock_path.exists() or lock_path.is_symlink():
245
+ try:
246
+ lock_path.unlink()
247
+ except OSError:
248
+ pass
249
+
250
+ chrome_bin = find_chrome_binary()
251
+ args = [
252
+ chrome_bin,
253
+ f"--user-data-dir={self.user_data_dir}",
254
+ f"--remote-debugging-port={self.cdp_port}",
255
+ "--remote-debugging-address=127.0.0.1",
256
+ "--remote-allow-origins=*",
257
+ f"--window-size={self.x11.width},{self.x11.height}",
258
+ "--window-position=0,0",
259
+ "--no-first-run",
260
+ "--no-default-browser-check",
261
+ "--disable-dev-shm-usage",
262
+ "--disable-gpu",
263
+ "--disable-software-rasterizer",
264
+ "--disable-background-networking",
265
+ "--disable-background-timer-throttling",
266
+ "--disable-backgrounding-occluded-windows",
267
+ "--disable-breakpad",
268
+ "--disable-component-update",
269
+ "--disable-default-apps",
270
+ "--disable-extensions",
271
+ "--disable-hang-monitor",
272
+ "--disable-popup-blocking",
273
+ "--disable-prompt-on-repost",
274
+ "--disable-sync",
275
+ "--disable-translate",
276
+ "--metrics-recording-only",
277
+ "--no-pings",
278
+ "--password-store=basic",
279
+ "--use-mock-keychain",
280
+ "--renderer-process-limit=1",
281
+ "--process-per-site",
282
+ "--js-flags=--max-old-space-size=256",
283
+ "--disable-blink-features=AutomationControlled",
284
+ "--lang=en-US",
285
+ ]
286
+ if os.geteuid() == 0:
287
+ args.extend(["--no-sandbox", "--disable-setuid-sandbox"])
288
+
289
+ self._active_proxy = want_proxy
290
+ if want_proxy:
291
+ args.append(f"--proxy-server={want_proxy}")
292
+
293
+ args.append("about:blank")
294
+ log.info("Launching Chrome: bin=%s display=%s proxy=%s", chrome_bin, self.x11.display, want_proxy or "direct")
295
+ self._stderr_file = self._stderr_path.open("w", encoding="utf-8")
296
+ self._chrome_proc = subprocess.Popen(
297
+ args,
298
+ env=self.x11.env,
299
+ stdout=subprocess.DEVNULL,
300
+ stderr=self._stderr_file,
301
+ )
302
+ t0 = time.time()
303
+ target = self._wait_for_cdp(timeout=12.0)
304
+ log.info(
305
+ "Chrome ready in %.2fs (PID=%s, proxy=%s, ws=%s)",
306
+ time.time() - t0,
307
+ self._chrome_proc.pid,
308
+ want_proxy or "direct",
309
+ target.get("webSocketDebuggerUrl", ""),
310
+ )
311
+
312
+ def _wait_for_cdp(self, timeout: float = 12.0) -> dict:
313
+ deadline = time.time() + timeout
314
+ list_url = f"http://127.0.0.1:{self.cdp_port}/json/list"
315
+ last_err = ""
316
+ while time.time() < deadline:
317
+ if self._chrome_proc is not None and self._chrome_proc.poll() is not None:
318
+ rc = self._chrome_proc.returncode
319
+ stderr_tail = self._read_chrome_stderr()
320
+ raise RuntimeError(f"Chrome exited prematurely (rc={rc}): {stderr_tail}")
321
+ try:
322
+ with urllib.request.urlopen(list_url, timeout=1.5) as resp:
323
+ targets = json.loads(resp.read().decode("utf-8"))
324
+ for t in targets:
325
+ if t.get("type") == "page" and t.get("webSocketDebuggerUrl"):
326
+ return t
327
+ req = urllib.request.Request(
328
+ f"http://127.0.0.1:{self.cdp_port}/json/new?about:blank", method="PUT"
329
+ )
330
+ with urllib.request.urlopen(req, timeout=1.5) as new_resp:
331
+ t = json.loads(new_resp.read().decode("utf-8"))
332
+ if t.get("webSocketDebuggerUrl"):
333
+ return t
334
+ except Exception as e: # noqa: BLE001
335
+ last_err = f"{type(e).__name__}: {e}"
336
+ time.sleep(0.25)
337
+ stderr_tail = self._read_chrome_stderr()
338
+ raise TimeoutError(
339
+ f"Chrome CDP port {self.cdp_port} not ready after {timeout}s (last_err={last_err}, stderr={stderr_tail})"
340
+ )
341
+
342
+ def _connect_page_cdp(self) -> StealthCDPConnection:
343
+ target = self._wait_for_cdp(timeout=6.0)
344
+ return StealthCDPConnection(target["webSocketDebuggerUrl"])
345
+
346
+ def _stop_chrome_only(self) -> None:
347
+ if self._chrome_proc and self._chrome_proc.poll() is None:
348
+ self._chrome_proc.terminate()
349
+ try:
350
+ self._chrome_proc.wait(timeout=3)
351
+ except subprocess.TimeoutExpired:
352
+ self._chrome_proc.kill()
353
+ self._chrome_proc = None
354
+ if self._stderr_file:
355
+ try:
356
+ self._stderr_file.close()
357
+ except OSError:
358
+ pass
359
+ self._stderr_file = None
360
+
361
+ def stop(self) -> None:
362
+ self._stop_chrome_only()
363
+ self.x11.stop()
364
+ for p in Path("/tmp").glob("Crashpad*"):
365
+ shutil.rmtree(p, ignore_errors=True)
366
+
367
+ def get_html_and_title(self) -> tuple[str, str, str]:
368
+ """Return (current_url, page_title, outer_html) using DOM.getOuterHTML (zero Runtime.enable)."""
369
+ cdp = self._connect_page_cdp()
370
+ try:
371
+ doc = cdp.call("DOM.getDocument", {"depth": 1}, timeout=12.0)
372
+ root = doc.get("root") or {}
373
+ node_id = root.get("nodeId", 1)
374
+ cur_url = root.get("documentURL") or ""
375
+ outer = cdp.call("DOM.getOuterHTML", {"nodeId": node_id}, timeout=20.0).get("outerHTML", "")
376
+ m = re.search(r"<title[^>]*>(.*?)</title>", outer, flags=re.IGNORECASE | re.DOTALL)
377
+ title = re.sub(r"\s+", " ", m.group(1)).strip() if m else ""
378
+ return cur_url, title, outer
379
+ finally:
380
+ cdp.close()
381
+
382
+ def get_cookies(self) -> list[dict]:
383
+ cdp = self._connect_page_cdp()
384
+ try:
385
+ res = cdp.call("Network.getAllCookies")
386
+ return res.get("cookies", [])
387
+ except Exception: # noqa: BLE001
388
+ return []
389
+ finally:
390
+ cdp.close()
391
+
392
+ def _locate_turnstile_screen_coords(self) -> tuple[int, int]:
393
+ """Locate the Cloudflare Turnstile checkbox in X11 screen coordinates."""
394
+ cdp = self._connect_page_cdp()
395
+ try:
396
+ doc = cdp.call("DOM.getDocument", {"depth": -1, "pierce": True})
397
+ root_id = (doc.get("root") or {}).get("nodeId", 1)
398
+ for selector in (
399
+ "iframe[src*='challenges.cloudflare.com']",
400
+ "iframe[id^='cf-chl-widget']",
401
+ ".cf-turnstile",
402
+ "#turnstile-wrapper",
403
+ "#challenge-stage",
404
+ ):
405
+ try:
406
+ q = cdp.call("DOM.querySelector", {"nodeId": root_id, "selector": selector})
407
+ nid = q.get("nodeId") or 0
408
+ if nid > 0:
409
+ box = cdp.call("DOM.getBoxModel", {"nodeId": nid}).get("model", {})
410
+ content = box.get("content") or []
411
+ if len(content) >= 6:
412
+ left, top = float(content[0]), float(content[1])
413
+ # Turnstile checkbox is ~28px from the left edge and vertically centered (~32px down)
414
+ sx = int(round(left + 28))
415
+ sy = int(round(top + 32 + self.chrome_top_offset))
416
+ if 10 < sx < self.x11.width - 10 and 40 < sy < self.x11.height - 10:
417
+ return sx, sy
418
+ except Exception: # noqa: BLE001
419
+ continue
420
+ except Exception: # noqa: BLE001
421
+ pass
422
+ finally:
423
+ cdp.close()
424
+
425
+ # Standard Cloudflare full-page interstitial Turnstile checkbox position on 1280x720 window
426
+ return (215, 290 + self.chrome_top_offset)
427
+
428
+ def solve_cloudflare_if_present(self, max_attempts: int = 3) -> bool:
429
+ """Detect Cloudflare challenge and solve it using human Bezier X11 mouse movement + click."""
430
+ _, title, html = self.get_html_and_title()
431
+ if not is_cloudflare_challenge(html, title):
432
+ return False
433
+
434
+ log.info("Cloudflare challenge detected (%r) — solving with real X11 mouse...", title)
435
+ for attempt in range(1, max_attempts + 1):
436
+ # Give Turnstile widget 2.5s to finish initializing its iframe
437
+ time.sleep(2.5)
438
+ _, title, html = self.get_html_and_title()
439
+ if not is_cloudflare_challenge(html, title):
440
+ log.info("Cloudflare challenge auto-resolved before click")
441
+ return True
442
+
443
+ sx, sy = self._locate_turnstile_screen_coords()
444
+ # Perform natural warm-up mouse movement across the viewport first
445
+ self.x11.move_mouse(
446
+ sx + 140,
447
+ max(120, sy - 60),
448
+ duration=0.25,
449
+ )
450
+ time.sleep(0.15)
451
+ log.info("Attempt %d/%d: clicking Turnstile checkbox at X11 (%d, %d)", attempt, max_attempts, sx, sy)
452
+ self.x11.click(sx, sy, jitter=3)
453
+ time.sleep(4.0)
454
+
455
+ _, title, html = self.get_html_and_title()
456
+ if not is_cloudflare_challenge(html, title):
457
+ log.info("Cloudflare Turnstile passed on attempt %d!", attempt)
458
+ return True
459
+
460
+ return False
461
+
462
+ def fetch_page(
463
+ self,
464
+ url: str,
465
+ wait_sec: float = 3.5,
466
+ solve_cloudflare: bool = True,
467
+ use_proxy: bool | None = None,
468
+ include_screenshot: bool = False,
469
+ ) -> dict:
470
+ """Open URL in real headful Chrome, solve Cloudflare if needed, and return rendered HTML & text."""
471
+ res = self._fetch_page_once(
472
+ url=url,
473
+ wait_sec=wait_sec,
474
+ solve_cloudflare=solve_cloudflare,
475
+ use_proxy=bool(use_proxy),
476
+ include_screenshot=include_screenshot,
477
+ )
478
+ # Auto-retry via local Cloudflare WARP proxy (127.0.0.1:40000) if direct datacenter IP was blocked
479
+ if (
480
+ use_proxy is None
481
+ and (res.get("blocked") or len(res.get("text") or "") < 160)
482
+ and is_warp_proxy_ready()
483
+ ):
484
+ log.info("Direct fetch blocked or empty for %s — retrying via WARP proxy (127.0.0.1:40000)", url)
485
+ res = self._fetch_page_once(
486
+ url=url,
487
+ wait_sec=wait_sec,
488
+ solve_cloudflare=solve_cloudflare,
489
+ use_proxy=True,
490
+ include_screenshot=include_screenshot,
491
+ )
492
+ return res
493
+
494
+ def _fetch_page_once(
495
+ self,
496
+ url: str,
497
+ wait_sec: float,
498
+ solve_cloudflare: bool,
499
+ use_proxy: bool,
500
+ include_screenshot: bool,
501
+ ) -> dict:
502
+ self.start(use_proxy=use_proxy)
503
+ self.last_used = time.time()
504
+
505
+ log.info("🌐 Navigating to %s (proxy=%s, wait=%.1fs)", url, self._active_proxy or "direct", wait_sec)
506
+ cdp = self._connect_page_cdp()
507
+ try:
508
+ try:
509
+ cdp.call("Page.navigate", {"url": url}, timeout=9.0)
510
+ except TimeoutError:
511
+ log.warning("⚠️ Page.navigate timed out on slow subresources for %s — stopping load & reading DOM", url)
512
+ try:
513
+ cdp.call("Page.stopLoading", timeout=3.0)
514
+ except Exception: # noqa: BLE001
515
+ pass
516
+ finally:
517
+ cdp.close()
518
+
519
+ time.sleep(max(1.0, float(wait_sec)))
520
+ cur_url, title, html = self.get_html_and_title()
521
+ log.info("📄 Initial DOM: url=%s title=%r html_len=%d cf=%s",
522
+ cur_url, title[:60], len(html), is_cloudflare_challenge(html, title))
523
+
524
+ cf_solved = False
525
+ if solve_cloudflare:
526
+ cf_solved = self.solve_cloudflare_if_present()
527
+ if cf_solved:
528
+ time.sleep(1.5)
529
+
530
+ self.x11.scroll(clicks=3, direction="down", x=self.x11.width // 2, y=self.x11.height // 2)
531
+ time.sleep(0.5)
532
+
533
+ final_url, title, html = self.get_html_and_title()
534
+ blocked = is_cloudflare_challenge(html, title)
535
+ text = _extract_text_from_html(html, final_url or url)
536
+ log.info(
537
+ "✅ Fetch finished: url=%s title=%r html_len=%d text_len=%d blocked=%s cf_solved=%s proxy=%s",
538
+ final_url or url, title[:60], len(html), len(text), blocked, cf_solved, self._active_proxy or "direct",
539
+ )
540
+ screenshot_b64 = ""
541
+ if include_screenshot:
542
+ try:
543
+ screenshot_b64 = base64.b64encode(self.x11.screenshot_bytes()).decode("ascii")
544
+ except Exception: # noqa: BLE001
545
+ pass
546
+
547
+ try:
548
+ cdp = self._connect_page_cdp()
549
+ try:
550
+ cdp.call("Page.navigate", {"url": "about:blank"})
551
+ finally:
552
+ cdp.close()
553
+ except Exception: # noqa: BLE001
554
+ pass
555
+
556
+ return {
557
+ "ok": not blocked and bool(html),
558
+ "blocked": blocked,
559
+ "cloudflare_solved": cf_solved,
560
+ "via_warp": bool(self._active_proxy),
561
+ "url": final_url or url,
562
+ "title": title,
563
+ "html": html,
564
+ "text": text,
565
+ "cookies": self.get_cookies(),
566
+ "screenshot_b64": screenshot_b64,
567
+ }
568
+
569
+
570
+ def _extract_text_from_html(html: str, url: str = "") -> str:
571
+ if not html:
572
+ return ""
573
+ try:
574
+ import trafilatura
575
+
576
+ extracted = trafilatura.extract(
577
+ html,
578
+ url=url or None,
579
+ include_comments=False,
580
+ include_tables=False,
581
+ favor_recall=True,
582
+ )
583
+ if extracted and len(extracted.strip()) >= 120:
584
+ return extracted.strip()
585
+ except Exception: # noqa: BLE001
586
+ pass
587
+
588
+ # Fallback plain-text strip
589
+ cleaned = re.sub(r"<(script|style|noscript|svg|header|footer|nav)[^>]*>.*?</\1>", " ", html, flags=re.I | re.S)
590
+ cleaned = re.sub(r"<[^>]+>", " ", cleaned)
591
+ return re.sub(r"\s+", " ", cleaned).strip()