veddata 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
veddata/chromium.py ADDED
@@ -0,0 +1,321 @@
1
+ """Chromium 启动 / 接管 —— 自起普通 Chrome + CDP 端口(DrissionPage 式)。
2
+
3
+ 为什么不用 Playwright 启动浏览器
4
+ --------------------------------
5
+ Playwright 用 ``--remote-debugging-pipe``,并附加它自己一长串启动开关;实测同一份
6
+ ``chrome.exe``:pipe → ``navigator.webdriver=true``(boss 直聘据此自毁页面);
7
+ 自起 Chrome + ``--remote-debugging-port`` → ``webdriver=false``、无注入痕迹、正常 UA。
8
+
9
+ 三条实用设计
10
+ ------------
11
+ 1. **profile 里记端口与 pid**(``.veddata-browser.json``):重连时先看已有实例是否还活着,
12
+ 活着就复用 —— MCP 服务重启、用户登录/过验证之后会话都不丢。
13
+ 2. **生命周期归自己管**:我们起的、以及档案里记着 pid 的那个,``close_running`` 能真关掉;
14
+ 只有 ``BROWSER_ADDRESS`` 指定的外部浏览器才只断开。
15
+ 3. **启动失败不重试**:profile 被另一个 Chrome 占用时,多试一次就是多开一个窗口 ——
16
+ 只起一次,失败就把事实(pid / 端口 / 退出码 / profile 路径)报出来。
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import json
22
+ import os
23
+ import socket
24
+ import subprocess
25
+ import sys
26
+ import time
27
+ import urllib.request
28
+ from pathlib import Path
29
+
30
+ PORT_FILE = ".veddata-browser.json"
31
+ _READY_TIMEOUT = 25.0
32
+ _CREATE_NO_WINDOW = 0x08000000 if sys.platform == "win32" else 0
33
+
34
+
35
+ class ChromiumError(RuntimeError):
36
+ """起不来 / 找不到浏览器 / profile 被占用。"""
37
+
38
+
39
+ def free_port() -> int:
40
+ """要一个当前空闲的本地端口(绑定后立刻释放,Chrome 起来前有极小竞态)。"""
41
+ with socket.socket() as sock:
42
+ sock.bind(("127.0.0.1", 0))
43
+ return int(sock.getsockname()[1])
44
+
45
+
46
+ def build_args(path: str, port: int, profile: Path, headless: bool) -> list[str]:
47
+ """自起 Chrome 的参数:**只要最少的**,绝不带自动化开关。"""
48
+ args = [
49
+ path,
50
+ f"--remote-debugging-port={port}",
51
+ f"--user-data-dir={profile}",
52
+ "--no-first-run",
53
+ "--no-default-browser-check",
54
+ ]
55
+ if headless:
56
+ args.append("--headless=new")
57
+ return args
58
+
59
+
60
+ # ---------------- 浏览器可执行文件查找 ----------------
61
+
62
+ _WIN_CHROME_RELATIVE = [
63
+ ("LOCALAPPDATA", r"Google\Chrome\Application\chrome.exe"),
64
+ ("ProgramFiles", r"Google\Chrome\Application\chrome.exe"),
65
+ ("ProgramFiles(x86)", r"Google\Chrome\Application\chrome.exe"),
66
+ ]
67
+ _WIN_EDGE_RELATIVE = [
68
+ ("ProgramFiles(x86)", r"Microsoft\Edge\Application\msedge.exe"),
69
+ ("ProgramFiles", r"Microsoft\Edge\Application\msedge.exe"),
70
+ ("LOCALAPPDATA", r"Microsoft\Edge\Application\msedge.exe"),
71
+ ]
72
+ _WIN_REGISTRY = [
73
+ r"SOFTWARE\Microsoft\Windows\CurrentVersion\App Paths\chrome.exe",
74
+ r"SOFTWARE\WOW6432Node\Microsoft\Windows\CurrentVersion\App Paths\chrome.exe",
75
+ ]
76
+ _WIN_REGISTRY_EDGE = [
77
+ r"SOFTWARE\Microsoft\Windows\CurrentVersion\App Paths\msedge.exe",
78
+ r"SOFTWARE\WOW6432Node\Microsoft\Windows\CurrentVersion\App Paths\msedge.exe",
79
+ ]
80
+
81
+
82
+ def _from_registry(keys: list[str]) -> str | None:
83
+ if sys.platform != "win32":
84
+ return None
85
+ try:
86
+ from winreg import HKEY_CURRENT_USER, HKEY_LOCAL_MACHINE, OpenKey, QueryValueEx
87
+ except ImportError: # pragma: no cover
88
+ return None
89
+ for root in (HKEY_LOCAL_MACHINE, HKEY_CURRENT_USER):
90
+ for key in keys:
91
+ try:
92
+ with OpenKey(root, key) as handle:
93
+ value, _ = QueryValueEx(handle, None)
94
+ if value and Path(value).exists():
95
+ return str(value)
96
+ except OSError:
97
+ continue
98
+ return None
99
+
100
+
101
+ def _from_relative(entries: list[tuple[str, str]]) -> str | None:
102
+ for env_name, tail in entries:
103
+ base = os.environ.get(env_name)
104
+ if not base:
105
+ continue
106
+ candidate = Path(base) / tail
107
+ if candidate.exists():
108
+ return str(candidate)
109
+ return None
110
+
111
+
112
+ def _from_path(names: list[str]) -> str | None:
113
+ for directory in os.environ.get("PATH", "").split(os.pathsep):
114
+ if not directory:
115
+ continue
116
+ for name in names:
117
+ candidate = Path(directory) / name
118
+ if candidate.exists():
119
+ return str(candidate)
120
+ return None
121
+
122
+
123
+ def find_chrome() -> str | None:
124
+ if sys.platform == "darwin":
125
+ for candidate in ("/Applications/Google Chrome.app/Contents/MacOS/Google Chrome",
126
+ "/Applications/Chromium.app/Contents/MacOS/Chromium"):
127
+ if Path(candidate).exists():
128
+ return candidate
129
+ return (_from_registry(_WIN_REGISTRY) or _from_relative(_WIN_CHROME_RELATIVE)
130
+ or _from_path(["chrome.exe" if sys.platform == "win32" else "google-chrome",
131
+ "chrome", "chromium"]))
132
+
133
+
134
+ def find_edge() -> str | None:
135
+ if sys.platform == "darwin":
136
+ candidate = "/Applications/Microsoft Edge.app/Contents/MacOS/Microsoft Edge"
137
+ return candidate if Path(candidate).exists() else None
138
+ return (_from_registry(_WIN_REGISTRY_EDGE) or _from_relative(_WIN_EDGE_RELATIVE)
139
+ or _from_path(["msedge.exe" if sys.platform == "win32" else "microsoft-edge"]))
140
+
141
+
142
+ def browser_path() -> str | None:
143
+ """按 ``BROWSER_PATH`` 决定用哪个浏览器(空/未知 → Chrome 优先,其次 Edge)。"""
144
+ raw = os.environ.get("BROWSER_PATH", "").strip()
145
+ if raw:
146
+ lowered = raw.lower()
147
+ if lowered in ("edge", "msedge"):
148
+ return find_edge()
149
+ if lowered not in ("chrome", "chrome.exe", "google-chrome"):
150
+ candidate = Path(raw).expanduser()
151
+ if candidate.exists():
152
+ return str(candidate)
153
+ return find_chrome() or find_edge()
154
+
155
+
156
+ # ---------------- 存活判定 / 端口档案 ----------------
157
+
158
+ def alive(port: int, timeout: float = 1.5) -> bool:
159
+ """CDP 端点是否响应(``/json/version``)。"""
160
+ try:
161
+ with urllib.request.urlopen(f"http://127.0.0.1:{port}/json/version", timeout=timeout) as response:
162
+ payload = json.loads(response.read().decode("utf-8", "replace"))
163
+ return bool(payload.get("webSocketDebuggerUrl"))
164
+ except Exception: # noqa: BLE001
165
+ return False
166
+
167
+
168
+ def _read_port_payload(profile: Path) -> dict:
169
+ try:
170
+ return json.loads((Path(profile) / PORT_FILE).read_text(encoding="utf-8"))
171
+ except Exception: # noqa: BLE001
172
+ return {}
173
+
174
+
175
+ def read_port_file(profile: Path) -> int | None:
176
+ try:
177
+ return int(_read_port_payload(profile)["port"])
178
+ except Exception: # noqa: BLE001
179
+ return None
180
+
181
+
182
+ def read_pid(profile: Path) -> int:
183
+ """档案里记的浏览器主进程 pid(没有就是 0)。"""
184
+ try:
185
+ return int(_read_port_payload(profile).get("pid") or 0)
186
+ except Exception: # noqa: BLE001
187
+ return 0
188
+
189
+
190
+ def write_port_file(profile: Path, port: int, pid: int | None = None) -> None:
191
+ target = Path(profile) / PORT_FILE
192
+ target.parent.mkdir(parents=True, exist_ok=True)
193
+ target.write_text(json.dumps({"port": port, "pid": pid}, ensure_ascii=False), encoding="utf-8")
194
+
195
+
196
+ def clear_port_file(profile: Path) -> None:
197
+ try:
198
+ (Path(profile) / PORT_FILE).unlink()
199
+ except OSError:
200
+ pass
201
+
202
+
203
+ def kill_tree(pid: int) -> bool:
204
+ """按 pid 结束整个进程树(Windows: taskkill /T /F)。返回是否成功。"""
205
+ if not pid:
206
+ return False
207
+ try:
208
+ if sys.platform == "win32":
209
+ done = subprocess.run(["taskkill", "/PID", str(pid), "/T", "/F"],
210
+ capture_output=True, text=True)
211
+ else:
212
+ done = subprocess.run(["kill", "-TERM", f"-{pid}"], capture_output=True, text=True)
213
+ return done.returncode == 0
214
+ except Exception: # noqa: BLE001
215
+ return False
216
+
217
+
218
+ def close_running(profile: Path) -> dict:
219
+ """关掉档案里记的那个浏览器(**哪怕它不是我起起的**)。
220
+
221
+ Returns:
222
+ ``{"port":…, "pid":…, "killed":bool}`` —— 只有事实,调用方照原样报告。
223
+ """
224
+ profile = Path(profile)
225
+ port = read_port_file(profile)
226
+ pid = read_pid(profile)
227
+ killed = kill_tree(pid)
228
+ clear_port_file(profile)
229
+ return {"port": port, "pid": pid, "killed": killed}
230
+
231
+
232
+ # ---------------- 启动器 ----------------
233
+
234
+ class Launcher:
235
+ """确保"有一个跑着的普通 Chrome 可以用",返回它的 CDP 地址。"""
236
+
237
+ def __init__(self, path: str | None = None) -> None:
238
+ self.path = path or browser_path()
239
+ self.process: subprocess.Popen | None = None
240
+ self.port: int | None = None
241
+ self.started = False
242
+
243
+ def ensure(self, profile: Path, headless: bool) -> str:
244
+ """返回 ``http://127.0.0.1:<port>``;能复用已有实例就复用。
245
+
246
+ Raises:
247
+ ChromiumError: 找不到浏览器,或起不来(**只尝试一次**,不重试 —— 重试只会多开窗口)。
248
+ """
249
+ profile = Path(profile)
250
+ profile.mkdir(parents=True, exist_ok=True)
251
+
252
+ existing = read_port_file(profile)
253
+ if existing and alive(existing):
254
+ self.port = existing
255
+ self.started = False
256
+ return f"http://127.0.0.1:{existing}"
257
+
258
+ if not self.path:
259
+ raise ChromiumError(
260
+ "未找到 Chrome/Edge:请安装,或用 BROWSER_PATH 指定浏览器可执行文件"
261
+ )
262
+
263
+ port = free_port()
264
+ args = build_args(self.path, port, profile, headless)
265
+ try:
266
+ process = subprocess.Popen(args, stdout=subprocess.DEVNULL,
267
+ stderr=subprocess.DEVNULL,
268
+ creationflags=_CREATE_NO_WINDOW)
269
+ except OSError as exc:
270
+ raise ChromiumError(f"无法启动 {self.path}:{exc}") from exc
271
+
272
+ if self._wait_ready(port, process):
273
+ self.process = process
274
+ self.port = port
275
+ self.started = True
276
+ write_port_file(profile, port, process.pid)
277
+ return f"http://127.0.0.1:{port}"
278
+
279
+ code = process.poll()
280
+ self._terminate(process)
281
+ raise ChromiumError(
282
+ f"Chrome 未能就绪(pid={process.pid}, port={port}, 退出码={code})—— "
283
+ f"profile 可能已被另一个 Chrome 占用:{profile}"
284
+ )
285
+
286
+ def stop(self, profile: Path | None = None, clear: bool = True) -> None:
287
+ """关掉**本进程起的**实例;接管来的不动。
288
+
289
+ Args:
290
+ profile: 档案所在的 profile 目录。
291
+ clear: 是否删掉端口档案。**detach(只断开)时必须传 False** ——
292
+ 删了就等于把"我们在用的那个浏览器"弄丢,下次会去起一个新的(profile 被占用→开一堆窗口)。
293
+ """
294
+ if self.started and self.process is not None:
295
+ self._terminate(self.process)
296
+ self.process = None
297
+ self.started = False
298
+ if profile is not None and clear:
299
+ clear_port_file(Path(profile))
300
+
301
+ @staticmethod
302
+ def _wait_ready(port: int, process: subprocess.Popen, timeout: float = _READY_TIMEOUT) -> bool:
303
+ deadline = time.time() + timeout
304
+ while time.time() < deadline:
305
+ if alive(port):
306
+ return True
307
+ if process.poll() is not None: # 进程自己退了,别傻等
308
+ return False
309
+ time.sleep(0.25)
310
+ return False
311
+
312
+ @staticmethod
313
+ def _terminate(process: subprocess.Popen) -> None:
314
+ try:
315
+ process.terminate()
316
+ process.wait(timeout=5)
317
+ except Exception: # noqa: BLE001
318
+ try:
319
+ process.kill()
320
+ except Exception: # noqa: BLE001
321
+ pass
veddata/dom.py ADDED
@@ -0,0 +1,298 @@
1
+ """DOM tree module — in-memory snapshot of the page structure.
2
+
3
+ The browser's live DOM is snapshotted once (after the page stabilizes)
4
+ into a DOMTree held in Python memory. Searches and path lookups run
5
+ against the snapshot without touching the browser again.
6
+ """
7
+
8
+ import time
9
+ from dataclasses import dataclass, field
10
+
11
+
12
+ # 递归遍历 DOM 生成可序列化快照。节点数上限 2000,超出截断。
13
+ SNAPSHOT_JS = """
14
+ () => {
15
+ const MAX_NODES = 2000;
16
+ let count = 0;
17
+ let truncated = false;
18
+ const ATTR_KEYS = ['href', 'src', 'placeholder', 'aria-label', 'title', 'name', 'type', 'value', 'role'];
19
+ function walk(el, depth) {
20
+ if (depth > 24) return null;
21
+ if (count >= MAX_NODES) { truncated = true; return null; }
22
+ count++;
23
+ const tag = el.tagName ? el.tagName.toLowerCase() : '';
24
+ const node = {
25
+ tag: tag,
26
+ id: el.id || '',
27
+ classes: Array.from(el.classList || []),
28
+ text: '',
29
+ attrs: {},
30
+ visible: el.offsetParent !== null || el === document.documentElement,
31
+ rect: null,
32
+ children: []
33
+ };
34
+ for (const k of ATTR_KEYS) {
35
+ const v = el.getAttribute(k);
36
+ if (v) node.attrs[k] = v.slice(0, 120);
37
+ }
38
+ if (node.visible) {
39
+ const r = el.getBoundingClientRect();
40
+ node.rect = { w: Math.round(r.width), h: Math.round(r.height), top: Math.round(r.top), left: Math.round(r.left) };
41
+ }
42
+ const kids = Array.from(el.children || []);
43
+ if (kids.length === 0) {
44
+ const t = (el.textContent || '').trim();
45
+ if (t) node.text = t.slice(0, 80);
46
+ } else {
47
+ for (const c of kids) {
48
+ const cn = walk(c, depth + 1);
49
+ if (cn) node.children.push(cn);
50
+ }
51
+ }
52
+ return node;
53
+ }
54
+ const root = walk(document.documentElement, 0);
55
+ return { root: root, truncated: truncated };
56
+ }
57
+ """
58
+
59
+
60
+ @dataclass
61
+ class DOMNode:
62
+ tag: str = ""
63
+ id: str = ""
64
+ classes: list[str] = field(default_factory=list)
65
+ text: str = ""
66
+ attrs: dict = field(default_factory=dict)
67
+ rect: dict | None = None
68
+ children: list["DOMNode"] = field(default_factory=list)
69
+ collapsed: bool = False
70
+ collapsed_count: int = 0
71
+
72
+ @property
73
+ def selector(self) -> str:
74
+ """Compact selector like div#app.feed or input#kw."""
75
+ s = self.tag or "?"
76
+ if self.id:
77
+ s += f"#{self.id}"
78
+ if self.classes:
79
+ s += "." + ".".join(self.classes[:3])
80
+ return s
81
+
82
+ def is_interactive(self) -> bool:
83
+ return self.tag in ("input", "button", "select") or (self.tag == "a" and "href" in self.attrs)
84
+
85
+
86
+ def _from_dict(d: dict) -> DOMNode:
87
+ node = DOMNode(
88
+ tag=d.get("tag", ""),
89
+ id=d.get("id", ""),
90
+ classes=d.get("classes", []),
91
+ text=d.get("text", ""),
92
+ attrs=d.get("attrs", {}),
93
+ rect=d.get("rect"),
94
+ )
95
+ for c in d.get("children", []):
96
+ node.children.append(_from_dict(c))
97
+ return node
98
+
99
+
100
+ class DOMTree:
101
+ """In-memory DOM snapshot with search / locate / format operations."""
102
+
103
+ def __init__(self, root: DOMNode, tab_id: str, page_url: str):
104
+ self.root = root
105
+ self.tab_id = tab_id
106
+ self.captured_at = time.time()
107
+ self.page_url = page_url
108
+ self.truncated = False
109
+
110
+ # ---- 折叠(连续同类兄弟合并;无信息量节点删除) ----
111
+
112
+ def collapse(self):
113
+ self._collapse_node(self.root)
114
+
115
+ @classmethod
116
+ def _collapse_node(cls, node: DOMNode) -> None:
117
+ if not node.children:
118
+ return
119
+ merged: list[DOMNode] = []
120
+ for child in node.children:
121
+ if merged and cls._same_shape(merged[-1], child):
122
+ merged[-1].collapsed = True
123
+ merged[-1].collapsed_count += 1
124
+ continue
125
+ merged.append(child)
126
+ node.children = merged
127
+ for child in node.children:
128
+ cls._collapse_node(child)
129
+
130
+ @staticmethod
131
+ def _same_shape(a: DOMNode, b: DOMNode) -> bool:
132
+ return (
133
+ a.tag == b.tag
134
+ and a.classes == b.classes
135
+ and not a.id and not b.id
136
+ and not a.attrs and not b.attrs
137
+ )
138
+
139
+ def prune_empty(self):
140
+ """删除不可见且无文本且无标识且无子节点的节点。"""
141
+ self._prune_node(self.root)
142
+
143
+ def _prune_node(self, node: DOMNode) -> None:
144
+ kept = []
145
+ for child in node.children:
146
+ self._prune_node(child)
147
+ if child.children or child.text or child.id or child.classes or child.attrs or child.collapsed:
148
+ kept.append(child)
149
+ node.children = kept
150
+
151
+ # ---- 查询 ----
152
+
153
+ def search(self, text: str) -> list[tuple[str, DOMNode]]:
154
+ """DFS 搜索文本/属性/id 包含关键字(大小写不敏感),返回 (路径, 节点)。"""
155
+ kw = text.lower()
156
+ results: list[tuple[str, DOMNode]] = []
157
+
158
+ def dfs(node: DOMNode, path: str):
159
+ haystack = (node.text + " " + node.id + " " + " ".join(str(v) for v in node.attrs.values())).lower()
160
+ if kw in haystack:
161
+ results.append((path, node))
162
+ for child in node.children:
163
+ dfs(child, f"{path} > {child.selector}")
164
+
165
+ dfs(self.root, self.root.selector or "html")
166
+ return results
167
+
168
+ def find_tag(self, tag: str) -> list[DOMNode]:
169
+ out: list[DOMNode] = []
170
+
171
+ def dfs(node: DOMNode):
172
+ if node.tag == tag:
173
+ out.append(node)
174
+ for c in node.children:
175
+ dfs(c)
176
+
177
+ dfs(self.root)
178
+ return out
179
+
180
+ def find_attr(self, key: str, value: str = "") -> list[DOMNode]:
181
+ out: list[DOMNode] = []
182
+
183
+ def dfs(node: DOMNode):
184
+ v = node.attrs.get(key)
185
+ if v is not None and (not value or value in v):
186
+ out.append(node)
187
+ for c in node.children:
188
+ dfs(c)
189
+
190
+ dfs(self.root)
191
+ return out
192
+
193
+ def locate(self, path: str) -> DOMNode | None:
194
+ """路径定位:`div.content > div.post-list > article:nth-child(3)`。
195
+
196
+ 段格式 `tag[.class][#id][:nth-child(n)]`,`>` 分隔。
197
+ """
198
+ segments = [s.strip() for s in path.split(">") if s.strip()]
199
+ if not segments:
200
+ return None
201
+
202
+ def match(node: DOMNode, spec: str) -> bool:
203
+ rest = spec
204
+ tag = rest.split(".", 1)[0].split("#", 1)[0].split(":", 1)[0]
205
+ if tag and node.tag != tag:
206
+ return False
207
+ rest = rest[len(tag):]
208
+ if rest.startswith("."):
209
+ cls = rest[1:].split("#", 1)[0].split(":", 1)[0]
210
+ if cls not in node.classes:
211
+ return False
212
+ rest = rest[len(cls) + 1:]
213
+ if rest.startswith("#"):
214
+ nid = rest[1:].split(":", 1)[0]
215
+ if node.id != nid:
216
+ return False
217
+ rest = rest[len(nid) + 1:]
218
+ if rest.startswith(":nth-child(") and rest.endswith(")"):
219
+ try:
220
+ n = int(rest[len(":nth-child("):-1])
221
+ except ValueError:
222
+ n = None
223
+ if n is not None and n >= 1:
224
+ parent = self._parent_of(node)
225
+ if parent is None or n > len(parent.children) or parent.children[n - 1] is not node:
226
+ return False
227
+ return True
228
+
229
+ def find(nodes: list[DOMNode], spec: str) -> DOMNode | None:
230
+ for n in nodes:
231
+ if match(n, spec):
232
+ return n
233
+ return None
234
+
235
+ current: DOMNode | None = None
236
+ for spec in segments:
237
+ if current is None:
238
+ current = find([self.root], spec)
239
+ else:
240
+ current = find(current.children, spec)
241
+ if current is None:
242
+ return None
243
+ return current
244
+
245
+ def _parent_of(self, node: DOMNode) -> DOMNode | None:
246
+ def dfs(n: DOMNode) -> DOMNode | None:
247
+ for c in n.children:
248
+ if c is node:
249
+ return n
250
+ r = dfs(c)
251
+ if r:
252
+ return r
253
+ return None
254
+ return dfs(self.root)
255
+
256
+ # ---- 输出 ----
257
+
258
+ def format(self, depth: int = 4) -> str:
259
+ lines: list[str] = []
260
+ self._format_node(self.root, "", "", 0, depth, lines)
261
+ if self.truncated:
262
+ lines.append("... (truncated)")
263
+ return "\n".join(lines)
264
+
265
+ def _format_node(self, node: DOMNode, prefix: str, connector: str, level: int, depth: int, lines: list[str]):
266
+ if level >= depth and node is not self.root:
267
+ return
268
+ line = prefix + connector + node.selector
269
+ if node.collapsed and node.collapsed_count >= 1:
270
+ line += f" [x{node.collapsed_count + 1}]"
271
+ if node.text:
272
+ line += f' "{node.text}"'
273
+ if node.is_interactive():
274
+ line += f" [{node.tag}]"
275
+ lines.append(line)
276
+
277
+ child_prefix = prefix + (" " if connector == "└── " else "│ ")
278
+ for i, child in enumerate(node.children):
279
+ last = (i == len(node.children) - 1)
280
+ c = "└── " if last else "├── "
281
+ self._format_node(child, child_prefix, c, level + 1, depth, lines)
282
+
283
+
284
+ async def snapshot_tree(page) -> DOMTree:
285
+ """Evaluate SNAPSHOT_JS on the page and build a DOMTree."""
286
+ data = await page.evaluate(SNAPSHOT_JS)
287
+ root_dict = (data or {}).get("root")
288
+ if not root_dict:
289
+ root_dict = {"tag": "html", "id": "", "classes": [], "text": "", "attrs": {}, "children": []}
290
+ try:
291
+ page_url = page.url
292
+ except Exception:
293
+ page_url = ""
294
+ tree = DOMTree(_from_dict(root_dict), "", page_url)
295
+ tree.truncated = bool((data or {}).get("truncated"))
296
+ tree.collapse()
297
+ tree.prune_empty()
298
+ return tree