veddata 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- veddata/__init__.py +15 -0
- veddata/browser.py +364 -0
- veddata/chromium.py +321 -0
- veddata/dom.py +298 -0
- veddata/downloads.py +193 -0
- veddata/export.py +221 -0
- veddata/gate.py +173 -0
- veddata/limits.py +188 -0
- veddata/naming.py +111 -0
- veddata/network_monitor.py +565 -0
- veddata/observation.py +181 -0
- veddata/paths.py +131 -0
- veddata/requester.py +129 -0
- veddata/scripts.py +167 -0
- veddata/server.py +113 -0
- veddata/state.py +76 -0
- veddata/tools/__init__.py +0 -0
- veddata/tools/act.py +207 -0
- veddata/tools/chain.py +372 -0
- veddata/tools/chain_tool.py +80 -0
- veddata/tools/discover.py +936 -0
- veddata/tools/navigate.py +389 -0
- veddata/tools/observe.py +402 -0
- veddata/tools/scan.py +172 -0
- veddata/watch_engine.py +359 -0
- veddata/watch_policy.py +106 -0
- veddata-0.1.0.dist-info/METADATA +198 -0
- veddata-0.1.0.dist-info/RECORD +32 -0
- veddata-0.1.0.dist-info/WHEEL +5 -0
- veddata-0.1.0.dist-info/entry_points.txt +2 -0
- veddata-0.1.0.dist-info/licenses/LICENSE +21 -0
- veddata-0.1.0.dist-info/top_level.txt +1 -0
veddata/chromium.py
ADDED
|
@@ -0,0 +1,321 @@
|
|
|
1
|
+
"""Chromium 启动 / 接管 —— 自起普通 Chrome + CDP 端口(DrissionPage 式)。
|
|
2
|
+
|
|
3
|
+
为什么不用 Playwright 启动浏览器
|
|
4
|
+
--------------------------------
|
|
5
|
+
Playwright 用 ``--remote-debugging-pipe``,并附加它自己一长串启动开关;实测同一份
|
|
6
|
+
``chrome.exe``:pipe → ``navigator.webdriver=true``(boss 直聘据此自毁页面);
|
|
7
|
+
自起 Chrome + ``--remote-debugging-port`` → ``webdriver=false``、无注入痕迹、正常 UA。
|
|
8
|
+
|
|
9
|
+
三条实用设计
|
|
10
|
+
------------
|
|
11
|
+
1. **profile 里记端口与 pid**(``.veddata-browser.json``):重连时先看已有实例是否还活着,
|
|
12
|
+
活着就复用 —— MCP 服务重启、用户登录/过验证之后会话都不丢。
|
|
13
|
+
2. **生命周期归自己管**:我们起的、以及档案里记着 pid 的那个,``close_running`` 能真关掉;
|
|
14
|
+
只有 ``BROWSER_ADDRESS`` 指定的外部浏览器才只断开。
|
|
15
|
+
3. **启动失败不重试**:profile 被另一个 Chrome 占用时,多试一次就是多开一个窗口 ——
|
|
16
|
+
只起一次,失败就把事实(pid / 端口 / 退出码 / profile 路径)报出来。
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import json
|
|
22
|
+
import os
|
|
23
|
+
import socket
|
|
24
|
+
import subprocess
|
|
25
|
+
import sys
|
|
26
|
+
import time
|
|
27
|
+
import urllib.request
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
|
|
30
|
+
PORT_FILE = ".veddata-browser.json"
|
|
31
|
+
_READY_TIMEOUT = 25.0
|
|
32
|
+
_CREATE_NO_WINDOW = 0x08000000 if sys.platform == "win32" else 0
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class ChromiumError(RuntimeError):
|
|
36
|
+
"""起不来 / 找不到浏览器 / profile 被占用。"""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def free_port() -> int:
|
|
40
|
+
"""要一个当前空闲的本地端口(绑定后立刻释放,Chrome 起来前有极小竞态)。"""
|
|
41
|
+
with socket.socket() as sock:
|
|
42
|
+
sock.bind(("127.0.0.1", 0))
|
|
43
|
+
return int(sock.getsockname()[1])
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def build_args(path: str, port: int, profile: Path, headless: bool) -> list[str]:
|
|
47
|
+
"""自起 Chrome 的参数:**只要最少的**,绝不带自动化开关。"""
|
|
48
|
+
args = [
|
|
49
|
+
path,
|
|
50
|
+
f"--remote-debugging-port={port}",
|
|
51
|
+
f"--user-data-dir={profile}",
|
|
52
|
+
"--no-first-run",
|
|
53
|
+
"--no-default-browser-check",
|
|
54
|
+
]
|
|
55
|
+
if headless:
|
|
56
|
+
args.append("--headless=new")
|
|
57
|
+
return args
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
# ---------------- 浏览器可执行文件查找 ----------------
|
|
61
|
+
|
|
62
|
+
_WIN_CHROME_RELATIVE = [
|
|
63
|
+
("LOCALAPPDATA", r"Google\Chrome\Application\chrome.exe"),
|
|
64
|
+
("ProgramFiles", r"Google\Chrome\Application\chrome.exe"),
|
|
65
|
+
("ProgramFiles(x86)", r"Google\Chrome\Application\chrome.exe"),
|
|
66
|
+
]
|
|
67
|
+
_WIN_EDGE_RELATIVE = [
|
|
68
|
+
("ProgramFiles(x86)", r"Microsoft\Edge\Application\msedge.exe"),
|
|
69
|
+
("ProgramFiles", r"Microsoft\Edge\Application\msedge.exe"),
|
|
70
|
+
("LOCALAPPDATA", r"Microsoft\Edge\Application\msedge.exe"),
|
|
71
|
+
]
|
|
72
|
+
_WIN_REGISTRY = [
|
|
73
|
+
r"SOFTWARE\Microsoft\Windows\CurrentVersion\App Paths\chrome.exe",
|
|
74
|
+
r"SOFTWARE\WOW6432Node\Microsoft\Windows\CurrentVersion\App Paths\chrome.exe",
|
|
75
|
+
]
|
|
76
|
+
_WIN_REGISTRY_EDGE = [
|
|
77
|
+
r"SOFTWARE\Microsoft\Windows\CurrentVersion\App Paths\msedge.exe",
|
|
78
|
+
r"SOFTWARE\WOW6432Node\Microsoft\Windows\CurrentVersion\App Paths\msedge.exe",
|
|
79
|
+
]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _from_registry(keys: list[str]) -> str | None:
|
|
83
|
+
if sys.platform != "win32":
|
|
84
|
+
return None
|
|
85
|
+
try:
|
|
86
|
+
from winreg import HKEY_CURRENT_USER, HKEY_LOCAL_MACHINE, OpenKey, QueryValueEx
|
|
87
|
+
except ImportError: # pragma: no cover
|
|
88
|
+
return None
|
|
89
|
+
for root in (HKEY_LOCAL_MACHINE, HKEY_CURRENT_USER):
|
|
90
|
+
for key in keys:
|
|
91
|
+
try:
|
|
92
|
+
with OpenKey(root, key) as handle:
|
|
93
|
+
value, _ = QueryValueEx(handle, None)
|
|
94
|
+
if value and Path(value).exists():
|
|
95
|
+
return str(value)
|
|
96
|
+
except OSError:
|
|
97
|
+
continue
|
|
98
|
+
return None
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _from_relative(entries: list[tuple[str, str]]) -> str | None:
|
|
102
|
+
for env_name, tail in entries:
|
|
103
|
+
base = os.environ.get(env_name)
|
|
104
|
+
if not base:
|
|
105
|
+
continue
|
|
106
|
+
candidate = Path(base) / tail
|
|
107
|
+
if candidate.exists():
|
|
108
|
+
return str(candidate)
|
|
109
|
+
return None
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _from_path(names: list[str]) -> str | None:
|
|
113
|
+
for directory in os.environ.get("PATH", "").split(os.pathsep):
|
|
114
|
+
if not directory:
|
|
115
|
+
continue
|
|
116
|
+
for name in names:
|
|
117
|
+
candidate = Path(directory) / name
|
|
118
|
+
if candidate.exists():
|
|
119
|
+
return str(candidate)
|
|
120
|
+
return None
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def find_chrome() -> str | None:
|
|
124
|
+
if sys.platform == "darwin":
|
|
125
|
+
for candidate in ("/Applications/Google Chrome.app/Contents/MacOS/Google Chrome",
|
|
126
|
+
"/Applications/Chromium.app/Contents/MacOS/Chromium"):
|
|
127
|
+
if Path(candidate).exists():
|
|
128
|
+
return candidate
|
|
129
|
+
return (_from_registry(_WIN_REGISTRY) or _from_relative(_WIN_CHROME_RELATIVE)
|
|
130
|
+
or _from_path(["chrome.exe" if sys.platform == "win32" else "google-chrome",
|
|
131
|
+
"chrome", "chromium"]))
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def find_edge() -> str | None:
|
|
135
|
+
if sys.platform == "darwin":
|
|
136
|
+
candidate = "/Applications/Microsoft Edge.app/Contents/MacOS/Microsoft Edge"
|
|
137
|
+
return candidate if Path(candidate).exists() else None
|
|
138
|
+
return (_from_registry(_WIN_REGISTRY_EDGE) or _from_relative(_WIN_EDGE_RELATIVE)
|
|
139
|
+
or _from_path(["msedge.exe" if sys.platform == "win32" else "microsoft-edge"]))
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def browser_path() -> str | None:
|
|
143
|
+
"""按 ``BROWSER_PATH`` 决定用哪个浏览器(空/未知 → Chrome 优先,其次 Edge)。"""
|
|
144
|
+
raw = os.environ.get("BROWSER_PATH", "").strip()
|
|
145
|
+
if raw:
|
|
146
|
+
lowered = raw.lower()
|
|
147
|
+
if lowered in ("edge", "msedge"):
|
|
148
|
+
return find_edge()
|
|
149
|
+
if lowered not in ("chrome", "chrome.exe", "google-chrome"):
|
|
150
|
+
candidate = Path(raw).expanduser()
|
|
151
|
+
if candidate.exists():
|
|
152
|
+
return str(candidate)
|
|
153
|
+
return find_chrome() or find_edge()
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
# ---------------- 存活判定 / 端口档案 ----------------
|
|
157
|
+
|
|
158
|
+
def alive(port: int, timeout: float = 1.5) -> bool:
|
|
159
|
+
"""CDP 端点是否响应(``/json/version``)。"""
|
|
160
|
+
try:
|
|
161
|
+
with urllib.request.urlopen(f"http://127.0.0.1:{port}/json/version", timeout=timeout) as response:
|
|
162
|
+
payload = json.loads(response.read().decode("utf-8", "replace"))
|
|
163
|
+
return bool(payload.get("webSocketDebuggerUrl"))
|
|
164
|
+
except Exception: # noqa: BLE001
|
|
165
|
+
return False
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _read_port_payload(profile: Path) -> dict:
|
|
169
|
+
try:
|
|
170
|
+
return json.loads((Path(profile) / PORT_FILE).read_text(encoding="utf-8"))
|
|
171
|
+
except Exception: # noqa: BLE001
|
|
172
|
+
return {}
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def read_port_file(profile: Path) -> int | None:
|
|
176
|
+
try:
|
|
177
|
+
return int(_read_port_payload(profile)["port"])
|
|
178
|
+
except Exception: # noqa: BLE001
|
|
179
|
+
return None
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def read_pid(profile: Path) -> int:
|
|
183
|
+
"""档案里记的浏览器主进程 pid(没有就是 0)。"""
|
|
184
|
+
try:
|
|
185
|
+
return int(_read_port_payload(profile).get("pid") or 0)
|
|
186
|
+
except Exception: # noqa: BLE001
|
|
187
|
+
return 0
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def write_port_file(profile: Path, port: int, pid: int | None = None) -> None:
|
|
191
|
+
target = Path(profile) / PORT_FILE
|
|
192
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
193
|
+
target.write_text(json.dumps({"port": port, "pid": pid}, ensure_ascii=False), encoding="utf-8")
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def clear_port_file(profile: Path) -> None:
|
|
197
|
+
try:
|
|
198
|
+
(Path(profile) / PORT_FILE).unlink()
|
|
199
|
+
except OSError:
|
|
200
|
+
pass
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def kill_tree(pid: int) -> bool:
|
|
204
|
+
"""按 pid 结束整个进程树(Windows: taskkill /T /F)。返回是否成功。"""
|
|
205
|
+
if not pid:
|
|
206
|
+
return False
|
|
207
|
+
try:
|
|
208
|
+
if sys.platform == "win32":
|
|
209
|
+
done = subprocess.run(["taskkill", "/PID", str(pid), "/T", "/F"],
|
|
210
|
+
capture_output=True, text=True)
|
|
211
|
+
else:
|
|
212
|
+
done = subprocess.run(["kill", "-TERM", f"-{pid}"], capture_output=True, text=True)
|
|
213
|
+
return done.returncode == 0
|
|
214
|
+
except Exception: # noqa: BLE001
|
|
215
|
+
return False
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def close_running(profile: Path) -> dict:
|
|
219
|
+
"""关掉档案里记的那个浏览器(**哪怕它不是我起起的**)。
|
|
220
|
+
|
|
221
|
+
Returns:
|
|
222
|
+
``{"port":…, "pid":…, "killed":bool}`` —— 只有事实,调用方照原样报告。
|
|
223
|
+
"""
|
|
224
|
+
profile = Path(profile)
|
|
225
|
+
port = read_port_file(profile)
|
|
226
|
+
pid = read_pid(profile)
|
|
227
|
+
killed = kill_tree(pid)
|
|
228
|
+
clear_port_file(profile)
|
|
229
|
+
return {"port": port, "pid": pid, "killed": killed}
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
# ---------------- 启动器 ----------------
|
|
233
|
+
|
|
234
|
+
class Launcher:
|
|
235
|
+
"""确保"有一个跑着的普通 Chrome 可以用",返回它的 CDP 地址。"""
|
|
236
|
+
|
|
237
|
+
def __init__(self, path: str | None = None) -> None:
|
|
238
|
+
self.path = path or browser_path()
|
|
239
|
+
self.process: subprocess.Popen | None = None
|
|
240
|
+
self.port: int | None = None
|
|
241
|
+
self.started = False
|
|
242
|
+
|
|
243
|
+
def ensure(self, profile: Path, headless: bool) -> str:
|
|
244
|
+
"""返回 ``http://127.0.0.1:<port>``;能复用已有实例就复用。
|
|
245
|
+
|
|
246
|
+
Raises:
|
|
247
|
+
ChromiumError: 找不到浏览器,或起不来(**只尝试一次**,不重试 —— 重试只会多开窗口)。
|
|
248
|
+
"""
|
|
249
|
+
profile = Path(profile)
|
|
250
|
+
profile.mkdir(parents=True, exist_ok=True)
|
|
251
|
+
|
|
252
|
+
existing = read_port_file(profile)
|
|
253
|
+
if existing and alive(existing):
|
|
254
|
+
self.port = existing
|
|
255
|
+
self.started = False
|
|
256
|
+
return f"http://127.0.0.1:{existing}"
|
|
257
|
+
|
|
258
|
+
if not self.path:
|
|
259
|
+
raise ChromiumError(
|
|
260
|
+
"未找到 Chrome/Edge:请安装,或用 BROWSER_PATH 指定浏览器可执行文件"
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
port = free_port()
|
|
264
|
+
args = build_args(self.path, port, profile, headless)
|
|
265
|
+
try:
|
|
266
|
+
process = subprocess.Popen(args, stdout=subprocess.DEVNULL,
|
|
267
|
+
stderr=subprocess.DEVNULL,
|
|
268
|
+
creationflags=_CREATE_NO_WINDOW)
|
|
269
|
+
except OSError as exc:
|
|
270
|
+
raise ChromiumError(f"无法启动 {self.path}:{exc}") from exc
|
|
271
|
+
|
|
272
|
+
if self._wait_ready(port, process):
|
|
273
|
+
self.process = process
|
|
274
|
+
self.port = port
|
|
275
|
+
self.started = True
|
|
276
|
+
write_port_file(profile, port, process.pid)
|
|
277
|
+
return f"http://127.0.0.1:{port}"
|
|
278
|
+
|
|
279
|
+
code = process.poll()
|
|
280
|
+
self._terminate(process)
|
|
281
|
+
raise ChromiumError(
|
|
282
|
+
f"Chrome 未能就绪(pid={process.pid}, port={port}, 退出码={code})—— "
|
|
283
|
+
f"profile 可能已被另一个 Chrome 占用:{profile}"
|
|
284
|
+
)
|
|
285
|
+
|
|
286
|
+
def stop(self, profile: Path | None = None, clear: bool = True) -> None:
|
|
287
|
+
"""关掉**本进程起的**实例;接管来的不动。
|
|
288
|
+
|
|
289
|
+
Args:
|
|
290
|
+
profile: 档案所在的 profile 目录。
|
|
291
|
+
clear: 是否删掉端口档案。**detach(只断开)时必须传 False** ——
|
|
292
|
+
删了就等于把"我们在用的那个浏览器"弄丢,下次会去起一个新的(profile 被占用→开一堆窗口)。
|
|
293
|
+
"""
|
|
294
|
+
if self.started and self.process is not None:
|
|
295
|
+
self._terminate(self.process)
|
|
296
|
+
self.process = None
|
|
297
|
+
self.started = False
|
|
298
|
+
if profile is not None and clear:
|
|
299
|
+
clear_port_file(Path(profile))
|
|
300
|
+
|
|
301
|
+
@staticmethod
|
|
302
|
+
def _wait_ready(port: int, process: subprocess.Popen, timeout: float = _READY_TIMEOUT) -> bool:
|
|
303
|
+
deadline = time.time() + timeout
|
|
304
|
+
while time.time() < deadline:
|
|
305
|
+
if alive(port):
|
|
306
|
+
return True
|
|
307
|
+
if process.poll() is not None: # 进程自己退了,别傻等
|
|
308
|
+
return False
|
|
309
|
+
time.sleep(0.25)
|
|
310
|
+
return False
|
|
311
|
+
|
|
312
|
+
@staticmethod
|
|
313
|
+
def _terminate(process: subprocess.Popen) -> None:
|
|
314
|
+
try:
|
|
315
|
+
process.terminate()
|
|
316
|
+
process.wait(timeout=5)
|
|
317
|
+
except Exception: # noqa: BLE001
|
|
318
|
+
try:
|
|
319
|
+
process.kill()
|
|
320
|
+
except Exception: # noqa: BLE001
|
|
321
|
+
pass
|
veddata/dom.py
ADDED
|
@@ -0,0 +1,298 @@
|
|
|
1
|
+
"""DOM tree module — in-memory snapshot of the page structure.
|
|
2
|
+
|
|
3
|
+
The browser's live DOM is snapshotted once (after the page stabilizes)
|
|
4
|
+
into a DOMTree held in Python memory. Searches and path lookups run
|
|
5
|
+
against the snapshot without touching the browser again.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import time
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
# 递归遍历 DOM 生成可序列化快照。节点数上限 2000,超出截断。
|
|
13
|
+
SNAPSHOT_JS = """
|
|
14
|
+
() => {
|
|
15
|
+
const MAX_NODES = 2000;
|
|
16
|
+
let count = 0;
|
|
17
|
+
let truncated = false;
|
|
18
|
+
const ATTR_KEYS = ['href', 'src', 'placeholder', 'aria-label', 'title', 'name', 'type', 'value', 'role'];
|
|
19
|
+
function walk(el, depth) {
|
|
20
|
+
if (depth > 24) return null;
|
|
21
|
+
if (count >= MAX_NODES) { truncated = true; return null; }
|
|
22
|
+
count++;
|
|
23
|
+
const tag = el.tagName ? el.tagName.toLowerCase() : '';
|
|
24
|
+
const node = {
|
|
25
|
+
tag: tag,
|
|
26
|
+
id: el.id || '',
|
|
27
|
+
classes: Array.from(el.classList || []),
|
|
28
|
+
text: '',
|
|
29
|
+
attrs: {},
|
|
30
|
+
visible: el.offsetParent !== null || el === document.documentElement,
|
|
31
|
+
rect: null,
|
|
32
|
+
children: []
|
|
33
|
+
};
|
|
34
|
+
for (const k of ATTR_KEYS) {
|
|
35
|
+
const v = el.getAttribute(k);
|
|
36
|
+
if (v) node.attrs[k] = v.slice(0, 120);
|
|
37
|
+
}
|
|
38
|
+
if (node.visible) {
|
|
39
|
+
const r = el.getBoundingClientRect();
|
|
40
|
+
node.rect = { w: Math.round(r.width), h: Math.round(r.height), top: Math.round(r.top), left: Math.round(r.left) };
|
|
41
|
+
}
|
|
42
|
+
const kids = Array.from(el.children || []);
|
|
43
|
+
if (kids.length === 0) {
|
|
44
|
+
const t = (el.textContent || '').trim();
|
|
45
|
+
if (t) node.text = t.slice(0, 80);
|
|
46
|
+
} else {
|
|
47
|
+
for (const c of kids) {
|
|
48
|
+
const cn = walk(c, depth + 1);
|
|
49
|
+
if (cn) node.children.push(cn);
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
return node;
|
|
53
|
+
}
|
|
54
|
+
const root = walk(document.documentElement, 0);
|
|
55
|
+
return { root: root, truncated: truncated };
|
|
56
|
+
}
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass
|
|
61
|
+
class DOMNode:
|
|
62
|
+
tag: str = ""
|
|
63
|
+
id: str = ""
|
|
64
|
+
classes: list[str] = field(default_factory=list)
|
|
65
|
+
text: str = ""
|
|
66
|
+
attrs: dict = field(default_factory=dict)
|
|
67
|
+
rect: dict | None = None
|
|
68
|
+
children: list["DOMNode"] = field(default_factory=list)
|
|
69
|
+
collapsed: bool = False
|
|
70
|
+
collapsed_count: int = 0
|
|
71
|
+
|
|
72
|
+
@property
|
|
73
|
+
def selector(self) -> str:
|
|
74
|
+
"""Compact selector like div#app.feed or input#kw."""
|
|
75
|
+
s = self.tag or "?"
|
|
76
|
+
if self.id:
|
|
77
|
+
s += f"#{self.id}"
|
|
78
|
+
if self.classes:
|
|
79
|
+
s += "." + ".".join(self.classes[:3])
|
|
80
|
+
return s
|
|
81
|
+
|
|
82
|
+
def is_interactive(self) -> bool:
|
|
83
|
+
return self.tag in ("input", "button", "select") or (self.tag == "a" and "href" in self.attrs)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _from_dict(d: dict) -> DOMNode:
|
|
87
|
+
node = DOMNode(
|
|
88
|
+
tag=d.get("tag", ""),
|
|
89
|
+
id=d.get("id", ""),
|
|
90
|
+
classes=d.get("classes", []),
|
|
91
|
+
text=d.get("text", ""),
|
|
92
|
+
attrs=d.get("attrs", {}),
|
|
93
|
+
rect=d.get("rect"),
|
|
94
|
+
)
|
|
95
|
+
for c in d.get("children", []):
|
|
96
|
+
node.children.append(_from_dict(c))
|
|
97
|
+
return node
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class DOMTree:
|
|
101
|
+
"""In-memory DOM snapshot with search / locate / format operations."""
|
|
102
|
+
|
|
103
|
+
def __init__(self, root: DOMNode, tab_id: str, page_url: str):
|
|
104
|
+
self.root = root
|
|
105
|
+
self.tab_id = tab_id
|
|
106
|
+
self.captured_at = time.time()
|
|
107
|
+
self.page_url = page_url
|
|
108
|
+
self.truncated = False
|
|
109
|
+
|
|
110
|
+
# ---- 折叠(连续同类兄弟合并;无信息量节点删除) ----
|
|
111
|
+
|
|
112
|
+
def collapse(self):
|
|
113
|
+
self._collapse_node(self.root)
|
|
114
|
+
|
|
115
|
+
@classmethod
|
|
116
|
+
def _collapse_node(cls, node: DOMNode) -> None:
|
|
117
|
+
if not node.children:
|
|
118
|
+
return
|
|
119
|
+
merged: list[DOMNode] = []
|
|
120
|
+
for child in node.children:
|
|
121
|
+
if merged and cls._same_shape(merged[-1], child):
|
|
122
|
+
merged[-1].collapsed = True
|
|
123
|
+
merged[-1].collapsed_count += 1
|
|
124
|
+
continue
|
|
125
|
+
merged.append(child)
|
|
126
|
+
node.children = merged
|
|
127
|
+
for child in node.children:
|
|
128
|
+
cls._collapse_node(child)
|
|
129
|
+
|
|
130
|
+
@staticmethod
|
|
131
|
+
def _same_shape(a: DOMNode, b: DOMNode) -> bool:
|
|
132
|
+
return (
|
|
133
|
+
a.tag == b.tag
|
|
134
|
+
and a.classes == b.classes
|
|
135
|
+
and not a.id and not b.id
|
|
136
|
+
and not a.attrs and not b.attrs
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
def prune_empty(self):
|
|
140
|
+
"""删除不可见且无文本且无标识且无子节点的节点。"""
|
|
141
|
+
self._prune_node(self.root)
|
|
142
|
+
|
|
143
|
+
def _prune_node(self, node: DOMNode) -> None:
|
|
144
|
+
kept = []
|
|
145
|
+
for child in node.children:
|
|
146
|
+
self._prune_node(child)
|
|
147
|
+
if child.children or child.text or child.id or child.classes or child.attrs or child.collapsed:
|
|
148
|
+
kept.append(child)
|
|
149
|
+
node.children = kept
|
|
150
|
+
|
|
151
|
+
# ---- 查询 ----
|
|
152
|
+
|
|
153
|
+
def search(self, text: str) -> list[tuple[str, DOMNode]]:
|
|
154
|
+
"""DFS 搜索文本/属性/id 包含关键字(大小写不敏感),返回 (路径, 节点)。"""
|
|
155
|
+
kw = text.lower()
|
|
156
|
+
results: list[tuple[str, DOMNode]] = []
|
|
157
|
+
|
|
158
|
+
def dfs(node: DOMNode, path: str):
|
|
159
|
+
haystack = (node.text + " " + node.id + " " + " ".join(str(v) for v in node.attrs.values())).lower()
|
|
160
|
+
if kw in haystack:
|
|
161
|
+
results.append((path, node))
|
|
162
|
+
for child in node.children:
|
|
163
|
+
dfs(child, f"{path} > {child.selector}")
|
|
164
|
+
|
|
165
|
+
dfs(self.root, self.root.selector or "html")
|
|
166
|
+
return results
|
|
167
|
+
|
|
168
|
+
def find_tag(self, tag: str) -> list[DOMNode]:
|
|
169
|
+
out: list[DOMNode] = []
|
|
170
|
+
|
|
171
|
+
def dfs(node: DOMNode):
|
|
172
|
+
if node.tag == tag:
|
|
173
|
+
out.append(node)
|
|
174
|
+
for c in node.children:
|
|
175
|
+
dfs(c)
|
|
176
|
+
|
|
177
|
+
dfs(self.root)
|
|
178
|
+
return out
|
|
179
|
+
|
|
180
|
+
def find_attr(self, key: str, value: str = "") -> list[DOMNode]:
|
|
181
|
+
out: list[DOMNode] = []
|
|
182
|
+
|
|
183
|
+
def dfs(node: DOMNode):
|
|
184
|
+
v = node.attrs.get(key)
|
|
185
|
+
if v is not None and (not value or value in v):
|
|
186
|
+
out.append(node)
|
|
187
|
+
for c in node.children:
|
|
188
|
+
dfs(c)
|
|
189
|
+
|
|
190
|
+
dfs(self.root)
|
|
191
|
+
return out
|
|
192
|
+
|
|
193
|
+
def locate(self, path: str) -> DOMNode | None:
|
|
194
|
+
"""路径定位:`div.content > div.post-list > article:nth-child(3)`。
|
|
195
|
+
|
|
196
|
+
段格式 `tag[.class][#id][:nth-child(n)]`,`>` 分隔。
|
|
197
|
+
"""
|
|
198
|
+
segments = [s.strip() for s in path.split(">") if s.strip()]
|
|
199
|
+
if not segments:
|
|
200
|
+
return None
|
|
201
|
+
|
|
202
|
+
def match(node: DOMNode, spec: str) -> bool:
|
|
203
|
+
rest = spec
|
|
204
|
+
tag = rest.split(".", 1)[0].split("#", 1)[0].split(":", 1)[0]
|
|
205
|
+
if tag and node.tag != tag:
|
|
206
|
+
return False
|
|
207
|
+
rest = rest[len(tag):]
|
|
208
|
+
if rest.startswith("."):
|
|
209
|
+
cls = rest[1:].split("#", 1)[0].split(":", 1)[0]
|
|
210
|
+
if cls not in node.classes:
|
|
211
|
+
return False
|
|
212
|
+
rest = rest[len(cls) + 1:]
|
|
213
|
+
if rest.startswith("#"):
|
|
214
|
+
nid = rest[1:].split(":", 1)[0]
|
|
215
|
+
if node.id != nid:
|
|
216
|
+
return False
|
|
217
|
+
rest = rest[len(nid) + 1:]
|
|
218
|
+
if rest.startswith(":nth-child(") and rest.endswith(")"):
|
|
219
|
+
try:
|
|
220
|
+
n = int(rest[len(":nth-child("):-1])
|
|
221
|
+
except ValueError:
|
|
222
|
+
n = None
|
|
223
|
+
if n is not None and n >= 1:
|
|
224
|
+
parent = self._parent_of(node)
|
|
225
|
+
if parent is None or n > len(parent.children) or parent.children[n - 1] is not node:
|
|
226
|
+
return False
|
|
227
|
+
return True
|
|
228
|
+
|
|
229
|
+
def find(nodes: list[DOMNode], spec: str) -> DOMNode | None:
|
|
230
|
+
for n in nodes:
|
|
231
|
+
if match(n, spec):
|
|
232
|
+
return n
|
|
233
|
+
return None
|
|
234
|
+
|
|
235
|
+
current: DOMNode | None = None
|
|
236
|
+
for spec in segments:
|
|
237
|
+
if current is None:
|
|
238
|
+
current = find([self.root], spec)
|
|
239
|
+
else:
|
|
240
|
+
current = find(current.children, spec)
|
|
241
|
+
if current is None:
|
|
242
|
+
return None
|
|
243
|
+
return current
|
|
244
|
+
|
|
245
|
+
def _parent_of(self, node: DOMNode) -> DOMNode | None:
|
|
246
|
+
def dfs(n: DOMNode) -> DOMNode | None:
|
|
247
|
+
for c in n.children:
|
|
248
|
+
if c is node:
|
|
249
|
+
return n
|
|
250
|
+
r = dfs(c)
|
|
251
|
+
if r:
|
|
252
|
+
return r
|
|
253
|
+
return None
|
|
254
|
+
return dfs(self.root)
|
|
255
|
+
|
|
256
|
+
# ---- 输出 ----
|
|
257
|
+
|
|
258
|
+
def format(self, depth: int = 4) -> str:
|
|
259
|
+
lines: list[str] = []
|
|
260
|
+
self._format_node(self.root, "", "", 0, depth, lines)
|
|
261
|
+
if self.truncated:
|
|
262
|
+
lines.append("... (truncated)")
|
|
263
|
+
return "\n".join(lines)
|
|
264
|
+
|
|
265
|
+
def _format_node(self, node: DOMNode, prefix: str, connector: str, level: int, depth: int, lines: list[str]):
|
|
266
|
+
if level >= depth and node is not self.root:
|
|
267
|
+
return
|
|
268
|
+
line = prefix + connector + node.selector
|
|
269
|
+
if node.collapsed and node.collapsed_count >= 1:
|
|
270
|
+
line += f" [x{node.collapsed_count + 1}]"
|
|
271
|
+
if node.text:
|
|
272
|
+
line += f' "{node.text}"'
|
|
273
|
+
if node.is_interactive():
|
|
274
|
+
line += f" [{node.tag}]"
|
|
275
|
+
lines.append(line)
|
|
276
|
+
|
|
277
|
+
child_prefix = prefix + (" " if connector == "└── " else "│ ")
|
|
278
|
+
for i, child in enumerate(node.children):
|
|
279
|
+
last = (i == len(node.children) - 1)
|
|
280
|
+
c = "└── " if last else "├── "
|
|
281
|
+
self._format_node(child, child_prefix, c, level + 1, depth, lines)
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
async def snapshot_tree(page) -> DOMTree:
|
|
285
|
+
"""Evaluate SNAPSHOT_JS on the page and build a DOMTree."""
|
|
286
|
+
data = await page.evaluate(SNAPSHOT_JS)
|
|
287
|
+
root_dict = (data or {}).get("root")
|
|
288
|
+
if not root_dict:
|
|
289
|
+
root_dict = {"tag": "html", "id": "", "classes": [], "text": "", "attrs": {}, "children": []}
|
|
290
|
+
try:
|
|
291
|
+
page_url = page.url
|
|
292
|
+
except Exception:
|
|
293
|
+
page_url = ""
|
|
294
|
+
tree = DOMTree(_from_dict(root_dict), "", page_url)
|
|
295
|
+
tree.truncated = bool((data or {}).get("truncated"))
|
|
296
|
+
tree.collapse()
|
|
297
|
+
tree.prune_empty()
|
|
298
|
+
return tree
|