html-reader-llm 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- html_reader_llm/__init__.py +39 -0
- html_reader_llm/browser.py +161 -0
- html_reader_llm/cli.py +563 -0
- html_reader_llm/llm.py +502 -0
- html_reader_llm/render_detect.py +415 -0
- html_reader_llm/selector_validator.py +177 -0
- html_reader_llm/settings.py +73 -0
- html_reader_llm/simplify.py +547 -0
- html_reader_llm-0.0.1.dist-info/METADATA +52 -0
- html_reader_llm-0.0.1.dist-info/RECORD +13 -0
- html_reader_llm-0.0.1.dist-info/WHEEL +4 -0
- html_reader_llm-0.0.1.dist-info/entry_points.txt +3 -0
- html_reader_llm-0.0.1.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,415 @@
|
|
|
1
|
+
"""Render detection — compare HTTP vs Chrome --dump-dom to decide if browser rendering is needed.
|
|
2
|
+
|
|
3
|
+
Three detection dimensions:
|
|
4
|
+
- DOM structure diff (node count ratio)
|
|
5
|
+
- Text content similarity (word-set Jaccard)
|
|
6
|
+
- SPA shell detection (empty body heuristic)
|
|
7
|
+
|
|
8
|
+
Output: RenderDetectResult with per-dimension scores + use_browser decision.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import hashlib
|
|
14
|
+
import logging
|
|
15
|
+
import os
|
|
16
|
+
import re
|
|
17
|
+
import subprocess
|
|
18
|
+
from typing import TypedDict
|
|
19
|
+
|
|
20
|
+
from selectolax.lexbor import LexborHTMLParser
|
|
21
|
+
from trafilatura import fetch_url
|
|
22
|
+
|
|
23
|
+
from html_reader_llm.browser import get_browser_path
|
|
24
|
+
from html_reader_llm.simplify import _walk_nodes
|
|
25
|
+
|
|
26
|
+
logger = logging.getLogger(__name__)
|
|
27
|
+
|
|
28
|
+
# --- ENV helpers ---
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _env_int(name: str, default: int) -> int:
|
|
32
|
+
val = os.environ.get(f"HTMLREADER_{name}")
|
|
33
|
+
if val is None:
|
|
34
|
+
return default
|
|
35
|
+
try:
|
|
36
|
+
return int(val)
|
|
37
|
+
except ValueError:
|
|
38
|
+
return default
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
# --- Defaults ---
|
|
42
|
+
|
|
43
|
+
HTTP_TIMEOUT = _env_int("HTTP_TIMEOUT", 15)
|
|
44
|
+
CHROME_TIMEOUT = _env_int("CHROME_TIMEOUT", 30)
|
|
45
|
+
CHROME_VIRTUAL_TIME_BUDGET = _env_int("CHROME_VIRTUAL_TIME_BUDGET", 5000)
|
|
46
|
+
|
|
47
|
+
_USER_AGENT = (
|
|
48
|
+
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
|
49
|
+
"(KHTML, like Gecko) Chrome/151.0.0.0 Safari/537.36"
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
# --- Confidence weights ---
|
|
53
|
+
|
|
54
|
+
W_DOM_RATIO = 0.3
|
|
55
|
+
W_TEXT_SIMILARITY = 0.4
|
|
56
|
+
W_SPA_SHELL = 0.3
|
|
57
|
+
USE_BROWSER_THRESHOLD = 0.7
|
|
58
|
+
|
|
59
|
+
# --- Cache dir for HTTP responses ---
|
|
60
|
+
_CACHE_DIR = os.path.join(".temp", "http_cache")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
# =========================================================================
|
|
64
|
+
# Result type
|
|
65
|
+
# =========================================================================
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class RenderDetectResult(TypedDict):
|
|
69
|
+
http_status: int | None
|
|
70
|
+
http_html: str | None
|
|
71
|
+
chrome_html: str | None
|
|
72
|
+
dom_ratio: float
|
|
73
|
+
text_similarity: float
|
|
74
|
+
spa_shell: bool
|
|
75
|
+
confidence: float
|
|
76
|
+
use_browser: bool
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
# =========================================================================
|
|
80
|
+
# 1. HTTP fetch (trafilatura + file cache)
|
|
81
|
+
# =========================================================================
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _cache_path(url: str) -> str:
|
|
85
|
+
"""Compute cache file path for a URL."""
|
|
86
|
+
url_hash = hashlib.md5(url.encode()).hexdigest()[:12]
|
|
87
|
+
return os.path.join(_CACHE_DIR, f"{url_hash}.html")
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _fetch_http(url: str, use_cache: bool = True) -> tuple[int | None, str | None]:
|
|
91
|
+
"""Fetch URL via trafilatura.fetch_url with file cache.
|
|
92
|
+
|
|
93
|
+
Returns (status_code, html_body) or (None, None) on error.
|
|
94
|
+
trafilatura handles gzip, charset, redirects, SSL internally.
|
|
95
|
+
"""
|
|
96
|
+
cache_file = _cache_path(url)
|
|
97
|
+
|
|
98
|
+
# Try cache first
|
|
99
|
+
if use_cache and os.path.isfile(cache_file):
|
|
100
|
+
try:
|
|
101
|
+
with open(cache_file, encoding="utf-8") as f:
|
|
102
|
+
html = f.read()
|
|
103
|
+
# Strip URL comment header if present
|
|
104
|
+
if html.startswith("<!-- source:"):
|
|
105
|
+
nl = html.find("\n")
|
|
106
|
+
if nl != -1:
|
|
107
|
+
html = html[nl + 1 :]
|
|
108
|
+
logger.debug("HTTP fetch %s -> cache hit (%d chars)", url, len(html))
|
|
109
|
+
return 200, html
|
|
110
|
+
except OSError:
|
|
111
|
+
pass
|
|
112
|
+
|
|
113
|
+
# Fetch via trafilatura
|
|
114
|
+
html = fetch_url(url)
|
|
115
|
+
if not html:
|
|
116
|
+
logger.debug("HTTP fetch %s -> failed", url)
|
|
117
|
+
return None, None
|
|
118
|
+
|
|
119
|
+
logger.debug("HTTP fetch %s -> %d chars", url, len(html))
|
|
120
|
+
|
|
121
|
+
# Save to cache
|
|
122
|
+
if use_cache:
|
|
123
|
+
try:
|
|
124
|
+
os.makedirs(_CACHE_DIR, exist_ok=True)
|
|
125
|
+
with open(cache_file, "w", encoding="utf-8") as f:
|
|
126
|
+
f.write(f"<!-- source: {url} -->\n{html}")
|
|
127
|
+
except OSError as e:
|
|
128
|
+
logger.debug("Cache write failed: %s", e)
|
|
129
|
+
|
|
130
|
+
return 200, html
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
# =========================================================================
|
|
134
|
+
# 2. Chrome dump-dom
|
|
135
|
+
# =========================================================================
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _fetch_chrome(
|
|
139
|
+
url: str,
|
|
140
|
+
chrome_path: str | None = None,
|
|
141
|
+
user_data_dir: str | None = None,
|
|
142
|
+
) -> str | None:
|
|
143
|
+
"""Fetch rendered HTML via Chrome --dump-dom.
|
|
144
|
+
|
|
145
|
+
Args:
|
|
146
|
+
url: URL to fetch.
|
|
147
|
+
chrome_path: Optional Chrome/Edge path. If None, auto-detects.
|
|
148
|
+
user_data_dir: Optional Chrome user-data-dir for caching. If None, not used.
|
|
149
|
+
|
|
150
|
+
Returns rendered HTML or None on error.
|
|
151
|
+
"""
|
|
152
|
+
if chrome_path is None:
|
|
153
|
+
chrome_path = get_browser_path()
|
|
154
|
+
if chrome_path is None:
|
|
155
|
+
logger.debug("Chrome dump-dom: no browser path found")
|
|
156
|
+
return None
|
|
157
|
+
|
|
158
|
+
cmd = [
|
|
159
|
+
chrome_path,
|
|
160
|
+
"--headless",
|
|
161
|
+
"--disable-gpu",
|
|
162
|
+
"--no-sandbox",
|
|
163
|
+
f"--virtual-time-budget={CHROME_VIRTUAL_TIME_BUDGET}",
|
|
164
|
+
"--blink-settings=imagesEnabled=false",
|
|
165
|
+
f"--user-agent={_USER_AGENT}",
|
|
166
|
+
]
|
|
167
|
+
|
|
168
|
+
if user_data_dir:
|
|
169
|
+
os.makedirs(user_data_dir, exist_ok=True)
|
|
170
|
+
cmd.append(f"--user-data-dir={user_data_dir}")
|
|
171
|
+
|
|
172
|
+
cmd.extend(["--dump-dom", url])
|
|
173
|
+
|
|
174
|
+
logger.debug("Chrome dump-dom: %s", " ".join(cmd[:3]) + " ... " + url)
|
|
175
|
+
|
|
176
|
+
try:
|
|
177
|
+
result = subprocess.run(
|
|
178
|
+
cmd,
|
|
179
|
+
capture_output=True,
|
|
180
|
+
timeout=CHROME_TIMEOUT,
|
|
181
|
+
text=True,
|
|
182
|
+
check=False,
|
|
183
|
+
)
|
|
184
|
+
if result.returncode != 0:
|
|
185
|
+
logger.debug(
|
|
186
|
+
"Chrome dump-dom: exit code %d, stderr: %s",
|
|
187
|
+
result.returncode,
|
|
188
|
+
result.stderr[:200] if result.stderr else "",
|
|
189
|
+
)
|
|
190
|
+
html = result.stdout
|
|
191
|
+
if html and len(html) > 100:
|
|
192
|
+
logger.debug("Chrome dump-dom: got %d chars", len(html))
|
|
193
|
+
return html
|
|
194
|
+
logger.debug("Chrome dump-dom: empty or too short output")
|
|
195
|
+
return None
|
|
196
|
+
except subprocess.TimeoutExpired:
|
|
197
|
+
logger.debug("Chrome dump-dom: timeout after %ds", CHROME_TIMEOUT)
|
|
198
|
+
return None
|
|
199
|
+
except FileNotFoundError:
|
|
200
|
+
logger.debug("Chrome dump-dom: binary not found: %s", chrome_path)
|
|
201
|
+
return None
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
# =========================================================================
|
|
205
|
+
# 3. Detection dimensions
|
|
206
|
+
# =========================================================================
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _count_element_nodes(html: str) -> int:
|
|
210
|
+
"""Count all element nodes in HTML."""
|
|
211
|
+
parser = LexborHTMLParser(html)
|
|
212
|
+
count = 0
|
|
213
|
+
for node in _walk_nodes(parser.root):
|
|
214
|
+
if node.is_element_node:
|
|
215
|
+
count += 1
|
|
216
|
+
return count
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _dom_ratio(html_a: str, html_b: str) -> float:
|
|
220
|
+
"""Compute DOM node count ratio: min(a,b) / max(a,b).
|
|
221
|
+
|
|
222
|
+
Returns 0.0 if either side has 0 nodes.
|
|
223
|
+
"""
|
|
224
|
+
count_a = _count_element_nodes(html_a)
|
|
225
|
+
count_b = _count_element_nodes(html_b)
|
|
226
|
+
if count_a == 0 and count_b == 0:
|
|
227
|
+
return 1.0
|
|
228
|
+
if count_a == 0 or count_b == 0:
|
|
229
|
+
return 0.0
|
|
230
|
+
return min(count_a, count_b) / max(count_a, count_b)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
_TAG_RE = re.compile(r"<[^>]+>")
|
|
234
|
+
_WS_RE = re.compile(r"\s+")
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def _extract_visible_text(html: str) -> str:
|
|
238
|
+
"""Strip HTML tags and collapse whitespace to get visible text."""
|
|
239
|
+
text = _TAG_RE.sub(" ", html)
|
|
240
|
+
text = _WS_RE.sub(" ", text).strip()
|
|
241
|
+
return text
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _text_similarity(html_a: str, html_b: str) -> float:
|
|
245
|
+
"""Compute Jaccard similarity of word sets from two HTML bodies.
|
|
246
|
+
|
|
247
|
+
For Chinese text, uses character-level tokens. For English, uses word-level.
|
|
248
|
+
"""
|
|
249
|
+
text_a = _extract_visible_text(html_a)
|
|
250
|
+
text_b = _extract_visible_text(html_b)
|
|
251
|
+
if not text_a and not text_b:
|
|
252
|
+
return 1.0
|
|
253
|
+
if not text_a or not text_b:
|
|
254
|
+
return 0.0
|
|
255
|
+
|
|
256
|
+
# Tokenize: split on whitespace, then further split Chinese chars
|
|
257
|
+
words_a = _tokenize(text_a)
|
|
258
|
+
words_b = _tokenize(text_b)
|
|
259
|
+
|
|
260
|
+
if not words_a and not words_b:
|
|
261
|
+
return 1.0
|
|
262
|
+
if not words_a or not words_b:
|
|
263
|
+
return 0.0
|
|
264
|
+
|
|
265
|
+
set_a = set(words_a)
|
|
266
|
+
set_b = set(words_b)
|
|
267
|
+
intersection = set_a & set_b
|
|
268
|
+
union = set_a | set_b
|
|
269
|
+
return len(intersection) / len(union)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
_CN_CHAR_RE = re.compile(r"[\u4e00-\u9fff\u3400-\u4dbf]")
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _tokenize(text: str) -> list[str]:
|
|
276
|
+
"""Tokenize text: English words as-is, Chinese chars individually."""
|
|
277
|
+
tokens: list[str] = []
|
|
278
|
+
for word in text.split():
|
|
279
|
+
if _CN_CHAR_RE.search(word):
|
|
280
|
+
# Chinese: split into individual characters
|
|
281
|
+
for ch in word:
|
|
282
|
+
if _CN_CHAR_RE.match(ch):
|
|
283
|
+
tokens.append(ch)
|
|
284
|
+
else:
|
|
285
|
+
tokens.append(ch)
|
|
286
|
+
else:
|
|
287
|
+
tokens.append(word.lower())
|
|
288
|
+
return tokens
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _is_spa_shell(html: str) -> bool:
|
|
292
|
+
"""Detect SPA shell: body contains only noscript and/or a single empty div."""
|
|
293
|
+
parser = LexborHTMLParser(html)
|
|
294
|
+
body = parser.css_first("body")
|
|
295
|
+
if body is None:
|
|
296
|
+
return False
|
|
297
|
+
|
|
298
|
+
# Count meaningful child elements of body
|
|
299
|
+
meaningful = 0
|
|
300
|
+
has_app_div = False
|
|
301
|
+
child = body.first_child
|
|
302
|
+
while child is not None:
|
|
303
|
+
if child.is_element_node:
|
|
304
|
+
tag = child.tag.lower()
|
|
305
|
+
if tag == "noscript":
|
|
306
|
+
# noscript is expected in SPA shells
|
|
307
|
+
meaningful += 1
|
|
308
|
+
elif tag in ("div", "section", "main"):
|
|
309
|
+
meaningful += 1
|
|
310
|
+
# Check if it's an app/root div (empty or near-empty)
|
|
311
|
+
text = child.text(deep=True).strip()
|
|
312
|
+
if len(text) < 50:
|
|
313
|
+
el_id = child.attributes.get("id", "").lower()
|
|
314
|
+
el_class = child.attributes.get("class", "").lower()
|
|
315
|
+
if (
|
|
316
|
+
el_id in ("app", "root", "__next", "__nuxt")
|
|
317
|
+
or "app" in el_class
|
|
318
|
+
):
|
|
319
|
+
has_app_div = True
|
|
320
|
+
else:
|
|
321
|
+
meaningful += 1
|
|
322
|
+
child = child.next
|
|
323
|
+
|
|
324
|
+
# SPA shell: only noscript + one div with app-like id
|
|
325
|
+
return has_app_div and meaningful <= 3
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
# =========================================================================
|
|
329
|
+
# 4. Confidence + decision
|
|
330
|
+
# =========================================================================
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def _compute_confidence(
|
|
334
|
+
dom_ratio: float, text_similarity: float, spa_shell: bool
|
|
335
|
+
) -> float:
|
|
336
|
+
"""Compute weighted confidence score. Lower = more likely needs browser."""
|
|
337
|
+
spa_score = 1.0 if spa_shell else 0.0
|
|
338
|
+
# Invert: high ratio/similarity = high confidence no browser needed
|
|
339
|
+
score = (
|
|
340
|
+
W_DOM_RATIO * dom_ratio
|
|
341
|
+
+ W_TEXT_SIMILARITY * text_similarity
|
|
342
|
+
+ W_SPA_SHELL * (1.0 - spa_score) # spa_shell=True means low confidence
|
|
343
|
+
)
|
|
344
|
+
return round(score, 4)
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
# =========================================================================
|
|
348
|
+
# 5. Main entry point
|
|
349
|
+
# =========================================================================
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def detect_render(
|
|
353
|
+
url: str,
|
|
354
|
+
chrome_path: str | None = None,
|
|
355
|
+
user_data_dir: str | None = None,
|
|
356
|
+
use_cache: bool = True,
|
|
357
|
+
) -> RenderDetectResult:
|
|
358
|
+
"""Compare urllib vs Chrome --dump-dom and decide if browser rendering is needed.
|
|
359
|
+
|
|
360
|
+
Args:
|
|
361
|
+
url: The URL to analyze.
|
|
362
|
+
chrome_path: Optional Chrome/Edge path. If None, auto-detects.
|
|
363
|
+
user_data_dir: Optional Chrome user-data-dir for caching. If None, not used.
|
|
364
|
+
use_cache: Whether to use HTTP response cache.
|
|
365
|
+
|
|
366
|
+
Returns:
|
|
367
|
+
RenderDetectResult with all metrics and use_browser decision.
|
|
368
|
+
"""
|
|
369
|
+
logger.debug("=== render detection start: %s ===", url)
|
|
370
|
+
|
|
371
|
+
# Fetch HTTP
|
|
372
|
+
http_status, http_html = _fetch_http(url, use_cache=use_cache)
|
|
373
|
+
|
|
374
|
+
# Fetch Chrome
|
|
375
|
+
chrome_html = _fetch_chrome(url, chrome_path, user_data_dir)
|
|
376
|
+
|
|
377
|
+
# Compute dimensions
|
|
378
|
+
if http_html and chrome_html:
|
|
379
|
+
dom_ratio_val = _dom_ratio(http_html, chrome_html)
|
|
380
|
+
text_sim_val = _text_similarity(http_html, chrome_html)
|
|
381
|
+
elif http_html:
|
|
382
|
+
# Chrome unavailable — can only check SPA shell
|
|
383
|
+
dom_ratio_val = 0.0
|
|
384
|
+
text_sim_val = 0.0
|
|
385
|
+
else:
|
|
386
|
+
dom_ratio_val = 0.0
|
|
387
|
+
text_sim_val = 0.0
|
|
388
|
+
|
|
389
|
+
spa_shell_val = _is_spa_shell(http_html) if http_html else False
|
|
390
|
+
|
|
391
|
+
# Confidence + decision
|
|
392
|
+
confidence_val = _compute_confidence(dom_ratio_val, text_sim_val, spa_shell_val)
|
|
393
|
+
use_browser_val = confidence_val < USE_BROWSER_THRESHOLD
|
|
394
|
+
|
|
395
|
+
result: RenderDetectResult = {
|
|
396
|
+
"http_status": http_status,
|
|
397
|
+
"http_html": http_html,
|
|
398
|
+
"chrome_html": chrome_html,
|
|
399
|
+
"dom_ratio": dom_ratio_val,
|
|
400
|
+
"text_similarity": text_sim_val,
|
|
401
|
+
"spa_shell": spa_shell_val,
|
|
402
|
+
"confidence": confidence_val,
|
|
403
|
+
"use_browser": use_browser_val,
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
logger.debug(
|
|
407
|
+
"=== render detection done: dom=%.3f text=%.3f spa=%s conf=%.3f use_browser=%s ===",
|
|
408
|
+
dom_ratio_val,
|
|
409
|
+
text_sim_val,
|
|
410
|
+
spa_shell_val,
|
|
411
|
+
confidence_val,
|
|
412
|
+
use_browser_val,
|
|
413
|
+
)
|
|
414
|
+
|
|
415
|
+
return result
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
"""CSS 选择器验证模块
|
|
2
|
+
|
|
3
|
+
用于验证 CSS 选择器是否能正确提取目标内容
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from selectolax.lexbor import LexborHTMLParser
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class SelectorValidator:
|
|
12
|
+
"""CSS 选择器验证器"""
|
|
13
|
+
|
|
14
|
+
def __init__(self, html: str):
|
|
15
|
+
"""初始化验证器
|
|
16
|
+
|
|
17
|
+
Args:
|
|
18
|
+
html: HTML 内容
|
|
19
|
+
"""
|
|
20
|
+
self.parser = LexborHTMLParser(html)
|
|
21
|
+
|
|
22
|
+
def validate_selector(
|
|
23
|
+
self, selector: str, method: str = "$text", expected_count: int | None = None
|
|
24
|
+
) -> dict[str, Any]:
|
|
25
|
+
"""验证单个选择器
|
|
26
|
+
|
|
27
|
+
Args:
|
|
28
|
+
selector: CSS 选择器
|
|
29
|
+
method: 取值方式 ($text, $html, @attr)
|
|
30
|
+
expected_count: 期望匹配的数量(None 表示不检查)
|
|
31
|
+
|
|
32
|
+
Returns:
|
|
33
|
+
验证结果字典,包含:
|
|
34
|
+
- valid: bool - 是否有效
|
|
35
|
+
- count: int - 匹配数量
|
|
36
|
+
- samples: list - 示例值(最多3个)
|
|
37
|
+
- error: str | None - 错误信息
|
|
38
|
+
"""
|
|
39
|
+
result = {"valid": False, "count": 0, "samples": [], "error": None}
|
|
40
|
+
|
|
41
|
+
try:
|
|
42
|
+
nodes = self.parser.css(selector)
|
|
43
|
+
result["count"] = len(nodes)
|
|
44
|
+
|
|
45
|
+
# 提取示例值
|
|
46
|
+
for node in nodes[:3]:
|
|
47
|
+
if method.startswith("@"):
|
|
48
|
+
# 属性值: @content, @href, @src 等
|
|
49
|
+
attr_name = method[1:]
|
|
50
|
+
value = node.attributes.get(attr_name, "")
|
|
51
|
+
elif method == "$html":
|
|
52
|
+
value = node.html
|
|
53
|
+
else:
|
|
54
|
+
# 默认 $text
|
|
55
|
+
value = node.text()
|
|
56
|
+
|
|
57
|
+
result["samples"].append(value[:100] if value else "")
|
|
58
|
+
|
|
59
|
+
# 检查是否有效
|
|
60
|
+
if result["count"] > 0:
|
|
61
|
+
result["valid"] = True
|
|
62
|
+
|
|
63
|
+
# 检查期望数量
|
|
64
|
+
if expected_count is not None and result["count"] != expected_count:
|
|
65
|
+
result["valid"] = False
|
|
66
|
+
result["error"] = (
|
|
67
|
+
f"期望 {expected_count} 个匹配,实际 {result['count']} 个"
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
except Exception as e:
|
|
71
|
+
result["error"] = str(e)
|
|
72
|
+
|
|
73
|
+
return result
|
|
74
|
+
|
|
75
|
+
def validate_rules(
|
|
76
|
+
self, rules: dict[str, dict], list_selector: str | None = None
|
|
77
|
+
) -> dict[str, dict]:
|
|
78
|
+
"""验证一组规则
|
|
79
|
+
|
|
80
|
+
Args:
|
|
81
|
+
rules: 规则字典,格式: {field_name: {css: str, method: str}}
|
|
82
|
+
list_selector: 列表选择器(如果 rules 是列表项的字段)
|
|
83
|
+
|
|
84
|
+
Returns:
|
|
85
|
+
验证结果字典,格式: {field_name: validation_result}
|
|
86
|
+
"""
|
|
87
|
+
results = {}
|
|
88
|
+
|
|
89
|
+
if list_selector:
|
|
90
|
+
# 验证列表选择器
|
|
91
|
+
list_result = self.validate_selector(list_selector)
|
|
92
|
+
results["_list"] = list_result
|
|
93
|
+
|
|
94
|
+
# 如果列表有效,验证列表项内的字段
|
|
95
|
+
if list_result["valid"] and list_result["count"] > 0:
|
|
96
|
+
list_nodes = self.parser.css(list_selector)
|
|
97
|
+
first_node = list_nodes[0]
|
|
98
|
+
|
|
99
|
+
# 创建临时验证器来验证列表项
|
|
100
|
+
item_html = first_node.html
|
|
101
|
+
if item_html:
|
|
102
|
+
item_validator = SelectorValidator(item_html)
|
|
103
|
+
for field_name, rule in rules.items():
|
|
104
|
+
results[field_name] = item_validator.validate_selector(
|
|
105
|
+
rule["css"], rule.get("method", "$text")
|
|
106
|
+
)
|
|
107
|
+
else:
|
|
108
|
+
# 直接验证字段
|
|
109
|
+
for field_name, rule in rules.items():
|
|
110
|
+
results[field_name] = self.validate_selector(
|
|
111
|
+
rule["css"], rule.get("method", "$text")
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
return results
|
|
115
|
+
|
|
116
|
+
def print_validation_report(
|
|
117
|
+
self, results: dict[str, dict], title: str = "验证报告"
|
|
118
|
+
) -> None:
|
|
119
|
+
"""打印验证报告
|
|
120
|
+
|
|
121
|
+
Args:
|
|
122
|
+
results: validate_rules 返回的结果
|
|
123
|
+
title: 报告标题
|
|
124
|
+
"""
|
|
125
|
+
print(f"\n=== {title} ===\n")
|
|
126
|
+
|
|
127
|
+
for field_name, result in results.items():
|
|
128
|
+
status = "✅" if result["valid"] else "❌"
|
|
129
|
+
print(f"{status} {field_name}")
|
|
130
|
+
|
|
131
|
+
if result["error"]:
|
|
132
|
+
print(f" 错误: {result['error']}")
|
|
133
|
+
else:
|
|
134
|
+
print(f" 匹配数量: {result['count']}")
|
|
135
|
+
if result["samples"]:
|
|
136
|
+
print(" 示例值:")
|
|
137
|
+
for i, sample in enumerate(result["samples"]):
|
|
138
|
+
print(f" {i + 1}. {sample}...")
|
|
139
|
+
|
|
140
|
+
print()
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def validate_url_selectors(
|
|
144
|
+
url: str, rules: dict[str, dict], use_chrome: bool = False
|
|
145
|
+
) -> dict[str, dict]:
|
|
146
|
+
"""验证 URL 的 CSS 选择器
|
|
147
|
+
|
|
148
|
+
Args:
|
|
149
|
+
url: 目标 URL
|
|
150
|
+
rules: 规则字典
|
|
151
|
+
use_chrome: 是否使用 Chrome 渲染
|
|
152
|
+
|
|
153
|
+
Returns:
|
|
154
|
+
验证结果
|
|
155
|
+
"""
|
|
156
|
+
import subprocess
|
|
157
|
+
|
|
158
|
+
# 获取 HTML
|
|
159
|
+
cmd = ["uv", "run", "html-reader-llm", "fetch", url]
|
|
160
|
+
if use_chrome:
|
|
161
|
+
cmd.append("--chrome")
|
|
162
|
+
|
|
163
|
+
result = subprocess.run( # noqa: S603
|
|
164
|
+
cmd, capture_output=True, text=True, encoding="utf-8"
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
if result.returncode != 0:
|
|
168
|
+
raise RuntimeError(f"获取页面失败: {result.stderr}")
|
|
169
|
+
|
|
170
|
+
html = result.stdout
|
|
171
|
+
|
|
172
|
+
# 验证选择器
|
|
173
|
+
validator = SelectorValidator(html)
|
|
174
|
+
validation_results = validator.validate_rules(rules)
|
|
175
|
+
validator.print_validation_report(validation_results)
|
|
176
|
+
|
|
177
|
+
return validation_results
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Environment-variable-driven configuration for html-reader-llm.
|
|
2
|
+
|
|
3
|
+
All constants read from HTMLREADER_* env vars with sensible defaults.
|
|
4
|
+
Loads .env file via python-dotenv if present (does not override existing env).
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import os
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from dotenv import load_dotenv
|
|
13
|
+
|
|
14
|
+
# Load .env from project root (if exists), does not override existing env vars
|
|
15
|
+
load_dotenv(Path(__file__).resolve().parent.parent.parent / ".env")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _env_int(name: str, default: int) -> int:
|
|
19
|
+
"""Read an integer from HTMLREADER_<name>, fallback to default."""
|
|
20
|
+
val = os.environ.get(f"HTMLREADER_{name}")
|
|
21
|
+
if val is None:
|
|
22
|
+
return default
|
|
23
|
+
try:
|
|
24
|
+
return int(val)
|
|
25
|
+
except ValueError:
|
|
26
|
+
return default
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _env_list(name: str, default: list[str]) -> list[str]:
|
|
30
|
+
"""Read a comma-separated list from HTMLREADER_<name>, fallback to default."""
|
|
31
|
+
val = os.environ.get(f"HTMLREADER_{name}")
|
|
32
|
+
if not val:
|
|
33
|
+
return default
|
|
34
|
+
return [item.strip() for item in val.split(",") if item.strip()]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
# --- Text truncation (units: Chinese chars / English words) ---
|
|
38
|
+
TEXT_HEAD = _env_int("TEXT_HEAD", 10)
|
|
39
|
+
TEXT_TAIL = _env_int("TEXT_TAIL", 5)
|
|
40
|
+
|
|
41
|
+
# --- Script inline code truncation ---
|
|
42
|
+
SCRIPT_HEAD = _env_int("SCRIPT_HEAD", 5)
|
|
43
|
+
SCRIPT_TAIL = _env_int("SCRIPT_TAIL", 3)
|
|
44
|
+
|
|
45
|
+
# --- Image alt truncation (chars) ---
|
|
46
|
+
MEDIA_ALT_HEAD = _env_int("MEDIA_ALT_HEAD", 10)
|
|
47
|
+
|
|
48
|
+
# --- Attribute mode: "blacklist" or "whitelist" ---
|
|
49
|
+
ATTR_MODE = os.environ.get("HTMLREADER_ATTR_MODE", "blacklist")
|
|
50
|
+
|
|
51
|
+
# --- Attribute lists (comma-separated) ---
|
|
52
|
+
KEEP_ATTRS: list[str] = _env_list(
|
|
53
|
+
"KEEP_ATTRS", ["id", "class", "href", "src", "alt", "title", "role"]
|
|
54
|
+
)
|
|
55
|
+
REMOVE_ATTRS: list[str] = _env_list("REMOVE_ATTRS", ["style"])
|
|
56
|
+
|
|
57
|
+
# --- CSS selectors for node removal / retention ---
|
|
58
|
+
REMOVE_SELECTORS: list[str] = _env_list("REMOVE_SELECTORS", [])
|
|
59
|
+
KEEP_SELECTORS: list[str] = _env_list("KEEP_SELECTORS", [])
|
|
60
|
+
|
|
61
|
+
# --- data-* attributes: always keep (0=off, 1=on) ---
|
|
62
|
+
KEEP_DATA_ATTRS = bool(_env_int("KEEP_DATA_ATTRS", 1))
|
|
63
|
+
|
|
64
|
+
# --- List truncation threshold ---
|
|
65
|
+
LIST_TRUNCATE_THRESHOLD = _env_int("LIST_TRUNCATE_THRESHOLD", 5)
|
|
66
|
+
LIST_TRUNCATE_KEEP_HEAD = _env_int("LIST_TRUNCATE_KEEP_HEAD", 3)
|
|
67
|
+
LIST_TRUNCATE_KEEP_TAIL = _env_int("LIST_TRUNCATE_KEEP_TAIL", 1)
|
|
68
|
+
|
|
69
|
+
# --- LLM configuration ---
|
|
70
|
+
LLM_BASE_URL = os.environ.get("HTMLREADER_LLM_BASE_URL", "")
|
|
71
|
+
LLM_API_KEY = os.environ.get("HTMLREADER_LLM_API_KEY", "")
|
|
72
|
+
LLM_MODEL = os.environ.get("HTMLREADER_LLM_MODEL", "")
|
|
73
|
+
LLM_MAX_TOKENS = _env_int("LLM_MAX_TOKENS", 512000)
|