html-reader-llm 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- html_reader_llm/__init__.py +39 -0
- html_reader_llm/browser.py +161 -0
- html_reader_llm/cli.py +563 -0
- html_reader_llm/llm.py +502 -0
- html_reader_llm/render_detect.py +415 -0
- html_reader_llm/selector_validator.py +177 -0
- html_reader_llm/settings.py +73 -0
- html_reader_llm/simplify.py +547 -0
- html_reader_llm-0.0.1.dist-info/METADATA +52 -0
- html_reader_llm-0.0.1.dist-info/RECORD +13 -0
- html_reader_llm-0.0.1.dist-info/WHEEL +4 -0
- html_reader_llm-0.0.1.dist-info/entry_points.txt +3 -0
- html_reader_llm-0.0.1.dist-info/licenses/LICENSE +21 -0
html_reader_llm/llm.py
ADDED
|
@@ -0,0 +1,502 @@
|
|
|
1
|
+
"""LLM analysis — page classification, metadata extraction, CSS selector rule generation.
|
|
2
|
+
|
|
3
|
+
Uses OpenAI-compatible API via urllib3. trafilatura provides metadata context.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import logging
|
|
10
|
+
import re
|
|
11
|
+
from typing import Any, TypedDict
|
|
12
|
+
|
|
13
|
+
import urllib3
|
|
14
|
+
from selectolax.lexbor import LexborHTMLParser
|
|
15
|
+
|
|
16
|
+
from html_reader_llm import settings
|
|
17
|
+
from html_reader_llm.render_detect import _fetch_http, detect_render
|
|
18
|
+
from html_reader_llm.simplify import simplify_html
|
|
19
|
+
|
|
20
|
+
logger = logging.getLogger(__name__)
|
|
21
|
+
|
|
22
|
+
# =========================================================================
|
|
23
|
+
# Result types
|
|
24
|
+
# =========================================================================
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class SelectorRule(TypedDict):
|
|
28
|
+
css: str
|
|
29
|
+
method: str # "$text", "$html", "@attr_name"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class ListRule(TypedDict):
|
|
33
|
+
name: str
|
|
34
|
+
is_main: bool
|
|
35
|
+
selectors: dict[str, SelectorRule]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class DetailRules(TypedDict, total=False):
|
|
39
|
+
title: SelectorRule | None
|
|
40
|
+
author: SelectorRule | None
|
|
41
|
+
date: SelectorRule | None
|
|
42
|
+
description: SelectorRule | None
|
|
43
|
+
image: SelectorRule | None
|
|
44
|
+
categories: SelectorRule | None
|
|
45
|
+
tags: SelectorRule | None
|
|
46
|
+
content: SelectorRule | None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class AnalysisResult(TypedDict):
|
|
50
|
+
page_type: str
|
|
51
|
+
metadata: dict[str, Any]
|
|
52
|
+
list_rules: list[ListRule] | None
|
|
53
|
+
detail_rules: DetailRules | None
|
|
54
|
+
raw_llm_response: str
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
# =========================================================================
|
|
58
|
+
# Token estimation
|
|
59
|
+
# =========================================================================
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _estimate_tokens(text: str) -> int:
|
|
63
|
+
"""Rough token count estimation (len/3 for CJK/English mix)."""
|
|
64
|
+
return len(text) // 3
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
# =========================================================================
|
|
68
|
+
# Prompt trimming
|
|
69
|
+
# =========================================================================
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _trim_to_budget(html: str, max_tokens: int) -> str:
|
|
73
|
+
"""Trim HTML to fit within 50% of max_tokens.
|
|
74
|
+
|
|
75
|
+
Keeps head and tail, truncates from middle.
|
|
76
|
+
"""
|
|
77
|
+
budget = max_tokens // 2
|
|
78
|
+
estimated = _estimate_tokens(html)
|
|
79
|
+
if estimated <= budget:
|
|
80
|
+
return html
|
|
81
|
+
|
|
82
|
+
# Keep 60% head, 40% tail
|
|
83
|
+
head_ratio = 0.6
|
|
84
|
+
head_chars = int(len(html) * budget / estimated * head_ratio)
|
|
85
|
+
tail_chars = int(len(html) * budget / estimated * (1 - head_ratio))
|
|
86
|
+
|
|
87
|
+
head = html[:head_chars]
|
|
88
|
+
tail = html[-tail_chars:] if tail_chars > 0 else ""
|
|
89
|
+
logger.debug(
|
|
90
|
+
"Prompt trimmed: %d -> %d tokens (budget=%d)",
|
|
91
|
+
estimated,
|
|
92
|
+
_estimate_tokens(head + tail),
|
|
93
|
+
budget,
|
|
94
|
+
)
|
|
95
|
+
return head + "\n[...truncated...]\n" + tail
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
# =========================================================================
|
|
99
|
+
# trafilatura metadata
|
|
100
|
+
# =========================================================================
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _extract_metadata(html: str) -> dict[str, Any]:
|
|
104
|
+
"""Extract metadata from raw HTML using trafilatura."""
|
|
105
|
+
try:
|
|
106
|
+
from trafilatura.metadata import extract_metadata as _extract
|
|
107
|
+
|
|
108
|
+
meta = _extract(html)
|
|
109
|
+
result: dict[str, Any] = {}
|
|
110
|
+
for field in (
|
|
111
|
+
"title",
|
|
112
|
+
"author",
|
|
113
|
+
"date",
|
|
114
|
+
"url",
|
|
115
|
+
"sitename",
|
|
116
|
+
"description",
|
|
117
|
+
"image",
|
|
118
|
+
"categories",
|
|
119
|
+
"tags",
|
|
120
|
+
):
|
|
121
|
+
value = getattr(meta, field, None)
|
|
122
|
+
if value:
|
|
123
|
+
result[field] = value
|
|
124
|
+
logger.debug("trafilatura metadata: %d fields extracted", len(result))
|
|
125
|
+
return result
|
|
126
|
+
except Exception as e: # noqa: BLE001
|
|
127
|
+
logger.warning("trafilatura failed: %s", e)
|
|
128
|
+
return {}
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
# =========================================================================
|
|
132
|
+
# LLM API call
|
|
133
|
+
# =========================================================================
|
|
134
|
+
|
|
135
|
+
_SYSTEM_PROMPT = """Analyze HTML and return JSON. Two fields: page_type + rules.
|
|
136
|
+
|
|
137
|
+
## Rules format
|
|
138
|
+
|
|
139
|
+
Each selector: {"css": "<selector>", "method": "<method>"}
|
|
140
|
+
|
|
141
|
+
Methods:
|
|
142
|
+
- `$text` — inner text (for h1, p, span, a, div, td, etc.)
|
|
143
|
+
- `$html` — inner HTML (for article body, content divs)
|
|
144
|
+
- `@attr` — attribute value (for meta tags, links, images, time, etc.)
|
|
145
|
+
|
|
146
|
+
**Standard CSS selectors ONLY.**
|
|
147
|
+
FORBIDDEN: :contains(), :has(), :first, :last, :eq(), :nth(), :not(), :gt(), :lt()
|
|
148
|
+
VALID: tag, #id, .class, [attr=value], tag.subclass, parent > child, a[href], meta[name="x"]
|
|
149
|
+
|
|
150
|
+
## page_type: "list" or "detail"
|
|
151
|
+
|
|
152
|
+
### List page output
|
|
153
|
+
```json
|
|
154
|
+
{"page_type":"list","list_rules":[
|
|
155
|
+
{"name":"main","is_main":true,"selectors":{
|
|
156
|
+
"title":{"css":".item-title","method":"$text"},
|
|
157
|
+
"link":{"css":".item-title","method":"@href"},
|
|
158
|
+
"pubtime":{"css":".item-date","method":"$text"}
|
|
159
|
+
}},
|
|
160
|
+
{"name":"sidebar","is_main":false,"selectors":{
|
|
161
|
+
"title":{"css":".side-title","method":"$text"},
|
|
162
|
+
"link":{"css":".side-title","method":"@href"}
|
|
163
|
+
}}
|
|
164
|
+
]}
|
|
165
|
+
```
|
|
166
|
+
link is REQUIRED in every list rule.
|
|
167
|
+
|
|
168
|
+
### Detail page output
|
|
169
|
+
```json
|
|
170
|
+
{"page_type":"detail","detail_rules":{
|
|
171
|
+
"title":{"css":"h1","method":"$text"},
|
|
172
|
+
"author":{"css":".author","method":"$text"},
|
|
173
|
+
"date":{"css":"meta[property='article:published_time']","method":"@content"},
|
|
174
|
+
"description":{"css":"meta[name='description']","method":"@content"},
|
|
175
|
+
"image":{"css":"meta[property='og:image']","method":"@content"},
|
|
176
|
+
"categories":{"css":".category","method":"$text"},
|
|
177
|
+
"tags":{"css":".tag","method":"$text"},
|
|
178
|
+
"content":{"css":"article .body","method":"$text"}
|
|
179
|
+
}}
|
|
180
|
+
```
|
|
181
|
+
Omit fields that don't exist on the page. Omit empty rules.
|
|
182
|
+
|
|
183
|
+
### More examples
|
|
184
|
+
|
|
185
|
+
Detail with time element:
|
|
186
|
+
```json
|
|
187
|
+
{"title":{"css":"h1.headline","method":"$text"},"date":{"css":"time","method":"@datetime"},"content":{"css":"article","method":"$text"}}
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
List with data attributes:
|
|
191
|
+
```json
|
|
192
|
+
{"link":{"css":"a.article-link","method":"@href"},"title":{"css":"a.article-link","method":"$text"},"pubtime":{"css":"span.date","method":"$text"}}
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
Return ONLY valid JSON. No markdown fences. No explanation."""
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _call_llm(
|
|
199
|
+
system_prompt: str,
|
|
200
|
+
user_prompt: str,
|
|
201
|
+
base_url: str,
|
|
202
|
+
api_key: str,
|
|
203
|
+
model: str,
|
|
204
|
+
max_completion_tokens: int = 8192,
|
|
205
|
+
temperature: float = 0.1,
|
|
206
|
+
) -> str:
|
|
207
|
+
"""Call OpenAI-compatible chat completions API.
|
|
208
|
+
|
|
209
|
+
Args:
|
|
210
|
+
max_completion_tokens: Max tokens for LLM completion (not prompt).
|
|
211
|
+
temperature: Sampling temperature.
|
|
212
|
+
"""
|
|
213
|
+
url = f"{base_url.rstrip('/')}/chat/completions"
|
|
214
|
+
headers = {
|
|
215
|
+
"Authorization": f"Bearer {api_key}",
|
|
216
|
+
"Content-Type": "application/json",
|
|
217
|
+
}
|
|
218
|
+
payload = {
|
|
219
|
+
"model": model,
|
|
220
|
+
"messages": [
|
|
221
|
+
{"role": "system", "content": system_prompt},
|
|
222
|
+
{"role": "user", "content": user_prompt},
|
|
223
|
+
],
|
|
224
|
+
"max_tokens": max_completion_tokens,
|
|
225
|
+
"temperature": temperature,
|
|
226
|
+
"response_format": {"type": "json_object"},
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
logger.debug("LLM request: model=%s, prompt=%d chars", model, len(user_prompt))
|
|
230
|
+
|
|
231
|
+
http = urllib3.PoolManager()
|
|
232
|
+
resp = http.request(
|
|
233
|
+
"POST",
|
|
234
|
+
url,
|
|
235
|
+
body=json.dumps(payload).encode("utf-8"),
|
|
236
|
+
headers=headers,
|
|
237
|
+
timeout=120,
|
|
238
|
+
)
|
|
239
|
+
if resp.status != 200:
|
|
240
|
+
raise urllib3.exceptions.HTTPError(f"HTTP {resp.status}")
|
|
241
|
+
data = json.loads(resp.data.decode("utf-8"))
|
|
242
|
+
|
|
243
|
+
usage = data.get("usage", {})
|
|
244
|
+
logger.debug(
|
|
245
|
+
"LLM response: prompt_tokens=%s, completion_tokens=%s, total=%s",
|
|
246
|
+
usage.get("prompt_tokens", "?"),
|
|
247
|
+
usage.get("completion_tokens", "?"),
|
|
248
|
+
usage.get("total_tokens", "?"),
|
|
249
|
+
)
|
|
250
|
+
|
|
251
|
+
content = data["choices"][0]["message"]["content"]
|
|
252
|
+
return content
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
# =========================================================================
|
|
256
|
+
# Response parsing
|
|
257
|
+
# =========================================================================
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _parse_llm_response(raw: str) -> dict[str, Any]:
|
|
261
|
+
"""Parse LLM JSON response with fallback."""
|
|
262
|
+
# Try direct parse
|
|
263
|
+
try:
|
|
264
|
+
return json.loads(raw)
|
|
265
|
+
except json.JSONDecodeError:
|
|
266
|
+
pass
|
|
267
|
+
|
|
268
|
+
# Try extracting JSON from markdown fences
|
|
269
|
+
match = re.search(r"```(?:json)?\s*\n?(.*?)\n?```", raw, re.DOTALL)
|
|
270
|
+
if match:
|
|
271
|
+
try:
|
|
272
|
+
return json.loads(match.group(1))
|
|
273
|
+
except json.JSONDecodeError:
|
|
274
|
+
pass
|
|
275
|
+
|
|
276
|
+
logger.warning("Failed to parse LLM response as JSON")
|
|
277
|
+
return {}
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
# =========================================================================
|
|
281
|
+
# Selector validation
|
|
282
|
+
# =========================================================================
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _validate_selector(parser: Any, selector: str | dict) -> bool:
|
|
286
|
+
"""Check if a CSS selector matches at least one node with extractable value.
|
|
287
|
+
|
|
288
|
+
Supports two formats:
|
|
289
|
+
- str (legacy): plain CSS selector, checks text + content attr
|
|
290
|
+
- dict (new): {"css": "...", "method": "$text|@attr"}
|
|
291
|
+
"""
|
|
292
|
+
if isinstance(selector, dict):
|
|
293
|
+
css = selector.get("css", "")
|
|
294
|
+
method = selector.get("method", "$text")
|
|
295
|
+
else:
|
|
296
|
+
css = selector
|
|
297
|
+
method = "$text"
|
|
298
|
+
|
|
299
|
+
if not css:
|
|
300
|
+
return False
|
|
301
|
+
|
|
302
|
+
try:
|
|
303
|
+
nodes = parser.css(css)
|
|
304
|
+
except Exception: # noqa: BLE001
|
|
305
|
+
return False
|
|
306
|
+
if not nodes:
|
|
307
|
+
return False
|
|
308
|
+
|
|
309
|
+
for node in nodes:
|
|
310
|
+
if method == "$text":
|
|
311
|
+
text = node.text(deep=True).strip()
|
|
312
|
+
if text:
|
|
313
|
+
return True
|
|
314
|
+
elif method == "$html":
|
|
315
|
+
html = node.html
|
|
316
|
+
if html and html.strip():
|
|
317
|
+
return True
|
|
318
|
+
elif method.startswith("@"):
|
|
319
|
+
attr_name = method[1:]
|
|
320
|
+
attr_val = (node.attributes.get(attr_name) or "").strip()
|
|
321
|
+
if attr_val:
|
|
322
|
+
return True
|
|
323
|
+
else:
|
|
324
|
+
# Unknown method, try text as fallback
|
|
325
|
+
text = node.text(deep=True).strip()
|
|
326
|
+
if text:
|
|
327
|
+
return True
|
|
328
|
+
return False
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def _validate_rules(parsed: dict[str, Any], html: str) -> dict[str, Any]:
|
|
332
|
+
"""Validate LLM-generated selectors against actual HTML.
|
|
333
|
+
|
|
334
|
+
Removes selectors that don't match or produce empty values.
|
|
335
|
+
For list rules: removes rules missing required 'link' selector.
|
|
336
|
+
"""
|
|
337
|
+
parser = LexborHTMLParser(html)
|
|
338
|
+
validated: dict[str, Any] = dict(parsed)
|
|
339
|
+
|
|
340
|
+
# Validate list rules
|
|
341
|
+
list_rules = parsed.get("list_rules")
|
|
342
|
+
if list_rules:
|
|
343
|
+
valid_rules = []
|
|
344
|
+
for rule in list_rules:
|
|
345
|
+
selectors = rule.get("selectors", {})
|
|
346
|
+
# link is required
|
|
347
|
+
link_sel = selectors.get("link")
|
|
348
|
+
if not link_sel or not _validate_selector(parser, link_sel):
|
|
349
|
+
logger.warning(
|
|
350
|
+
"List rule '%s' rejected: link selector '%s' not found or empty",
|
|
351
|
+
rule.get("name", "?"),
|
|
352
|
+
link_sel,
|
|
353
|
+
)
|
|
354
|
+
continue
|
|
355
|
+
# Validate other selectors, remove invalid ones
|
|
356
|
+
clean_selectors: dict[str, Any] = {"link": link_sel}
|
|
357
|
+
for field in ("title", "pubtime"):
|
|
358
|
+
sel = selectors.get(field)
|
|
359
|
+
if sel and _validate_selector(parser, sel):
|
|
360
|
+
clean_selectors[field] = sel
|
|
361
|
+
elif sel:
|
|
362
|
+
logger.warning(
|
|
363
|
+
"List rule '%s': %s selector '%s' removed (not found)",
|
|
364
|
+
rule.get("name", "?"),
|
|
365
|
+
field,
|
|
366
|
+
sel,
|
|
367
|
+
)
|
|
368
|
+
rule["selectors"] = clean_selectors
|
|
369
|
+
valid_rules.append(rule)
|
|
370
|
+
validated["list_rules"] = valid_rules
|
|
371
|
+
|
|
372
|
+
# Validate detail rules
|
|
373
|
+
detail_rules = parsed.get("detail_rules")
|
|
374
|
+
if detail_rules:
|
|
375
|
+
clean_detail: dict[str, Any] = {}
|
|
376
|
+
for field, sel in detail_rules.items():
|
|
377
|
+
if sel and _validate_selector(parser, sel):
|
|
378
|
+
clean_detail[field] = sel
|
|
379
|
+
elif sel:
|
|
380
|
+
logger.warning(
|
|
381
|
+
"Detail rule: %s selector '%s' removed (not found)",
|
|
382
|
+
field,
|
|
383
|
+
sel,
|
|
384
|
+
)
|
|
385
|
+
validated["detail_rules"] = clean_detail if clean_detail else None
|
|
386
|
+
|
|
387
|
+
return validated
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
# =========================================================================
|
|
391
|
+
# Main entry point
|
|
392
|
+
# =========================================================================
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def analyze_page(
|
|
396
|
+
url: str,
|
|
397
|
+
base_url: str | None = None,
|
|
398
|
+
api_key: str | None = None,
|
|
399
|
+
model: str | None = None,
|
|
400
|
+
max_tokens: int | None = None,
|
|
401
|
+
use_cache: bool = True,
|
|
402
|
+
) -> AnalysisResult:
|
|
403
|
+
"""Analyze a URL: classify page type and generate CSS selector rules.
|
|
404
|
+
|
|
405
|
+
Pipeline: fetch HTML → simplify → trafilatura metadata → LLM analysis.
|
|
406
|
+
|
|
407
|
+
Args:
|
|
408
|
+
url: URL to analyze.
|
|
409
|
+
base_url: LLM API base URL. Falls back to HTMLREADER_LLM_BASE_URL.
|
|
410
|
+
api_key: LLM API key. Falls back to HTMLREADER_LLM_API_KEY.
|
|
411
|
+
model: LLM model name. Falls back to HTMLREADER_LLM_MODEL.
|
|
412
|
+
max_tokens: Model context window. Falls back to HTMLREADER_LLM_MAX_TOKENS.
|
|
413
|
+
use_cache: Whether to use HTTP response cache.
|
|
414
|
+
|
|
415
|
+
Returns:
|
|
416
|
+
AnalysisResult with page_type, metadata, rules, and raw response.
|
|
417
|
+
"""
|
|
418
|
+
# Resolve config
|
|
419
|
+
_base_url = base_url or settings.LLM_BASE_URL
|
|
420
|
+
_api_key = api_key or settings.LLM_API_KEY
|
|
421
|
+
_model = model or settings.LLM_MODEL
|
|
422
|
+
_max_tokens = max_tokens or settings.LLM_MAX_TOKENS
|
|
423
|
+
|
|
424
|
+
if not _base_url or not _api_key:
|
|
425
|
+
raise ValueError(
|
|
426
|
+
"LLM config required: set HTMLREADER_LLM_BASE_URL + HTMLREADER_LLM_API_KEY "
|
|
427
|
+
"or pass base_url + api_key parameters"
|
|
428
|
+
)
|
|
429
|
+
|
|
430
|
+
logger.debug("=== analyze_page: %s ===", url)
|
|
431
|
+
|
|
432
|
+
# 1. Fetch raw HTML
|
|
433
|
+
status, raw_html = _fetch_http(url, use_cache=use_cache)
|
|
434
|
+
if not raw_html:
|
|
435
|
+
raise ValueError(f"Failed to fetch URL: {url} (status={status})")
|
|
436
|
+
|
|
437
|
+
# 1.5 Detect render need
|
|
438
|
+
render_result = detect_render(url, use_cache=use_cache)
|
|
439
|
+
use_browser = render_result["use_browser"]
|
|
440
|
+
chrome_html = render_result["chrome_html"]
|
|
441
|
+
logger.debug(
|
|
442
|
+
"Render detection: use_browser=%s (dom=%.3f, text=%.3f)",
|
|
443
|
+
use_browser,
|
|
444
|
+
render_result["dom_ratio"],
|
|
445
|
+
render_result["text_similarity"],
|
|
446
|
+
)
|
|
447
|
+
|
|
448
|
+
# 2. Extract metadata (from raw HTML, before simplification)
|
|
449
|
+
metadata = _extract_metadata(raw_html)
|
|
450
|
+
|
|
451
|
+
# 3. Choose HTML for LLM: Chrome rendered if use_browser, otherwise HTTP
|
|
452
|
+
llm_source_html = chrome_html if (use_browser and chrome_html) else raw_html
|
|
453
|
+
simplified = simplify_html(llm_source_html)["html"]
|
|
454
|
+
logger.debug(
|
|
455
|
+
"Simplified: %d -> %d chars (source=%s)",
|
|
456
|
+
len(llm_source_html),
|
|
457
|
+
len(simplified),
|
|
458
|
+
"chrome" if llm_source_html is chrome_html else "http",
|
|
459
|
+
)
|
|
460
|
+
|
|
461
|
+
# 4. Trim to budget
|
|
462
|
+
trimmed = _trim_to_budget(simplified, _max_tokens)
|
|
463
|
+
|
|
464
|
+
# 5. Build prompt
|
|
465
|
+
meta_str = (
|
|
466
|
+
json.dumps(metadata, ensure_ascii=False, indent=2) if metadata else "None"
|
|
467
|
+
)
|
|
468
|
+
user_prompt = f"Metadata (from trafilatura):\n{meta_str}\n\nHTML:\n{trimmed}"
|
|
469
|
+
|
|
470
|
+
# 6. Call LLM
|
|
471
|
+
raw_response = _call_llm(
|
|
472
|
+
system_prompt=_SYSTEM_PROMPT,
|
|
473
|
+
user_prompt=user_prompt,
|
|
474
|
+
base_url=_base_url,
|
|
475
|
+
api_key=_api_key,
|
|
476
|
+
model=_model,
|
|
477
|
+
)
|
|
478
|
+
|
|
479
|
+
# 7. Parse response
|
|
480
|
+
parsed = _parse_llm_response(raw_response)
|
|
481
|
+
|
|
482
|
+
# 7.5 Validate selectors against the same HTML used for LLM analysis
|
|
483
|
+
validate_html = llm_source_html
|
|
484
|
+
parsed = _validate_rules(parsed, validate_html)
|
|
485
|
+
|
|
486
|
+
# 8. Build result
|
|
487
|
+
page_type = parsed.get("page_type", "unknown")
|
|
488
|
+
result: AnalysisResult = {
|
|
489
|
+
"page_type": page_type,
|
|
490
|
+
"metadata": metadata,
|
|
491
|
+
"list_rules": parsed.get("list_rules") if page_type == "list" else None,
|
|
492
|
+
"detail_rules": parsed.get("detail_rules") if page_type == "detail" else None,
|
|
493
|
+
"raw_llm_response": raw_response,
|
|
494
|
+
}
|
|
495
|
+
|
|
496
|
+
logger.debug(
|
|
497
|
+
"=== analyze_page done: type=%s, metadata=%d fields ===",
|
|
498
|
+
page_type,
|
|
499
|
+
len(metadata),
|
|
500
|
+
)
|
|
501
|
+
|
|
502
|
+
return result
|