html-reader-llm 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
html_reader_llm/llm.py ADDED
@@ -0,0 +1,502 @@
1
+ """LLM analysis — page classification, metadata extraction, CSS selector rule generation.
2
+
3
+ Uses OpenAI-compatible API via urllib3. trafilatura provides metadata context.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import json
9
+ import logging
10
+ import re
11
+ from typing import Any, TypedDict
12
+
13
+ import urllib3
14
+ from selectolax.lexbor import LexborHTMLParser
15
+
16
+ from html_reader_llm import settings
17
+ from html_reader_llm.render_detect import _fetch_http, detect_render
18
+ from html_reader_llm.simplify import simplify_html
19
+
20
+ logger = logging.getLogger(__name__)
21
+
22
+ # =========================================================================
23
+ # Result types
24
+ # =========================================================================
25
+
26
+
27
+ class SelectorRule(TypedDict):
28
+ css: str
29
+ method: str # "$text", "$html", "@attr_name"
30
+
31
+
32
+ class ListRule(TypedDict):
33
+ name: str
34
+ is_main: bool
35
+ selectors: dict[str, SelectorRule]
36
+
37
+
38
+ class DetailRules(TypedDict, total=False):
39
+ title: SelectorRule | None
40
+ author: SelectorRule | None
41
+ date: SelectorRule | None
42
+ description: SelectorRule | None
43
+ image: SelectorRule | None
44
+ categories: SelectorRule | None
45
+ tags: SelectorRule | None
46
+ content: SelectorRule | None
47
+
48
+
49
+ class AnalysisResult(TypedDict):
50
+ page_type: str
51
+ metadata: dict[str, Any]
52
+ list_rules: list[ListRule] | None
53
+ detail_rules: DetailRules | None
54
+ raw_llm_response: str
55
+
56
+
57
+ # =========================================================================
58
+ # Token estimation
59
+ # =========================================================================
60
+
61
+
62
+ def _estimate_tokens(text: str) -> int:
63
+ """Rough token count estimation (len/3 for CJK/English mix)."""
64
+ return len(text) // 3
65
+
66
+
67
+ # =========================================================================
68
+ # Prompt trimming
69
+ # =========================================================================
70
+
71
+
72
+ def _trim_to_budget(html: str, max_tokens: int) -> str:
73
+ """Trim HTML to fit within 50% of max_tokens.
74
+
75
+ Keeps head and tail, truncates from middle.
76
+ """
77
+ budget = max_tokens // 2
78
+ estimated = _estimate_tokens(html)
79
+ if estimated <= budget:
80
+ return html
81
+
82
+ # Keep 60% head, 40% tail
83
+ head_ratio = 0.6
84
+ head_chars = int(len(html) * budget / estimated * head_ratio)
85
+ tail_chars = int(len(html) * budget / estimated * (1 - head_ratio))
86
+
87
+ head = html[:head_chars]
88
+ tail = html[-tail_chars:] if tail_chars > 0 else ""
89
+ logger.debug(
90
+ "Prompt trimmed: %d -> %d tokens (budget=%d)",
91
+ estimated,
92
+ _estimate_tokens(head + tail),
93
+ budget,
94
+ )
95
+ return head + "\n[...truncated...]\n" + tail
96
+
97
+
98
+ # =========================================================================
99
+ # trafilatura metadata
100
+ # =========================================================================
101
+
102
+
103
+ def _extract_metadata(html: str) -> dict[str, Any]:
104
+ """Extract metadata from raw HTML using trafilatura."""
105
+ try:
106
+ from trafilatura.metadata import extract_metadata as _extract
107
+
108
+ meta = _extract(html)
109
+ result: dict[str, Any] = {}
110
+ for field in (
111
+ "title",
112
+ "author",
113
+ "date",
114
+ "url",
115
+ "sitename",
116
+ "description",
117
+ "image",
118
+ "categories",
119
+ "tags",
120
+ ):
121
+ value = getattr(meta, field, None)
122
+ if value:
123
+ result[field] = value
124
+ logger.debug("trafilatura metadata: %d fields extracted", len(result))
125
+ return result
126
+ except Exception as e: # noqa: BLE001
127
+ logger.warning("trafilatura failed: %s", e)
128
+ return {}
129
+
130
+
131
+ # =========================================================================
132
+ # LLM API call
133
+ # =========================================================================
134
+
135
+ _SYSTEM_PROMPT = """Analyze HTML and return JSON. Two fields: page_type + rules.
136
+
137
+ ## Rules format
138
+
139
+ Each selector: {"css": "<selector>", "method": "<method>"}
140
+
141
+ Methods:
142
+ - `$text` — inner text (for h1, p, span, a, div, td, etc.)
143
+ - `$html` — inner HTML (for article body, content divs)
144
+ - `@attr` — attribute value (for meta tags, links, images, time, etc.)
145
+
146
+ **Standard CSS selectors ONLY.**
147
+ FORBIDDEN: :contains(), :has(), :first, :last, :eq(), :nth(), :not(), :gt(), :lt()
148
+ VALID: tag, #id, .class, [attr=value], tag.subclass, parent > child, a[href], meta[name="x"]
149
+
150
+ ## page_type: "list" or "detail"
151
+
152
+ ### List page output
153
+ ```json
154
+ {"page_type":"list","list_rules":[
155
+ {"name":"main","is_main":true,"selectors":{
156
+ "title":{"css":".item-title","method":"$text"},
157
+ "link":{"css":".item-title","method":"@href"},
158
+ "pubtime":{"css":".item-date","method":"$text"}
159
+ }},
160
+ {"name":"sidebar","is_main":false,"selectors":{
161
+ "title":{"css":".side-title","method":"$text"},
162
+ "link":{"css":".side-title","method":"@href"}
163
+ }}
164
+ ]}
165
+ ```
166
+ link is REQUIRED in every list rule.
167
+
168
+ ### Detail page output
169
+ ```json
170
+ {"page_type":"detail","detail_rules":{
171
+ "title":{"css":"h1","method":"$text"},
172
+ "author":{"css":".author","method":"$text"},
173
+ "date":{"css":"meta[property='article:published_time']","method":"@content"},
174
+ "description":{"css":"meta[name='description']","method":"@content"},
175
+ "image":{"css":"meta[property='og:image']","method":"@content"},
176
+ "categories":{"css":".category","method":"$text"},
177
+ "tags":{"css":".tag","method":"$text"},
178
+ "content":{"css":"article .body","method":"$text"}
179
+ }}
180
+ ```
181
+ Omit fields that don't exist on the page. Omit empty rules.
182
+
183
+ ### More examples
184
+
185
+ Detail with time element:
186
+ ```json
187
+ {"title":{"css":"h1.headline","method":"$text"},"date":{"css":"time","method":"@datetime"},"content":{"css":"article","method":"$text"}}
188
+ ```
189
+
190
+ List with data attributes:
191
+ ```json
192
+ {"link":{"css":"a.article-link","method":"@href"},"title":{"css":"a.article-link","method":"$text"},"pubtime":{"css":"span.date","method":"$text"}}
193
+ ```
194
+
195
+ Return ONLY valid JSON. No markdown fences. No explanation."""
196
+
197
+
198
+ def _call_llm(
199
+ system_prompt: str,
200
+ user_prompt: str,
201
+ base_url: str,
202
+ api_key: str,
203
+ model: str,
204
+ max_completion_tokens: int = 8192,
205
+ temperature: float = 0.1,
206
+ ) -> str:
207
+ """Call OpenAI-compatible chat completions API.
208
+
209
+ Args:
210
+ max_completion_tokens: Max tokens for LLM completion (not prompt).
211
+ temperature: Sampling temperature.
212
+ """
213
+ url = f"{base_url.rstrip('/')}/chat/completions"
214
+ headers = {
215
+ "Authorization": f"Bearer {api_key}",
216
+ "Content-Type": "application/json",
217
+ }
218
+ payload = {
219
+ "model": model,
220
+ "messages": [
221
+ {"role": "system", "content": system_prompt},
222
+ {"role": "user", "content": user_prompt},
223
+ ],
224
+ "max_tokens": max_completion_tokens,
225
+ "temperature": temperature,
226
+ "response_format": {"type": "json_object"},
227
+ }
228
+
229
+ logger.debug("LLM request: model=%s, prompt=%d chars", model, len(user_prompt))
230
+
231
+ http = urllib3.PoolManager()
232
+ resp = http.request(
233
+ "POST",
234
+ url,
235
+ body=json.dumps(payload).encode("utf-8"),
236
+ headers=headers,
237
+ timeout=120,
238
+ )
239
+ if resp.status != 200:
240
+ raise urllib3.exceptions.HTTPError(f"HTTP {resp.status}")
241
+ data = json.loads(resp.data.decode("utf-8"))
242
+
243
+ usage = data.get("usage", {})
244
+ logger.debug(
245
+ "LLM response: prompt_tokens=%s, completion_tokens=%s, total=%s",
246
+ usage.get("prompt_tokens", "?"),
247
+ usage.get("completion_tokens", "?"),
248
+ usage.get("total_tokens", "?"),
249
+ )
250
+
251
+ content = data["choices"][0]["message"]["content"]
252
+ return content
253
+
254
+
255
+ # =========================================================================
256
+ # Response parsing
257
+ # =========================================================================
258
+
259
+
260
+ def _parse_llm_response(raw: str) -> dict[str, Any]:
261
+ """Parse LLM JSON response with fallback."""
262
+ # Try direct parse
263
+ try:
264
+ return json.loads(raw)
265
+ except json.JSONDecodeError:
266
+ pass
267
+
268
+ # Try extracting JSON from markdown fences
269
+ match = re.search(r"```(?:json)?\s*\n?(.*?)\n?```", raw, re.DOTALL)
270
+ if match:
271
+ try:
272
+ return json.loads(match.group(1))
273
+ except json.JSONDecodeError:
274
+ pass
275
+
276
+ logger.warning("Failed to parse LLM response as JSON")
277
+ return {}
278
+
279
+
280
+ # =========================================================================
281
+ # Selector validation
282
+ # =========================================================================
283
+
284
+
285
+ def _validate_selector(parser: Any, selector: str | dict) -> bool:
286
+ """Check if a CSS selector matches at least one node with extractable value.
287
+
288
+ Supports two formats:
289
+ - str (legacy): plain CSS selector, checks text + content attr
290
+ - dict (new): {"css": "...", "method": "$text|@attr"}
291
+ """
292
+ if isinstance(selector, dict):
293
+ css = selector.get("css", "")
294
+ method = selector.get("method", "$text")
295
+ else:
296
+ css = selector
297
+ method = "$text"
298
+
299
+ if not css:
300
+ return False
301
+
302
+ try:
303
+ nodes = parser.css(css)
304
+ except Exception: # noqa: BLE001
305
+ return False
306
+ if not nodes:
307
+ return False
308
+
309
+ for node in nodes:
310
+ if method == "$text":
311
+ text = node.text(deep=True).strip()
312
+ if text:
313
+ return True
314
+ elif method == "$html":
315
+ html = node.html
316
+ if html and html.strip():
317
+ return True
318
+ elif method.startswith("@"):
319
+ attr_name = method[1:]
320
+ attr_val = (node.attributes.get(attr_name) or "").strip()
321
+ if attr_val:
322
+ return True
323
+ else:
324
+ # Unknown method, try text as fallback
325
+ text = node.text(deep=True).strip()
326
+ if text:
327
+ return True
328
+ return False
329
+
330
+
331
+ def _validate_rules(parsed: dict[str, Any], html: str) -> dict[str, Any]:
332
+ """Validate LLM-generated selectors against actual HTML.
333
+
334
+ Removes selectors that don't match or produce empty values.
335
+ For list rules: removes rules missing required 'link' selector.
336
+ """
337
+ parser = LexborHTMLParser(html)
338
+ validated: dict[str, Any] = dict(parsed)
339
+
340
+ # Validate list rules
341
+ list_rules = parsed.get("list_rules")
342
+ if list_rules:
343
+ valid_rules = []
344
+ for rule in list_rules:
345
+ selectors = rule.get("selectors", {})
346
+ # link is required
347
+ link_sel = selectors.get("link")
348
+ if not link_sel or not _validate_selector(parser, link_sel):
349
+ logger.warning(
350
+ "List rule '%s' rejected: link selector '%s' not found or empty",
351
+ rule.get("name", "?"),
352
+ link_sel,
353
+ )
354
+ continue
355
+ # Validate other selectors, remove invalid ones
356
+ clean_selectors: dict[str, Any] = {"link": link_sel}
357
+ for field in ("title", "pubtime"):
358
+ sel = selectors.get(field)
359
+ if sel and _validate_selector(parser, sel):
360
+ clean_selectors[field] = sel
361
+ elif sel:
362
+ logger.warning(
363
+ "List rule '%s': %s selector '%s' removed (not found)",
364
+ rule.get("name", "?"),
365
+ field,
366
+ sel,
367
+ )
368
+ rule["selectors"] = clean_selectors
369
+ valid_rules.append(rule)
370
+ validated["list_rules"] = valid_rules
371
+
372
+ # Validate detail rules
373
+ detail_rules = parsed.get("detail_rules")
374
+ if detail_rules:
375
+ clean_detail: dict[str, Any] = {}
376
+ for field, sel in detail_rules.items():
377
+ if sel and _validate_selector(parser, sel):
378
+ clean_detail[field] = sel
379
+ elif sel:
380
+ logger.warning(
381
+ "Detail rule: %s selector '%s' removed (not found)",
382
+ field,
383
+ sel,
384
+ )
385
+ validated["detail_rules"] = clean_detail if clean_detail else None
386
+
387
+ return validated
388
+
389
+
390
+ # =========================================================================
391
+ # Main entry point
392
+ # =========================================================================
393
+
394
+
395
+ def analyze_page(
396
+ url: str,
397
+ base_url: str | None = None,
398
+ api_key: str | None = None,
399
+ model: str | None = None,
400
+ max_tokens: int | None = None,
401
+ use_cache: bool = True,
402
+ ) -> AnalysisResult:
403
+ """Analyze a URL: classify page type and generate CSS selector rules.
404
+
405
+ Pipeline: fetch HTML → simplify → trafilatura metadata → LLM analysis.
406
+
407
+ Args:
408
+ url: URL to analyze.
409
+ base_url: LLM API base URL. Falls back to HTMLREADER_LLM_BASE_URL.
410
+ api_key: LLM API key. Falls back to HTMLREADER_LLM_API_KEY.
411
+ model: LLM model name. Falls back to HTMLREADER_LLM_MODEL.
412
+ max_tokens: Model context window. Falls back to HTMLREADER_LLM_MAX_TOKENS.
413
+ use_cache: Whether to use HTTP response cache.
414
+
415
+ Returns:
416
+ AnalysisResult with page_type, metadata, rules, and raw response.
417
+ """
418
+ # Resolve config
419
+ _base_url = base_url or settings.LLM_BASE_URL
420
+ _api_key = api_key or settings.LLM_API_KEY
421
+ _model = model or settings.LLM_MODEL
422
+ _max_tokens = max_tokens or settings.LLM_MAX_TOKENS
423
+
424
+ if not _base_url or not _api_key:
425
+ raise ValueError(
426
+ "LLM config required: set HTMLREADER_LLM_BASE_URL + HTMLREADER_LLM_API_KEY "
427
+ "or pass base_url + api_key parameters"
428
+ )
429
+
430
+ logger.debug("=== analyze_page: %s ===", url)
431
+
432
+ # 1. Fetch raw HTML
433
+ status, raw_html = _fetch_http(url, use_cache=use_cache)
434
+ if not raw_html:
435
+ raise ValueError(f"Failed to fetch URL: {url} (status={status})")
436
+
437
+ # 1.5 Detect render need
438
+ render_result = detect_render(url, use_cache=use_cache)
439
+ use_browser = render_result["use_browser"]
440
+ chrome_html = render_result["chrome_html"]
441
+ logger.debug(
442
+ "Render detection: use_browser=%s (dom=%.3f, text=%.3f)",
443
+ use_browser,
444
+ render_result["dom_ratio"],
445
+ render_result["text_similarity"],
446
+ )
447
+
448
+ # 2. Extract metadata (from raw HTML, before simplification)
449
+ metadata = _extract_metadata(raw_html)
450
+
451
+ # 3. Choose HTML for LLM: Chrome rendered if use_browser, otherwise HTTP
452
+ llm_source_html = chrome_html if (use_browser and chrome_html) else raw_html
453
+ simplified = simplify_html(llm_source_html)["html"]
454
+ logger.debug(
455
+ "Simplified: %d -> %d chars (source=%s)",
456
+ len(llm_source_html),
457
+ len(simplified),
458
+ "chrome" if llm_source_html is chrome_html else "http",
459
+ )
460
+
461
+ # 4. Trim to budget
462
+ trimmed = _trim_to_budget(simplified, _max_tokens)
463
+
464
+ # 5. Build prompt
465
+ meta_str = (
466
+ json.dumps(metadata, ensure_ascii=False, indent=2) if metadata else "None"
467
+ )
468
+ user_prompt = f"Metadata (from trafilatura):\n{meta_str}\n\nHTML:\n{trimmed}"
469
+
470
+ # 6. Call LLM
471
+ raw_response = _call_llm(
472
+ system_prompt=_SYSTEM_PROMPT,
473
+ user_prompt=user_prompt,
474
+ base_url=_base_url,
475
+ api_key=_api_key,
476
+ model=_model,
477
+ )
478
+
479
+ # 7. Parse response
480
+ parsed = _parse_llm_response(raw_response)
481
+
482
+ # 7.5 Validate selectors against the same HTML used for LLM analysis
483
+ validate_html = llm_source_html
484
+ parsed = _validate_rules(parsed, validate_html)
485
+
486
+ # 8. Build result
487
+ page_type = parsed.get("page_type", "unknown")
488
+ result: AnalysisResult = {
489
+ "page_type": page_type,
490
+ "metadata": metadata,
491
+ "list_rules": parsed.get("list_rules") if page_type == "list" else None,
492
+ "detail_rules": parsed.get("detail_rules") if page_type == "detail" else None,
493
+ "raw_llm_response": raw_response,
494
+ }
495
+
496
+ logger.debug(
497
+ "=== analyze_page done: type=%s, metadata=%d fields ===",
498
+ page_type,
499
+ len(metadata),
500
+ )
501
+
502
+ return result