html-reader-llm 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,415 @@
1
+ """Render detection — compare HTTP vs Chrome --dump-dom to decide if browser rendering is needed.
2
+
3
+ Three detection dimensions:
4
+ - DOM structure diff (node count ratio)
5
+ - Text content similarity (word-set Jaccard)
6
+ - SPA shell detection (empty body heuristic)
7
+
8
+ Output: RenderDetectResult with per-dimension scores + use_browser decision.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import hashlib
14
+ import logging
15
+ import os
16
+ import re
17
+ import subprocess
18
+ from typing import TypedDict
19
+
20
+ from selectolax.lexbor import LexborHTMLParser
21
+ from trafilatura import fetch_url
22
+
23
+ from html_reader_llm.browser import get_browser_path
24
+ from html_reader_llm.simplify import _walk_nodes
25
+
26
+ logger = logging.getLogger(__name__)
27
+
28
+ # --- ENV helpers ---
29
+
30
+
31
+ def _env_int(name: str, default: int) -> int:
32
+ val = os.environ.get(f"HTMLREADER_{name}")
33
+ if val is None:
34
+ return default
35
+ try:
36
+ return int(val)
37
+ except ValueError:
38
+ return default
39
+
40
+
41
+ # --- Defaults ---
42
+
43
+ HTTP_TIMEOUT = _env_int("HTTP_TIMEOUT", 15)
44
+ CHROME_TIMEOUT = _env_int("CHROME_TIMEOUT", 30)
45
+ CHROME_VIRTUAL_TIME_BUDGET = _env_int("CHROME_VIRTUAL_TIME_BUDGET", 5000)
46
+
47
+ _USER_AGENT = (
48
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
49
+ "(KHTML, like Gecko) Chrome/151.0.0.0 Safari/537.36"
50
+ )
51
+
52
+ # --- Confidence weights ---
53
+
54
+ W_DOM_RATIO = 0.3
55
+ W_TEXT_SIMILARITY = 0.4
56
+ W_SPA_SHELL = 0.3
57
+ USE_BROWSER_THRESHOLD = 0.7
58
+
59
+ # --- Cache dir for HTTP responses ---
60
+ _CACHE_DIR = os.path.join(".temp", "http_cache")
61
+
62
+
63
+ # =========================================================================
64
+ # Result type
65
+ # =========================================================================
66
+
67
+
68
+ class RenderDetectResult(TypedDict):
69
+ http_status: int | None
70
+ http_html: str | None
71
+ chrome_html: str | None
72
+ dom_ratio: float
73
+ text_similarity: float
74
+ spa_shell: bool
75
+ confidence: float
76
+ use_browser: bool
77
+
78
+
79
+ # =========================================================================
80
+ # 1. HTTP fetch (trafilatura + file cache)
81
+ # =========================================================================
82
+
83
+
84
+ def _cache_path(url: str) -> str:
85
+ """Compute cache file path for a URL."""
86
+ url_hash = hashlib.md5(url.encode()).hexdigest()[:12]
87
+ return os.path.join(_CACHE_DIR, f"{url_hash}.html")
88
+
89
+
90
+ def _fetch_http(url: str, use_cache: bool = True) -> tuple[int | None, str | None]:
91
+ """Fetch URL via trafilatura.fetch_url with file cache.
92
+
93
+ Returns (status_code, html_body) or (None, None) on error.
94
+ trafilatura handles gzip, charset, redirects, SSL internally.
95
+ """
96
+ cache_file = _cache_path(url)
97
+
98
+ # Try cache first
99
+ if use_cache and os.path.isfile(cache_file):
100
+ try:
101
+ with open(cache_file, encoding="utf-8") as f:
102
+ html = f.read()
103
+ # Strip URL comment header if present
104
+ if html.startswith("<!-- source:"):
105
+ nl = html.find("\n")
106
+ if nl != -1:
107
+ html = html[nl + 1 :]
108
+ logger.debug("HTTP fetch %s -> cache hit (%d chars)", url, len(html))
109
+ return 200, html
110
+ except OSError:
111
+ pass
112
+
113
+ # Fetch via trafilatura
114
+ html = fetch_url(url)
115
+ if not html:
116
+ logger.debug("HTTP fetch %s -> failed", url)
117
+ return None, None
118
+
119
+ logger.debug("HTTP fetch %s -> %d chars", url, len(html))
120
+
121
+ # Save to cache
122
+ if use_cache:
123
+ try:
124
+ os.makedirs(_CACHE_DIR, exist_ok=True)
125
+ with open(cache_file, "w", encoding="utf-8") as f:
126
+ f.write(f"<!-- source: {url} -->\n{html}")
127
+ except OSError as e:
128
+ logger.debug("Cache write failed: %s", e)
129
+
130
+ return 200, html
131
+
132
+
133
+ # =========================================================================
134
+ # 2. Chrome dump-dom
135
+ # =========================================================================
136
+
137
+
138
+ def _fetch_chrome(
139
+ url: str,
140
+ chrome_path: str | None = None,
141
+ user_data_dir: str | None = None,
142
+ ) -> str | None:
143
+ """Fetch rendered HTML via Chrome --dump-dom.
144
+
145
+ Args:
146
+ url: URL to fetch.
147
+ chrome_path: Optional Chrome/Edge path. If None, auto-detects.
148
+ user_data_dir: Optional Chrome user-data-dir for caching. If None, not used.
149
+
150
+ Returns rendered HTML or None on error.
151
+ """
152
+ if chrome_path is None:
153
+ chrome_path = get_browser_path()
154
+ if chrome_path is None:
155
+ logger.debug("Chrome dump-dom: no browser path found")
156
+ return None
157
+
158
+ cmd = [
159
+ chrome_path,
160
+ "--headless",
161
+ "--disable-gpu",
162
+ "--no-sandbox",
163
+ f"--virtual-time-budget={CHROME_VIRTUAL_TIME_BUDGET}",
164
+ "--blink-settings=imagesEnabled=false",
165
+ f"--user-agent={_USER_AGENT}",
166
+ ]
167
+
168
+ if user_data_dir:
169
+ os.makedirs(user_data_dir, exist_ok=True)
170
+ cmd.append(f"--user-data-dir={user_data_dir}")
171
+
172
+ cmd.extend(["--dump-dom", url])
173
+
174
+ logger.debug("Chrome dump-dom: %s", " ".join(cmd[:3]) + " ... " + url)
175
+
176
+ try:
177
+ result = subprocess.run(
178
+ cmd,
179
+ capture_output=True,
180
+ timeout=CHROME_TIMEOUT,
181
+ text=True,
182
+ check=False,
183
+ )
184
+ if result.returncode != 0:
185
+ logger.debug(
186
+ "Chrome dump-dom: exit code %d, stderr: %s",
187
+ result.returncode,
188
+ result.stderr[:200] if result.stderr else "",
189
+ )
190
+ html = result.stdout
191
+ if html and len(html) > 100:
192
+ logger.debug("Chrome dump-dom: got %d chars", len(html))
193
+ return html
194
+ logger.debug("Chrome dump-dom: empty or too short output")
195
+ return None
196
+ except subprocess.TimeoutExpired:
197
+ logger.debug("Chrome dump-dom: timeout after %ds", CHROME_TIMEOUT)
198
+ return None
199
+ except FileNotFoundError:
200
+ logger.debug("Chrome dump-dom: binary not found: %s", chrome_path)
201
+ return None
202
+
203
+
204
+ # =========================================================================
205
+ # 3. Detection dimensions
206
+ # =========================================================================
207
+
208
+
209
+ def _count_element_nodes(html: str) -> int:
210
+ """Count all element nodes in HTML."""
211
+ parser = LexborHTMLParser(html)
212
+ count = 0
213
+ for node in _walk_nodes(parser.root):
214
+ if node.is_element_node:
215
+ count += 1
216
+ return count
217
+
218
+
219
+ def _dom_ratio(html_a: str, html_b: str) -> float:
220
+ """Compute DOM node count ratio: min(a,b) / max(a,b).
221
+
222
+ Returns 0.0 if either side has 0 nodes.
223
+ """
224
+ count_a = _count_element_nodes(html_a)
225
+ count_b = _count_element_nodes(html_b)
226
+ if count_a == 0 and count_b == 0:
227
+ return 1.0
228
+ if count_a == 0 or count_b == 0:
229
+ return 0.0
230
+ return min(count_a, count_b) / max(count_a, count_b)
231
+
232
+
233
+ _TAG_RE = re.compile(r"<[^>]+>")
234
+ _WS_RE = re.compile(r"\s+")
235
+
236
+
237
+ def _extract_visible_text(html: str) -> str:
238
+ """Strip HTML tags and collapse whitespace to get visible text."""
239
+ text = _TAG_RE.sub(" ", html)
240
+ text = _WS_RE.sub(" ", text).strip()
241
+ return text
242
+
243
+
244
+ def _text_similarity(html_a: str, html_b: str) -> float:
245
+ """Compute Jaccard similarity of word sets from two HTML bodies.
246
+
247
+ For Chinese text, uses character-level tokens. For English, uses word-level.
248
+ """
249
+ text_a = _extract_visible_text(html_a)
250
+ text_b = _extract_visible_text(html_b)
251
+ if not text_a and not text_b:
252
+ return 1.0
253
+ if not text_a or not text_b:
254
+ return 0.0
255
+
256
+ # Tokenize: split on whitespace, then further split Chinese chars
257
+ words_a = _tokenize(text_a)
258
+ words_b = _tokenize(text_b)
259
+
260
+ if not words_a and not words_b:
261
+ return 1.0
262
+ if not words_a or not words_b:
263
+ return 0.0
264
+
265
+ set_a = set(words_a)
266
+ set_b = set(words_b)
267
+ intersection = set_a & set_b
268
+ union = set_a | set_b
269
+ return len(intersection) / len(union)
270
+
271
+
272
+ _CN_CHAR_RE = re.compile(r"[\u4e00-\u9fff\u3400-\u4dbf]")
273
+
274
+
275
+ def _tokenize(text: str) -> list[str]:
276
+ """Tokenize text: English words as-is, Chinese chars individually."""
277
+ tokens: list[str] = []
278
+ for word in text.split():
279
+ if _CN_CHAR_RE.search(word):
280
+ # Chinese: split into individual characters
281
+ for ch in word:
282
+ if _CN_CHAR_RE.match(ch):
283
+ tokens.append(ch)
284
+ else:
285
+ tokens.append(ch)
286
+ else:
287
+ tokens.append(word.lower())
288
+ return tokens
289
+
290
+
291
+ def _is_spa_shell(html: str) -> bool:
292
+ """Detect SPA shell: body contains only noscript and/or a single empty div."""
293
+ parser = LexborHTMLParser(html)
294
+ body = parser.css_first("body")
295
+ if body is None:
296
+ return False
297
+
298
+ # Count meaningful child elements of body
299
+ meaningful = 0
300
+ has_app_div = False
301
+ child = body.first_child
302
+ while child is not None:
303
+ if child.is_element_node:
304
+ tag = child.tag.lower()
305
+ if tag == "noscript":
306
+ # noscript is expected in SPA shells
307
+ meaningful += 1
308
+ elif tag in ("div", "section", "main"):
309
+ meaningful += 1
310
+ # Check if it's an app/root div (empty or near-empty)
311
+ text = child.text(deep=True).strip()
312
+ if len(text) < 50:
313
+ el_id = child.attributes.get("id", "").lower()
314
+ el_class = child.attributes.get("class", "").lower()
315
+ if (
316
+ el_id in ("app", "root", "__next", "__nuxt")
317
+ or "app" in el_class
318
+ ):
319
+ has_app_div = True
320
+ else:
321
+ meaningful += 1
322
+ child = child.next
323
+
324
+ # SPA shell: only noscript + one div with app-like id
325
+ return has_app_div and meaningful <= 3
326
+
327
+
328
+ # =========================================================================
329
+ # 4. Confidence + decision
330
+ # =========================================================================
331
+
332
+
333
+ def _compute_confidence(
334
+ dom_ratio: float, text_similarity: float, spa_shell: bool
335
+ ) -> float:
336
+ """Compute weighted confidence score. Lower = more likely needs browser."""
337
+ spa_score = 1.0 if spa_shell else 0.0
338
+ # Invert: high ratio/similarity = high confidence no browser needed
339
+ score = (
340
+ W_DOM_RATIO * dom_ratio
341
+ + W_TEXT_SIMILARITY * text_similarity
342
+ + W_SPA_SHELL * (1.0 - spa_score) # spa_shell=True means low confidence
343
+ )
344
+ return round(score, 4)
345
+
346
+
347
+ # =========================================================================
348
+ # 5. Main entry point
349
+ # =========================================================================
350
+
351
+
352
+ def detect_render(
353
+ url: str,
354
+ chrome_path: str | None = None,
355
+ user_data_dir: str | None = None,
356
+ use_cache: bool = True,
357
+ ) -> RenderDetectResult:
358
+ """Compare urllib vs Chrome --dump-dom and decide if browser rendering is needed.
359
+
360
+ Args:
361
+ url: The URL to analyze.
362
+ chrome_path: Optional Chrome/Edge path. If None, auto-detects.
363
+ user_data_dir: Optional Chrome user-data-dir for caching. If None, not used.
364
+ use_cache: Whether to use HTTP response cache.
365
+
366
+ Returns:
367
+ RenderDetectResult with all metrics and use_browser decision.
368
+ """
369
+ logger.debug("=== render detection start: %s ===", url)
370
+
371
+ # Fetch HTTP
372
+ http_status, http_html = _fetch_http(url, use_cache=use_cache)
373
+
374
+ # Fetch Chrome
375
+ chrome_html = _fetch_chrome(url, chrome_path, user_data_dir)
376
+
377
+ # Compute dimensions
378
+ if http_html and chrome_html:
379
+ dom_ratio_val = _dom_ratio(http_html, chrome_html)
380
+ text_sim_val = _text_similarity(http_html, chrome_html)
381
+ elif http_html:
382
+ # Chrome unavailable — can only check SPA shell
383
+ dom_ratio_val = 0.0
384
+ text_sim_val = 0.0
385
+ else:
386
+ dom_ratio_val = 0.0
387
+ text_sim_val = 0.0
388
+
389
+ spa_shell_val = _is_spa_shell(http_html) if http_html else False
390
+
391
+ # Confidence + decision
392
+ confidence_val = _compute_confidence(dom_ratio_val, text_sim_val, spa_shell_val)
393
+ use_browser_val = confidence_val < USE_BROWSER_THRESHOLD
394
+
395
+ result: RenderDetectResult = {
396
+ "http_status": http_status,
397
+ "http_html": http_html,
398
+ "chrome_html": chrome_html,
399
+ "dom_ratio": dom_ratio_val,
400
+ "text_similarity": text_sim_val,
401
+ "spa_shell": spa_shell_val,
402
+ "confidence": confidence_val,
403
+ "use_browser": use_browser_val,
404
+ }
405
+
406
+ logger.debug(
407
+ "=== render detection done: dom=%.3f text=%.3f spa=%s conf=%.3f use_browser=%s ===",
408
+ dom_ratio_val,
409
+ text_sim_val,
410
+ spa_shell_val,
411
+ confidence_val,
412
+ use_browser_val,
413
+ )
414
+
415
+ return result
@@ -0,0 +1,177 @@
1
+ """CSS 选择器验证模块
2
+
3
+ 用于验证 CSS 选择器是否能正确提取目标内容
4
+ """
5
+
6
+ from typing import Any
7
+
8
+ from selectolax.lexbor import LexborHTMLParser
9
+
10
+
11
+ class SelectorValidator:
12
+ """CSS 选择器验证器"""
13
+
14
+ def __init__(self, html: str):
15
+ """初始化验证器
16
+
17
+ Args:
18
+ html: HTML 内容
19
+ """
20
+ self.parser = LexborHTMLParser(html)
21
+
22
+ def validate_selector(
23
+ self, selector: str, method: str = "$text", expected_count: int | None = None
24
+ ) -> dict[str, Any]:
25
+ """验证单个选择器
26
+
27
+ Args:
28
+ selector: CSS 选择器
29
+ method: 取值方式 ($text, $html, @attr)
30
+ expected_count: 期望匹配的数量(None 表示不检查)
31
+
32
+ Returns:
33
+ 验证结果字典,包含:
34
+ - valid: bool - 是否有效
35
+ - count: int - 匹配数量
36
+ - samples: list - 示例值(最多3个)
37
+ - error: str | None - 错误信息
38
+ """
39
+ result = {"valid": False, "count": 0, "samples": [], "error": None}
40
+
41
+ try:
42
+ nodes = self.parser.css(selector)
43
+ result["count"] = len(nodes)
44
+
45
+ # 提取示例值
46
+ for node in nodes[:3]:
47
+ if method.startswith("@"):
48
+ # 属性值: @content, @href, @src 等
49
+ attr_name = method[1:]
50
+ value = node.attributes.get(attr_name, "")
51
+ elif method == "$html":
52
+ value = node.html
53
+ else:
54
+ # 默认 $text
55
+ value = node.text()
56
+
57
+ result["samples"].append(value[:100] if value else "")
58
+
59
+ # 检查是否有效
60
+ if result["count"] > 0:
61
+ result["valid"] = True
62
+
63
+ # 检查期望数量
64
+ if expected_count is not None and result["count"] != expected_count:
65
+ result["valid"] = False
66
+ result["error"] = (
67
+ f"期望 {expected_count} 个匹配,实际 {result['count']} 个"
68
+ )
69
+
70
+ except Exception as e:
71
+ result["error"] = str(e)
72
+
73
+ return result
74
+
75
+ def validate_rules(
76
+ self, rules: dict[str, dict], list_selector: str | None = None
77
+ ) -> dict[str, dict]:
78
+ """验证一组规则
79
+
80
+ Args:
81
+ rules: 规则字典,格式: {field_name: {css: str, method: str}}
82
+ list_selector: 列表选择器(如果 rules 是列表项的字段)
83
+
84
+ Returns:
85
+ 验证结果字典,格式: {field_name: validation_result}
86
+ """
87
+ results = {}
88
+
89
+ if list_selector:
90
+ # 验证列表选择器
91
+ list_result = self.validate_selector(list_selector)
92
+ results["_list"] = list_result
93
+
94
+ # 如果列表有效,验证列表项内的字段
95
+ if list_result["valid"] and list_result["count"] > 0:
96
+ list_nodes = self.parser.css(list_selector)
97
+ first_node = list_nodes[0]
98
+
99
+ # 创建临时验证器来验证列表项
100
+ item_html = first_node.html
101
+ if item_html:
102
+ item_validator = SelectorValidator(item_html)
103
+ for field_name, rule in rules.items():
104
+ results[field_name] = item_validator.validate_selector(
105
+ rule["css"], rule.get("method", "$text")
106
+ )
107
+ else:
108
+ # 直接验证字段
109
+ for field_name, rule in rules.items():
110
+ results[field_name] = self.validate_selector(
111
+ rule["css"], rule.get("method", "$text")
112
+ )
113
+
114
+ return results
115
+
116
+ def print_validation_report(
117
+ self, results: dict[str, dict], title: str = "验证报告"
118
+ ) -> None:
119
+ """打印验证报告
120
+
121
+ Args:
122
+ results: validate_rules 返回的结果
123
+ title: 报告标题
124
+ """
125
+ print(f"\n=== {title} ===\n")
126
+
127
+ for field_name, result in results.items():
128
+ status = "✅" if result["valid"] else "❌"
129
+ print(f"{status} {field_name}")
130
+
131
+ if result["error"]:
132
+ print(f" 错误: {result['error']}")
133
+ else:
134
+ print(f" 匹配数量: {result['count']}")
135
+ if result["samples"]:
136
+ print(" 示例值:")
137
+ for i, sample in enumerate(result["samples"]):
138
+ print(f" {i + 1}. {sample}...")
139
+
140
+ print()
141
+
142
+
143
+ def validate_url_selectors(
144
+ url: str, rules: dict[str, dict], use_chrome: bool = False
145
+ ) -> dict[str, dict]:
146
+ """验证 URL 的 CSS 选择器
147
+
148
+ Args:
149
+ url: 目标 URL
150
+ rules: 规则字典
151
+ use_chrome: 是否使用 Chrome 渲染
152
+
153
+ Returns:
154
+ 验证结果
155
+ """
156
+ import subprocess
157
+
158
+ # 获取 HTML
159
+ cmd = ["uv", "run", "html-reader-llm", "fetch", url]
160
+ if use_chrome:
161
+ cmd.append("--chrome")
162
+
163
+ result = subprocess.run( # noqa: S603
164
+ cmd, capture_output=True, text=True, encoding="utf-8"
165
+ )
166
+
167
+ if result.returncode != 0:
168
+ raise RuntimeError(f"获取页面失败: {result.stderr}")
169
+
170
+ html = result.stdout
171
+
172
+ # 验证选择器
173
+ validator = SelectorValidator(html)
174
+ validation_results = validator.validate_rules(rules)
175
+ validator.print_validation_report(validation_results)
176
+
177
+ return validation_results
@@ -0,0 +1,73 @@
1
+ """Environment-variable-driven configuration for html-reader-llm.
2
+
3
+ All constants read from HTMLREADER_* env vars with sensible defaults.
4
+ Loads .env file via python-dotenv if present (does not override existing env).
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import os
10
+ from pathlib import Path
11
+
12
+ from dotenv import load_dotenv
13
+
14
+ # Load .env from project root (if exists), does not override existing env vars
15
+ load_dotenv(Path(__file__).resolve().parent.parent.parent / ".env")
16
+
17
+
18
+ def _env_int(name: str, default: int) -> int:
19
+ """Read an integer from HTMLREADER_<name>, fallback to default."""
20
+ val = os.environ.get(f"HTMLREADER_{name}")
21
+ if val is None:
22
+ return default
23
+ try:
24
+ return int(val)
25
+ except ValueError:
26
+ return default
27
+
28
+
29
+ def _env_list(name: str, default: list[str]) -> list[str]:
30
+ """Read a comma-separated list from HTMLREADER_<name>, fallback to default."""
31
+ val = os.environ.get(f"HTMLREADER_{name}")
32
+ if not val:
33
+ return default
34
+ return [item.strip() for item in val.split(",") if item.strip()]
35
+
36
+
37
+ # --- Text truncation (units: Chinese chars / English words) ---
38
+ TEXT_HEAD = _env_int("TEXT_HEAD", 10)
39
+ TEXT_TAIL = _env_int("TEXT_TAIL", 5)
40
+
41
+ # --- Script inline code truncation ---
42
+ SCRIPT_HEAD = _env_int("SCRIPT_HEAD", 5)
43
+ SCRIPT_TAIL = _env_int("SCRIPT_TAIL", 3)
44
+
45
+ # --- Image alt truncation (chars) ---
46
+ MEDIA_ALT_HEAD = _env_int("MEDIA_ALT_HEAD", 10)
47
+
48
+ # --- Attribute mode: "blacklist" or "whitelist" ---
49
+ ATTR_MODE = os.environ.get("HTMLREADER_ATTR_MODE", "blacklist")
50
+
51
+ # --- Attribute lists (comma-separated) ---
52
+ KEEP_ATTRS: list[str] = _env_list(
53
+ "KEEP_ATTRS", ["id", "class", "href", "src", "alt", "title", "role"]
54
+ )
55
+ REMOVE_ATTRS: list[str] = _env_list("REMOVE_ATTRS", ["style"])
56
+
57
+ # --- CSS selectors for node removal / retention ---
58
+ REMOVE_SELECTORS: list[str] = _env_list("REMOVE_SELECTORS", [])
59
+ KEEP_SELECTORS: list[str] = _env_list("KEEP_SELECTORS", [])
60
+
61
+ # --- data-* attributes: always keep (0=off, 1=on) ---
62
+ KEEP_DATA_ATTRS = bool(_env_int("KEEP_DATA_ATTRS", 1))
63
+
64
+ # --- List truncation threshold ---
65
+ LIST_TRUNCATE_THRESHOLD = _env_int("LIST_TRUNCATE_THRESHOLD", 5)
66
+ LIST_TRUNCATE_KEEP_HEAD = _env_int("LIST_TRUNCATE_KEEP_HEAD", 3)
67
+ LIST_TRUNCATE_KEEP_TAIL = _env_int("LIST_TRUNCATE_KEEP_TAIL", 1)
68
+
69
+ # --- LLM configuration ---
70
+ LLM_BASE_URL = os.environ.get("HTMLREADER_LLM_BASE_URL", "")
71
+ LLM_API_KEY = os.environ.get("HTMLREADER_LLM_API_KEY", "")
72
+ LLM_MODEL = os.environ.get("HTMLREADER_LLM_MODEL", "")
73
+ LLM_MAX_TOKENS = _env_int("LLM_MAX_TOKENS", 512000)