html-reader-llm 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- html_reader_llm/__init__.py +39 -0
- html_reader_llm/browser.py +161 -0
- html_reader_llm/cli.py +563 -0
- html_reader_llm/llm.py +502 -0
- html_reader_llm/render_detect.py +415 -0
- html_reader_llm/selector_validator.py +177 -0
- html_reader_llm/settings.py +73 -0
- html_reader_llm/simplify.py +547 -0
- html_reader_llm-0.0.1.dist-info/METADATA +52 -0
- html_reader_llm-0.0.1.dist-info/RECORD +13 -0
- html_reader_llm-0.0.1.dist-info/WHEEL +4 -0
- html_reader_llm-0.0.1.dist-info/entry_points.txt +3 -0
- html_reader_llm-0.0.1.dist-info/licenses/LICENSE +21 -0
html_reader_llm/cli.py
ADDED
|
@@ -0,0 +1,563 @@
|
|
|
1
|
+
"""html-reader-llm — HTML simplification and intelligent extraction for LLM.
|
|
2
|
+
|
|
3
|
+
Provides three subcommands:
|
|
4
|
+
- simplify: Clean and shorten HTML for LLM consumption
|
|
5
|
+
- detect: Compare HTTP vs Chrome rendering to decide if browser is needed
|
|
6
|
+
- analyze: Full pipeline — fetch, simplify, extract metadata, LLM classification + rules
|
|
7
|
+
|
|
8
|
+
Output:
|
|
9
|
+
- simplify: cleaned HTML to stdout, stats to stderr (--stats)
|
|
10
|
+
- detect: JSON summary to stdout, HTML bodies to files (--output-http/--output-chrome)
|
|
11
|
+
- analyze: JSON with page_type + rules to stdout, raw LLM response to stderr (--raw)
|
|
12
|
+
|
|
13
|
+
Logging goes to stderr. stdout is reserved for programmatic output (piping, agent use).
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import argparse
|
|
19
|
+
import json
|
|
20
|
+
import sys
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def cmd_simplify(args: argparse.Namespace) -> None:
|
|
24
|
+
"""Run HTML simplification pipeline."""
|
|
25
|
+
from html_reader_llm.simplify import simplify_html
|
|
26
|
+
|
|
27
|
+
if args.file:
|
|
28
|
+
with open(args.file, encoding="utf-8") as f:
|
|
29
|
+
html = f.read()
|
|
30
|
+
elif not sys.stdin.isatty():
|
|
31
|
+
html = sys.stdin.read()
|
|
32
|
+
else:
|
|
33
|
+
print("Error: provide HTML via --file or stdin", file=sys.stderr)
|
|
34
|
+
sys.exit(1)
|
|
35
|
+
|
|
36
|
+
result = simplify_html(html)
|
|
37
|
+
|
|
38
|
+
if args.output:
|
|
39
|
+
with open(args.output, "w", encoding="utf-8") as f:
|
|
40
|
+
f.write(result["html"])
|
|
41
|
+
print(f"Simplified HTML written to {args.output}", file=sys.stderr)
|
|
42
|
+
else:
|
|
43
|
+
print(result["html"])
|
|
44
|
+
|
|
45
|
+
if args.stats:
|
|
46
|
+
print("\n--- Stats ---", file=sys.stderr)
|
|
47
|
+
for step, s in result["stats"].items():
|
|
48
|
+
delta = s["char_before"] - s["char_after"]
|
|
49
|
+
pct = (delta / s["char_before"] * 100) if s["char_before"] else 0
|
|
50
|
+
print(
|
|
51
|
+
f" {step}: {s['char_before']} -> {s['char_after']} ({delta} chars, {pct:.1f}%)",
|
|
52
|
+
file=sys.stderr,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def cmd_detect(args: argparse.Namespace) -> None:
|
|
57
|
+
"""Run render detection."""
|
|
58
|
+
from html_reader_llm.render_detect import detect_render
|
|
59
|
+
|
|
60
|
+
if args.list_browsers:
|
|
61
|
+
from html_reader_llm.browser import detect_browser_paths, get_browser_path
|
|
62
|
+
|
|
63
|
+
paths = detect_browser_paths()
|
|
64
|
+
auto = get_browser_path()
|
|
65
|
+
print(f"Detected browsers ({len(paths)}):", file=sys.stderr)
|
|
66
|
+
for i, p in enumerate(paths):
|
|
67
|
+
marker = " <-- auto-selected" if p == auto else ""
|
|
68
|
+
print(f" [{i}] {p}{marker}", file=sys.stderr)
|
|
69
|
+
if not paths:
|
|
70
|
+
print(" (none found)", file=sys.stderr)
|
|
71
|
+
return
|
|
72
|
+
|
|
73
|
+
result = detect_render(
|
|
74
|
+
url=args.url,
|
|
75
|
+
chrome_path=args.chrome_path,
|
|
76
|
+
user_data_dir=args.user_data_dir,
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
summary = {k: v for k, v in result.items() if k not in ("http_html", "chrome_html")}
|
|
80
|
+
print(json.dumps(summary, indent=2, ensure_ascii=False))
|
|
81
|
+
|
|
82
|
+
if args.output_http:
|
|
83
|
+
with open(args.output_http, "w", encoding="utf-8") as f:
|
|
84
|
+
f.write(result["http_html"] or "")
|
|
85
|
+
print(f"HTTP HTML written to {args.output_http}", file=sys.stderr)
|
|
86
|
+
|
|
87
|
+
if args.output_chrome:
|
|
88
|
+
with open(args.output_chrome, "w", encoding="utf-8") as f:
|
|
89
|
+
f.write(result["chrome_html"] or "")
|
|
90
|
+
print(f"Chrome HTML written to {args.output_chrome}", file=sys.stderr)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def cmd_analyze(args: argparse.Namespace) -> None:
|
|
94
|
+
"""Run full analysis pipeline: fetch -> simplify -> metadata -> LLM."""
|
|
95
|
+
if args.dry_run:
|
|
96
|
+
cmd_analyze_dry_run(args)
|
|
97
|
+
return
|
|
98
|
+
|
|
99
|
+
from html_reader_llm.llm import analyze_page
|
|
100
|
+
|
|
101
|
+
result = analyze_page(
|
|
102
|
+
url=args.url,
|
|
103
|
+
base_url=args.base_url,
|
|
104
|
+
api_key=args.api_key,
|
|
105
|
+
model=args.model,
|
|
106
|
+
use_cache=not args.no_cache,
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
output = {k: v for k, v in result.items() if k != "raw_llm_response"}
|
|
110
|
+
print(json.dumps(output, indent=2, ensure_ascii=False))
|
|
111
|
+
|
|
112
|
+
if args.raw:
|
|
113
|
+
print("\n--- Raw LLM Response ---", file=sys.stderr)
|
|
114
|
+
print(result["raw_llm_response"], file=sys.stderr)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def cmd_analyze_dry_run(args: argparse.Namespace) -> None:
|
|
118
|
+
"""Show the prompt that would be sent to LLM, without calling LLM."""
|
|
119
|
+
from html_reader_llm import settings
|
|
120
|
+
from html_reader_llm.llm import _extract_metadata, _trim_to_budget
|
|
121
|
+
from html_reader_llm.render_detect import _fetch_http, detect_render
|
|
122
|
+
from html_reader_llm.simplify import simplify_html
|
|
123
|
+
|
|
124
|
+
_max_tokens = args.max_tokens or settings.LLM_MAX_TOKENS
|
|
125
|
+
|
|
126
|
+
status, raw_html = _fetch_http(args.url, use_cache=not args.no_cache)
|
|
127
|
+
if not raw_html:
|
|
128
|
+
print(f"Error: failed to fetch {args.url}", file=sys.stderr)
|
|
129
|
+
sys.exit(1)
|
|
130
|
+
|
|
131
|
+
render_result = detect_render(args.url, use_cache=not args.no_cache)
|
|
132
|
+
use_browser = render_result["use_browser"]
|
|
133
|
+
chrome_html = render_result["chrome_html"]
|
|
134
|
+
llm_source = chrome_html if (use_browser and chrome_html) else raw_html
|
|
135
|
+
|
|
136
|
+
metadata = _extract_metadata(raw_html)
|
|
137
|
+
simplified = simplify_html(llm_source)["html"]
|
|
138
|
+
trimmed = _trim_to_budget(simplified, _max_tokens)
|
|
139
|
+
|
|
140
|
+
meta_str = (
|
|
141
|
+
json.dumps(metadata, ensure_ascii=False, indent=2) if metadata else "None"
|
|
142
|
+
)
|
|
143
|
+
user_prompt = f"Metadata (from trafilatura):\n{meta_str}\n\nHTML:\n{trimmed}"
|
|
144
|
+
|
|
145
|
+
print(f"=== Dry Run: {args.url} ===", file=sys.stderr)
|
|
146
|
+
print(
|
|
147
|
+
f"use_browser={use_browser}, source={'chrome' if llm_source is chrome_html else 'http'}",
|
|
148
|
+
file=sys.stderr,
|
|
149
|
+
)
|
|
150
|
+
print(
|
|
151
|
+
f"raw_html={len(raw_html)} chars, simplified={len(simplified)} chars",
|
|
152
|
+
file=sys.stderr,
|
|
153
|
+
)
|
|
154
|
+
print(
|
|
155
|
+
f"prompt={len(user_prompt)} chars, est_tokens={len(user_prompt) // 3}",
|
|
156
|
+
file=sys.stderr,
|
|
157
|
+
)
|
|
158
|
+
print(f"max_tokens={_max_tokens}, budget={_max_tokens // 2}", file=sys.stderr)
|
|
159
|
+
print(file=sys.stderr)
|
|
160
|
+
print(user_prompt)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def cmd_validate(args: argparse.Namespace) -> None:
|
|
164
|
+
"""Validate CSS selectors against a URL or HTML file."""
|
|
165
|
+
import json as json_mod
|
|
166
|
+
|
|
167
|
+
from html_reader_llm.selector_validator import SelectorValidator
|
|
168
|
+
|
|
169
|
+
# Get HTML content
|
|
170
|
+
html = None
|
|
171
|
+
if args.html_file:
|
|
172
|
+
with open(args.html_file, encoding="utf-8") as f:
|
|
173
|
+
html = f.read()
|
|
174
|
+
print(f"Loaded HTML from file: {args.html_file}", file=sys.stderr)
|
|
175
|
+
elif args.url:
|
|
176
|
+
from html_reader_llm.render_detect import _fetch_chrome, _fetch_http
|
|
177
|
+
|
|
178
|
+
if args.chrome:
|
|
179
|
+
html = _fetch_chrome(args.url)
|
|
180
|
+
if not html:
|
|
181
|
+
print("Error: Chrome dump-dom failed", file=sys.stderr)
|
|
182
|
+
sys.exit(1)
|
|
183
|
+
print(f"Fetched via Chrome: {len(html)} chars", file=sys.stderr)
|
|
184
|
+
else:
|
|
185
|
+
status, html = _fetch_http(args.url, use_cache=not args.no_cache)
|
|
186
|
+
if not html:
|
|
187
|
+
print(f"Error: HTTP fetch failed (status={status})", file=sys.stderr)
|
|
188
|
+
sys.exit(1)
|
|
189
|
+
print(f"Fetched via HTTP: {len(html)} chars", file=sys.stderr)
|
|
190
|
+
else:
|
|
191
|
+
print("Error: provide URL or --html-file", file=sys.stderr)
|
|
192
|
+
sys.exit(1)
|
|
193
|
+
|
|
194
|
+
# Get rules
|
|
195
|
+
rules = None
|
|
196
|
+
if args.rules_file:
|
|
197
|
+
with open(args.rules_file, encoding="utf-8") as f:
|
|
198
|
+
rules = json_mod.load(f)
|
|
199
|
+
print(f"Loaded rules from: {args.rules_file}", file=sys.stderr)
|
|
200
|
+
else:
|
|
201
|
+
print("Error: provide --rules-file", file=sys.stderr)
|
|
202
|
+
sys.exit(1)
|
|
203
|
+
|
|
204
|
+
# Validate
|
|
205
|
+
validator = SelectorValidator(html)
|
|
206
|
+
all_results = {}
|
|
207
|
+
|
|
208
|
+
# Validate metadata
|
|
209
|
+
if "metadata" in rules:
|
|
210
|
+
results = validator.validate_rules(rules["metadata"])
|
|
211
|
+
all_results["metadata"] = results
|
|
212
|
+
|
|
213
|
+
# Validate question info
|
|
214
|
+
if "question" in rules:
|
|
215
|
+
results = validator.validate_rules(rules["question"])
|
|
216
|
+
all_results["question"] = results
|
|
217
|
+
|
|
218
|
+
# Validate answers
|
|
219
|
+
if "answers" in rules:
|
|
220
|
+
answer_rules = rules["answers"]
|
|
221
|
+
results = validator.validate_rules(
|
|
222
|
+
answer_rules["fields"], list_selector=answer_rules["list_selector"]
|
|
223
|
+
)
|
|
224
|
+
all_results["answers"] = results
|
|
225
|
+
|
|
226
|
+
# Validate page
|
|
227
|
+
if "page" in rules:
|
|
228
|
+
results = validator.validate_rules(rules["page"])
|
|
229
|
+
all_results["page"] = results
|
|
230
|
+
|
|
231
|
+
if args.json:
|
|
232
|
+
# JSON output
|
|
233
|
+
output = {
|
|
234
|
+
"valid": all(
|
|
235
|
+
r["valid"] for group in all_results.values() for r in group.values()
|
|
236
|
+
),
|
|
237
|
+
"groups": all_results,
|
|
238
|
+
}
|
|
239
|
+
print(json_mod.dumps(output, indent=2, ensure_ascii=False))
|
|
240
|
+
else:
|
|
241
|
+
# Human-readable output
|
|
242
|
+
for group_name, results in all_results.items():
|
|
243
|
+
validator.print_validation_report(results, f"{group_name} 验证")
|
|
244
|
+
|
|
245
|
+
# Summary
|
|
246
|
+
total = sum(len(r) for r in all_results.values())
|
|
247
|
+
passed = sum(
|
|
248
|
+
1 for group in all_results.values() for r in group.values() if r["valid"]
|
|
249
|
+
)
|
|
250
|
+
print("\n=== 总结 ===")
|
|
251
|
+
print(f"总选择器: {total}, 通过: {passed}, 失败: {total - passed}")
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def cmd_fetch(args: argparse.Namespace) -> None:
|
|
255
|
+
"""Fetch URL and output HTML to stdout."""
|
|
256
|
+
from html_reader_llm.render_detect import _fetch_chrome, _fetch_http, detect_render
|
|
257
|
+
|
|
258
|
+
use_chrome = args.chrome is not None # --chrome or --chrome=<path>
|
|
259
|
+
chrome_path = args.chrome if use_chrome and args.chrome != "auto" else None
|
|
260
|
+
|
|
261
|
+
if args.json or args.detect_only:
|
|
262
|
+
# JSON mode: run full detection and output structured result
|
|
263
|
+
result = detect_render(
|
|
264
|
+
args.url,
|
|
265
|
+
chrome_path=chrome_path,
|
|
266
|
+
use_cache=not args.no_cache,
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
output: dict[str, object] = {
|
|
270
|
+
"url": args.url,
|
|
271
|
+
"use_browser": result["use_browser"],
|
|
272
|
+
"http_status": result["http_status"],
|
|
273
|
+
"dom_ratio": result["dom_ratio"],
|
|
274
|
+
"text_similarity": result["text_similarity"],
|
|
275
|
+
"spa_shell": result["spa_shell"],
|
|
276
|
+
"confidence": result["confidence"],
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
if not args.detect_only:
|
|
280
|
+
# Include HTML based on use_browser decision
|
|
281
|
+
if result["use_browser"]:
|
|
282
|
+
output["html"] = result["chrome_html"] or ""
|
|
283
|
+
else:
|
|
284
|
+
output["html"] = result["http_html"] or ""
|
|
285
|
+
output["html_chars"] = len(output["html"])
|
|
286
|
+
|
|
287
|
+
# Include both raw HTMLs if --verbose
|
|
288
|
+
if args.verbose:
|
|
289
|
+
output["http_html"] = result["http_html"] or ""
|
|
290
|
+
output["chrome_html"] = result["chrome_html"] or ""
|
|
291
|
+
|
|
292
|
+
print(json.dumps(output, indent=2, ensure_ascii=False))
|
|
293
|
+
elif use_chrome:
|
|
294
|
+
# Chrome mode (explicit --chrome)
|
|
295
|
+
html = _fetch_chrome(args.url, chrome_path=chrome_path)
|
|
296
|
+
if not html:
|
|
297
|
+
print("Error: Chrome dump-dom failed", file=sys.stderr)
|
|
298
|
+
sys.exit(1)
|
|
299
|
+
print(html)
|
|
300
|
+
print(f"Fetched via Chrome: {len(html)} chars", file=sys.stderr)
|
|
301
|
+
else:
|
|
302
|
+
# Default HTTP mode (trafilatura)
|
|
303
|
+
status, html = _fetch_http(args.url, use_cache=not args.no_cache)
|
|
304
|
+
if not html:
|
|
305
|
+
print(f"Error: HTTP fetch failed (status={status})", file=sys.stderr)
|
|
306
|
+
sys.exit(1)
|
|
307
|
+
print(html)
|
|
308
|
+
print(f"Fetched via HTTP: {len(html)} chars", file=sys.stderr)
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def main() -> None:
|
|
312
|
+
parser = argparse.ArgumentParser(
|
|
313
|
+
prog="html-reader-llm",
|
|
314
|
+
description="HTML simplification and intelligent extraction for LLM.",
|
|
315
|
+
epilog="stdout: programmatic output (HTML/JSON). stderr: logs/diagnostics.",
|
|
316
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
317
|
+
)
|
|
318
|
+
sub = parser.add_subparsers(dest="command")
|
|
319
|
+
|
|
320
|
+
# --- simplify ---
|
|
321
|
+
p_simplify = sub.add_parser(
|
|
322
|
+
"simplify",
|
|
323
|
+
help="Clean and shorten HTML for LLM consumption",
|
|
324
|
+
description=(
|
|
325
|
+
"7-step pipeline: strip noise -> media placeholders -> "
|
|
326
|
+
"clean attrs -> truncate lists -> truncate text -> "
|
|
327
|
+
"unwrap bare -> normalize whitespace."
|
|
328
|
+
),
|
|
329
|
+
epilog="""
|
|
330
|
+
examples:
|
|
331
|
+
html-reader-llm simplify < page.html
|
|
332
|
+
cat page.html | html-reader-llm simplify --stats
|
|
333
|
+
html-reader-llm simplify --file page.html --output clean.html
|
|
334
|
+
|
|
335
|
+
input: HTML via stdin or --file
|
|
336
|
+
output: cleaned HTML to stdout
|
|
337
|
+
per-step char stats to stderr (--stats)
|
|
338
|
+
""",
|
|
339
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
340
|
+
)
|
|
341
|
+
p_simplify.add_argument("--file", "-f", help="Input HTML file (default: stdin)")
|
|
342
|
+
p_simplify.add_argument("--output", "-o", help="Output file (default: stdout)")
|
|
343
|
+
p_simplify.add_argument(
|
|
344
|
+
"--stats",
|
|
345
|
+
"-s",
|
|
346
|
+
action="store_true",
|
|
347
|
+
help="Print per-step character statistics to stderr",
|
|
348
|
+
)
|
|
349
|
+
|
|
350
|
+
# --- detect ---
|
|
351
|
+
p_detect = sub.add_parser(
|
|
352
|
+
"detect",
|
|
353
|
+
help="Detect if URL needs browser rendering",
|
|
354
|
+
description="Compare HTTP vs Chrome --dump-dom: DOM ratio, text Jaccard similarity, SPA shell detection.",
|
|
355
|
+
epilog="""
|
|
356
|
+
examples:
|
|
357
|
+
html-reader-llm detect https://example.com
|
|
358
|
+
html-reader-llm detect --list-browsers
|
|
359
|
+
html-reader-llm detect https://example.com --chrome-path "C:/chrome.exe"
|
|
360
|
+
|
|
361
|
+
input: URL (positional)
|
|
362
|
+
output: JSON to stdout:
|
|
363
|
+
{"http_status":200, "dom_ratio":0.983, "text_similarity":0.929,
|
|
364
|
+
"spa_shell":false, "confidence":0.967, "use_browser":false}
|
|
365
|
+
""",
|
|
366
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
367
|
+
)
|
|
368
|
+
p_detect.add_argument("url", nargs="?", help="URL to analyze")
|
|
369
|
+
p_detect.add_argument(
|
|
370
|
+
"--chrome-path", help="Chrome/Edge executable path (default: auto-detect)"
|
|
371
|
+
)
|
|
372
|
+
p_detect.add_argument(
|
|
373
|
+
"--user-data-dir", help="Chrome user-data-dir for caching (default: none)"
|
|
374
|
+
)
|
|
375
|
+
p_detect.add_argument(
|
|
376
|
+
"--list-browsers", action="store_true", help="List detected browsers and exit"
|
|
377
|
+
)
|
|
378
|
+
p_detect.add_argument("--output-http", help="Save HTTP-fetched HTML to file")
|
|
379
|
+
p_detect.add_argument("--output-chrome", help="Save Chrome-rendered HTML to file")
|
|
380
|
+
|
|
381
|
+
# --- analyze ---
|
|
382
|
+
p_analyze = sub.add_parser(
|
|
383
|
+
"analyze",
|
|
384
|
+
help="Full analysis: fetch, simplify, metadata, LLM classification + rules",
|
|
385
|
+
description=(
|
|
386
|
+
"End-to-end: fetch -> detect render -> simplify -> "
|
|
387
|
+
"trafilatura metadata -> LLM classification (list/detail) "
|
|
388
|
+
"+ CSS selector rule generation."
|
|
389
|
+
),
|
|
390
|
+
epilog="""
|
|
391
|
+
requires: HTMLREADER_LLM_BASE_URL + HTMLREADER_LLM_API_KEY in .env or --base-url/--api-key
|
|
392
|
+
|
|
393
|
+
examples:
|
|
394
|
+
html-reader-llm analyze https://example.com
|
|
395
|
+
html-reader-llm analyze https://example.com --dry-run
|
|
396
|
+
html-reader-llm analyze https://example.com --raw
|
|
397
|
+
|
|
398
|
+
input: URL (positional)
|
|
399
|
+
output: JSON to stdout:
|
|
400
|
+
{"page_type":"detail",
|
|
401
|
+
"metadata":{"title":"...", "author":"...", "date":"..."},
|
|
402
|
+
"detail_rules":{
|
|
403
|
+
"title":{"css":"h1", "method":"$text"},
|
|
404
|
+
"content":{"css":".article-body", "method":"$text"}
|
|
405
|
+
}}
|
|
406
|
+
""",
|
|
407
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
408
|
+
)
|
|
409
|
+
p_analyze.add_argument("url", help="URL to analyze")
|
|
410
|
+
p_analyze.add_argument(
|
|
411
|
+
"--base-url", help="LLM API base URL (default: env HTMLREADER_LLM_BASE_URL)"
|
|
412
|
+
)
|
|
413
|
+
p_analyze.add_argument(
|
|
414
|
+
"--api-key", help="LLM API key (default: env HTMLREADER_LLM_API_KEY)"
|
|
415
|
+
)
|
|
416
|
+
p_analyze.add_argument(
|
|
417
|
+
"--model", help="LLM model name (default: env HTMLREADER_LLM_MODEL)"
|
|
418
|
+
)
|
|
419
|
+
p_analyze.add_argument(
|
|
420
|
+
"--max-tokens",
|
|
421
|
+
type=int,
|
|
422
|
+
help="Model context window (default: env HTMLREADER_LLM_MAX_TOKENS)",
|
|
423
|
+
)
|
|
424
|
+
p_analyze.add_argument(
|
|
425
|
+
"--no-cache", action="store_true", help="Bypass HTTP response cache"
|
|
426
|
+
)
|
|
427
|
+
p_analyze.add_argument(
|
|
428
|
+
"--dry-run",
|
|
429
|
+
action="store_true",
|
|
430
|
+
help="Show prompt to stdout without calling LLM",
|
|
431
|
+
)
|
|
432
|
+
p_analyze.add_argument(
|
|
433
|
+
"--raw", action="store_true", help="Print raw LLM response to stderr"
|
|
434
|
+
)
|
|
435
|
+
|
|
436
|
+
# --- fetch ---
|
|
437
|
+
p_fetch = sub.add_parser(
|
|
438
|
+
"fetch",
|
|
439
|
+
help="Download HTML from a URL",
|
|
440
|
+
description=(
|
|
441
|
+
"Fetch a URL and output raw HTML to stdout. "
|
|
442
|
+
"Default: HTTP via trafilatura. "
|
|
443
|
+
"Use --chrome for headless Chrome."
|
|
444
|
+
),
|
|
445
|
+
epilog="""
|
|
446
|
+
examples:
|
|
447
|
+
html-reader-llm fetch https://example.com
|
|
448
|
+
html-reader-llm fetch https://example.com --chrome=auto
|
|
449
|
+
html-reader-llm fetch https://example.com --chrome="C:/chrome.exe"
|
|
450
|
+
html-reader-llm fetch https://example.com -H "Accept-Language: zh-CN"
|
|
451
|
+
html-reader-llm fetch https://example.com --timeout 30
|
|
452
|
+
html-reader-llm fetch https://example.com --json
|
|
453
|
+
html-reader-llm fetch https://example.com --json --verbose
|
|
454
|
+
html-reader-llm fetch https://example.com --json --detect-only
|
|
455
|
+
|
|
456
|
+
input: URL (positional)
|
|
457
|
+
output: raw HTML to stdout (default)
|
|
458
|
+
fetch method + char count to stderr
|
|
459
|
+
JSON with detection + optimal HTML to stdout (--json)
|
|
460
|
+
JSON with both http/chrome HTML to stdout (--json --verbose)
|
|
461
|
+
""",
|
|
462
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
463
|
+
)
|
|
464
|
+
p_fetch.add_argument("url", help="URL to fetch")
|
|
465
|
+
p_fetch.add_argument(
|
|
466
|
+
"-H",
|
|
467
|
+
"--header",
|
|
468
|
+
action="append",
|
|
469
|
+
dest="headers",
|
|
470
|
+
help="Custom header (repeatable), format: 'Name: Value'",
|
|
471
|
+
)
|
|
472
|
+
p_fetch.add_argument("--timeout", type=int, help="Request timeout in seconds")
|
|
473
|
+
p_fetch.add_argument(
|
|
474
|
+
"--chrome",
|
|
475
|
+
nargs="?",
|
|
476
|
+
const="auto",
|
|
477
|
+
default=None,
|
|
478
|
+
help="Use Chrome --dump-dom (default path or specify path)",
|
|
479
|
+
)
|
|
480
|
+
p_fetch.add_argument(
|
|
481
|
+
"--no-cache", action="store_true", help="Bypass HTTP response cache"
|
|
482
|
+
)
|
|
483
|
+
p_fetch.add_argument(
|
|
484
|
+
"--json",
|
|
485
|
+
action="store_true",
|
|
486
|
+
help="Output JSON with detection results and optimal HTML",
|
|
487
|
+
)
|
|
488
|
+
p_fetch.add_argument(
|
|
489
|
+
"--detect-only",
|
|
490
|
+
action="store_true",
|
|
491
|
+
help="Only detect if browser is needed, without outputting HTML (use with --json)",
|
|
492
|
+
)
|
|
493
|
+
p_fetch.add_argument(
|
|
494
|
+
"--verbose",
|
|
495
|
+
"-v",
|
|
496
|
+
action="store_true",
|
|
497
|
+
help="Include both http_html and chrome_html in JSON output",
|
|
498
|
+
)
|
|
499
|
+
|
|
500
|
+
# --- validate ---
|
|
501
|
+
p_validate = sub.add_parser(
|
|
502
|
+
"validate",
|
|
503
|
+
help="Validate CSS selectors against a URL or HTML file",
|
|
504
|
+
description=(
|
|
505
|
+
"Test if CSS selectors can extract target content from a page. "
|
|
506
|
+
"Supports QQ News Q&A page selectors out of the box."
|
|
507
|
+
),
|
|
508
|
+
epilog="""
|
|
509
|
+
examples:
|
|
510
|
+
html-reader-llm validate https://example.com --rules-file selectors.json
|
|
511
|
+
html-reader-llm validate --html-file page.html --rules-file selectors.json
|
|
512
|
+
html-reader-llm validate https://example.com --chrome --rules-file selectors.json
|
|
513
|
+
|
|
514
|
+
input: URL (positional) or --html-file
|
|
515
|
+
output: validation report to stderr
|
|
516
|
+
JSON results to stdout (--json)
|
|
517
|
+
""",
|
|
518
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
519
|
+
)
|
|
520
|
+
p_validate.add_argument("url", nargs="?", help="URL to validate against")
|
|
521
|
+
p_validate.add_argument(
|
|
522
|
+
"--html-file", help="HTML file to validate against (instead of URL)"
|
|
523
|
+
)
|
|
524
|
+
p_validate.add_argument(
|
|
525
|
+
"--chrome",
|
|
526
|
+
action="store_true",
|
|
527
|
+
help="Use Chrome to fetch URL (for SPA pages)",
|
|
528
|
+
)
|
|
529
|
+
p_validate.add_argument(
|
|
530
|
+
"--rules-file",
|
|
531
|
+
required=True,
|
|
532
|
+
help="JSON file with selectors to validate",
|
|
533
|
+
)
|
|
534
|
+
p_validate.add_argument(
|
|
535
|
+
"--json",
|
|
536
|
+
action="store_true",
|
|
537
|
+
help="Output validation results as JSON",
|
|
538
|
+
)
|
|
539
|
+
p_validate.add_argument(
|
|
540
|
+
"--no-cache", action="store_true", help="Bypass HTTP response cache"
|
|
541
|
+
)
|
|
542
|
+
|
|
543
|
+
args = parser.parse_args()
|
|
544
|
+
|
|
545
|
+
if args.command == "simplify":
|
|
546
|
+
cmd_simplify(args)
|
|
547
|
+
elif args.command == "detect":
|
|
548
|
+
if not args.url and not args.list_browsers:
|
|
549
|
+
p_detect.error("url is required (or use --list-browsers)")
|
|
550
|
+
cmd_detect(args)
|
|
551
|
+
elif args.command == "analyze":
|
|
552
|
+
cmd_analyze(args)
|
|
553
|
+
elif args.command == "fetch":
|
|
554
|
+
cmd_fetch(args)
|
|
555
|
+
elif args.command == "validate":
|
|
556
|
+
cmd_validate(args)
|
|
557
|
+
else:
|
|
558
|
+
parser.print_help()
|
|
559
|
+
sys.exit(1)
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
if __name__ == "__main__":
|
|
563
|
+
main()
|