html-reader-llm 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
html_reader_llm/cli.py ADDED
@@ -0,0 +1,563 @@
1
+ """html-reader-llm — HTML simplification and intelligent extraction for LLM.
2
+
3
+ Provides three subcommands:
4
+ - simplify: Clean and shorten HTML for LLM consumption
5
+ - detect: Compare HTTP vs Chrome rendering to decide if browser is needed
6
+ - analyze: Full pipeline — fetch, simplify, extract metadata, LLM classification + rules
7
+
8
+ Output:
9
+ - simplify: cleaned HTML to stdout, stats to stderr (--stats)
10
+ - detect: JSON summary to stdout, HTML bodies to files (--output-http/--output-chrome)
11
+ - analyze: JSON with page_type + rules to stdout, raw LLM response to stderr (--raw)
12
+
13
+ Logging goes to stderr. stdout is reserved for programmatic output (piping, agent use).
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import argparse
19
+ import json
20
+ import sys
21
+
22
+
23
+ def cmd_simplify(args: argparse.Namespace) -> None:
24
+ """Run HTML simplification pipeline."""
25
+ from html_reader_llm.simplify import simplify_html
26
+
27
+ if args.file:
28
+ with open(args.file, encoding="utf-8") as f:
29
+ html = f.read()
30
+ elif not sys.stdin.isatty():
31
+ html = sys.stdin.read()
32
+ else:
33
+ print("Error: provide HTML via --file or stdin", file=sys.stderr)
34
+ sys.exit(1)
35
+
36
+ result = simplify_html(html)
37
+
38
+ if args.output:
39
+ with open(args.output, "w", encoding="utf-8") as f:
40
+ f.write(result["html"])
41
+ print(f"Simplified HTML written to {args.output}", file=sys.stderr)
42
+ else:
43
+ print(result["html"])
44
+
45
+ if args.stats:
46
+ print("\n--- Stats ---", file=sys.stderr)
47
+ for step, s in result["stats"].items():
48
+ delta = s["char_before"] - s["char_after"]
49
+ pct = (delta / s["char_before"] * 100) if s["char_before"] else 0
50
+ print(
51
+ f" {step}: {s['char_before']} -> {s['char_after']} ({delta} chars, {pct:.1f}%)",
52
+ file=sys.stderr,
53
+ )
54
+
55
+
56
+ def cmd_detect(args: argparse.Namespace) -> None:
57
+ """Run render detection."""
58
+ from html_reader_llm.render_detect import detect_render
59
+
60
+ if args.list_browsers:
61
+ from html_reader_llm.browser import detect_browser_paths, get_browser_path
62
+
63
+ paths = detect_browser_paths()
64
+ auto = get_browser_path()
65
+ print(f"Detected browsers ({len(paths)}):", file=sys.stderr)
66
+ for i, p in enumerate(paths):
67
+ marker = " <-- auto-selected" if p == auto else ""
68
+ print(f" [{i}] {p}{marker}", file=sys.stderr)
69
+ if not paths:
70
+ print(" (none found)", file=sys.stderr)
71
+ return
72
+
73
+ result = detect_render(
74
+ url=args.url,
75
+ chrome_path=args.chrome_path,
76
+ user_data_dir=args.user_data_dir,
77
+ )
78
+
79
+ summary = {k: v for k, v in result.items() if k not in ("http_html", "chrome_html")}
80
+ print(json.dumps(summary, indent=2, ensure_ascii=False))
81
+
82
+ if args.output_http:
83
+ with open(args.output_http, "w", encoding="utf-8") as f:
84
+ f.write(result["http_html"] or "")
85
+ print(f"HTTP HTML written to {args.output_http}", file=sys.stderr)
86
+
87
+ if args.output_chrome:
88
+ with open(args.output_chrome, "w", encoding="utf-8") as f:
89
+ f.write(result["chrome_html"] or "")
90
+ print(f"Chrome HTML written to {args.output_chrome}", file=sys.stderr)
91
+
92
+
93
+ def cmd_analyze(args: argparse.Namespace) -> None:
94
+ """Run full analysis pipeline: fetch -> simplify -> metadata -> LLM."""
95
+ if args.dry_run:
96
+ cmd_analyze_dry_run(args)
97
+ return
98
+
99
+ from html_reader_llm.llm import analyze_page
100
+
101
+ result = analyze_page(
102
+ url=args.url,
103
+ base_url=args.base_url,
104
+ api_key=args.api_key,
105
+ model=args.model,
106
+ use_cache=not args.no_cache,
107
+ )
108
+
109
+ output = {k: v for k, v in result.items() if k != "raw_llm_response"}
110
+ print(json.dumps(output, indent=2, ensure_ascii=False))
111
+
112
+ if args.raw:
113
+ print("\n--- Raw LLM Response ---", file=sys.stderr)
114
+ print(result["raw_llm_response"], file=sys.stderr)
115
+
116
+
117
+ def cmd_analyze_dry_run(args: argparse.Namespace) -> None:
118
+ """Show the prompt that would be sent to LLM, without calling LLM."""
119
+ from html_reader_llm import settings
120
+ from html_reader_llm.llm import _extract_metadata, _trim_to_budget
121
+ from html_reader_llm.render_detect import _fetch_http, detect_render
122
+ from html_reader_llm.simplify import simplify_html
123
+
124
+ _max_tokens = args.max_tokens or settings.LLM_MAX_TOKENS
125
+
126
+ status, raw_html = _fetch_http(args.url, use_cache=not args.no_cache)
127
+ if not raw_html:
128
+ print(f"Error: failed to fetch {args.url}", file=sys.stderr)
129
+ sys.exit(1)
130
+
131
+ render_result = detect_render(args.url, use_cache=not args.no_cache)
132
+ use_browser = render_result["use_browser"]
133
+ chrome_html = render_result["chrome_html"]
134
+ llm_source = chrome_html if (use_browser and chrome_html) else raw_html
135
+
136
+ metadata = _extract_metadata(raw_html)
137
+ simplified = simplify_html(llm_source)["html"]
138
+ trimmed = _trim_to_budget(simplified, _max_tokens)
139
+
140
+ meta_str = (
141
+ json.dumps(metadata, ensure_ascii=False, indent=2) if metadata else "None"
142
+ )
143
+ user_prompt = f"Metadata (from trafilatura):\n{meta_str}\n\nHTML:\n{trimmed}"
144
+
145
+ print(f"=== Dry Run: {args.url} ===", file=sys.stderr)
146
+ print(
147
+ f"use_browser={use_browser}, source={'chrome' if llm_source is chrome_html else 'http'}",
148
+ file=sys.stderr,
149
+ )
150
+ print(
151
+ f"raw_html={len(raw_html)} chars, simplified={len(simplified)} chars",
152
+ file=sys.stderr,
153
+ )
154
+ print(
155
+ f"prompt={len(user_prompt)} chars, est_tokens={len(user_prompt) // 3}",
156
+ file=sys.stderr,
157
+ )
158
+ print(f"max_tokens={_max_tokens}, budget={_max_tokens // 2}", file=sys.stderr)
159
+ print(file=sys.stderr)
160
+ print(user_prompt)
161
+
162
+
163
+ def cmd_validate(args: argparse.Namespace) -> None:
164
+ """Validate CSS selectors against a URL or HTML file."""
165
+ import json as json_mod
166
+
167
+ from html_reader_llm.selector_validator import SelectorValidator
168
+
169
+ # Get HTML content
170
+ html = None
171
+ if args.html_file:
172
+ with open(args.html_file, encoding="utf-8") as f:
173
+ html = f.read()
174
+ print(f"Loaded HTML from file: {args.html_file}", file=sys.stderr)
175
+ elif args.url:
176
+ from html_reader_llm.render_detect import _fetch_chrome, _fetch_http
177
+
178
+ if args.chrome:
179
+ html = _fetch_chrome(args.url)
180
+ if not html:
181
+ print("Error: Chrome dump-dom failed", file=sys.stderr)
182
+ sys.exit(1)
183
+ print(f"Fetched via Chrome: {len(html)} chars", file=sys.stderr)
184
+ else:
185
+ status, html = _fetch_http(args.url, use_cache=not args.no_cache)
186
+ if not html:
187
+ print(f"Error: HTTP fetch failed (status={status})", file=sys.stderr)
188
+ sys.exit(1)
189
+ print(f"Fetched via HTTP: {len(html)} chars", file=sys.stderr)
190
+ else:
191
+ print("Error: provide URL or --html-file", file=sys.stderr)
192
+ sys.exit(1)
193
+
194
+ # Get rules
195
+ rules = None
196
+ if args.rules_file:
197
+ with open(args.rules_file, encoding="utf-8") as f:
198
+ rules = json_mod.load(f)
199
+ print(f"Loaded rules from: {args.rules_file}", file=sys.stderr)
200
+ else:
201
+ print("Error: provide --rules-file", file=sys.stderr)
202
+ sys.exit(1)
203
+
204
+ # Validate
205
+ validator = SelectorValidator(html)
206
+ all_results = {}
207
+
208
+ # Validate metadata
209
+ if "metadata" in rules:
210
+ results = validator.validate_rules(rules["metadata"])
211
+ all_results["metadata"] = results
212
+
213
+ # Validate question info
214
+ if "question" in rules:
215
+ results = validator.validate_rules(rules["question"])
216
+ all_results["question"] = results
217
+
218
+ # Validate answers
219
+ if "answers" in rules:
220
+ answer_rules = rules["answers"]
221
+ results = validator.validate_rules(
222
+ answer_rules["fields"], list_selector=answer_rules["list_selector"]
223
+ )
224
+ all_results["answers"] = results
225
+
226
+ # Validate page
227
+ if "page" in rules:
228
+ results = validator.validate_rules(rules["page"])
229
+ all_results["page"] = results
230
+
231
+ if args.json:
232
+ # JSON output
233
+ output = {
234
+ "valid": all(
235
+ r["valid"] for group in all_results.values() for r in group.values()
236
+ ),
237
+ "groups": all_results,
238
+ }
239
+ print(json_mod.dumps(output, indent=2, ensure_ascii=False))
240
+ else:
241
+ # Human-readable output
242
+ for group_name, results in all_results.items():
243
+ validator.print_validation_report(results, f"{group_name} 验证")
244
+
245
+ # Summary
246
+ total = sum(len(r) for r in all_results.values())
247
+ passed = sum(
248
+ 1 for group in all_results.values() for r in group.values() if r["valid"]
249
+ )
250
+ print("\n=== 总结 ===")
251
+ print(f"总选择器: {total}, 通过: {passed}, 失败: {total - passed}")
252
+
253
+
254
+ def cmd_fetch(args: argparse.Namespace) -> None:
255
+ """Fetch URL and output HTML to stdout."""
256
+ from html_reader_llm.render_detect import _fetch_chrome, _fetch_http, detect_render
257
+
258
+ use_chrome = args.chrome is not None # --chrome or --chrome=<path>
259
+ chrome_path = args.chrome if use_chrome and args.chrome != "auto" else None
260
+
261
+ if args.json or args.detect_only:
262
+ # JSON mode: run full detection and output structured result
263
+ result = detect_render(
264
+ args.url,
265
+ chrome_path=chrome_path,
266
+ use_cache=not args.no_cache,
267
+ )
268
+
269
+ output: dict[str, object] = {
270
+ "url": args.url,
271
+ "use_browser": result["use_browser"],
272
+ "http_status": result["http_status"],
273
+ "dom_ratio": result["dom_ratio"],
274
+ "text_similarity": result["text_similarity"],
275
+ "spa_shell": result["spa_shell"],
276
+ "confidence": result["confidence"],
277
+ }
278
+
279
+ if not args.detect_only:
280
+ # Include HTML based on use_browser decision
281
+ if result["use_browser"]:
282
+ output["html"] = result["chrome_html"] or ""
283
+ else:
284
+ output["html"] = result["http_html"] or ""
285
+ output["html_chars"] = len(output["html"])
286
+
287
+ # Include both raw HTMLs if --verbose
288
+ if args.verbose:
289
+ output["http_html"] = result["http_html"] or ""
290
+ output["chrome_html"] = result["chrome_html"] or ""
291
+
292
+ print(json.dumps(output, indent=2, ensure_ascii=False))
293
+ elif use_chrome:
294
+ # Chrome mode (explicit --chrome)
295
+ html = _fetch_chrome(args.url, chrome_path=chrome_path)
296
+ if not html:
297
+ print("Error: Chrome dump-dom failed", file=sys.stderr)
298
+ sys.exit(1)
299
+ print(html)
300
+ print(f"Fetched via Chrome: {len(html)} chars", file=sys.stderr)
301
+ else:
302
+ # Default HTTP mode (trafilatura)
303
+ status, html = _fetch_http(args.url, use_cache=not args.no_cache)
304
+ if not html:
305
+ print(f"Error: HTTP fetch failed (status={status})", file=sys.stderr)
306
+ sys.exit(1)
307
+ print(html)
308
+ print(f"Fetched via HTTP: {len(html)} chars", file=sys.stderr)
309
+
310
+
311
+ def main() -> None:
312
+ parser = argparse.ArgumentParser(
313
+ prog="html-reader-llm",
314
+ description="HTML simplification and intelligent extraction for LLM.",
315
+ epilog="stdout: programmatic output (HTML/JSON). stderr: logs/diagnostics.",
316
+ formatter_class=argparse.RawDescriptionHelpFormatter,
317
+ )
318
+ sub = parser.add_subparsers(dest="command")
319
+
320
+ # --- simplify ---
321
+ p_simplify = sub.add_parser(
322
+ "simplify",
323
+ help="Clean and shorten HTML for LLM consumption",
324
+ description=(
325
+ "7-step pipeline: strip noise -> media placeholders -> "
326
+ "clean attrs -> truncate lists -> truncate text -> "
327
+ "unwrap bare -> normalize whitespace."
328
+ ),
329
+ epilog="""
330
+ examples:
331
+ html-reader-llm simplify < page.html
332
+ cat page.html | html-reader-llm simplify --stats
333
+ html-reader-llm simplify --file page.html --output clean.html
334
+
335
+ input: HTML via stdin or --file
336
+ output: cleaned HTML to stdout
337
+ per-step char stats to stderr (--stats)
338
+ """,
339
+ formatter_class=argparse.RawDescriptionHelpFormatter,
340
+ )
341
+ p_simplify.add_argument("--file", "-f", help="Input HTML file (default: stdin)")
342
+ p_simplify.add_argument("--output", "-o", help="Output file (default: stdout)")
343
+ p_simplify.add_argument(
344
+ "--stats",
345
+ "-s",
346
+ action="store_true",
347
+ help="Print per-step character statistics to stderr",
348
+ )
349
+
350
+ # --- detect ---
351
+ p_detect = sub.add_parser(
352
+ "detect",
353
+ help="Detect if URL needs browser rendering",
354
+ description="Compare HTTP vs Chrome --dump-dom: DOM ratio, text Jaccard similarity, SPA shell detection.",
355
+ epilog="""
356
+ examples:
357
+ html-reader-llm detect https://example.com
358
+ html-reader-llm detect --list-browsers
359
+ html-reader-llm detect https://example.com --chrome-path "C:/chrome.exe"
360
+
361
+ input: URL (positional)
362
+ output: JSON to stdout:
363
+ {"http_status":200, "dom_ratio":0.983, "text_similarity":0.929,
364
+ "spa_shell":false, "confidence":0.967, "use_browser":false}
365
+ """,
366
+ formatter_class=argparse.RawDescriptionHelpFormatter,
367
+ )
368
+ p_detect.add_argument("url", nargs="?", help="URL to analyze")
369
+ p_detect.add_argument(
370
+ "--chrome-path", help="Chrome/Edge executable path (default: auto-detect)"
371
+ )
372
+ p_detect.add_argument(
373
+ "--user-data-dir", help="Chrome user-data-dir for caching (default: none)"
374
+ )
375
+ p_detect.add_argument(
376
+ "--list-browsers", action="store_true", help="List detected browsers and exit"
377
+ )
378
+ p_detect.add_argument("--output-http", help="Save HTTP-fetched HTML to file")
379
+ p_detect.add_argument("--output-chrome", help="Save Chrome-rendered HTML to file")
380
+
381
+ # --- analyze ---
382
+ p_analyze = sub.add_parser(
383
+ "analyze",
384
+ help="Full analysis: fetch, simplify, metadata, LLM classification + rules",
385
+ description=(
386
+ "End-to-end: fetch -> detect render -> simplify -> "
387
+ "trafilatura metadata -> LLM classification (list/detail) "
388
+ "+ CSS selector rule generation."
389
+ ),
390
+ epilog="""
391
+ requires: HTMLREADER_LLM_BASE_URL + HTMLREADER_LLM_API_KEY in .env or --base-url/--api-key
392
+
393
+ examples:
394
+ html-reader-llm analyze https://example.com
395
+ html-reader-llm analyze https://example.com --dry-run
396
+ html-reader-llm analyze https://example.com --raw
397
+
398
+ input: URL (positional)
399
+ output: JSON to stdout:
400
+ {"page_type":"detail",
401
+ "metadata":{"title":"...", "author":"...", "date":"..."},
402
+ "detail_rules":{
403
+ "title":{"css":"h1", "method":"$text"},
404
+ "content":{"css":".article-body", "method":"$text"}
405
+ }}
406
+ """,
407
+ formatter_class=argparse.RawDescriptionHelpFormatter,
408
+ )
409
+ p_analyze.add_argument("url", help="URL to analyze")
410
+ p_analyze.add_argument(
411
+ "--base-url", help="LLM API base URL (default: env HTMLREADER_LLM_BASE_URL)"
412
+ )
413
+ p_analyze.add_argument(
414
+ "--api-key", help="LLM API key (default: env HTMLREADER_LLM_API_KEY)"
415
+ )
416
+ p_analyze.add_argument(
417
+ "--model", help="LLM model name (default: env HTMLREADER_LLM_MODEL)"
418
+ )
419
+ p_analyze.add_argument(
420
+ "--max-tokens",
421
+ type=int,
422
+ help="Model context window (default: env HTMLREADER_LLM_MAX_TOKENS)",
423
+ )
424
+ p_analyze.add_argument(
425
+ "--no-cache", action="store_true", help="Bypass HTTP response cache"
426
+ )
427
+ p_analyze.add_argument(
428
+ "--dry-run",
429
+ action="store_true",
430
+ help="Show prompt to stdout without calling LLM",
431
+ )
432
+ p_analyze.add_argument(
433
+ "--raw", action="store_true", help="Print raw LLM response to stderr"
434
+ )
435
+
436
+ # --- fetch ---
437
+ p_fetch = sub.add_parser(
438
+ "fetch",
439
+ help="Download HTML from a URL",
440
+ description=(
441
+ "Fetch a URL and output raw HTML to stdout. "
442
+ "Default: HTTP via trafilatura. "
443
+ "Use --chrome for headless Chrome."
444
+ ),
445
+ epilog="""
446
+ examples:
447
+ html-reader-llm fetch https://example.com
448
+ html-reader-llm fetch https://example.com --chrome=auto
449
+ html-reader-llm fetch https://example.com --chrome="C:/chrome.exe"
450
+ html-reader-llm fetch https://example.com -H "Accept-Language: zh-CN"
451
+ html-reader-llm fetch https://example.com --timeout 30
452
+ html-reader-llm fetch https://example.com --json
453
+ html-reader-llm fetch https://example.com --json --verbose
454
+ html-reader-llm fetch https://example.com --json --detect-only
455
+
456
+ input: URL (positional)
457
+ output: raw HTML to stdout (default)
458
+ fetch method + char count to stderr
459
+ JSON with detection + optimal HTML to stdout (--json)
460
+ JSON with both http/chrome HTML to stdout (--json --verbose)
461
+ """,
462
+ formatter_class=argparse.RawDescriptionHelpFormatter,
463
+ )
464
+ p_fetch.add_argument("url", help="URL to fetch")
465
+ p_fetch.add_argument(
466
+ "-H",
467
+ "--header",
468
+ action="append",
469
+ dest="headers",
470
+ help="Custom header (repeatable), format: 'Name: Value'",
471
+ )
472
+ p_fetch.add_argument("--timeout", type=int, help="Request timeout in seconds")
473
+ p_fetch.add_argument(
474
+ "--chrome",
475
+ nargs="?",
476
+ const="auto",
477
+ default=None,
478
+ help="Use Chrome --dump-dom (default path or specify path)",
479
+ )
480
+ p_fetch.add_argument(
481
+ "--no-cache", action="store_true", help="Bypass HTTP response cache"
482
+ )
483
+ p_fetch.add_argument(
484
+ "--json",
485
+ action="store_true",
486
+ help="Output JSON with detection results and optimal HTML",
487
+ )
488
+ p_fetch.add_argument(
489
+ "--detect-only",
490
+ action="store_true",
491
+ help="Only detect if browser is needed, without outputting HTML (use with --json)",
492
+ )
493
+ p_fetch.add_argument(
494
+ "--verbose",
495
+ "-v",
496
+ action="store_true",
497
+ help="Include both http_html and chrome_html in JSON output",
498
+ )
499
+
500
+ # --- validate ---
501
+ p_validate = sub.add_parser(
502
+ "validate",
503
+ help="Validate CSS selectors against a URL or HTML file",
504
+ description=(
505
+ "Test if CSS selectors can extract target content from a page. "
506
+ "Supports QQ News Q&A page selectors out of the box."
507
+ ),
508
+ epilog="""
509
+ examples:
510
+ html-reader-llm validate https://example.com --rules-file selectors.json
511
+ html-reader-llm validate --html-file page.html --rules-file selectors.json
512
+ html-reader-llm validate https://example.com --chrome --rules-file selectors.json
513
+
514
+ input: URL (positional) or --html-file
515
+ output: validation report to stderr
516
+ JSON results to stdout (--json)
517
+ """,
518
+ formatter_class=argparse.RawDescriptionHelpFormatter,
519
+ )
520
+ p_validate.add_argument("url", nargs="?", help="URL to validate against")
521
+ p_validate.add_argument(
522
+ "--html-file", help="HTML file to validate against (instead of URL)"
523
+ )
524
+ p_validate.add_argument(
525
+ "--chrome",
526
+ action="store_true",
527
+ help="Use Chrome to fetch URL (for SPA pages)",
528
+ )
529
+ p_validate.add_argument(
530
+ "--rules-file",
531
+ required=True,
532
+ help="JSON file with selectors to validate",
533
+ )
534
+ p_validate.add_argument(
535
+ "--json",
536
+ action="store_true",
537
+ help="Output validation results as JSON",
538
+ )
539
+ p_validate.add_argument(
540
+ "--no-cache", action="store_true", help="Bypass HTTP response cache"
541
+ )
542
+
543
+ args = parser.parse_args()
544
+
545
+ if args.command == "simplify":
546
+ cmd_simplify(args)
547
+ elif args.command == "detect":
548
+ if not args.url and not args.list_browsers:
549
+ p_detect.error("url is required (or use --list-browsers)")
550
+ cmd_detect(args)
551
+ elif args.command == "analyze":
552
+ cmd_analyze(args)
553
+ elif args.command == "fetch":
554
+ cmd_fetch(args)
555
+ elif args.command == "validate":
556
+ cmd_validate(args)
557
+ else:
558
+ parser.print_help()
559
+ sys.exit(1)
560
+
561
+
562
+ if __name__ == "__main__":
563
+ main()