html-reader-llm 0.0.2__tar.gz → 0.0.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/AGENTS.md +39 -19
  2. html_reader_llm-0.0.3/PKG-INFO +140 -0
  3. html_reader_llm-0.0.3/README.md +125 -0
  4. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/pyproject.toml +2 -1
  5. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/__init__.py +17 -1
  6. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/cli.py +64 -33
  7. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/llm.py +144 -40
  8. html_reader_llm-0.0.3/src/html_reader_llm/mcp_server.py +196 -0
  9. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/render_detect.py +53 -7
  10. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/selector_validator.py +60 -0
  11. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/tests/test_browser.py +7 -7
  12. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/tests/test_cli.py +4 -4
  13. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/tests/test_llm.py +10 -11
  14. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/tests/test_render_detect.py +15 -15
  15. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/tests/test_simplify.py +7 -7
  16. html_reader_llm-0.0.3/uv.lock +1865 -0
  17. html_reader_llm-0.0.2/PKG-INFO +0 -52
  18. html_reader_llm-0.0.2/README.md +0 -39
  19. html_reader_llm-0.0.2/uv.lock +0 -581
  20. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/.env.example +0 -0
  21. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/.gitignore +0 -0
  22. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/.python-version +0 -0
  23. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/LICENSE +0 -0
  24. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/chrome-path-detection/.openspec.yaml +0 -0
  25. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/chrome-path-detection/design.md +0 -0
  26. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/chrome-path-detection/proposal.md +0 -0
  27. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/chrome-path-detection/specs/chrome-path-detection/spec.md +0 -0
  28. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/chrome-path-detection/tasks.md +0 -0
  29. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/dotenv-config/.openspec.yaml +0 -0
  30. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/llm-analysis/.openspec.yaml +0 -0
  31. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/llm-analysis/design.md +0 -0
  32. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/llm-analysis/proposal.md +0 -0
  33. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/llm-analysis/specs/llm-analysis/spec.md +0 -0
  34. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/llm-analysis/tasks.md +0 -0
  35. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/render-detection/.openspec.yaml +0 -0
  36. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/render-detection/design.md +0 -0
  37. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/render-detection/proposal.md +0 -0
  38. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/render-detection/specs/render-detection/spec.md +0 -0
  39. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/render-detection/tasks.md +0 -0
  40. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/simplify-pipeline/.openspec.yaml +0 -0
  41. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/simplify-pipeline/design.md +0 -0
  42. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/simplify-pipeline/proposal.md +0 -0
  43. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/simplify-pipeline/specs/html-simplification/spec.md +0 -0
  44. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/simplify-pipeline/tasks.md +0 -0
  45. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/config.yaml +0 -0
  46. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/browser.py +0 -0
  47. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/settings.py +0 -0
  48. {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/simplify.py +0 -0
@@ -2,16 +2,16 @@
2
2
 
3
3
  ## 基本信息
4
4
 
5
- - **项目**:htmlreader — 智能 HTML 解析与提取库
5
+ - **项目**:html-reader-llm — 智能 HTML 解析与提取库
6
6
  - **阶段**:Phase 1(核心功能开发)
7
- - **技术栈**:Python 3.12+ / uv / httpx2 / selectolax
7
+ - **技术栈**:Python 3.12+ / uv / selectolax / trafilatura / fastmcp-slim
8
8
  - **语言**:中文优先
9
9
 
10
10
  ## 常用命令
11
11
 
12
12
  ```bash
13
13
  uv venv && uv sync # 安装依赖
14
- uv run htmlreader # CLI
14
+ uv run html-reader-llm # CLI
15
15
  uv build # 构建
16
16
  ```
17
17
 
@@ -20,21 +20,23 @@ uv build # 构建
20
20
  ```
21
21
  输入 URL/HTML
22
22
  │
23
- ├─ 自定义简化流程(后续参考草稿+开源代码)─→ minify-html ─→ LLM 分析结构
23
+ ├─ HTML 简化(7 步流水线)─→ LLM 友好格式
24
24
  │
25
25
  ├─ 普通请求 vs 浏览器渲染 ─→ 差异度计算 + 文本统计
26
26
  │
27
- ├─ httpx2 请求 ─→ LLM(页面分类 / CSS 选择器规则生成)
28
- │ └─ tranfilatura 辅助元数据提取
27
+ ├─ urllib3 请求 ─→ LLM(页面分类 / CSS 选择器规则生成)
28
+ │ └─ trafilatura 辅助元数据提取
29
+ │
30
+ ├─ MCP Server(--mcp)─→ stdio JSON-RPC ─→ 3 个工具
29
31
  │
30
32
  └─ 自定义 CSS 提取语法 ─→ 低代码提取工作流
31
33
  ```
32
34
 
33
- ## 设计决策(探索阶段)
35
+ ## 设计决策
34
36
 
35
37
  ### HTML 简化
38
+ - 7 步流水线:strip noise → media placeholders → clean attrs → truncate lists → truncate text → unwrap bare → normalize whitespace
36
39
  - minify-html 是喂给 LLM 之前的最后一步压缩
37
- - 中间有自定义简化步骤,后续提供草稿代码 + 大量开源代码学习后确定
38
40
 
39
41
  ### 浏览器渲染判断
40
42
  - 不用框架,用 Chrome 原生 `--dump-dom` 管道获取渲染后 HTML
@@ -42,23 +44,32 @@ uv build # 构建
42
44
  - 关键参数:
43
45
  - `--virtual-time-budget=` 预估等待 JS 异步加载
44
46
  - `--disable-gpu`
45
- - `--run-all-compositor-stages-before-draw`
46
- - `--user-agent="指定的 UA"`
47
47
  - `--headless`
48
48
  - `--timeout=毫秒数`
49
49
  - `--blink-settings=imagesEnabled=false` 禁用图片减少流量
50
50
  - `--user-data-dir=` 缓存用户数据,减少重复加载
51
+ - `--proxy-server=` 代理
51
52
  - 纯管道模式,不走 CDP 协议(后续可扩展)
52
- - 渲染后 HTML 与 httpx2 结果对比,统计文本差异度
53
+ - 渲染后 HTML 与 HTTP 结果对比,统计文本差异度
54
+
55
+ ### LLM 通信
56
+ - 使用 urllib3 发送 HTTP 请求与 LLM 通信(OpenAI 兼容 API)
57
+ - python-dotenv 加载 .env 配置
53
58
 
54
- ### CSS 提取语法(草稿)
59
+ ### CSS 提取语法
55
60
  - 三个核心字段:
56
61
  - `selector`:CSS 选择器
57
62
  - `value_mode`:取值方式
58
63
  - `$html`:取元素内部完整 HTML
59
64
  - `$text`:取纯文本(去标签)
60
65
  - `@attribute`:取指定属性值(如 `@href`、`@src`)
61
- - 交互形式(YAML / 链式 API / LLM 生成)后续再定
66
+ - 交互形式:YAML / 链式 API / LLM 生成
67
+
68
+ ### MCP Server
69
+ - 入口:`uvx html-reader-llm --mcp`
70
+ - 协议:MCP stdio(JSON-RPC 2.0)
71
+ - 工具只接收 HTML 字符串,不发起网络请求
72
+ - 调用方负责获取 HTML
62
73
 
63
74
  ## 核心需求
64
75
 
@@ -73,14 +84,20 @@ uv build # 构建
73
84
  - 输出决策:`use_browser: bool`
74
85
 
75
86
  ### 3. LLM 驱动的智能分析
76
- - 使用 httpx2 发送 HTTP 请求与 LLM 通信
87
+ - 使用 urllib3 发送 HTTP 请求与 LLM 通信
77
88
  - **页面分类**:识别 `list`(列表页)/ `detail`(详情页)
78
89
  - **CSS 选择器规则生成**:
79
- - list 页:提取列表规则
80
- - detail 页:提取正文 + 元数据规则
81
- - **tranfilatura 协作**:提取元数据作为 LLM 上下文,提高选择器准确度
90
+ - list 页:提取列表规则(title, link, pubtime)
91
+ - detail 页:提取正文 + 元数据规则(title, author, date, description, image, categories, tags, content)
92
+ - **trafilatura 协作**:提取元数据作为 LLM 上下文,提高选择器准确度
93
+
94
+ ### 4. MCP 工具集成
95
+ - `simplify_html`:7 步清洗 HTML
96
+ - `prepare_analysis`:组装 system_prompt + user_prompt(不调 LLM)
97
+ - `validate_selectors`:校验 CSS 选择器是否有效
98
+ - `extract_data`:用规则提取数据(detail 返回字段值,list 返回条目列表)
82
99
 
83
- ### 4. CSS 提取语法工作流
100
+ ### 5. CSS 提取语法工作流
84
101
  - 自定义一套文本化的 CSS 提取规则语法
85
102
  - 低代码方式描述网页内容提取逻辑
86
103
  - 支持组合、嵌套、条件选择
@@ -90,9 +107,11 @@ uv build # 构建
90
107
 
91
108
  | 依赖 | 用途 |
92
109
  |---|---|
93
- | `httpx2` | 异步 HTTP 客户端,用于下载页面及 LLM API 通信 |
94
110
  | `selectolax` | 高性能 HTML 解析(基于 Modest/Lexbor) |
111
+ | `trafilatura` | HTTP fetch + 元数据提取 |
95
112
  | `minify-html` | HTML 极致压缩,缩短 LLM 提示词长度 |
113
+ | `python-dotenv` | 加载 .env 配置 |
114
+ | `fastmcp-slim[server]` | MCP stdio server |
96
115
 
97
116
  ## 关键概念
98
117
 
@@ -100,3 +119,4 @@ uv build # 构建
100
119
  - **差异度**:普通请求与浏览器渲染结果的 DOM/文本相似度
101
120
  - **提取规则**:文本化 CSS 选择器组合,描述从 HTML 中提取目标数据的逻辑
102
121
  - **简化格式**:去除样式噪声后的 HTML,保留结构语义供 LLM 理解
122
+ - **MCP**:Model Context Protocol,LLM 工具集成协议
@@ -0,0 +1,140 @@
1
+ Metadata-Version: 2.4
2
+ Name: html-reader-llm
3
+ Version: 0.0.3
4
+ Summary: HTML simplification and intelligent extraction for LLM
5
+ Author-email: Clericpy <clericpy@gmail.com>
6
+ Requires-Python: >=3.12
7
+ Description-Content-Type: text/markdown
8
+ License-File: LICENSE
9
+ Requires-Dist: fastmcp-slim[server]>=3.4.7
10
+ Requires-Dist: minify-html>=0.18.1
11
+ Requires-Dist: python-dotenv>=1.2.2
12
+ Requires-Dist: selectolax>=0.4.7
13
+ Requires-Dist: trafilatura>=2.2.0
14
+
15
+ # html-reader-llm
16
+
17
+ HTML simplification and intelligent extraction for LLM.
18
+
19
+ ## Features
20
+
21
+ - **HTML Simplification** — 7-step pipeline to clean HTML for LLM consumption
22
+ - **Render Detection** — Compare HTTP vs Chrome to decide if browser rendering is needed
23
+ - **LLM Analysis** — Page classification (list/detail) + CSS selector rule generation
24
+ - **CSS Selector Validation** — Verify selectors against actual HTML
25
+ - **MCP Server** — Expose tools to LLM agents via stdio protocol
26
+
27
+ ## Installation
28
+
29
+ ```bash
30
+ pip install html-reader-llm
31
+ # or
32
+ uv tool install html-reader-llm
33
+ ```
34
+
35
+ ## CLI Usage
36
+
37
+ ```bash
38
+ # Simplify HTML
39
+ html-reader-llm simplify < page.html
40
+ html-reader-llm simplify --file page.html --stats
41
+
42
+ # Detect if browser rendering is needed
43
+ html-reader-llm detect https://example.com
44
+ html-reader-llm detect https://example.com --proxy http://127.0.0.1:7890
45
+
46
+ # Full analysis with LLM
47
+ html-reader-llm analyze https://example.com
48
+ html-reader-llm analyze https://example.com --dry-run
49
+ html-reader-llm analyze https://example.com --base-url https://api.openai.com/v1 --api-key sk-xxx
50
+
51
+ # Fetch HTML
52
+ html-reader-llm fetch https://example.com
53
+ html-reader-llm fetch https://example.com --chrome
54
+ html-reader-llm fetch https://example.com --json --detect-only
55
+
56
+ # Validate CSS selectors
57
+ html-reader-llm validate https://example.com --rules-file selectors.json
58
+ html-reader-llm validate --html-file page.html --rules-file selectors.json --json
59
+ ```
60
+
61
+ ## MCP Mode
62
+
63
+ Expose tools to LLM agents via stdio protocol:
64
+
65
+ ```bash
66
+ # Start MCP server
67
+ uvx html-reader-llm --mcp
68
+ ```
69
+
70
+ ### MCP Client Configuration
71
+
72
+ **Claude Desktop** (`claude_desktop_config.json`):
73
+ ```json
74
+ {
75
+ "mcpServers": {
76
+ "html-reader-llm": {
77
+ "command": "uvx",
78
+ "args": ["html-reader-llm", "--mcp"]
79
+ }
80
+ }
81
+ }
82
+ ```
83
+
84
+ **Cursor** (`.cursor/mcp.json`):
85
+ ```json
86
+ {
87
+ "mcpServers": {
88
+ "html-reader-llm": {
89
+ "command": "uvx",
90
+ "args": ["html-reader-llm", "--mcp"]
91
+ }
92
+ }
93
+ }
94
+ ```
95
+
96
+ ### Available MCP Tools
97
+
98
+ | Tool | Input | Output |
99
+ |---|---|---|
100
+ | `simplify_html` | `html` | Simplified HTML + per-step stats |
101
+ | `prepare_analysis` | `html, max_tokens?` | System prompt + user prompt for LLM analysis |
102
+ | `validate_selectors` | `html, rules` | Per-selector validation results |
103
+ | `extract_data` | `html, rules` | Extracted data (detail) or list of items (list) |
104
+
105
+ All tools accept HTML strings directly — no network calls. The caller is responsible for fetching HTML.
106
+
107
+ ### Agent Workflow
108
+
109
+ ```
110
+ Agent fetches HTML (browser tool, curl, etc.)
111
+ │
112
+ ├─ simplify_html(html)
113
+ │ → Cleaned HTML for analysis
114
+ │
115
+ ├─ prepare_analysis(html)
116
+ │ → System prompt + user prompt
117
+ │ → Agent analyzes page, generates extraction rules
118
+ │
119
+ ├─ validate_selectors(html, rules)
120
+ │ → Verify rules work against actual HTML
121
+ │
122
+ └─ extract_data(html, rules)
123
+ → Get extracted content (detail values or list items)
124
+ ```
125
+
126
+ ## Proxy Support
127
+
128
+ All network-dependent commands support `--proxy`:
129
+
130
+ ```bash
131
+ html-reader-llm fetch https://example.com --proxy http://127.0.0.1:7890
132
+ html-reader-llm detect https://example.com --proxy http://127.0.0.1:7890
133
+ html-reader-llm analyze https://example.com --proxy http://127.0.0.1:7890
134
+ html-reader-llm validate https://example.com --rules-file rules.json --proxy http://127.0.0.1:7890
135
+ ```
136
+
137
+ ## License
138
+
139
+ MIT
140
+
@@ -0,0 +1,125 @@
1
+ # html-reader-llm
2
+
3
+ HTML simplification and intelligent extraction for LLM.
4
+
5
+ ## Features
6
+
7
+ - **HTML Simplification** — 7-step pipeline to clean HTML for LLM consumption
8
+ - **Render Detection** — Compare HTTP vs Chrome to decide if browser rendering is needed
9
+ - **LLM Analysis** — Page classification (list/detail) + CSS selector rule generation
10
+ - **CSS Selector Validation** — Verify selectors against actual HTML
11
+ - **MCP Server** — Expose tools to LLM agents via stdio protocol
12
+
13
+ ## Installation
14
+
15
+ ```bash
16
+ pip install html-reader-llm
17
+ # or
18
+ uv tool install html-reader-llm
19
+ ```
20
+
21
+ ## CLI Usage
22
+
23
+ ```bash
24
+ # Simplify HTML
25
+ html-reader-llm simplify < page.html
26
+ html-reader-llm simplify --file page.html --stats
27
+
28
+ # Detect if browser rendering is needed
29
+ html-reader-llm detect https://example.com
30
+ html-reader-llm detect https://example.com --proxy http://127.0.0.1:7890
31
+
32
+ # Full analysis with LLM
33
+ html-reader-llm analyze https://example.com
34
+ html-reader-llm analyze https://example.com --dry-run
35
+ html-reader-llm analyze https://example.com --base-url https://api.openai.com/v1 --api-key sk-xxx
36
+
37
+ # Fetch HTML
38
+ html-reader-llm fetch https://example.com
39
+ html-reader-llm fetch https://example.com --chrome
40
+ html-reader-llm fetch https://example.com --json --detect-only
41
+
42
+ # Validate CSS selectors
43
+ html-reader-llm validate https://example.com --rules-file selectors.json
44
+ html-reader-llm validate --html-file page.html --rules-file selectors.json --json
45
+ ```
46
+
47
+ ## MCP Mode
48
+
49
+ Expose tools to LLM agents via stdio protocol:
50
+
51
+ ```bash
52
+ # Start MCP server
53
+ uvx html-reader-llm --mcp
54
+ ```
55
+
56
+ ### MCP Client Configuration
57
+
58
+ **Claude Desktop** (`claude_desktop_config.json`):
59
+ ```json
60
+ {
61
+ "mcpServers": {
62
+ "html-reader-llm": {
63
+ "command": "uvx",
64
+ "args": ["html-reader-llm", "--mcp"]
65
+ }
66
+ }
67
+ }
68
+ ```
69
+
70
+ **Cursor** (`.cursor/mcp.json`):
71
+ ```json
72
+ {
73
+ "mcpServers": {
74
+ "html-reader-llm": {
75
+ "command": "uvx",
76
+ "args": ["html-reader-llm", "--mcp"]
77
+ }
78
+ }
79
+ }
80
+ ```
81
+
82
+ ### Available MCP Tools
83
+
84
+ | Tool | Input | Output |
85
+ |---|---|---|
86
+ | `simplify_html` | `html` | Simplified HTML + per-step stats |
87
+ | `prepare_analysis` | `html, max_tokens?` | System prompt + user prompt for LLM analysis |
88
+ | `validate_selectors` | `html, rules` | Per-selector validation results |
89
+ | `extract_data` | `html, rules` | Extracted data (detail) or list of items (list) |
90
+
91
+ All tools accept HTML strings directly — no network calls. The caller is responsible for fetching HTML.
92
+
93
+ ### Agent Workflow
94
+
95
+ ```
96
+ Agent fetches HTML (browser tool, curl, etc.)
97
+ │
98
+ ├─ simplify_html(html)
99
+ │ → Cleaned HTML for analysis
100
+ │
101
+ ├─ prepare_analysis(html)
102
+ │ → System prompt + user prompt
103
+ │ → Agent analyzes page, generates extraction rules
104
+ │
105
+ ├─ validate_selectors(html, rules)
106
+ │ → Verify rules work against actual HTML
107
+ │
108
+ └─ extract_data(html, rules)
109
+ → Get extracted content (detail values or list items)
110
+ ```
111
+
112
+ ## Proxy Support
113
+
114
+ All network-dependent commands support `--proxy`:
115
+
116
+ ```bash
117
+ html-reader-llm fetch https://example.com --proxy http://127.0.0.1:7890
118
+ html-reader-llm detect https://example.com --proxy http://127.0.0.1:7890
119
+ html-reader-llm analyze https://example.com --proxy http://127.0.0.1:7890
120
+ html-reader-llm validate https://example.com --rules-file rules.json --proxy http://127.0.0.1:7890
121
+ ```
122
+
123
+ ## License
124
+
125
+ MIT
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "html-reader-llm"
3
- version = "0.0.2"
3
+ version = "0.0.3"
4
4
  description = "HTML simplification and intelligent extraction for LLM"
5
5
  readme = "README.md"
6
6
  authors = [
@@ -8,6 +8,7 @@ authors = [
8
8
  ]
9
9
  requires-python = ">=3.12"
10
10
  dependencies = [
11
+ "fastmcp-slim[server]>=3.4.7",
11
12
  "minify-html>=0.18.1",
12
13
  "python-dotenv>=1.2.2",
13
14
  "selectolax>=0.4.7",
@@ -19,7 +19,13 @@ from html_reader_llm.browser import (
19
19
  get_browser_path,
20
20
  validate_browser_path,
21
21
  )
22
- from html_reader_llm.llm import AnalysisResult, analyze_page
22
+ from html_reader_llm.llm import (
23
+ AnalysisResult,
24
+ PreparedPrompt,
25
+ analyze_page,
26
+ prepare_analysis_from_html,
27
+ prepare_analysis_prompt,
28
+ )
23
29
  from html_reader_llm.render_detect import RenderDetectResult, detect_render
24
30
  from html_reader_llm.selector_validator import SelectorValidator, validate_url_selectors
25
31
  from html_reader_llm.simplify import simplify_html
@@ -27,13 +33,23 @@ from html_reader_llm.simplify import simplify_html
27
33
  __all__ = [
28
34
  "AnalysisResult",
29
35
  "BrowserNotFoundError",
36
+ "PreparedPrompt",
30
37
  "RenderDetectResult",
31
38
  "SelectorValidator",
32
39
  "analyze_page",
33
40
  "detect_browser_paths",
34
41
  "detect_render",
35
42
  "get_browser_path",
43
+ "prepare_analysis_from_html",
44
+ "prepare_analysis_prompt",
36
45
  "simplify_html",
37
46
  "validate_browser_path",
38
47
  "validate_url_selectors",
39
48
  ]
49
+
50
+
51
+ def run_mcp_server() -> None:
52
+ """Start the MCP stdio server for LLM tool integration."""
53
+ from html_reader_llm.mcp_server import run_stdio
54
+
55
+ run_stdio()
@@ -1,9 +1,14 @@
1
1
  """html-reader-llm — HTML simplification and intelligent extraction for LLM.
2
2
 
3
- Provides three subcommands:
3
+ Provides five subcommands:
4
4
  - simplify: Clean and shorten HTML for LLM consumption
5
5
  - detect: Compare HTTP vs Chrome rendering to decide if browser is needed
6
- - analyze: Full pipeline — fetch, simplify, extract metadata, LLM classification + rules
6
+ - analyze: Full pipeline — fetch, simplify, metadata, LLM classification + rules
7
+ - fetch: Download HTML from a URL
8
+ - validate: Validate CSS selectors against a URL or HTML file
9
+
10
+ MCP mode:
11
+ - --mcp: Start MCP stdio server (for LLM tool integration via uvx html-reader-llm --mcp)
7
12
 
8
13
  Output:
9
14
  - simplify: cleaned HTML to stdout, stats to stderr (--stats)
@@ -74,6 +79,7 @@ def cmd_detect(args: argparse.Namespace) -> None:
74
79
  url=args.url,
75
80
  chrome_path=args.chrome_path,
76
81
  user_data_dir=args.user_data_dir,
82
+ proxy=getattr(args, "proxy", None),
77
83
  )
78
84
 
79
85
  summary = {k: v for k, v in result.items() if k not in ("http_html", "chrome_html")}
@@ -104,6 +110,7 @@ def cmd_analyze(args: argparse.Namespace) -> None:
104
110
  api_key=args.api_key,
105
111
  model=args.model,
106
112
  use_cache=not args.no_cache,
113
+ proxy=getattr(args, "proxy", None),
107
114
  )
108
115
 
109
116
  output = {k: v for k, v in result.items() if k != "raw_llm_response"}
@@ -116,48 +123,40 @@ def cmd_analyze(args: argparse.Namespace) -> None:
116
123
 
117
124
  def cmd_analyze_dry_run(args: argparse.Namespace) -> None:
118
125
  """Show the prompt that would be sent to LLM, without calling LLM."""
119
- from html_reader_llm import settings
120
- from html_reader_llm.llm import _extract_metadata, _trim_to_budget
121
- from html_reader_llm.render_detect import _fetch_http, detect_render
122
- from html_reader_llm.simplify import simplify_html
126
+ from html_reader_llm.llm import prepare_analysis_prompt
123
127
 
124
- _max_tokens = args.max_tokens or settings.LLM_MAX_TOKENS
128
+ proxy = getattr(args, "proxy", None)
125
129
 
126
- status, raw_html = _fetch_http(args.url, use_cache=not args.no_cache)
127
- if not raw_html:
128
- print(f"Error: failed to fetch {args.url}", file=sys.stderr)
130
+ try:
131
+ prepared = prepare_analysis_prompt(
132
+ url=args.url,
133
+ max_tokens=args.max_tokens,
134
+ use_cache=not args.no_cache,
135
+ proxy=proxy,
136
+ )
137
+ except ValueError as e:
138
+ print(f"Error: {e}", file=sys.stderr)
129
139
  sys.exit(1)
130
140
 
131
- render_result = detect_render(args.url, use_cache=not args.no_cache)
132
- use_browser = render_result["use_browser"]
133
- chrome_html = render_result["chrome_html"]
134
- llm_source = chrome_html if (use_browser and chrome_html) else raw_html
135
-
136
- metadata = _extract_metadata(raw_html)
137
- simplified = simplify_html(llm_source)["html"]
138
- trimmed = _trim_to_budget(simplified, _max_tokens)
139
-
140
- meta_str = (
141
- json.dumps(metadata, ensure_ascii=False, indent=2) if metadata else "None"
142
- )
143
- user_prompt = f"Metadata (from trafilatura):\n{meta_str}\n\nHTML:\n{trimmed}"
144
-
145
141
  print(f"=== Dry Run: {args.url} ===", file=sys.stderr)
146
142
  print(
147
- f"use_browser={use_browser}, source={'chrome' if llm_source is chrome_html else 'http'}",
143
+ f"use_browser={prepared['use_browser']}, source={'chrome' if prepared['use_browser'] else 'http'}",
144
+ file=sys.stderr,
145
+ )
146
+ print(
147
+ f"raw_html={len(prepared['raw_html'])} chars",
148
148
  file=sys.stderr,
149
149
  )
150
150
  print(
151
- f"raw_html={len(raw_html)} chars, simplified={len(simplified)} chars",
151
+ f"prompt={len(prepared['user_prompt'])} chars, est_tokens={len(prepared['user_prompt']) // 3}",
152
152
  file=sys.stderr,
153
153
  )
154
154
  print(
155
- f"prompt={len(user_prompt)} chars, est_tokens={len(user_prompt) // 3}",
155
+ f"max_tokens={prepared['max_tokens']}, budget={prepared['max_tokens'] // 2}",
156
156
  file=sys.stderr,
157
157
  )
158
- print(f"max_tokens={_max_tokens}, budget={_max_tokens // 2}", file=sys.stderr)
159
158
  print(file=sys.stderr)
160
- print(user_prompt)
159
+ print(prepared["user_prompt"])
161
160
 
162
161
 
163
162
  def cmd_validate(args: argparse.Namespace) -> None:
@@ -175,14 +174,17 @@ def cmd_validate(args: argparse.Namespace) -> None:
175
174
  elif args.url:
176
175
  from html_reader_llm.render_detect import _fetch_chrome, _fetch_http
177
176
 
177
+ proxy = getattr(args, "proxy", None)
178
178
  if args.chrome:
179
- html = _fetch_chrome(args.url)
179
+ html = _fetch_chrome(args.url, proxy=proxy)
180
180
  if not html:
181
181
  print("Error: Chrome dump-dom failed", file=sys.stderr)
182
182
  sys.exit(1)
183
183
  print(f"Fetched via Chrome: {len(html)} chars", file=sys.stderr)
184
184
  else:
185
- status, html = _fetch_http(args.url, use_cache=not args.no_cache)
185
+ status, html = _fetch_http(
186
+ args.url, use_cache=not args.no_cache, proxy=proxy
187
+ )
186
188
  if not html:
187
189
  print(f"Error: HTTP fetch failed (status={status})", file=sys.stderr)
188
190
  sys.exit(1)
@@ -257,6 +259,7 @@ def cmd_fetch(args: argparse.Namespace) -> None:
257
259
 
258
260
  use_chrome = args.chrome is not None # --chrome or --chrome=<path>
259
261
  chrome_path = args.chrome if use_chrome and args.chrome != "auto" else None
262
+ proxy = getattr(args, "proxy", None)
260
263
 
261
264
  if args.json or args.detect_only:
262
265
  # JSON mode: run full detection and output structured result
@@ -264,6 +267,7 @@ def cmd_fetch(args: argparse.Namespace) -> None:
264
267
  args.url,
265
268
  chrome_path=chrome_path,
266
269
  use_cache=not args.no_cache,
270
+ proxy=proxy,
267
271
  )
268
272
 
269
273
  output: dict[str, object] = {
@@ -292,7 +296,7 @@ def cmd_fetch(args: argparse.Namespace) -> None:
292
296
  print(json.dumps(output, indent=2, ensure_ascii=False))
293
297
  elif use_chrome:
294
298
  # Chrome mode (explicit --chrome)
295
- html = _fetch_chrome(args.url, chrome_path=chrome_path)
299
+ html = _fetch_chrome(args.url, chrome_path=chrome_path, proxy=proxy)
296
300
  if not html:
297
301
  print("Error: Chrome dump-dom failed", file=sys.stderr)
298
302
  sys.exit(1)
@@ -300,7 +304,7 @@ def cmd_fetch(args: argparse.Namespace) -> None:
300
304
  print(f"Fetched via Chrome: {len(html)} chars", file=sys.stderr)
301
305
  else:
302
306
  # Default HTTP mode (trafilatura)
303
- status, html = _fetch_http(args.url, use_cache=not args.no_cache)
307
+ status, html = _fetch_http(args.url, use_cache=not args.no_cache, proxy=proxy)
304
308
  if not html:
305
309
  print(f"Error: HTTP fetch failed (status={status})", file=sys.stderr)
306
310
  sys.exit(1)
@@ -315,6 +319,11 @@ def main() -> None:
315
319
  epilog="stdout: programmatic output (HTML/JSON). stderr: logs/diagnostics.",
316
320
  formatter_class=argparse.RawDescriptionHelpFormatter,
317
321
  )
322
+ parser.add_argument(
323
+ "--mcp",
324
+ action="store_true",
325
+ help="Start MCP stdio server (for LLM tool integration)",
326
+ )
318
327
  sub = parser.add_subparsers(dest="command")
319
328
 
320
329
  # --- simplify ---
@@ -377,6 +386,10 @@ output: JSON to stdout:
377
386
  )
378
387
  p_detect.add_argument("--output-http", help="Save HTTP-fetched HTML to file")
379
388
  p_detect.add_argument("--output-chrome", help="Save Chrome-rendered HTML to file")
389
+ p_detect.add_argument(
390
+ "--proxy",
391
+ help="Proxy URL for HTTP/Chrome requests (e.g. 'http://127.0.0.1:7890')",
392
+ )
380
393
 
381
394
  # --- analyze ---
382
395
  p_analyze = sub.add_parser(
@@ -432,6 +445,10 @@ output: JSON to stdout:
432
445
  p_analyze.add_argument(
433
446
  "--raw", action="store_true", help="Print raw LLM response to stderr"
434
447
  )
448
+ p_analyze.add_argument(
449
+ "--proxy",
450
+ help="Proxy URL for HTTP/Chrome requests (e.g. 'http://127.0.0.1:7890')",
451
+ )
435
452
 
436
453
  # --- fetch ---
437
454
  p_fetch = sub.add_parser(
@@ -470,6 +487,10 @@ output: raw HTML to stdout (default)
470
487
  help="Custom header (repeatable), format: 'Name: Value'",
471
488
  )
472
489
  p_fetch.add_argument("--timeout", type=int, help="Request timeout in seconds")
490
+ p_fetch.add_argument(
491
+ "--proxy",
492
+ help="Proxy URL for HTTP/Chrome requests (e.g. 'http://127.0.0.1:7890')",
493
+ )
473
494
  p_fetch.add_argument(
474
495
  "--chrome",
475
496
  nargs="?",
@@ -539,9 +560,19 @@ output: validation report to stderr
539
560
  p_validate.add_argument(
540
561
  "--no-cache", action="store_true", help="Bypass HTTP response cache"
541
562
  )
563
+ p_validate.add_argument(
564
+ "--proxy",
565
+ help="Proxy URL for HTTP/Chrome requests (e.g. 'http://127.0.0.1:7890')",
566
+ )
542
567
 
543
568
  args = parser.parse_args()
544
569
 
570
+ if args.mcp:
571
+ from html_reader_llm.mcp_server import run_stdio
572
+
573
+ run_stdio()
574
+ return
575
+
545
576
  if args.command == "simplify":
546
577
  cmd_simplify(args)
547
578
  elif args.command == "detect":