html-reader-llm 0.0.2__tar.gz → 0.0.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/AGENTS.md +39 -19
- html_reader_llm-0.0.3/PKG-INFO +140 -0
- html_reader_llm-0.0.3/README.md +125 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/pyproject.toml +2 -1
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/__init__.py +17 -1
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/cli.py +64 -33
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/llm.py +144 -40
- html_reader_llm-0.0.3/src/html_reader_llm/mcp_server.py +196 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/render_detect.py +53 -7
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/selector_validator.py +60 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/tests/test_browser.py +7 -7
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/tests/test_cli.py +4 -4
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/tests/test_llm.py +10 -11
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/tests/test_render_detect.py +15 -15
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/tests/test_simplify.py +7 -7
- html_reader_llm-0.0.3/uv.lock +1865 -0
- html_reader_llm-0.0.2/PKG-INFO +0 -52
- html_reader_llm-0.0.2/README.md +0 -39
- html_reader_llm-0.0.2/uv.lock +0 -581
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/.env.example +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/.gitignore +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/.python-version +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/LICENSE +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/chrome-path-detection/.openspec.yaml +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/chrome-path-detection/design.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/chrome-path-detection/proposal.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/chrome-path-detection/specs/chrome-path-detection/spec.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/chrome-path-detection/tasks.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/dotenv-config/.openspec.yaml +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/llm-analysis/.openspec.yaml +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/llm-analysis/design.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/llm-analysis/proposal.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/llm-analysis/specs/llm-analysis/spec.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/llm-analysis/tasks.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/render-detection/.openspec.yaml +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/render-detection/design.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/render-detection/proposal.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/render-detection/specs/render-detection/spec.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/render-detection/tasks.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/simplify-pipeline/.openspec.yaml +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/simplify-pipeline/design.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/simplify-pipeline/proposal.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/simplify-pipeline/specs/html-simplification/spec.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/changes/simplify-pipeline/tasks.md +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/openspec/config.yaml +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/browser.py +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/settings.py +0 -0
- {html_reader_llm-0.0.2 → html_reader_llm-0.0.3}/src/html_reader_llm/simplify.py +0 -0
|
@@ -2,16 +2,16 @@
|
|
|
2
2
|
|
|
3
3
|
## 基本信息
|
|
4
4
|
|
|
5
|
-
- **项目**:
|
|
5
|
+
- **项目**:html-reader-llm — 智能 HTML 解析与提取库
|
|
6
6
|
- **阶段**:Phase 1(核心功能开发)
|
|
7
|
-
- **技术栈**:Python 3.12+ / uv /
|
|
7
|
+
- **技术栈**:Python 3.12+ / uv / selectolax / trafilatura / fastmcp-slim
|
|
8
8
|
- **语言**:中文优先
|
|
9
9
|
|
|
10
10
|
## 常用命令
|
|
11
11
|
|
|
12
12
|
```bash
|
|
13
13
|
uv venv && uv sync # 安装依赖
|
|
14
|
-
uv run
|
|
14
|
+
uv run html-reader-llm # CLI
|
|
15
15
|
uv build # 构建
|
|
16
16
|
```
|
|
17
17
|
|
|
@@ -20,21 +20,23 @@ uv build # 构建
|
|
|
20
20
|
```
|
|
21
21
|
输入 URL/HTML
|
|
22
22
|
│
|
|
23
|
-
├─
|
|
23
|
+
├─ HTML 简化(7 步流水线)─→ LLM 友好格式
|
|
24
24
|
│
|
|
25
25
|
├─ 普通请求 vs 浏览器渲染 ─→ 差异度计算 + 文本统计
|
|
26
26
|
│
|
|
27
|
-
├─
|
|
28
|
-
│ └─
|
|
27
|
+
├─ urllib3 请求 ─→ LLM(页面分类 / CSS 选择器规则生成)
|
|
28
|
+
│ └─ trafilatura 辅助元数据提取
|
|
29
|
+
│
|
|
30
|
+
├─ MCP Server(--mcp)─→ stdio JSON-RPC ─→ 3 个工具
|
|
29
31
|
│
|
|
30
32
|
└─ 自定义 CSS 提取语法 ─→ 低代码提取工作流
|
|
31
33
|
```
|
|
32
34
|
|
|
33
|
-
##
|
|
35
|
+
## 设计决策
|
|
34
36
|
|
|
35
37
|
### HTML 简化
|
|
38
|
+
- 7 步流水线:strip noise → media placeholders → clean attrs → truncate lists → truncate text → unwrap bare → normalize whitespace
|
|
36
39
|
- minify-html 是喂给 LLM 之前的最后一步压缩
|
|
37
|
-
- 中间有自定义简化步骤,后续提供草稿代码 + 大量开源代码学习后确定
|
|
38
40
|
|
|
39
41
|
### 浏览器渲染判断
|
|
40
42
|
- 不用框架,用 Chrome 原生 `--dump-dom` 管道获取渲染后 HTML
|
|
@@ -42,23 +44,32 @@ uv build # 构建
|
|
|
42
44
|
- 关键参数:
|
|
43
45
|
- `--virtual-time-budget=` 预估等待 JS 异步加载
|
|
44
46
|
- `--disable-gpu`
|
|
45
|
-
- `--run-all-compositor-stages-before-draw`
|
|
46
|
-
- `--user-agent="指定的 UA"`
|
|
47
47
|
- `--headless`
|
|
48
48
|
- `--timeout=毫秒数`
|
|
49
49
|
- `--blink-settings=imagesEnabled=false` 禁用图片减少流量
|
|
50
50
|
- `--user-data-dir=` 缓存用户数据,减少重复加载
|
|
51
|
+
- `--proxy-server=` 代理
|
|
51
52
|
- 纯管道模式,不走 CDP 协议(后续可扩展)
|
|
52
|
-
- 渲染后 HTML 与
|
|
53
|
+
- 渲染后 HTML 与 HTTP 结果对比,统计文本差异度
|
|
54
|
+
|
|
55
|
+
### LLM 通信
|
|
56
|
+
- 使用 urllib3 发送 HTTP 请求与 LLM 通信(OpenAI 兼容 API)
|
|
57
|
+
- python-dotenv 加载 .env 配置
|
|
53
58
|
|
|
54
|
-
### CSS
|
|
59
|
+
### CSS 提取语法
|
|
55
60
|
- 三个核心字段:
|
|
56
61
|
- `selector`:CSS 选择器
|
|
57
62
|
- `value_mode`:取值方式
|
|
58
63
|
- `$html`:取元素内部完整 HTML
|
|
59
64
|
- `$text`:取纯文本(去标签)
|
|
60
65
|
- `@attribute`:取指定属性值(如 `@href`、`@src`)
|
|
61
|
-
-
|
|
66
|
+
- 交互形式:YAML / 链式 API / LLM 生成
|
|
67
|
+
|
|
68
|
+
### MCP Server
|
|
69
|
+
- 入口:`uvx html-reader-llm --mcp`
|
|
70
|
+
- 协议:MCP stdio(JSON-RPC 2.0)
|
|
71
|
+
- 工具只接收 HTML 字符串,不发起网络请求
|
|
72
|
+
- 调用方负责获取 HTML
|
|
62
73
|
|
|
63
74
|
## 核心需求
|
|
64
75
|
|
|
@@ -73,14 +84,20 @@ uv build # 构建
|
|
|
73
84
|
- 输出决策:`use_browser: bool`
|
|
74
85
|
|
|
75
86
|
### 3. LLM 驱动的智能分析
|
|
76
|
-
- 使用
|
|
87
|
+
- 使用 urllib3 发送 HTTP 请求与 LLM 通信
|
|
77
88
|
- **页面分类**:识别 `list`(列表页)/ `detail`(详情页)
|
|
78
89
|
- **CSS 选择器规则生成**:
|
|
79
|
-
- list
|
|
80
|
-
- detail 页:提取正文 +
|
|
81
|
-
- **
|
|
90
|
+
- list 页:提取列表规则(title, link, pubtime)
|
|
91
|
+
- detail 页:提取正文 + 元数据规则(title, author, date, description, image, categories, tags, content)
|
|
92
|
+
- **trafilatura 协作**:提取元数据作为 LLM 上下文,提高选择器准确度
|
|
93
|
+
|
|
94
|
+
### 4. MCP 工具集成
|
|
95
|
+
- `simplify_html`:7 步清洗 HTML
|
|
96
|
+
- `prepare_analysis`:组装 system_prompt + user_prompt(不调 LLM)
|
|
97
|
+
- `validate_selectors`:校验 CSS 选择器是否有效
|
|
98
|
+
- `extract_data`:用规则提取数据(detail 返回字段值,list 返回条目列表)
|
|
82
99
|
|
|
83
|
-
###
|
|
100
|
+
### 5. CSS 提取语法工作流
|
|
84
101
|
- 自定义一套文本化的 CSS 提取规则语法
|
|
85
102
|
- 低代码方式描述网页内容提取逻辑
|
|
86
103
|
- 支持组合、嵌套、条件选择
|
|
@@ -90,9 +107,11 @@ uv build # 构建
|
|
|
90
107
|
|
|
91
108
|
| 依赖 | 用途 |
|
|
92
109
|
|---|---|
|
|
93
|
-
| `httpx2` | 异步 HTTP 客户端,用于下载页面及 LLM API 通信 |
|
|
94
110
|
| `selectolax` | 高性能 HTML 解析(基于 Modest/Lexbor) |
|
|
111
|
+
| `trafilatura` | HTTP fetch + 元数据提取 |
|
|
95
112
|
| `minify-html` | HTML 极致压缩,缩短 LLM 提示词长度 |
|
|
113
|
+
| `python-dotenv` | 加载 .env 配置 |
|
|
114
|
+
| `fastmcp-slim[server]` | MCP stdio server |
|
|
96
115
|
|
|
97
116
|
## 关键概念
|
|
98
117
|
|
|
@@ -100,3 +119,4 @@ uv build # 构建
|
|
|
100
119
|
- **差异度**:普通请求与浏览器渲染结果的 DOM/文本相似度
|
|
101
120
|
- **提取规则**:文本化 CSS 选择器组合,描述从 HTML 中提取目标数据的逻辑
|
|
102
121
|
- **简化格式**:去除样式噪声后的 HTML,保留结构语义供 LLM 理解
|
|
122
|
+
- **MCP**:Model Context Protocol,LLM 工具集成协议
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: html-reader-llm
|
|
3
|
+
Version: 0.0.3
|
|
4
|
+
Summary: HTML simplification and intelligent extraction for LLM
|
|
5
|
+
Author-email: Clericpy <clericpy@gmail.com>
|
|
6
|
+
Requires-Python: >=3.12
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Requires-Dist: fastmcp-slim[server]>=3.4.7
|
|
10
|
+
Requires-Dist: minify-html>=0.18.1
|
|
11
|
+
Requires-Dist: python-dotenv>=1.2.2
|
|
12
|
+
Requires-Dist: selectolax>=0.4.7
|
|
13
|
+
Requires-Dist: trafilatura>=2.2.0
|
|
14
|
+
|
|
15
|
+
# html-reader-llm
|
|
16
|
+
|
|
17
|
+
HTML simplification and intelligent extraction for LLM.
|
|
18
|
+
|
|
19
|
+
## Features
|
|
20
|
+
|
|
21
|
+
- **HTML Simplification** — 7-step pipeline to clean HTML for LLM consumption
|
|
22
|
+
- **Render Detection** — Compare HTTP vs Chrome to decide if browser rendering is needed
|
|
23
|
+
- **LLM Analysis** — Page classification (list/detail) + CSS selector rule generation
|
|
24
|
+
- **CSS Selector Validation** — Verify selectors against actual HTML
|
|
25
|
+
- **MCP Server** — Expose tools to LLM agents via stdio protocol
|
|
26
|
+
|
|
27
|
+
## Installation
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
pip install html-reader-llm
|
|
31
|
+
# or
|
|
32
|
+
uv tool install html-reader-llm
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
## CLI Usage
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
# Simplify HTML
|
|
39
|
+
html-reader-llm simplify < page.html
|
|
40
|
+
html-reader-llm simplify --file page.html --stats
|
|
41
|
+
|
|
42
|
+
# Detect if browser rendering is needed
|
|
43
|
+
html-reader-llm detect https://example.com
|
|
44
|
+
html-reader-llm detect https://example.com --proxy http://127.0.0.1:7890
|
|
45
|
+
|
|
46
|
+
# Full analysis with LLM
|
|
47
|
+
html-reader-llm analyze https://example.com
|
|
48
|
+
html-reader-llm analyze https://example.com --dry-run
|
|
49
|
+
html-reader-llm analyze https://example.com --base-url https://api.openai.com/v1 --api-key sk-xxx
|
|
50
|
+
|
|
51
|
+
# Fetch HTML
|
|
52
|
+
html-reader-llm fetch https://example.com
|
|
53
|
+
html-reader-llm fetch https://example.com --chrome
|
|
54
|
+
html-reader-llm fetch https://example.com --json --detect-only
|
|
55
|
+
|
|
56
|
+
# Validate CSS selectors
|
|
57
|
+
html-reader-llm validate https://example.com --rules-file selectors.json
|
|
58
|
+
html-reader-llm validate --html-file page.html --rules-file selectors.json --json
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## MCP Mode
|
|
62
|
+
|
|
63
|
+
Expose tools to LLM agents via stdio protocol:
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
# Start MCP server
|
|
67
|
+
uvx html-reader-llm --mcp
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
### MCP Client Configuration
|
|
71
|
+
|
|
72
|
+
**Claude Desktop** (`claude_desktop_config.json`):
|
|
73
|
+
```json
|
|
74
|
+
{
|
|
75
|
+
"mcpServers": {
|
|
76
|
+
"html-reader-llm": {
|
|
77
|
+
"command": "uvx",
|
|
78
|
+
"args": ["html-reader-llm", "--mcp"]
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
**Cursor** (`.cursor/mcp.json`):
|
|
85
|
+
```json
|
|
86
|
+
{
|
|
87
|
+
"mcpServers": {
|
|
88
|
+
"html-reader-llm": {
|
|
89
|
+
"command": "uvx",
|
|
90
|
+
"args": ["html-reader-llm", "--mcp"]
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
### Available MCP Tools
|
|
97
|
+
|
|
98
|
+
| Tool | Input | Output |
|
|
99
|
+
|---|---|---|
|
|
100
|
+
| `simplify_html` | `html` | Simplified HTML + per-step stats |
|
|
101
|
+
| `prepare_analysis` | `html, max_tokens?` | System prompt + user prompt for LLM analysis |
|
|
102
|
+
| `validate_selectors` | `html, rules` | Per-selector validation results |
|
|
103
|
+
| `extract_data` | `html, rules` | Extracted data (detail) or list of items (list) |
|
|
104
|
+
|
|
105
|
+
All tools accept HTML strings directly — no network calls. The caller is responsible for fetching HTML.
|
|
106
|
+
|
|
107
|
+
### Agent Workflow
|
|
108
|
+
|
|
109
|
+
```
|
|
110
|
+
Agent fetches HTML (browser tool, curl, etc.)
|
|
111
|
+
│
|
|
112
|
+
├─ simplify_html(html)
|
|
113
|
+
│ → Cleaned HTML for analysis
|
|
114
|
+
│
|
|
115
|
+
├─ prepare_analysis(html)
|
|
116
|
+
│ → System prompt + user prompt
|
|
117
|
+
│ → Agent analyzes page, generates extraction rules
|
|
118
|
+
│
|
|
119
|
+
├─ validate_selectors(html, rules)
|
|
120
|
+
│ → Verify rules work against actual HTML
|
|
121
|
+
│
|
|
122
|
+
└─ extract_data(html, rules)
|
|
123
|
+
→ Get extracted content (detail values or list items)
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
## Proxy Support
|
|
127
|
+
|
|
128
|
+
All network-dependent commands support `--proxy`:
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
html-reader-llm fetch https://example.com --proxy http://127.0.0.1:7890
|
|
132
|
+
html-reader-llm detect https://example.com --proxy http://127.0.0.1:7890
|
|
133
|
+
html-reader-llm analyze https://example.com --proxy http://127.0.0.1:7890
|
|
134
|
+
html-reader-llm validate https://example.com --rules-file rules.json --proxy http://127.0.0.1:7890
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
## License
|
|
138
|
+
|
|
139
|
+
MIT
|
|
140
|
+
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
# html-reader-llm
|
|
2
|
+
|
|
3
|
+
HTML simplification and intelligent extraction for LLM.
|
|
4
|
+
|
|
5
|
+
## Features
|
|
6
|
+
|
|
7
|
+
- **HTML Simplification** — 7-step pipeline to clean HTML for LLM consumption
|
|
8
|
+
- **Render Detection** — Compare HTTP vs Chrome to decide if browser rendering is needed
|
|
9
|
+
- **LLM Analysis** — Page classification (list/detail) + CSS selector rule generation
|
|
10
|
+
- **CSS Selector Validation** — Verify selectors against actual HTML
|
|
11
|
+
- **MCP Server** — Expose tools to LLM agents via stdio protocol
|
|
12
|
+
|
|
13
|
+
## Installation
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
pip install html-reader-llm
|
|
17
|
+
# or
|
|
18
|
+
uv tool install html-reader-llm
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
## CLI Usage
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
# Simplify HTML
|
|
25
|
+
html-reader-llm simplify < page.html
|
|
26
|
+
html-reader-llm simplify --file page.html --stats
|
|
27
|
+
|
|
28
|
+
# Detect if browser rendering is needed
|
|
29
|
+
html-reader-llm detect https://example.com
|
|
30
|
+
html-reader-llm detect https://example.com --proxy http://127.0.0.1:7890
|
|
31
|
+
|
|
32
|
+
# Full analysis with LLM
|
|
33
|
+
html-reader-llm analyze https://example.com
|
|
34
|
+
html-reader-llm analyze https://example.com --dry-run
|
|
35
|
+
html-reader-llm analyze https://example.com --base-url https://api.openai.com/v1 --api-key sk-xxx
|
|
36
|
+
|
|
37
|
+
# Fetch HTML
|
|
38
|
+
html-reader-llm fetch https://example.com
|
|
39
|
+
html-reader-llm fetch https://example.com --chrome
|
|
40
|
+
html-reader-llm fetch https://example.com --json --detect-only
|
|
41
|
+
|
|
42
|
+
# Validate CSS selectors
|
|
43
|
+
html-reader-llm validate https://example.com --rules-file selectors.json
|
|
44
|
+
html-reader-llm validate --html-file page.html --rules-file selectors.json --json
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## MCP Mode
|
|
48
|
+
|
|
49
|
+
Expose tools to LLM agents via stdio protocol:
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
# Start MCP server
|
|
53
|
+
uvx html-reader-llm --mcp
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
### MCP Client Configuration
|
|
57
|
+
|
|
58
|
+
**Claude Desktop** (`claude_desktop_config.json`):
|
|
59
|
+
```json
|
|
60
|
+
{
|
|
61
|
+
"mcpServers": {
|
|
62
|
+
"html-reader-llm": {
|
|
63
|
+
"command": "uvx",
|
|
64
|
+
"args": ["html-reader-llm", "--mcp"]
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
**Cursor** (`.cursor/mcp.json`):
|
|
71
|
+
```json
|
|
72
|
+
{
|
|
73
|
+
"mcpServers": {
|
|
74
|
+
"html-reader-llm": {
|
|
75
|
+
"command": "uvx",
|
|
76
|
+
"args": ["html-reader-llm", "--mcp"]
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
### Available MCP Tools
|
|
83
|
+
|
|
84
|
+
| Tool | Input | Output |
|
|
85
|
+
|---|---|---|
|
|
86
|
+
| `simplify_html` | `html` | Simplified HTML + per-step stats |
|
|
87
|
+
| `prepare_analysis` | `html, max_tokens?` | System prompt + user prompt for LLM analysis |
|
|
88
|
+
| `validate_selectors` | `html, rules` | Per-selector validation results |
|
|
89
|
+
| `extract_data` | `html, rules` | Extracted data (detail) or list of items (list) |
|
|
90
|
+
|
|
91
|
+
All tools accept HTML strings directly — no network calls. The caller is responsible for fetching HTML.
|
|
92
|
+
|
|
93
|
+
### Agent Workflow
|
|
94
|
+
|
|
95
|
+
```
|
|
96
|
+
Agent fetches HTML (browser tool, curl, etc.)
|
|
97
|
+
│
|
|
98
|
+
├─ simplify_html(html)
|
|
99
|
+
│ → Cleaned HTML for analysis
|
|
100
|
+
│
|
|
101
|
+
├─ prepare_analysis(html)
|
|
102
|
+
│ → System prompt + user prompt
|
|
103
|
+
│ → Agent analyzes page, generates extraction rules
|
|
104
|
+
│
|
|
105
|
+
├─ validate_selectors(html, rules)
|
|
106
|
+
│ → Verify rules work against actual HTML
|
|
107
|
+
│
|
|
108
|
+
└─ extract_data(html, rules)
|
|
109
|
+
→ Get extracted content (detail values or list items)
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
## Proxy Support
|
|
113
|
+
|
|
114
|
+
All network-dependent commands support `--proxy`:
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
html-reader-llm fetch https://example.com --proxy http://127.0.0.1:7890
|
|
118
|
+
html-reader-llm detect https://example.com --proxy http://127.0.0.1:7890
|
|
119
|
+
html-reader-llm analyze https://example.com --proxy http://127.0.0.1:7890
|
|
120
|
+
html-reader-llm validate https://example.com --rules-file rules.json --proxy http://127.0.0.1:7890
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
## License
|
|
124
|
+
|
|
125
|
+
MIT
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "html-reader-llm"
|
|
3
|
-
version = "0.0.
|
|
3
|
+
version = "0.0.3"
|
|
4
4
|
description = "HTML simplification and intelligent extraction for LLM"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
authors = [
|
|
@@ -8,6 +8,7 @@ authors = [
|
|
|
8
8
|
]
|
|
9
9
|
requires-python = ">=3.12"
|
|
10
10
|
dependencies = [
|
|
11
|
+
"fastmcp-slim[server]>=3.4.7",
|
|
11
12
|
"minify-html>=0.18.1",
|
|
12
13
|
"python-dotenv>=1.2.2",
|
|
13
14
|
"selectolax>=0.4.7",
|
|
@@ -19,7 +19,13 @@ from html_reader_llm.browser import (
|
|
|
19
19
|
get_browser_path,
|
|
20
20
|
validate_browser_path,
|
|
21
21
|
)
|
|
22
|
-
from html_reader_llm.llm import
|
|
22
|
+
from html_reader_llm.llm import (
|
|
23
|
+
AnalysisResult,
|
|
24
|
+
PreparedPrompt,
|
|
25
|
+
analyze_page,
|
|
26
|
+
prepare_analysis_from_html,
|
|
27
|
+
prepare_analysis_prompt,
|
|
28
|
+
)
|
|
23
29
|
from html_reader_llm.render_detect import RenderDetectResult, detect_render
|
|
24
30
|
from html_reader_llm.selector_validator import SelectorValidator, validate_url_selectors
|
|
25
31
|
from html_reader_llm.simplify import simplify_html
|
|
@@ -27,13 +33,23 @@ from html_reader_llm.simplify import simplify_html
|
|
|
27
33
|
__all__ = [
|
|
28
34
|
"AnalysisResult",
|
|
29
35
|
"BrowserNotFoundError",
|
|
36
|
+
"PreparedPrompt",
|
|
30
37
|
"RenderDetectResult",
|
|
31
38
|
"SelectorValidator",
|
|
32
39
|
"analyze_page",
|
|
33
40
|
"detect_browser_paths",
|
|
34
41
|
"detect_render",
|
|
35
42
|
"get_browser_path",
|
|
43
|
+
"prepare_analysis_from_html",
|
|
44
|
+
"prepare_analysis_prompt",
|
|
36
45
|
"simplify_html",
|
|
37
46
|
"validate_browser_path",
|
|
38
47
|
"validate_url_selectors",
|
|
39
48
|
]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def run_mcp_server() -> None:
|
|
52
|
+
"""Start the MCP stdio server for LLM tool integration."""
|
|
53
|
+
from html_reader_llm.mcp_server import run_stdio
|
|
54
|
+
|
|
55
|
+
run_stdio()
|
|
@@ -1,9 +1,14 @@
|
|
|
1
1
|
"""html-reader-llm — HTML simplification and intelligent extraction for LLM.
|
|
2
2
|
|
|
3
|
-
Provides
|
|
3
|
+
Provides five subcommands:
|
|
4
4
|
- simplify: Clean and shorten HTML for LLM consumption
|
|
5
5
|
- detect: Compare HTTP vs Chrome rendering to decide if browser is needed
|
|
6
|
-
- analyze: Full pipeline — fetch, simplify,
|
|
6
|
+
- analyze: Full pipeline — fetch, simplify, metadata, LLM classification + rules
|
|
7
|
+
- fetch: Download HTML from a URL
|
|
8
|
+
- validate: Validate CSS selectors against a URL or HTML file
|
|
9
|
+
|
|
10
|
+
MCP mode:
|
|
11
|
+
- --mcp: Start MCP stdio server (for LLM tool integration via uvx html-reader-llm --mcp)
|
|
7
12
|
|
|
8
13
|
Output:
|
|
9
14
|
- simplify: cleaned HTML to stdout, stats to stderr (--stats)
|
|
@@ -74,6 +79,7 @@ def cmd_detect(args: argparse.Namespace) -> None:
|
|
|
74
79
|
url=args.url,
|
|
75
80
|
chrome_path=args.chrome_path,
|
|
76
81
|
user_data_dir=args.user_data_dir,
|
|
82
|
+
proxy=getattr(args, "proxy", None),
|
|
77
83
|
)
|
|
78
84
|
|
|
79
85
|
summary = {k: v for k, v in result.items() if k not in ("http_html", "chrome_html")}
|
|
@@ -104,6 +110,7 @@ def cmd_analyze(args: argparse.Namespace) -> None:
|
|
|
104
110
|
api_key=args.api_key,
|
|
105
111
|
model=args.model,
|
|
106
112
|
use_cache=not args.no_cache,
|
|
113
|
+
proxy=getattr(args, "proxy", None),
|
|
107
114
|
)
|
|
108
115
|
|
|
109
116
|
output = {k: v for k, v in result.items() if k != "raw_llm_response"}
|
|
@@ -116,48 +123,40 @@ def cmd_analyze(args: argparse.Namespace) -> None:
|
|
|
116
123
|
|
|
117
124
|
def cmd_analyze_dry_run(args: argparse.Namespace) -> None:
|
|
118
125
|
"""Show the prompt that would be sent to LLM, without calling LLM."""
|
|
119
|
-
from html_reader_llm import
|
|
120
|
-
from html_reader_llm.llm import _extract_metadata, _trim_to_budget
|
|
121
|
-
from html_reader_llm.render_detect import _fetch_http, detect_render
|
|
122
|
-
from html_reader_llm.simplify import simplify_html
|
|
126
|
+
from html_reader_llm.llm import prepare_analysis_prompt
|
|
123
127
|
|
|
124
|
-
|
|
128
|
+
proxy = getattr(args, "proxy", None)
|
|
125
129
|
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
130
|
+
try:
|
|
131
|
+
prepared = prepare_analysis_prompt(
|
|
132
|
+
url=args.url,
|
|
133
|
+
max_tokens=args.max_tokens,
|
|
134
|
+
use_cache=not args.no_cache,
|
|
135
|
+
proxy=proxy,
|
|
136
|
+
)
|
|
137
|
+
except ValueError as e:
|
|
138
|
+
print(f"Error: {e}", file=sys.stderr)
|
|
129
139
|
sys.exit(1)
|
|
130
140
|
|
|
131
|
-
render_result = detect_render(args.url, use_cache=not args.no_cache)
|
|
132
|
-
use_browser = render_result["use_browser"]
|
|
133
|
-
chrome_html = render_result["chrome_html"]
|
|
134
|
-
llm_source = chrome_html if (use_browser and chrome_html) else raw_html
|
|
135
|
-
|
|
136
|
-
metadata = _extract_metadata(raw_html)
|
|
137
|
-
simplified = simplify_html(llm_source)["html"]
|
|
138
|
-
trimmed = _trim_to_budget(simplified, _max_tokens)
|
|
139
|
-
|
|
140
|
-
meta_str = (
|
|
141
|
-
json.dumps(metadata, ensure_ascii=False, indent=2) if metadata else "None"
|
|
142
|
-
)
|
|
143
|
-
user_prompt = f"Metadata (from trafilatura):\n{meta_str}\n\nHTML:\n{trimmed}"
|
|
144
|
-
|
|
145
141
|
print(f"=== Dry Run: {args.url} ===", file=sys.stderr)
|
|
146
142
|
print(
|
|
147
|
-
f"use_browser={use_browser}, source={'chrome' if
|
|
143
|
+
f"use_browser={prepared['use_browser']}, source={'chrome' if prepared['use_browser'] else 'http'}",
|
|
144
|
+
file=sys.stderr,
|
|
145
|
+
)
|
|
146
|
+
print(
|
|
147
|
+
f"raw_html={len(prepared['raw_html'])} chars",
|
|
148
148
|
file=sys.stderr,
|
|
149
149
|
)
|
|
150
150
|
print(
|
|
151
|
-
f"
|
|
151
|
+
f"prompt={len(prepared['user_prompt'])} chars, est_tokens={len(prepared['user_prompt']) // 3}",
|
|
152
152
|
file=sys.stderr,
|
|
153
153
|
)
|
|
154
154
|
print(
|
|
155
|
-
f"
|
|
155
|
+
f"max_tokens={prepared['max_tokens']}, budget={prepared['max_tokens'] // 2}",
|
|
156
156
|
file=sys.stderr,
|
|
157
157
|
)
|
|
158
|
-
print(f"max_tokens={_max_tokens}, budget={_max_tokens // 2}", file=sys.stderr)
|
|
159
158
|
print(file=sys.stderr)
|
|
160
|
-
print(user_prompt)
|
|
159
|
+
print(prepared["user_prompt"])
|
|
161
160
|
|
|
162
161
|
|
|
163
162
|
def cmd_validate(args: argparse.Namespace) -> None:
|
|
@@ -175,14 +174,17 @@ def cmd_validate(args: argparse.Namespace) -> None:
|
|
|
175
174
|
elif args.url:
|
|
176
175
|
from html_reader_llm.render_detect import _fetch_chrome, _fetch_http
|
|
177
176
|
|
|
177
|
+
proxy = getattr(args, "proxy", None)
|
|
178
178
|
if args.chrome:
|
|
179
|
-
html = _fetch_chrome(args.url)
|
|
179
|
+
html = _fetch_chrome(args.url, proxy=proxy)
|
|
180
180
|
if not html:
|
|
181
181
|
print("Error: Chrome dump-dom failed", file=sys.stderr)
|
|
182
182
|
sys.exit(1)
|
|
183
183
|
print(f"Fetched via Chrome: {len(html)} chars", file=sys.stderr)
|
|
184
184
|
else:
|
|
185
|
-
status, html = _fetch_http(
|
|
185
|
+
status, html = _fetch_http(
|
|
186
|
+
args.url, use_cache=not args.no_cache, proxy=proxy
|
|
187
|
+
)
|
|
186
188
|
if not html:
|
|
187
189
|
print(f"Error: HTTP fetch failed (status={status})", file=sys.stderr)
|
|
188
190
|
sys.exit(1)
|
|
@@ -257,6 +259,7 @@ def cmd_fetch(args: argparse.Namespace) -> None:
|
|
|
257
259
|
|
|
258
260
|
use_chrome = args.chrome is not None # --chrome or --chrome=<path>
|
|
259
261
|
chrome_path = args.chrome if use_chrome and args.chrome != "auto" else None
|
|
262
|
+
proxy = getattr(args, "proxy", None)
|
|
260
263
|
|
|
261
264
|
if args.json or args.detect_only:
|
|
262
265
|
# JSON mode: run full detection and output structured result
|
|
@@ -264,6 +267,7 @@ def cmd_fetch(args: argparse.Namespace) -> None:
|
|
|
264
267
|
args.url,
|
|
265
268
|
chrome_path=chrome_path,
|
|
266
269
|
use_cache=not args.no_cache,
|
|
270
|
+
proxy=proxy,
|
|
267
271
|
)
|
|
268
272
|
|
|
269
273
|
output: dict[str, object] = {
|
|
@@ -292,7 +296,7 @@ def cmd_fetch(args: argparse.Namespace) -> None:
|
|
|
292
296
|
print(json.dumps(output, indent=2, ensure_ascii=False))
|
|
293
297
|
elif use_chrome:
|
|
294
298
|
# Chrome mode (explicit --chrome)
|
|
295
|
-
html = _fetch_chrome(args.url, chrome_path=chrome_path)
|
|
299
|
+
html = _fetch_chrome(args.url, chrome_path=chrome_path, proxy=proxy)
|
|
296
300
|
if not html:
|
|
297
301
|
print("Error: Chrome dump-dom failed", file=sys.stderr)
|
|
298
302
|
sys.exit(1)
|
|
@@ -300,7 +304,7 @@ def cmd_fetch(args: argparse.Namespace) -> None:
|
|
|
300
304
|
print(f"Fetched via Chrome: {len(html)} chars", file=sys.stderr)
|
|
301
305
|
else:
|
|
302
306
|
# Default HTTP mode (trafilatura)
|
|
303
|
-
status, html = _fetch_http(args.url, use_cache=not args.no_cache)
|
|
307
|
+
status, html = _fetch_http(args.url, use_cache=not args.no_cache, proxy=proxy)
|
|
304
308
|
if not html:
|
|
305
309
|
print(f"Error: HTTP fetch failed (status={status})", file=sys.stderr)
|
|
306
310
|
sys.exit(1)
|
|
@@ -315,6 +319,11 @@ def main() -> None:
|
|
|
315
319
|
epilog="stdout: programmatic output (HTML/JSON). stderr: logs/diagnostics.",
|
|
316
320
|
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
317
321
|
)
|
|
322
|
+
parser.add_argument(
|
|
323
|
+
"--mcp",
|
|
324
|
+
action="store_true",
|
|
325
|
+
help="Start MCP stdio server (for LLM tool integration)",
|
|
326
|
+
)
|
|
318
327
|
sub = parser.add_subparsers(dest="command")
|
|
319
328
|
|
|
320
329
|
# --- simplify ---
|
|
@@ -377,6 +386,10 @@ output: JSON to stdout:
|
|
|
377
386
|
)
|
|
378
387
|
p_detect.add_argument("--output-http", help="Save HTTP-fetched HTML to file")
|
|
379
388
|
p_detect.add_argument("--output-chrome", help="Save Chrome-rendered HTML to file")
|
|
389
|
+
p_detect.add_argument(
|
|
390
|
+
"--proxy",
|
|
391
|
+
help="Proxy URL for HTTP/Chrome requests (e.g. 'http://127.0.0.1:7890')",
|
|
392
|
+
)
|
|
380
393
|
|
|
381
394
|
# --- analyze ---
|
|
382
395
|
p_analyze = sub.add_parser(
|
|
@@ -432,6 +445,10 @@ output: JSON to stdout:
|
|
|
432
445
|
p_analyze.add_argument(
|
|
433
446
|
"--raw", action="store_true", help="Print raw LLM response to stderr"
|
|
434
447
|
)
|
|
448
|
+
p_analyze.add_argument(
|
|
449
|
+
"--proxy",
|
|
450
|
+
help="Proxy URL for HTTP/Chrome requests (e.g. 'http://127.0.0.1:7890')",
|
|
451
|
+
)
|
|
435
452
|
|
|
436
453
|
# --- fetch ---
|
|
437
454
|
p_fetch = sub.add_parser(
|
|
@@ -470,6 +487,10 @@ output: raw HTML to stdout (default)
|
|
|
470
487
|
help="Custom header (repeatable), format: 'Name: Value'",
|
|
471
488
|
)
|
|
472
489
|
p_fetch.add_argument("--timeout", type=int, help="Request timeout in seconds")
|
|
490
|
+
p_fetch.add_argument(
|
|
491
|
+
"--proxy",
|
|
492
|
+
help="Proxy URL for HTTP/Chrome requests (e.g. 'http://127.0.0.1:7890')",
|
|
493
|
+
)
|
|
473
494
|
p_fetch.add_argument(
|
|
474
495
|
"--chrome",
|
|
475
496
|
nargs="?",
|
|
@@ -539,9 +560,19 @@ output: validation report to stderr
|
|
|
539
560
|
p_validate.add_argument(
|
|
540
561
|
"--no-cache", action="store_true", help="Bypass HTTP response cache"
|
|
541
562
|
)
|
|
563
|
+
p_validate.add_argument(
|
|
564
|
+
"--proxy",
|
|
565
|
+
help="Proxy URL for HTTP/Chrome requests (e.g. 'http://127.0.0.1:7890')",
|
|
566
|
+
)
|
|
542
567
|
|
|
543
568
|
args = parser.parse_args()
|
|
544
569
|
|
|
570
|
+
if args.mcp:
|
|
571
|
+
from html_reader_llm.mcp_server import run_stdio
|
|
572
|
+
|
|
573
|
+
run_stdio()
|
|
574
|
+
return
|
|
575
|
+
|
|
545
576
|
if args.command == "simplify":
|
|
546
577
|
cmd_simplify(args)
|
|
547
578
|
elif args.command == "detect":
|