internal-web-reader 0.2.3__tar.gz → 1.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {internal_web_reader-0.2.3 → internal_web_reader-1.0.1}/PKG-INFO +10 -7
- internal_web_reader-1.0.1/README.md +27 -0
- {internal_web_reader-0.2.3 → internal_web_reader-1.0.1}/pyproject.toml +4 -3
- internal_web_reader-1.0.1/src/internal_web_reader/__main__.py +53 -0
- internal_web_reader-1.0.1/src/internal_web_reader/reader.py +334 -0
- {internal_web_reader-0.2.3 → internal_web_reader-1.0.1}/src/internal_web_reader.egg-info/PKG-INFO +10 -7
- {internal_web_reader-0.2.3 → internal_web_reader-1.0.1}/src/internal_web_reader.egg-info/requires.txt +2 -1
- internal_web_reader-0.2.3/README.md +0 -25
- internal_web_reader-0.2.3/src/internal_web_reader/__main__.py +0 -54
- internal_web_reader-0.2.3/src/internal_web_reader/reader.py +0 -182
- {internal_web_reader-0.2.3 → internal_web_reader-1.0.1}/setup.cfg +0 -0
- {internal_web_reader-0.2.3 → internal_web_reader-1.0.1}/src/internal_web_reader/__init__.py +0 -0
- {internal_web_reader-0.2.3 → internal_web_reader-1.0.1}/src/internal_web_reader.egg-info/SOURCES.txt +0 -0
- {internal_web_reader-0.2.3 → internal_web_reader-1.0.1}/src/internal_web_reader.egg-info/dependency_links.txt +0 -0
- {internal_web_reader-0.2.3 → internal_web_reader-1.0.1}/src/internal_web_reader.egg-info/entry_points.txt +0 -0
- {internal_web_reader-0.2.3 → internal_web_reader-1.0.1}/src/internal_web_reader.egg-info/top_level.txt +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: internal-web-reader
|
|
3
|
-
Version: 0.
|
|
4
|
-
Summary: MCP Server -
|
|
3
|
+
Version: 1.0.1
|
|
4
|
+
Summary: MCP Server - read internal web pages through your own Chrome browser
|
|
5
5
|
Author: zhanghaoran18006
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Keywords: mcp,claude,internal,wiki,confluence,gitlab
|
|
@@ -12,13 +12,16 @@ Classifier: Topic :: Software Development :: Libraries
|
|
|
12
12
|
Requires-Python: >=3.10
|
|
13
13
|
Description-Content-Type: text/markdown
|
|
14
14
|
Requires-Dist: mcp[cli]<2.0.0,>=1.2.0
|
|
15
|
-
Requires-Dist:
|
|
15
|
+
Requires-Dist: httpx>=0.27.0
|
|
16
|
+
Requires-Dist: websocket-client>=1.7.0
|
|
16
17
|
Requires-Dist: beautifulsoup4>=4.12.0
|
|
17
18
|
Requires-Dist: html2text>=2024.2.26
|
|
18
19
|
|
|
19
20
|
# Internal Web Reader
|
|
20
21
|
|
|
21
|
-
让 Claude Code
|
|
22
|
+
让 Claude Code 用你自己的 Chrome 浏览器读内网网页。
|
|
23
|
+
|
|
24
|
+
原理:通过 Chrome DevTools Protocol (CDP) 连接到你正在运行的 Chrome,复用已有的登录 session。不需要重新登录。
|
|
22
25
|
|
|
23
26
|
## 安装
|
|
24
27
|
|
|
@@ -34,10 +37,10 @@ claude mcp add -s user internal-web-reader -- internal-web-reader
|
|
|
34
37
|
|
|
35
38
|
## 使用
|
|
36
39
|
|
|
37
|
-
|
|
40
|
+
重启 Claude Code,然后给链接:
|
|
38
41
|
|
|
39
42
|
```
|
|
40
|
-
>
|
|
43
|
+
> 用 read_web_page 读 https://wiki.yourcompany.com/pages/xxx
|
|
41
44
|
```
|
|
42
45
|
|
|
43
|
-
|
|
46
|
+
首次使用时,如果 Chrome 没有启用调试端口,工具会自动重启 Chrome(你的标签页会自动恢复)。之后就直接用了。
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# Internal Web Reader
|
|
2
|
+
|
|
3
|
+
让 Claude Code 用你自己的 Chrome 浏览器读内网网页。
|
|
4
|
+
|
|
5
|
+
原理:通过 Chrome DevTools Protocol (CDP) 连接到你正在运行的 Chrome,复用已有的登录 session。不需要重新登录。
|
|
6
|
+
|
|
7
|
+
## 安装
|
|
8
|
+
|
|
9
|
+
```
|
|
10
|
+
pip install internal-web-reader
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## 配置 Claude Code
|
|
14
|
+
|
|
15
|
+
```
|
|
16
|
+
claude mcp add -s user internal-web-reader -- internal-web-reader
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
## 使用
|
|
20
|
+
|
|
21
|
+
重启 Claude Code,然后给链接:
|
|
22
|
+
|
|
23
|
+
```
|
|
24
|
+
> 用 read_web_page 读 https://wiki.yourcompany.com/pages/xxx
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
首次使用时,如果 Chrome 没有启用调试端口,工具会自动重启 Chrome(你的标签页会自动恢复)。之后就直接用了。
|
|
@@ -4,8 +4,8 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "internal-web-reader"
|
|
7
|
-
version = "0.
|
|
8
|
-
description = "MCP Server -
|
|
7
|
+
version = "1.0.1"
|
|
8
|
+
description = "MCP Server - read internal web pages through your own Chrome browser"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
11
|
license = "MIT"
|
|
@@ -19,7 +19,8 @@ classifiers = [
|
|
|
19
19
|
]
|
|
20
20
|
dependencies = [
|
|
21
21
|
"mcp[cli]>=1.2.0,<2.0.0",
|
|
22
|
-
"
|
|
22
|
+
"httpx>=0.27.0",
|
|
23
|
+
"websocket-client>=1.7.0",
|
|
23
24
|
"beautifulsoup4>=4.12.0",
|
|
24
25
|
"html2text>=2024.2.26",
|
|
25
26
|
]
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""MCP Server — 通过 CDP 连接用户自己的 Chrome 来读网页。"""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
|
|
5
|
+
from mcp.server.fastmcp import FastMCP
|
|
6
|
+
|
|
7
|
+
from .reader import read_page
|
|
8
|
+
|
|
9
|
+
logging.basicConfig(
|
|
10
|
+
level=logging.INFO,
|
|
11
|
+
format="[web-reader] %(message)s",
|
|
12
|
+
handlers=[logging.StreamHandler()],
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
mcp = FastMCP("internal-web-reader")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@mcp.tool()
|
|
19
|
+
def read_web_page(url: str) -> str:
|
|
20
|
+
"""[READ WEB PAGES] Read any URL using the user's own Chrome browser.
|
|
21
|
+
Reuses your existing Chrome login sessions — no re-login needed.
|
|
22
|
+
If Chrome is not running with debug mode, it will auto-restart Chrome
|
|
23
|
+
(your tabs will be restored). Use this for ALL web pages, especially
|
|
24
|
+
internal/intranet sites (GitLab, Wiki, docs requiring VPN/SSO).
|
|
25
|
+
"""
|
|
26
|
+
page = read_page(url)
|
|
27
|
+
|
|
28
|
+
lines = [f"# {page.title}", f"> {page.url}", "", page.content]
|
|
29
|
+
if 0 < len(page.links) <= 30:
|
|
30
|
+
lines += ["", "---", "## Links"]
|
|
31
|
+
for lk in page.links:
|
|
32
|
+
lines.append(f"- [{lk['text']}]({lk['href']})")
|
|
33
|
+
return "\n".join(lines)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@mcp.tool()
|
|
37
|
+
def list_web_links(url: str) -> str:
|
|
38
|
+
"""List all links on a web page."""
|
|
39
|
+
page = read_page(url)
|
|
40
|
+
if not page.links:
|
|
41
|
+
return f"No links found on {url}."
|
|
42
|
+
lines = [f"# {page.title}", f"> {len(page.links)} links", ""]
|
|
43
|
+
for lk in page.links:
|
|
44
|
+
lines.append(f"- [{lk['text']}]({lk['href']})")
|
|
45
|
+
return "\n".join(lines)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def main():
|
|
49
|
+
mcp.run()
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
if __name__ == "__main__":
|
|
53
|
+
main()
|
|
@@ -0,0 +1,334 @@
|
|
|
1
|
+
"""通过 CDP 连接用户自己的 Chrome,复用已有登录 session。
|
|
2
|
+
|
|
3
|
+
流程:
|
|
4
|
+
1. 检测 Chrome 是否在运行
|
|
5
|
+
2. 如果没启用调试端口 → 自动重启 Chrome 加上 --remote-debugging-port=0
|
|
6
|
+
3. 从 DevToolsActivePort 文件读取端口
|
|
7
|
+
4. 通过 CDP 协议连接,打开 URL,拿页面内容
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import logging
|
|
12
|
+
import os
|
|
13
|
+
import subprocess
|
|
14
|
+
import sys
|
|
15
|
+
import time
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from urllib.parse import urljoin
|
|
18
|
+
|
|
19
|
+
import httpx
|
|
20
|
+
from bs4 import BeautifulSoup
|
|
21
|
+
import html2text
|
|
22
|
+
|
|
23
|
+
log = logging.getLogger("internal_web_reader")
|
|
24
|
+
|
|
25
|
+
_H2T = html2text.HTML2Text()
|
|
26
|
+
_H2T.body_width = 0
|
|
27
|
+
_H2T.ignore_links = False
|
|
28
|
+
_H2T.ignore_images = False
|
|
29
|
+
_H2T.protect_links = True
|
|
30
|
+
_H2T.wrap_links = False
|
|
31
|
+
_H2T.unicode_snob = True
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class PageResult:
|
|
36
|
+
url: str
|
|
37
|
+
title: str
|
|
38
|
+
content: str
|
|
39
|
+
links: list[dict] = field(default_factory=list)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _stderr(msg: str):
|
|
43
|
+
sys.stderr.write(f"\n[web-reader] {msg}\n")
|
|
44
|
+
sys.stderr.flush()
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
# ── Chrome detection & launch ────────────────────────────
|
|
48
|
+
|
|
49
|
+
def _chrome_exe() -> str:
|
|
50
|
+
"""Find Chrome executable path."""
|
|
51
|
+
candidates = [
|
|
52
|
+
os.path.expandvars(r"%PROGRAMFILES%\Google\Chrome\Application\chrome.exe"),
|
|
53
|
+
os.path.expandvars(r"%PROGRAMFILES(X86)%\Google\Chrome\Application\chrome.exe"),
|
|
54
|
+
os.path.expandvars(r"%LOCALAPPDATA%\Google\Chrome\Application\chrome.exe"),
|
|
55
|
+
]
|
|
56
|
+
for p in candidates:
|
|
57
|
+
if os.path.isfile(p):
|
|
58
|
+
return p
|
|
59
|
+
# Fallback: hope it's on PATH
|
|
60
|
+
return "chrome"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _user_data_dir() -> str:
|
|
64
|
+
return os.path.join(
|
|
65
|
+
os.environ.get("LOCALAPPDATA", ""),
|
|
66
|
+
"Google", "Chrome", "User Data",
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _chrome_running() -> bool:
|
|
71
|
+
if sys.platform != "win32":
|
|
72
|
+
return "chrome" in subprocess.getoutput("pgrep -la chrome 2>/dev/null")
|
|
73
|
+
result = subprocess.run(
|
|
74
|
+
["tasklist", "/FI", "IMAGENAME eq chrome.exe", "/NH"],
|
|
75
|
+
capture_output=True, text=True, timeout=5,
|
|
76
|
+
)
|
|
77
|
+
return "chrome.exe" in result.stdout
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _get_debug_port() -> int | None:
|
|
81
|
+
"""Read DevToolsActivePort file to get the debugging port."""
|
|
82
|
+
port_file = os.path.join(_user_data_dir(), "DevToolsActivePort")
|
|
83
|
+
if not os.path.isfile(port_file):
|
|
84
|
+
return None
|
|
85
|
+
try:
|
|
86
|
+
with open(port_file) as f:
|
|
87
|
+
lines = f.read().strip().split("\n")
|
|
88
|
+
return int(lines[0])
|
|
89
|
+
except Exception:
|
|
90
|
+
return None
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _restart_chrome_with_debug():
|
|
94
|
+
"""Kill Chrome and relaunch with --remote-debugging-port=0."""
|
|
95
|
+
_stderr("正在重启 Chrome 以启用调试端口(你的标签页会自动恢复)...")
|
|
96
|
+
|
|
97
|
+
# Kill Chrome
|
|
98
|
+
if sys.platform == "win32":
|
|
99
|
+
subprocess.run(["taskkill", "/F", "/IM", "chrome.exe"],
|
|
100
|
+
capture_output=True, timeout=10)
|
|
101
|
+
else:
|
|
102
|
+
subprocess.run(["pkill", "-f", "chrome"], capture_output=True, timeout=10)
|
|
103
|
+
|
|
104
|
+
time.sleep(2)
|
|
105
|
+
|
|
106
|
+
# Relaunch with debugging port
|
|
107
|
+
chrome = _chrome_exe()
|
|
108
|
+
subprocess.Popen(
|
|
109
|
+
[chrome, "--remote-debugging-port=0", "--restore-last-session"],
|
|
110
|
+
creationflags=subprocess.CREATE_NO_WINDOW if sys.platform == "win32" else 0,
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
# Wait for DevToolsActivePort to appear
|
|
114
|
+
_stderr("等待 Chrome 启动...")
|
|
115
|
+
for _ in range(30):
|
|
116
|
+
time.sleep(1)
|
|
117
|
+
port = _get_debug_port()
|
|
118
|
+
if port:
|
|
119
|
+
_stderr(f"Chrome 调试端口就绪: {port}")
|
|
120
|
+
return port
|
|
121
|
+
|
|
122
|
+
raise RuntimeError("Chrome 启动超时,无法获取调试端口")
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _ensure_debug_port() -> int:
|
|
126
|
+
"""Ensure Chrome has a debugging port available."""
|
|
127
|
+
port = _get_debug_port()
|
|
128
|
+
if port:
|
|
129
|
+
return port
|
|
130
|
+
|
|
131
|
+
if _chrome_running():
|
|
132
|
+
_stderr("Chrome 正在运行但未启用调试端口,需要重启...")
|
|
133
|
+
return _restart_chrome_with_debug()
|
|
134
|
+
else:
|
|
135
|
+
_stderr("Chrome 未运行,正在启动...")
|
|
136
|
+
chrome = _chrome_exe()
|
|
137
|
+
subprocess.Popen(
|
|
138
|
+
[chrome, "--remote-debugging-port=0"],
|
|
139
|
+
creationflags=subprocess.CREATE_NO_WINDOW if sys.platform == "win32" else 0,
|
|
140
|
+
)
|
|
141
|
+
for _ in range(30):
|
|
142
|
+
time.sleep(1)
|
|
143
|
+
port = _get_debug_port()
|
|
144
|
+
if port:
|
|
145
|
+
_stderr(f"Chrome 调试端口就绪: {port}")
|
|
146
|
+
return port
|
|
147
|
+
raise RuntimeError("Chrome 启动超时")
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
# ── CDP communication ────────────────────────────────────
|
|
151
|
+
|
|
152
|
+
def _cdp_request(port: int, method: str, params: dict | None = None, session_id: str | None = None) -> dict:
|
|
153
|
+
"""Send a CDP command via HTTP (simple target-level commands)."""
|
|
154
|
+
# For simple commands we can use the HTTP endpoint
|
|
155
|
+
url = f"http://127.0.0.1:{port}/json"
|
|
156
|
+
resp = httpx.get(url, timeout=5)
|
|
157
|
+
return resp.json()
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _cdp_ws_command(port: int, target_ws_url: str, method: str, params: dict = None, msg_id: int = 1) -> dict:
|
|
161
|
+
"""Send a CDP command via WebSocket and get the response."""
|
|
162
|
+
import websocket
|
|
163
|
+
|
|
164
|
+
ws = websocket.create_connection(target_ws_url, timeout=30)
|
|
165
|
+
cmd = {"id": msg_id, "method": method}
|
|
166
|
+
if params:
|
|
167
|
+
cmd["params"] = params
|
|
168
|
+
ws.send(json.dumps(cmd))
|
|
169
|
+
|
|
170
|
+
while True:
|
|
171
|
+
raw = ws.recv()
|
|
172
|
+
data = json.loads(raw)
|
|
173
|
+
if data.get("id") == msg_id:
|
|
174
|
+
ws.close()
|
|
175
|
+
if "error" in data:
|
|
176
|
+
raise RuntimeError(f"CDP error: {data['error']}")
|
|
177
|
+
return data.get("result", {})
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _cdp_navigate_and_get(port: int, url: str, timeout: float = 60) -> tuple[str, str, str]:
|
|
181
|
+
"""
|
|
182
|
+
Navigate to URL in a new CDP target, wait for load, get HTML + title.
|
|
183
|
+
Returns (final_url, title, html).
|
|
184
|
+
"""
|
|
185
|
+
import websocket
|
|
186
|
+
|
|
187
|
+
# Create a new tab/target
|
|
188
|
+
resp = httpx.get(f"http://127.0.0.1:{port}/json/new?{url}", timeout=10)
|
|
189
|
+
target = resp.json()
|
|
190
|
+
ws_url = target["webSocketDebuggerUrl"]
|
|
191
|
+
tab_id = target["id"]
|
|
192
|
+
|
|
193
|
+
try:
|
|
194
|
+
ws = websocket.create_connection(ws_url, timeout=timeout)
|
|
195
|
+
|
|
196
|
+
# Enable page events
|
|
197
|
+
ws.send(json.dumps({"id": 1, "method": "Page.enable"}))
|
|
198
|
+
_recv_response(ws, 1)
|
|
199
|
+
|
|
200
|
+
# Navigate
|
|
201
|
+
ws.send(json.dumps({"id": 2, "method": "Page.navigate", "params": {"url": url}}))
|
|
202
|
+
nav_result = _recv_response(ws, 2)
|
|
203
|
+
|
|
204
|
+
# Wait for page load + SPA rendering
|
|
205
|
+
deadline = time.time() + timeout
|
|
206
|
+
load_fired = False
|
|
207
|
+
while time.time() < deadline:
|
|
208
|
+
try:
|
|
209
|
+
ws.settimeout(2)
|
|
210
|
+
raw = ws.recv()
|
|
211
|
+
data = json.loads(raw)
|
|
212
|
+
if data.get("method") == "Page.loadEventFired":
|
|
213
|
+
load_fired = True
|
|
214
|
+
break
|
|
215
|
+
except Exception:
|
|
216
|
+
continue
|
|
217
|
+
|
|
218
|
+
# Wait for SPA/dynamic content to render
|
|
219
|
+
if load_fired:
|
|
220
|
+
time.sleep(3)
|
|
221
|
+
# Check if page has meaningful content, if not wait more
|
|
222
|
+
for attempt in range(5):
|
|
223
|
+
ws.send(json.dumps({"id": 20 + attempt, "method": "Runtime.evaluate",
|
|
224
|
+
"params": {"expression": "document.body.innerText.length"}}))
|
|
225
|
+
try:
|
|
226
|
+
check = _recv_response(ws, 20 + attempt)
|
|
227
|
+
text_len = check.get("result", {}).get("value", 0)
|
|
228
|
+
if text_len > 200:
|
|
229
|
+
break
|
|
230
|
+
except Exception:
|
|
231
|
+
pass
|
|
232
|
+
time.sleep(2)
|
|
233
|
+
|
|
234
|
+
# Get page info
|
|
235
|
+
ws.settimeout(10)
|
|
236
|
+
ws.send(json.dumps({"id": 10, "method": "Runtime.evaluate",
|
|
237
|
+
"params": {"expression": "document.title"}}))
|
|
238
|
+
title_resp = _recv_response(ws, 10)
|
|
239
|
+
title = title_resp.get("result", {}).get("value", "") or url
|
|
240
|
+
|
|
241
|
+
ws.send(json.dumps({"id": 11, "method": "Runtime.evaluate",
|
|
242
|
+
"params": {"expression": "document.URL"}}))
|
|
243
|
+
url_resp = _recv_response(ws, 11)
|
|
244
|
+
final_url = url_resp.get("result", {}).get("value", url)
|
|
245
|
+
|
|
246
|
+
ws.send(json.dumps({"id": 12, "method": "Runtime.evaluate",
|
|
247
|
+
"params": {"expression": "document.documentElement.outerHTML",
|
|
248
|
+
"returnByValue": True}}))
|
|
249
|
+
html_resp = _recv_response(ws, 12)
|
|
250
|
+
html = html_resp.get("result", {}).get("value", "")
|
|
251
|
+
|
|
252
|
+
ws.close()
|
|
253
|
+
return final_url, title, html
|
|
254
|
+
|
|
255
|
+
finally:
|
|
256
|
+
# Close the tab
|
|
257
|
+
try:
|
|
258
|
+
httpx.get(f"http://127.0.0.1:{port}/json/close/{tab_id}", timeout=5)
|
|
259
|
+
except Exception:
|
|
260
|
+
pass
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _recv_response(ws, expected_id: int) -> dict:
|
|
264
|
+
"""Read WebSocket messages until we get the response for our command ID."""
|
|
265
|
+
deadline = time.time() + 30
|
|
266
|
+
while time.time() < deadline:
|
|
267
|
+
try:
|
|
268
|
+
ws.settimeout(5)
|
|
269
|
+
raw = ws.recv()
|
|
270
|
+
data = json.loads(raw)
|
|
271
|
+
if data.get("id") == expected_id:
|
|
272
|
+
if "error" in data:
|
|
273
|
+
raise RuntimeError(f"CDP: {data['error']}")
|
|
274
|
+
return data.get("result", {})
|
|
275
|
+
except Exception as e:
|
|
276
|
+
if "timed out" in str(e).lower():
|
|
277
|
+
continue
|
|
278
|
+
raise
|
|
279
|
+
raise TimeoutError(f"CDP response timeout for id={expected_id}")
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
# ── Main API ─────────────────────────────────────────────
|
|
283
|
+
|
|
284
|
+
def read_page(url: str, timeout: float = 60) -> PageResult:
|
|
285
|
+
"""Read a web page using the user's running Chrome via CDP."""
|
|
286
|
+
port = _ensure_debug_port()
|
|
287
|
+
_stderr(f"正在通过 Chrome 读取: {url}")
|
|
288
|
+
|
|
289
|
+
final_url, title, html = _cdp_navigate_and_get(port, url, timeout)
|
|
290
|
+
|
|
291
|
+
if not html:
|
|
292
|
+
return PageResult(url=final_url, title=title, content="[empty page]")
|
|
293
|
+
|
|
294
|
+
soup = BeautifulSoup(html, "html.parser")
|
|
295
|
+
|
|
296
|
+
for tag in soup.find_all(["script", "style", "noscript", "iframe", "svg"]):
|
|
297
|
+
tag.decompose()
|
|
298
|
+
|
|
299
|
+
main = None
|
|
300
|
+
for sel in ("article", "main", '[role="main"]', ".content", ".article-content",
|
|
301
|
+
".markdown-body", "#content", "#main"):
|
|
302
|
+
el = soup.select_one(sel)
|
|
303
|
+
if el and len(el.get_text(strip=True)) > 100:
|
|
304
|
+
main = el
|
|
305
|
+
break
|
|
306
|
+
target = main or soup.body or soup
|
|
307
|
+
|
|
308
|
+
links = _extract_links(target, final_url)
|
|
309
|
+
markdown = _H2T.handle(str(target)).strip()
|
|
310
|
+
|
|
311
|
+
if len(markdown) > 200_000:
|
|
312
|
+
markdown = markdown[:200_000] + "\n\n... [truncated]"
|
|
313
|
+
|
|
314
|
+
_stderr(f"读取完成: {title}")
|
|
315
|
+
return PageResult(url=final_url, title=title, content=markdown, links=links)
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _extract_links(soup, base_url: str) -> list[dict]:
|
|
319
|
+
seen: set[str] = set()
|
|
320
|
+
links: list[dict] = []
|
|
321
|
+
for a in soup.find_all("a", href=True):
|
|
322
|
+
href = a["href"].strip()
|
|
323
|
+
if not href or href.startswith("#") or href.startswith("javascript:") or href.startswith("mailto:"):
|
|
324
|
+
continue
|
|
325
|
+
try:
|
|
326
|
+
absolute = urljoin(base_url, href)
|
|
327
|
+
except Exception:
|
|
328
|
+
continue
|
|
329
|
+
if absolute in seen:
|
|
330
|
+
continue
|
|
331
|
+
seen.add(absolute)
|
|
332
|
+
text = a.get_text(strip=True)[:100] or href
|
|
333
|
+
links.append({"text": text, "href": absolute})
|
|
334
|
+
return links
|
{internal_web_reader-0.2.3 → internal_web_reader-1.0.1}/src/internal_web_reader.egg-info/PKG-INFO
RENAMED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: internal-web-reader
|
|
3
|
-
Version: 0.
|
|
4
|
-
Summary: MCP Server -
|
|
3
|
+
Version: 1.0.1
|
|
4
|
+
Summary: MCP Server - read internal web pages through your own Chrome browser
|
|
5
5
|
Author: zhanghaoran18006
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Keywords: mcp,claude,internal,wiki,confluence,gitlab
|
|
@@ -12,13 +12,16 @@ Classifier: Topic :: Software Development :: Libraries
|
|
|
12
12
|
Requires-Python: >=3.10
|
|
13
13
|
Description-Content-Type: text/markdown
|
|
14
14
|
Requires-Dist: mcp[cli]<2.0.0,>=1.2.0
|
|
15
|
-
Requires-Dist:
|
|
15
|
+
Requires-Dist: httpx>=0.27.0
|
|
16
|
+
Requires-Dist: websocket-client>=1.7.0
|
|
16
17
|
Requires-Dist: beautifulsoup4>=4.12.0
|
|
17
18
|
Requires-Dist: html2text>=2024.2.26
|
|
18
19
|
|
|
19
20
|
# Internal Web Reader
|
|
20
21
|
|
|
21
|
-
让 Claude Code
|
|
22
|
+
让 Claude Code 用你自己的 Chrome 浏览器读内网网页。
|
|
23
|
+
|
|
24
|
+
原理:通过 Chrome DevTools Protocol (CDP) 连接到你正在运行的 Chrome,复用已有的登录 session。不需要重新登录。
|
|
22
25
|
|
|
23
26
|
## 安装
|
|
24
27
|
|
|
@@ -34,10 +37,10 @@ claude mcp add -s user internal-web-reader -- internal-web-reader
|
|
|
34
37
|
|
|
35
38
|
## 使用
|
|
36
39
|
|
|
37
|
-
|
|
40
|
+
重启 Claude Code,然后给链接:
|
|
38
41
|
|
|
39
42
|
```
|
|
40
|
-
>
|
|
43
|
+
> 用 read_web_page 读 https://wiki.yourcompany.com/pages/xxx
|
|
41
44
|
```
|
|
42
45
|
|
|
43
|
-
|
|
46
|
+
首次使用时,如果 Chrome 没有启用调试端口,工具会自动重启 Chrome(你的标签页会自动恢复)。之后就直接用了。
|
|
@@ -1,25 +0,0 @@
|
|
|
1
|
-
# Internal Web Reader
|
|
2
|
-
|
|
3
|
-
让 Claude Code 读内网网页。弹个浏览器窗口,登录一次就行。
|
|
4
|
-
|
|
5
|
-
## 安装
|
|
6
|
-
|
|
7
|
-
```
|
|
8
|
-
pip install internal-web-reader
|
|
9
|
-
```
|
|
10
|
-
|
|
11
|
-
## 配置 Claude Code
|
|
12
|
-
|
|
13
|
-
```
|
|
14
|
-
claude mcp add -s user internal-web-reader -- internal-web-reader
|
|
15
|
-
```
|
|
16
|
-
|
|
17
|
-
## 使用
|
|
18
|
-
|
|
19
|
-
在 Claude Code 里给链接:
|
|
20
|
-
|
|
21
|
-
```
|
|
22
|
-
> 看看 https://doc.autohome.com.cn/docapi/page/share/share_1Hx8joJJIfo
|
|
23
|
-
```
|
|
24
|
-
|
|
25
|
-
第一次访问需要登录的网站,会弹出 Chrome 窗口。登录一下,之后就不用再登了。
|
|
@@ -1,54 +0,0 @@
|
|
|
1
|
-
"""MCP Server 入口。"""
|
|
2
|
-
|
|
3
|
-
import atexit
|
|
4
|
-
import logging
|
|
5
|
-
|
|
6
|
-
from mcp.server.fastmcp import FastMCP
|
|
7
|
-
|
|
8
|
-
from .reader import read_page, close_browser
|
|
9
|
-
|
|
10
|
-
logging.basicConfig(
|
|
11
|
-
level=logging.INFO,
|
|
12
|
-
format="[web-reader] %(message)s",
|
|
13
|
-
handlers=[logging.StreamHandler()],
|
|
14
|
-
)
|
|
15
|
-
|
|
16
|
-
mcp = FastMCP("internal-web-reader")
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
@mcp.tool()
|
|
20
|
-
async def read_web_page(url: str) -> str:
|
|
21
|
-
"""[READ WEB PAGES] Open any URL in a local Chrome browser and return Markdown content.
|
|
22
|
-
USE THIS TOOL for all web page reading, especially internal/intranet sites
|
|
23
|
-
(GitLab, Wiki, docs requiring VPN/SSO). Do NOT use the built-in Fetch tool for web pages.
|
|
24
|
-
First visit to a login-required site will pop up a browser window for you to log in.
|
|
25
|
-
Session is saved automatically for future visits.
|
|
26
|
-
"""
|
|
27
|
-
page = await read_page(url)
|
|
28
|
-
|
|
29
|
-
lines = [f"# {page.title}", f"> {page.url}", "", page.content]
|
|
30
|
-
if 0 < len(page.links) <= 30:
|
|
31
|
-
lines += ["", "---", "## Links"]
|
|
32
|
-
for lk in page.links:
|
|
33
|
-
lines.append(f"- [{lk['text']}]({lk['href']})")
|
|
34
|
-
return "\n".join(lines)
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
@mcp.tool()
|
|
38
|
-
async def list_web_links(url: str) -> str:
|
|
39
|
-
"""List all links on a web page for navigation."""
|
|
40
|
-
page = await read_page(url)
|
|
41
|
-
if not page.links:
|
|
42
|
-
return f"No links found on {url}."
|
|
43
|
-
lines = [f"# {page.title} - Links", f"> {len(page.links)} links", ""]
|
|
44
|
-
for lk in page.links:
|
|
45
|
-
lines.append(f"- [{lk['text']}]({lk['href']})")
|
|
46
|
-
return "\n".join(lines)
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
def main():
|
|
50
|
-
mcp.run()
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
if __name__ == "__main__":
|
|
54
|
-
main()
|
|
@@ -1,182 +0,0 @@
|
|
|
1
|
-
"""用 Playwright (async) 启动浏览器读网页。登录一次,session 自动保存。"""
|
|
2
|
-
|
|
3
|
-
import asyncio
|
|
4
|
-
import logging
|
|
5
|
-
import os
|
|
6
|
-
import sys
|
|
7
|
-
from dataclasses import dataclass, field
|
|
8
|
-
from urllib.parse import urljoin
|
|
9
|
-
|
|
10
|
-
from bs4 import BeautifulSoup
|
|
11
|
-
import html2text
|
|
12
|
-
|
|
13
|
-
log = logging.getLogger("internal_web_reader")
|
|
14
|
-
|
|
15
|
-
_H2T = html2text.HTML2Text()
|
|
16
|
-
_H2T.body_width = 0
|
|
17
|
-
_H2T.ignore_links = False
|
|
18
|
-
_H2T.ignore_images = False
|
|
19
|
-
_H2T.protect_links = True
|
|
20
|
-
_H2T.wrap_links = False
|
|
21
|
-
_H2T.unicode_snob = True
|
|
22
|
-
|
|
23
|
-
_USER_DATA_DIR = os.path.join(
|
|
24
|
-
os.environ.get("APPDATA", os.path.expanduser("~")),
|
|
25
|
-
"internal-web-reader",
|
|
26
|
-
"browser-profile",
|
|
27
|
-
)
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
@dataclass
|
|
31
|
-
class PageResult:
|
|
32
|
-
url: str
|
|
33
|
-
title: str
|
|
34
|
-
content: str
|
|
35
|
-
links: list[dict] = field(default_factory=list)
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
_context = None
|
|
39
|
-
_pw = None
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
def _stderr(msg: str):
|
|
43
|
-
sys.stderr.write(f"\n{'='*60}\n{msg}\n{'='*60}\n")
|
|
44
|
-
sys.stderr.flush()
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
async def _ensure_browser():
|
|
48
|
-
global _context, _pw
|
|
49
|
-
if _context is not None:
|
|
50
|
-
return _context
|
|
51
|
-
|
|
52
|
-
from playwright.async_api import async_playwright
|
|
53
|
-
|
|
54
|
-
_pw = await async_playwright().start()
|
|
55
|
-
os.makedirs(_USER_DATA_DIR, exist_ok=True)
|
|
56
|
-
|
|
57
|
-
_context = await _pw.chromium.launch_persistent_context(
|
|
58
|
-
_USER_DATA_DIR,
|
|
59
|
-
headless=False,
|
|
60
|
-
channel="chrome",
|
|
61
|
-
viewport={"width": 1280, "height": 900},
|
|
62
|
-
locale="zh-CN",
|
|
63
|
-
args=["--disable-blink-features=AutomationControlled"],
|
|
64
|
-
)
|
|
65
|
-
return _context
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
def _is_login_url(url: str) -> bool:
|
|
69
|
-
url_lower = url.lower()
|
|
70
|
-
keywords = ["/login", "/signin", "/sso", "/cas", "/oauth", "/auth", "passport"]
|
|
71
|
-
return any(k in url_lower for k in keywords)
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
async def _wait_for_login(page, timeout_s: int = 300):
|
|
75
|
-
_stderr(
|
|
76
|
-
" 需要登录!请切换到弹出的浏览器窗口完成登录。\n"
|
|
77
|
-
" 登录成功后会自动继续,不用回来操作终端。"
|
|
78
|
-
)
|
|
79
|
-
|
|
80
|
-
await page.evaluate("""() => {
|
|
81
|
-
const b = document.createElement('div');
|
|
82
|
-
b.id = '__iwr__';
|
|
83
|
-
b.style.cssText = 'position:fixed;top:0;left:0;right:0;z-index:999999;'
|
|
84
|
-
+ 'background:#ff6b35;color:white;padding:16px;text-align:center;'
|
|
85
|
-
+ 'font-size:18px;font-weight:bold;font-family:sans-serif;'
|
|
86
|
-
+ 'box-shadow:0 4px 12px rgba(0,0,0,0.3);';
|
|
87
|
-
b.textContent = 'Internal Web Reader: Please log in here. Auto-continuing after login...';
|
|
88
|
-
document.body.prepend(b);
|
|
89
|
-
}""")
|
|
90
|
-
|
|
91
|
-
elapsed = 0
|
|
92
|
-
while elapsed < timeout_s:
|
|
93
|
-
await asyncio.sleep(2)
|
|
94
|
-
elapsed += 2
|
|
95
|
-
|
|
96
|
-
current_url = page.url.lower()
|
|
97
|
-
has_pw = await page.query_selector('input[type="password"]')
|
|
98
|
-
|
|
99
|
-
if not _is_login_url(current_url) and not has_pw:
|
|
100
|
-
await page.evaluate("() => { const e = document.getElementById('__iwr__'); if(e) e.remove(); }")
|
|
101
|
-
_stderr(" 登录成功!正在读取页面...")
|
|
102
|
-
await page.wait_for_load_state("domcontentloaded", timeout=30_000)
|
|
103
|
-
return True
|
|
104
|
-
|
|
105
|
-
_stderr(" 登录等待超时,尝试继续...")
|
|
106
|
-
return False
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
async def read_page(url: str, timeout: float = 60) -> PageResult:
|
|
110
|
-
ctx = await _ensure_browser()
|
|
111
|
-
page = ctx.pages[0] if ctx.pages else await ctx.new_page()
|
|
112
|
-
|
|
113
|
-
try:
|
|
114
|
-
await page.bring_to_front()
|
|
115
|
-
except Exception:
|
|
116
|
-
pass
|
|
117
|
-
|
|
118
|
-
await page.goto(url, wait_until="domcontentloaded", timeout=timeout * 1000)
|
|
119
|
-
|
|
120
|
-
# 登录检测
|
|
121
|
-
if _is_login_url(page.url) or await page.query_selector('input[type="password"]'):
|
|
122
|
-
await _wait_for_login(page)
|
|
123
|
-
|
|
124
|
-
final_url = page.url
|
|
125
|
-
html = await page.content()
|
|
126
|
-
title = await page.title() or url
|
|
127
|
-
|
|
128
|
-
soup = BeautifulSoup(html, "html.parser")
|
|
129
|
-
|
|
130
|
-
for tag in soup.find_all(["script", "style", "noscript", "iframe", "svg", "nav", "footer", "aside"]):
|
|
131
|
-
tag.decompose()
|
|
132
|
-
|
|
133
|
-
main = None
|
|
134
|
-
for sel in ("article", "main", '[role="main"]', ".content", ".article-content", ".markdown-body", "#content", "#main"):
|
|
135
|
-
el = soup.select_one(sel)
|
|
136
|
-
if el and len(el.get_text(strip=True)) > 100:
|
|
137
|
-
main = el
|
|
138
|
-
break
|
|
139
|
-
target = main or soup.body or soup
|
|
140
|
-
|
|
141
|
-
links = _extract_links(target, final_url)
|
|
142
|
-
markdown = _H2T.handle(str(target)).strip()
|
|
143
|
-
|
|
144
|
-
if len(markdown) > 200_000:
|
|
145
|
-
markdown = markdown[:200_000] + "\n\n... [truncated]"
|
|
146
|
-
|
|
147
|
-
return PageResult(url=final_url, title=title, content=markdown, links=links)
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
async def close_browser():
|
|
151
|
-
global _context, _pw
|
|
152
|
-
if _context:
|
|
153
|
-
try:
|
|
154
|
-
await _context.close()
|
|
155
|
-
except Exception:
|
|
156
|
-
pass
|
|
157
|
-
if _pw:
|
|
158
|
-
try:
|
|
159
|
-
await _pw.stop()
|
|
160
|
-
except Exception:
|
|
161
|
-
pass
|
|
162
|
-
_context = None
|
|
163
|
-
_pw = None
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
def _extract_links(soup, base_url: str) -> list[dict]:
|
|
167
|
-
seen: set[str] = set()
|
|
168
|
-
links: list[dict] = []
|
|
169
|
-
for a in soup.find_all("a", href=True):
|
|
170
|
-
href = a["href"].strip()
|
|
171
|
-
if not href or href.startswith("#") or href.startswith("javascript:") or href.startswith("mailto:"):
|
|
172
|
-
continue
|
|
173
|
-
try:
|
|
174
|
-
absolute = urljoin(base_url, href)
|
|
175
|
-
except Exception:
|
|
176
|
-
continue
|
|
177
|
-
if absolute in seen:
|
|
178
|
-
continue
|
|
179
|
-
seen.add(absolute)
|
|
180
|
-
text = a.get_text(strip=True)[:100] or href
|
|
181
|
-
links.append({"text": text, "href": absolute})
|
|
182
|
-
return links
|
|
File without changes
|
|
File without changes
|
{internal_web_reader-0.2.3 → internal_web_reader-1.0.1}/src/internal_web_reader.egg-info/SOURCES.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|