internal-web-reader 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: internal-web-reader
3
- Version: 0.1.0
3
+ Version: 0.2.0
4
4
  Summary: MCP Server - let Claude Code read internal web pages via browser cookies
5
5
  Author: zhanghaoran18006
6
6
  License-Expression: MIT
@@ -12,10 +12,9 @@ Classifier: Topic :: Software Development :: Libraries
12
12
  Requires-Python: >=3.10
13
13
  Description-Content-Type: text/markdown
14
14
  Requires-Dist: mcp[cli]<2.0.0,>=1.2.0
15
- Requires-Dist: httpx>=0.27.0
15
+ Requires-Dist: playwright>=1.40.0
16
16
  Requires-Dist: beautifulsoup4>=4.12.0
17
17
  Requires-Dist: html2text>=2024.2.26
18
- Requires-Dist: browser_cookie3>=0.19.1
19
18
 
20
19
  # Internal Web Reader
21
20
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "internal-web-reader"
7
- version = "0.1.0"
7
+ version = "0.2.0"
8
8
  description = "MCP Server - let Claude Code read internal web pages via browser cookies"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -19,10 +19,9 @@ classifiers = [
19
19
  ]
20
20
  dependencies = [
21
21
  "mcp[cli]>=1.2.0,<2.0.0",
22
- "httpx>=0.27.0",
22
+ "playwright>=1.40.0",
23
23
  "beautifulsoup4>=4.12.0",
24
24
  "html2text>=2024.2.26",
25
- "browser_cookie3>=0.19.1",
26
25
  ]
27
26
 
28
27
  [project.scripts]
@@ -0,0 +1,55 @@
1
+ """MCP Server 入口。用 Playwright 浏览器读网页,登录一次自动保存。"""
2
+
3
+ import atexit
4
+ import logging
5
+
6
+ from mcp.server.fastmcp import FastMCP
7
+
8
+ from .reader import read_page, close_browser
9
+
10
+ logging.basicConfig(
11
+ level=logging.INFO,
12
+ format="[web-reader] %(message)s",
13
+ handlers=[logging.StreamHandler()],
14
+ )
15
+ log = logging.getLogger("internal_web_reader")
16
+
17
+ atexit.register(close_browser)
18
+
19
+ mcp = FastMCP("internal-web-reader")
20
+
21
+
22
+ @mcp.tool()
23
+ def read_page_tool(url: str) -> str:
24
+ """读取任意网页,返回干净的 Markdown。
25
+ 首次访问需要登录的网站时,会弹出浏览器窗口让你登录。
26
+ 登录一次后 session 自动保存,后续无需重复登录。
27
+ """
28
+ page = read_page(url)
29
+
30
+ lines = [f"# {page.title}", f"> {page.url}", "", page.content]
31
+ if 0 < len(page.links) <= 30:
32
+ lines += ["", "---", "## 页面链接"]
33
+ for lk in page.links:
34
+ lines.append(f"- [{lk['text']}]({lk['href']})")
35
+ return "\n".join(lines)
36
+
37
+
38
+ @mcp.tool()
39
+ def list_links(url: str) -> str:
40
+ """列出页面上的所有链接,用于浏览内网站点。"""
41
+ page = read_page(url)
42
+ if not page.links:
43
+ return f"页面 {url} 上未找到链接。"
44
+ lines = [f"# {page.title} 的链接", f"> {len(page.links)} 个", ""]
45
+ for lk in page.links:
46
+ lines.append(f"- [{lk['text']}]({lk['href']})")
47
+ return "\n".join(lines)
48
+
49
+
50
+ def main():
51
+ mcp.run()
52
+
53
+
54
+ if __name__ == "__main__":
55
+ main()
@@ -0,0 +1,166 @@
1
+ """用 Playwright 启动浏览器读网页。登录一次,session 自动保存。"""
2
+
3
+ import logging
4
+ import os
5
+ import re
6
+ from dataclasses import dataclass, field
7
+ from pathlib import Path
8
+ from urllib.parse import urljoin
9
+
10
+ from bs4 import BeautifulSoup
11
+ import html2text
12
+
13
+ log = logging.getLogger("internal_web_reader")
14
+
15
+ _H2T = html2text.HTML2Text()
16
+ _H2T.body_width = 0
17
+ _H2T.ignore_links = False
18
+ _H2T.ignore_images = False
19
+ _H2T.protect_links = True
20
+ _H2T.wrap_links = False
21
+ _H2T.unicode_snob = True
22
+
23
+ # 持久化浏览器 profile 目录
24
+ _USER_DATA_DIR = os.path.join(
25
+ os.environ.get("APPDATA", os.path.expanduser("~")),
26
+ "internal-web-reader",
27
+ "browser-profile",
28
+ )
29
+
30
+
31
+ @dataclass
32
+ class PageResult:
33
+ url: str
34
+ title: str
35
+ content: str
36
+ links: list[dict] = field(default_factory=list)
37
+
38
+
39
+ # 全局浏览器实例(进程内复用)
40
+ _browser = None
41
+ _context = None
42
+ _pw = None
43
+
44
+
45
+ def _ensure_browser():
46
+ """启动 Playwright 浏览器(进程内单例)。"""
47
+ global _browser, _context, _pw
48
+ if _context is not None:
49
+ return _context
50
+
51
+ from playwright.sync_api import sync_playwright
52
+
53
+ _pw = sync_playwright().start()
54
+ os.makedirs(_USER_DATA_DIR, exist_ok=True)
55
+
56
+ _context = _pw.chromium.launch_persistent_context(
57
+ _USER_DATA_DIR,
58
+ headless=False,
59
+ channel="chrome",
60
+ viewport={"width": 1280, "height": 900},
61
+ locale="zh-CN",
62
+ args=["--disable-blink-features=AutomationControlled"],
63
+ )
64
+ log.info("浏览器已启动")
65
+ return _context
66
+
67
+
68
+ def _is_login_page(page) -> bool:
69
+ """判断当前页面是否是登录页。"""
70
+ url = page.url.lower()
71
+ if any(k in url for k in ("/login", "/signin", "/sso", "/auth", "/cas/", "/oauth")):
72
+ return True
73
+ # 检查页面是否有密码输入框
74
+ pw_inputs = page.query_selector_all('input[type="password"]')
75
+ if pw_inputs and len(pw_inputs) > 0:
76
+ return True
77
+ return False
78
+
79
+
80
+ def read_page(url: str, timeout: float = 60) -> PageResult:
81
+ """用浏览器打开 URL,返回清洗后的 Markdown。"""
82
+ ctx = _ensure_browser()
83
+
84
+ # 用现有 page 或新建
85
+ page = ctx.pages[0] if ctx.pages else ctx.new_page()
86
+
87
+ page.goto(url, wait_until="domcontentloaded", timeout=timeout * 1000)
88
+
89
+ # 如果是登录页,等用户手动登录
90
+ if _is_login_page(page):
91
+ log.info("检测到登录页面,请在浏览器窗口中完成登录...")
92
+ # 等 URL 变化(登录成功后会跳转)或者最多等 120 秒
93
+ try:
94
+ page.wait_for_url(
95
+ lambda u: u != page.url and "login" not in u.lower(),
96
+ timeout=120_000,
97
+ )
98
+ # 等页面加载完
99
+ page.wait_for_load_state("domcontentloaded", timeout=30_000)
100
+ log.info(f"登录成功,当前页面: {page.url}")
101
+ except Exception:
102
+ log.warning("等待登录超时,尝试继续...")
103
+
104
+ # 获取页面内容
105
+ final_url = page.url
106
+ html = page.content()
107
+ title = page.title() or url
108
+
109
+ soup = BeautifulSoup(html, "html.parser")
110
+
111
+ # 去噪
112
+ for tag in soup.find_all(["script", "style", "noscript", "iframe", "svg", "nav", "footer", "aside"]):
113
+ tag.decompose()
114
+
115
+ # 找主体
116
+ main = None
117
+ for sel in ("article", "main", '[role="main"]', ".content", ".article-content", ".markdown-body", "#content", "#main"):
118
+ el = soup.select_one(sel)
119
+ if el and len(el.get_text(strip=True)) > 100:
120
+ main = el
121
+ break
122
+ target = main or soup.body or soup
123
+
124
+ links = _extract_links(target, final_url)
125
+ markdown = _H2T.handle(str(target)).strip()
126
+
127
+ if len(markdown) > 200_000:
128
+ markdown = markdown[:200_000] + "\n\n... [已截断]"
129
+
130
+ return PageResult(url=final_url, title=title, content=markdown, links=links)
131
+
132
+
133
+ def close_browser():
134
+ """关闭浏览器(进程退出时调用)。"""
135
+ global _browser, _context, _pw
136
+ if _context:
137
+ try:
138
+ _context.close()
139
+ except Exception:
140
+ pass
141
+ if _pw:
142
+ try:
143
+ _pw.stop()
144
+ except Exception:
145
+ pass
146
+ _context = None
147
+ _pw = None
148
+
149
+
150
+ def _extract_links(soup, base_url: str) -> list[dict]:
151
+ seen: set[str] = set()
152
+ links: list[dict] = []
153
+ for a in soup.find_all("a", href=True):
154
+ href = a["href"].strip()
155
+ if not href or href.startswith("#") or href.startswith("javascript:") or href.startswith("mailto:"):
156
+ continue
157
+ try:
158
+ absolute = urljoin(base_url, href)
159
+ except Exception:
160
+ continue
161
+ if absolute in seen:
162
+ continue
163
+ seen.add(absolute)
164
+ text = a.get_text(strip=True)[:100] or href
165
+ links.append({"text": text, "href": absolute})
166
+ return links
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: internal-web-reader
3
- Version: 0.1.0
3
+ Version: 0.2.0
4
4
  Summary: MCP Server - let Claude Code read internal web pages via browser cookies
5
5
  Author: zhanghaoran18006
6
6
  License-Expression: MIT
@@ -12,10 +12,9 @@ Classifier: Topic :: Software Development :: Libraries
12
12
  Requires-Python: >=3.10
13
13
  Description-Content-Type: text/markdown
14
14
  Requires-Dist: mcp[cli]<2.0.0,>=1.2.0
15
- Requires-Dist: httpx>=0.27.0
15
+ Requires-Dist: playwright>=1.40.0
16
16
  Requires-Dist: beautifulsoup4>=4.12.0
17
17
  Requires-Dist: html2text>=2024.2.26
18
- Requires-Dist: browser_cookie3>=0.19.1
19
18
 
20
19
  # Internal Web Reader
21
20
 
@@ -2,7 +2,6 @@ README.md
2
2
  pyproject.toml
3
3
  src/internal_web_reader/__init__.py
4
4
  src/internal_web_reader/__main__.py
5
- src/internal_web_reader/cookies.py
6
5
  src/internal_web_reader/reader.py
7
6
  src/internal_web_reader.egg-info/PKG-INFO
8
7
  src/internal_web_reader.egg-info/SOURCES.txt
@@ -1,5 +1,4 @@
1
1
  mcp[cli]<2.0.0,>=1.2.0
2
- httpx>=0.27.0
2
+ playwright>=1.40.0
3
3
  beautifulsoup4>=4.12.0
4
4
  html2text>=2024.2.26
5
- browser_cookie3>=0.19.1
@@ -1,129 +0,0 @@
1
- """MCP Server 入口。Cookie 懒加载:首次访问某域名时自动从浏览器提取。"""
2
-
3
- import logging
4
- from urllib.parse import urlparse
5
-
6
- from mcp.server.fastmcp import FastMCP
7
-
8
- from .cookies import get_cookie_header
9
- from .reader import read_page
10
-
11
- logging.basicConfig(
12
- level=logging.INFO,
13
- format="[web-reader] %(message)s",
14
- handlers=[logging.StreamHandler()],
15
- )
16
- log = logging.getLogger("internal_web_reader")
17
-
18
- # domain → cookie header string(内存缓存)
19
- _cookie_cache: dict[str, str] = {}
20
-
21
- mcp = FastMCP("internal-web-reader")
22
-
23
-
24
- def _get_cookie(url: str) -> str | None:
25
- """获取 URL 对应的 Cookie,没有就自动从浏览器提取并缓存。"""
26
- try:
27
- host = urlparse(url).hostname or ""
28
- except Exception:
29
- return None
30
-
31
- if not host:
32
- return None
33
-
34
- # 1. 缓存命中
35
- if host in _cookie_cache:
36
- return _cookie_cache[host]
37
-
38
- # 2. 父域名
39
- parts = host.split(".")
40
- for i in range(1, len(parts) - 1):
41
- parent = ".".join(parts[i:])
42
- if parent in _cookie_cache:
43
- return _cookie_cache[parent]
44
-
45
- # 3. 自动提取
46
- header = get_cookie_header(host)
47
- if header:
48
- _cookie_cache[host] = header
49
- return header
50
-
51
-
52
- def _refresh_cookie(host: str) -> str | None:
53
- """强制刷新某域名的 Cookie(401/403 时调用)。"""
54
- _cookie_cache.pop(host, None)
55
- parts = host.split(".")
56
- for i in range(1, len(parts) - 1):
57
- _cookie_cache.pop(".".join(parts[i:]), None)
58
-
59
- header = get_cookie_header(host)
60
- if header:
61
- _cookie_cache[host] = header
62
- return header
63
-
64
-
65
- # ── Tools ────────────────────────────────────────────────
66
-
67
-
68
- @mcp.tool()
69
- def read_page_tool(url: str) -> str:
70
- """读取任意网页,返回干净的 Markdown。
71
- Cookie 自动从浏览器提取,不需要预配置。
72
- 支持内网 GitLab、Wiki、文档中心等。
73
- """
74
- cookie = _get_cookie(url)
75
-
76
- try:
77
- page = read_page(url, cookie=cookie)
78
- except Exception as e:
79
- # 401/403 → 刷新 Cookie 重试
80
- msg = str(e)
81
- if any(k in msg for k in ("401", "403", "Unauthorized", "Forbidden")):
82
- try:
83
- host = urlparse(url).hostname or ""
84
- except Exception:
85
- raise
86
- log.info(f"鉴权失败,刷新 {host} Cookie...")
87
- cookie = _refresh_cookie(host)
88
- if cookie:
89
- log.info("重试中...")
90
- page = read_page(url, cookie=cookie)
91
- else:
92
- raise
93
- else:
94
- raise
95
-
96
- lines = [f"# {page.title}", f"> {page.url}", "", page.content]
97
-
98
- if 0 < len(page.links) <= 30:
99
- lines += ["", "---", "## 页面链接"]
100
- for lk in page.links:
101
- lines.append(f"- [{lk['text']}]({lk['href']})")
102
-
103
- return "\n".join(lines)
104
-
105
-
106
- @mcp.tool()
107
- def list_links(url: str) -> str:
108
- """列出页面上的所有链接,用于浏览内网站点。"""
109
- cookie = _get_cookie(url)
110
- page = read_page(url, cookie=cookie)
111
-
112
- if not page.links:
113
- return f"页面 {url} 上未找到链接。"
114
-
115
- lines = [f"# {page.title} 的链接", f"> {len(page.links)} 个", ""]
116
- for lk in page.links:
117
- lines.append(f"- [{lk['text']}]({lk['href']})")
118
- return "\n".join(lines)
119
-
120
-
121
- # ── Entry ────────────────────────────────────────────────
122
-
123
-
124
- def main():
125
- mcp.run()
126
-
127
-
128
- if __name__ == "__main__":
129
- main()
@@ -1,44 +0,0 @@
1
- """从本地浏览器自动提取 Cookie。"""
2
-
3
- import logging
4
-
5
- log = logging.getLogger("internal_web_reader")
6
-
7
-
8
- def get_cookie_header(domain: str) -> str | None:
9
- """
10
- 从 Chrome / Edge / Firefox 提取指定域名的 Cookie,
11
- 返回 "name1=val1; name2=val2" 格式的字符串。
12
- 找不到返回 None。
13
- """
14
- cookie_jars = []
15
-
16
- # 依次尝试各浏览器
17
- try:
18
- import browser_cookie3
19
-
20
- for loader in (browser_cookie3.chrome, browser_cookie3.edge, browser_cookie3.firefox):
21
- try:
22
- cj = loader(domain_name=domain)
23
- cookie_jars.append(cj)
24
- except Exception as e:
25
- log.debug(f"{loader.__name__} 跳过: {e}")
26
- except ImportError:
27
- log.warning("browser_cookie3 未安装,无法自动提取 Cookie")
28
- return None
29
-
30
- # 合并 + 去重
31
- seen: dict[str, str] = {}
32
- for cj in cookie_jars:
33
- for c in cj:
34
- # 匹配: 精确域名 或 父域名 (.example.com 匹配 sub.example.com)
35
- cdomain = c.domain.lstrip(".")
36
- if domain == cdomain or domain.endswith("." + cdomain):
37
- seen[c.name] = c.value
38
-
39
- if not seen:
40
- return None
41
-
42
- header = "; ".join(f"{k}={v}" for k, v in seen.items())
43
- log.info(f"从浏览器提取 {len(seen)} 个 Cookie ({domain})")
44
- return header
@@ -1,114 +0,0 @@
1
- """抓取网页,HTML → 干净 Markdown。"""
2
-
3
- import logging
4
- from dataclasses import dataclass, field
5
- from urllib.parse import urljoin
6
-
7
- import httpx
8
- from bs4 import BeautifulSoup
9
- import html2text
10
-
11
- log = logging.getLogger("internal_web_reader")
12
-
13
- _H2T = html2text.HTML2Text()
14
- _H2T.body_width = 0 # 不自动换行
15
- _H2T.ignore_links = False
16
- _H2T.ignore_images = False
17
- _H2T.protect_links = True
18
- _H2T.wrap_links = False
19
- _H2T.unicode_snob = True
20
-
21
- _UA = (
22
- "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
23
- "AppleWebKit/537.36 (KHTML, like Gecko) "
24
- "Chrome/124.0.0.0 Safari/537.36"
25
- )
26
-
27
-
28
- @dataclass
29
- class PageResult:
30
- url: str
31
- title: str
32
- content: str
33
- links: list[dict] = field(default_factory=list)
34
-
35
-
36
- def read_page(url: str, cookie: str | None = None, timeout: float = 30) -> PageResult:
37
- """抓取一个 URL,返回清洗后的 Markdown。"""
38
-
39
- headers = {
40
- "User-Agent": _UA,
41
- "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
42
- "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
43
- }
44
- if cookie:
45
- headers["Cookie"] = cookie
46
-
47
- resp = httpx.get(url, headers=headers, follow_redirects=True, timeout=timeout)
48
- resp.raise_for_status()
49
-
50
- ct = resp.headers.get("content-type", "")
51
- if "html" not in ct and "text" not in ct:
52
- return PageResult(url=url, title=url, content=f"[非 HTML] Content-Type: {ct}")
53
-
54
- # 检测编码
55
- resp.encoding = resp.charset_encoding or "utf-8"
56
- html = resp.text
57
-
58
- soup = BeautifulSoup(html, "html.parser")
59
-
60
- # 标题
61
- title = ""
62
- if soup.title and soup.title.string:
63
- title = soup.title.string.strip()
64
- if not title:
65
- h1 = soup.find("h1")
66
- if h1:
67
- title = h1.get_text(strip=True)
68
- title = title or url
69
-
70
- # 去噪
71
- for tag in soup.find_all(["script", "style", "noscript", "iframe", "svg", "nav", "footer", "aside"]):
72
- tag.decompose()
73
-
74
- # 找主体内容
75
- main = None
76
- for sel in ("article", "main", '[role="main"]', ".content", ".article-content", ".markdown-body", "#content", "#main"):
77
- el = soup.select_one(sel)
78
- if el and len(el.get_text(strip=True)) > 100:
79
- main = el
80
- break
81
-
82
- target = main or soup.body or soup
83
-
84
- # 提取链接
85
- links = _extract_links(target, url)
86
-
87
- # HTML → Markdown
88
- content_html = str(target)
89
- markdown = _H2T.handle(content_html).strip()
90
-
91
- # 截断
92
- if len(markdown) > 200_000:
93
- markdown = markdown[:200_000] + "\n\n... [内容过长,已截断]"
94
-
95
- return PageResult(url=url, title=title, content=markdown, links=links)
96
-
97
-
98
- def _extract_links(soup, base_url: str) -> list[dict]:
99
- seen: set[str] = set()
100
- links: list[dict] = []
101
- for a in soup.find_all("a", href=True):
102
- href = a["href"].strip()
103
- if not href or href.startswith("#") or href.startswith("javascript:") or href.startswith("mailto:"):
104
- continue
105
- try:
106
- absolute = urljoin(base_url, href)
107
- except Exception:
108
- continue
109
- if absolute in seen:
110
- continue
111
- seen.add(absolute)
112
- text = a.get_text(strip=True)[:100] or href
113
- links.append({"text": text, "href": absolute})
114
- return links