ai-browser-mcp 0.12.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. ai_browser_mcp-0.12.0/.github/workflows/publish.yml +41 -0
  2. ai_browser_mcp-0.12.0/.gitignore +12 -0
  3. ai_browser_mcp-0.12.0/CHANGELOG.md +108 -0
  4. ai_browser_mcp-0.12.0/LICENSE +21 -0
  5. ai_browser_mcp-0.12.0/PKG-INFO +393 -0
  6. ai_browser_mcp-0.12.0/README.en.md +377 -0
  7. ai_browser_mcp-0.12.0/README.md +354 -0
  8. ai_browser_mcp-0.12.0/docs/architecture.md +513 -0
  9. ai_browser_mcp-0.12.0/docs/implementation-plan.md +354 -0
  10. ai_browser_mcp-0.12.0/extension/background.js +3483 -0
  11. ai_browser_mcp-0.12.0/extension/comment_sessions.js +206 -0
  12. ai_browser_mcp-0.12.0/extension/content_bridge.js +19 -0
  13. ai_browser_mcp-0.12.0/extension/content_inject.js +71 -0
  14. ai_browser_mcp-0.12.0/extension/douyin_content_bridge.js +19 -0
  15. ai_browser_mcp-0.12.0/extension/douyin_content_inject.js +70 -0
  16. ai_browser_mcp-0.12.0/extension/manifest.json +42 -0
  17. ai_browser_mcp-0.12.0/extension/options.html +20 -0
  18. ai_browser_mcp-0.12.0/extension/options.js +19 -0
  19. ai_browser_mcp-0.12.0/pyproject.toml +63 -0
  20. ai_browser_mcp-0.12.0/scripts/smoke_real.py +719 -0
  21. ai_browser_mcp-0.12.0/src/browser_mcp/__init__.py +3 -0
  22. ai_browser_mcp-0.12.0/src/browser_mcp/__main__.py +6 -0
  23. ai_browser_mcp-0.12.0/src/browser_mcp/application/__init__.py +5 -0
  24. ai_browser_mcp-0.12.0/src/browser_mcp/application/browser_service.py +169 -0
  25. ai_browser_mcp-0.12.0/src/browser_mcp/bridge/__init__.py +5 -0
  26. ai_browser_mcp-0.12.0/src/browser_mcp/bridge/bundle.py +150 -0
  27. ai_browser_mcp-0.12.0/src/browser_mcp/bridge/manager.py +524 -0
  28. ai_browser_mcp-0.12.0/src/browser_mcp/bridge/protocol.py +38 -0
  29. ai_browser_mcp-0.12.0/src/browser_mcp/cli.py +64 -0
  30. ai_browser_mcp-0.12.0/src/browser_mcp/config.py +138 -0
  31. ai_browser_mcp-0.12.0/src/browser_mcp/extension/__init__.py +1 -0
  32. ai_browser_mcp-0.12.0/src/browser_mcp/extraction.py +181 -0
  33. ai_browser_mcp-0.12.0/src/browser_mcp/logging_config.py +18 -0
  34. ai_browser_mcp-0.12.0/src/browser_mcp/mcp/__init__.py +5 -0
  35. ai_browser_mcp-0.12.0/src/browser_mcp/mcp/server.py +1075 -0
  36. ai_browser_mcp-0.12.0/src/browser_mcp/models.py +249 -0
  37. ai_browser_mcp-0.12.0/src/browser_mcp/process_lifecycle.py +118 -0
  38. ai_browser_mcp-0.12.0/src/browser_mcp/security/__init__.py +5 -0
  39. ai_browser_mcp-0.12.0/src/browser_mcp/security/url_policy.py +222 -0
  40. ai_browser_mcp-0.12.0/src/browser_mcp/sites/__init__.py +5 -0
  41. ai_browser_mcp-0.12.0/src/browser_mcp/sites/auth.py +207 -0
  42. ai_browser_mcp-0.12.0/src/browser_mcp/sites/bilibili.py +452 -0
  43. ai_browser_mcp-0.12.0/src/browser_mcp/sites/bilibili_media.py +256 -0
  44. ai_browser_mcp-0.12.0/src/browser_mcp/sites/douyin.py +464 -0
  45. ai_browser_mcp-0.12.0/src/browser_mcp/sites/html_utils.py +62 -0
  46. ai_browser_mcp-0.12.0/src/browser_mcp/sites/media.py +339 -0
  47. ai_browser_mcp-0.12.0/src/browser_mcp/sites/models.py +787 -0
  48. ai_browser_mcp-0.12.0/src/browser_mcp/sites/reddit.py +162 -0
  49. ai_browser_mcp-0.12.0/src/browser_mcp/sites/search_engines.py +197 -0
  50. ai_browser_mcp-0.12.0/src/browser_mcp/sites/service.py +613 -0
  51. ai_browser_mcp-0.12.0/src/browser_mcp/sites/snapshot.py +163 -0
  52. ai_browser_mcp-0.12.0/src/browser_mcp/sites/x.py +163 -0
  53. ai_browser_mcp-0.12.0/src/browser_mcp/sites/xhs.py +432 -0
  54. ai_browser_mcp-0.12.0/src/browser_mcp/sites/zhihu.py +442 -0
  55. ai_browser_mcp-0.12.0/src/browser_mcp/snapshot/__init__.py +5 -0
  56. ai_browser_mcp-0.12.0/src/browser_mcp/snapshot/store.py +176 -0
  57. ai_browser_mcp-0.12.0/src/browser_mcp/upgrade.py +404 -0
  58. ai_browser_mcp-0.12.0/tests/__init__.py +1 -0
  59. ai_browser_mcp-0.12.0/tests/helpers.py +203 -0
  60. ai_browser_mcp-0.12.0/tests/test_application.py +75 -0
  61. ai_browser_mcp-0.12.0/tests/test_bilibili.py +237 -0
  62. ai_browser_mcp-0.12.0/tests/test_bilibili_media.py +85 -0
  63. ai_browser_mcp-0.12.0/tests/test_bridge_manager.py +568 -0
  64. ai_browser_mcp-0.12.0/tests/test_config.py +61 -0
  65. ai_browser_mcp-0.12.0/tests/test_douyin.py +223 -0
  66. ai_browser_mcp-0.12.0/tests/test_extension_bundle.py +109 -0
  67. ai_browser_mcp-0.12.0/tests/test_extraction.py +51 -0
  68. ai_browser_mcp-0.12.0/tests/test_interactions.py +77 -0
  69. ai_browser_mcp-0.12.0/tests/test_media.py +203 -0
  70. ai_browser_mcp-0.12.0/tests/test_process_lifecycle.py +116 -0
  71. ai_browser_mcp-0.12.0/tests/test_reddit.py +92 -0
  72. ai_browser_mcp-0.12.0/tests/test_search_engines.py +89 -0
  73. ai_browser_mcp-0.12.0/tests/test_server.py +217 -0
  74. ai_browser_mcp-0.12.0/tests/test_site_auth.py +145 -0
  75. ai_browser_mcp-0.12.0/tests/test_site_service.py +650 -0
  76. ai_browser_mcp-0.12.0/tests/test_snapshot_store.py +68 -0
  77. ai_browser_mcp-0.12.0/tests/test_stdio_contract.py +94 -0
  78. ai_browser_mcp-0.12.0/tests/test_upgrade.py +147 -0
  79. ai_browser_mcp-0.12.0/tests/test_url_policy.py +146 -0
  80. ai_browser_mcp-0.12.0/tests/test_version_consistency.py +27 -0
  81. ai_browser_mcp-0.12.0/tests/test_x.py +66 -0
  82. ai_browser_mcp-0.12.0/tests/test_xhs.py +326 -0
  83. ai_browser_mcp-0.12.0/tests/test_zhihu.py +222 -0
  84. ai_browser_mcp-0.12.0/uv.lock +1027 -0
@@ -0,0 +1,41 @@
1
+ name: Publish to PyPI
2
+
3
+ on:
4
+ push:
5
+ tags:
6
+ - "v*"
7
+ workflow_dispatch: {}
8
+
9
+ permissions:
10
+ contents: read
11
+
12
+ jobs:
13
+ publish:
14
+ runs-on: ubuntu-latest
15
+ steps:
16
+ - name: Checkout
17
+ uses: actions/checkout@v4
18
+
19
+ - name: Check tag matches package version
20
+ id: version
21
+ shell: bash
22
+ run: |
23
+ PKG_VERSION=$(python -c "import tomllib;print(tomllib.load(open('pyproject.toml','rb'))['project']['version'])")
24
+ TAG_VERSION=${GITHUB_REF_NAME#v}
25
+ echo "package=${PKG_VERSION}"
26
+ echo "tag=${TAG_VERSION}"
27
+ if [ "$PKG_VERSION" != "$TAG_VERSION" ]; then
28
+ echo "::error::package version ($PKG_VERSION) != tag ($TAG_VERSION). Bump pyproject.toml to match and re-tag."
29
+ exit 1
30
+ fi
31
+
32
+ - name: Install uv
33
+ uses: astral-sh/setup-uv@v5
34
+
35
+ - name: Build
36
+ run: uv build
37
+
38
+ - name: Publish to PyPI
39
+ uses: pypa/gh-action-pypi-publish@release/v1
40
+ with:
41
+ password: ${{ secrets.PYPI_API_TOKEN }}
@@ -0,0 +1,12 @@
1
+ # Python, test, and build outputs.
2
+ .venv/
3
+ __pycache__/
4
+ *.py[cod]
5
+ .pytest_cache/
6
+ .ruff_cache/
7
+ .pyright/
8
+ dist/
9
+
10
+ # Local extension pairing artifacts contain machine-specific metadata and secrets.
11
+ extension/build-info.js
12
+ extension/pairing.json
@@ -0,0 +1,108 @@
1
+ # 变更日志
2
+
3
+ 本项目按[语义化版本](https://semver.org/lang/zh-CN/)维护版本号。
4
+
5
+ ## [0.12.0] - 2026-08-24
6
+
7
+ ### 发布与文档
8
+
9
+ - 包名改为 `ai-browser-mcp`(PyPI 原名 `browser-mcp` 已被占用),并新增
10
+ `.github/workflows/publish.yml`:打 `v*` tag 时用仓库的 `PYPI_API_TOKEN` 自动构建并发布。
11
+ - README 重写为 pitch 前置:开头突出「让你的 AI 真正搜遍全网、登录能访问的站点都能抓、还能后台
12
+ 自动化操作」,并新增英文版 `README.en.md`;PyPI 页面以此为 `readme`。
13
+
14
+ ### 改进
15
+
16
+ - `xhs_comments` 与 `douyin_comments` 改为限时采集:一次调用在 `time_budget_seconds`
17
+ (默认 40 秒)到点后挂起而不是超时失败,返回本次新抓到的评论、`collected_total` 进度和
18
+ 可续抓的 `session_id`;带上该 `session_id` 再次调用即从上次的滚动位置、去重集合和分页
19
+ 状态继续,不重复已抓过的评论。此前热门作品的评论采集会在 MCP 客户端 60 秒超时时整批丢弃。
20
+ - 新增 `extension/comment_sessions.js`:评论采集会话的窗口、去重集合与循环状态在调用之间
21
+ 存活,镜像到 `chrome.storage.session` 以便 service worker 回收后恢复或清理,闲置 5 分钟
22
+ 由 alarm 关闭,配对进程退出时立即全部释放。
23
+ - 评论采集的 bridge 超时不再是固定 180 秒,而是按本次预算加 15 秒余量,触发它明确表示扩展
24
+ 卡死而非采集慢。
25
+
26
+ ### 修复
27
+
28
+ - 通用 `browser_click` 不再调用会产生 `isTrusted=false` 事件的 `element.click()`,改为在
29
+ debugger 导航守卫存续期间解析可命中的 DOM 坐标,并只发送一次 CDP 可信鼠标点击。
30
+ - 截图坐标现在按 JPEG 实际像素尺寸映射到 CSS 视口,修复 Retina 和页面缩放环境下的点击
31
+ 偏移;视口快照同时返回截图宽高,调用方仍可显式选择 CSS viewport 坐标。
32
+ - 元素引用点击会在滚动后重新计算可见区域并检查遮挡;无法命中的目标直接返回诊断错误,
33
+ 不再把未产生页面效果的合成事件误报为成功。
34
+
35
+ ## [0.11.0] - 2026-08-20
36
+
37
+ ### 新增
38
+
39
+ - 新增 `bilibili_search` 与 `bilibili_video`,通过当前 Chrome 会话搜索 B 站视频并读取
40
+ BV/AV 标识、内容 meta、作者、标签、互动统计及分 P 信息。
41
+ - 新增 `bilibili_download_video` 与 `bilibili_download_audio`,支持指定 `?p=N` 下载分 P
42
+ 视频,或只保存最佳兼容音轨。
43
+ - 新增独立 `bilibili.fetch` 扩展协议和 `sites/bilibili*.py` adapter,不把 B 站 API、
44
+ DASH 选择或 FFmpeg 编排混入其他平台模块。
45
+
46
+ ### 改进
47
+
48
+ - B 站 DASH 下载优先最高可用画质和 AVC 兼容轨;检测到 FFmpeg 时使用 stream copy 无损
49
+ 合并画面与音频,未安装 FFmpeg 时明确返回两个独立轨道。
50
+ - B 站公开搜索 API 遇到 HTTP 412 风控时自动回退到真实搜索结果页,并对重复渲染卡片按
51
+ BV 标识去重。
52
+ - 共享媒体下载器增加 B 站 CDN、Referer/Origin、音频 MP4 容器和 `.m4a` 扩展名兼容,
53
+ 保留逐跳 URL 校验、大小限制、SHA-256 与原子发布。
54
+ - MCP 服务新增 owner watchdog:自动跳过 `uv` wrapper 监控真实宿主,宿主异常退出且 stdin
55
+ 未正常关闭时仍能终止服务并释放 bridge 端口。
56
+ - 正常关闭会先发送 `bridge.shutdown`;扩展在显式关闭或 WebSocket 断开后按端口清理隔离
57
+ 交互窗口、队列和 debugger,不关闭用户原有标签页。
58
+
59
+ ### 验证
60
+
61
+ - 已用真实扩展会话完成搜索、meta、视频和纯音频下载;测试 MP4 经 `ffprobe` 确认包含
62
+ H.264 + AAC,纯音频 M4A 只包含 AAC。
63
+
64
+ ## [0.10.0] - 2026-08-19
65
+
66
+ ### 新增
67
+
68
+ - 新增 `xhs_like`、`xhs_collect`、`douyin_like`、`douyin_collect` 四个窄工具,支持通过
69
+ `enabled` 设置点赞/收藏的期望状态。
70
+ - 新增站点专用 `xhs.mutate` 与 `douyin.mutate` 扩展协议,写操作不进入通用读取命名空间。
71
+
72
+ ### 改进
73
+
74
+ - 点赞与收藏先读取当前页面状态,仅在不一致时发送一次可信点击;重复调用同一状态为 no-op。
75
+ - 点击后只轮询验证最终状态,验证失败不会重试点击,避免页面延迟导致反向切换。
76
+
77
+ ### 安全性
78
+
79
+ - 四个变更型工具均声明非只读、可撤销型 destructive 和幂等风险标记,并在工具说明中要求
80
+ MCP 客户端在调用前立即取得用户明确确认。
81
+ - 小红书只定位详情底栏的语义图标,抖音只定位作品级 `data-e2e` 控件,避免误点评论点赞。
82
+ - 真实媒体 smoke test 改为运行时搜索并获取有效 `xsec_token`,不再把临时访问参数写入源码,
83
+ 同时在控制台摘要中对该参数脱敏。
84
+
85
+ ## [0.9.0] - 2026-08-19
86
+
87
+ ### 新增
88
+
89
+ - 新增 `douyin_search`、`douyin_video`、`douyin_comments`,通过当前 Chrome 登录态完成
90
+ 抖音搜索、视频/图文详情读取,以及主评论和展开回复采集。
91
+ - 新增 `xhs_download`、`douyin_download`,支持按 `images`、`video` 或 `all` 下载媒体,
92
+ 可指定绝对输出目录、覆盖策略和单文件大小上限。
93
+ - `site_login_status` 新增抖音登录状态检查。
94
+ - 新增抖音独立页面注入与桥接脚本,避免平台实现污染通用页面桥接。
95
+
96
+ ### 改进
97
+
98
+ - 抖音评论采集滚动作评论流而不是页面 `body`,并支持展开回复及 root/reply 分页完成判定。
99
+ - 抖音图文详情在页面未请求 `/aweme/detail/` 时,从服务端渲染的 React Flight 状态读取
100
+ 作品与图片信息。
101
+ - 小红书视频解析兼容 `EF4`、`EF5`、`EF6` 等不透明流分组以及 snake/camel URL 字段。
102
+ - 下载结果返回最终路径、字节数、Content-Type 和 SHA-256;默认避免覆盖同名文件。
103
+
104
+ ### 安全性
105
+
106
+ - 平台请求继续由页面自身生成签名,Browser MCP 不复制签名算法,也不返回或持久化 Cookie。
107
+ - 媒体下载限制为平台 CDN,并对每次重定向执行公共地址校验,拒绝 HTML/JSON 伪媒体响应。
108
+ - 下载采用流式大小限制、同目录 `.part` 临时文件和原子落盘,失败时不发布不完整文件。
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 ywleeo
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,393 @@
1
+ Metadata-Version: 2.5
2
+ Name: ai-browser-mcp
3
+ Version: 0.12.0
4
+ Summary: Local MCP server for reading and interacting with web pages through real Chrome.
5
+ Author: ywleeo
6
+ License: MIT
7
+ License-File: LICENSE
8
+ Requires-Python: >=3.12
9
+ Requires-Dist: httpx<1,>=0.28
10
+ Requires-Dist: mcp==2.0.0
11
+ Requires-Dist: platformdirs<5,>=4.4
12
+ Requires-Dist: pydantic<3,>=2.12
13
+ Requires-Dist: readability-lxml<0.9,>=0.8.4.1
14
+ Requires-Dist: websockets<16,>=15
15
+ Description-Content-Type: text/markdown
16
+
17
+ # Browser MCP
18
+
19
+ [![M8ven Verified](https://m8ven.ai/badge/mcp/ywleeo-browser-mcp-1fpz06)](https://m8ven.ai/mcp/ywleeo-browser-mcp-1fpz06)
20
+
21
+ [English](README.en.md) · [中文](README.md)
22
+
23
+ > **Let your AI really search the whole web.**
24
+ > Any site you can reach with a login — Xiaohongshu, Zhihu, X, Douyin, Bilibili, Reddit, or any
25
+ > logged-in site — its content can be captured, and it can be **automated** in the background: no
26
+ > headless browser, no reverse-engineering, no cookie theft.
27
+
28
+ Browser MCP is a local MCP server. It lets **any MCP-capable AI assistant**, inside the session state of
29
+ your **real Chrome**:
30
+
31
+ - **Search and capture the content of any site on the web** — public pages, JavaScript-rendered pages,
32
+ and pages that are **only visible after login**;
33
+ - **Automate interactions in a background Chrome window** — click, scroll, type, press keys, select —
34
+ **without switching you away from the page you're on**.
35
+
36
+ | Built-in search / scraper | **Browser MCP** |
37
+ | --- | --- |
38
+ | Only gets what a search engine has or what's publicly crawlable | **The whole web — if you can log in and reach it, it can be captured** |
39
+ | Can't reach logged-in content | **Uses your logged-in state and takes it directly** |
40
+ | Mostly read-only | **Can click, scroll, type, press keys, select** |
41
+ | Breaks because it depends on reverse-engineered APIs the moment the site changes | **Drives the real UI and reads the real rendered DOM — resilient to redesigns** |
42
+ | Easily blocked by anti-bot, leaks cookies | **No reverse-engineering, extension never requests `cookies`, downloads verified by SHA-256, side-effect actions need confirmation** |
43
+ | Only works with a couple of agents | **Standard MCP: Codex / Claude Desktop / Cursor / Claude Code… all work** |
44
+
45
+ **Fully compliant**: it doesn't reverse-engineer internal APIs, bypass CAPTCHAs, steal cookies, or
46
+ bulk-scrape — it behaves exactly like a normal logged-in user browsing. Your login state is only used
47
+ normally inside your own Chrome; it's never sent back or returned through MCP results.
48
+
49
+ ## What you can do
50
+
51
+ In one line: **if you can reach it with a login, it can be captured.** Your AI can search, collect, and
52
+ operate on any site — Zhihu, Xiaohongshu, X, Douyin, Bilibili, Reddit and beyond — using your logged-in
53
+ state to get the platform's own data.
54
+
55
+ - **Search directly inside the real platforms**: Zhihu, Xiaohongshu, X, Douyin, Bilibili, Reddit — with your logged-in state, getting the platforms' own data, not your agent's built-in search.
56
+ - **Read any page**: article body, visible text, JS-rendered content, data returned by page requests, and content that's **only visible after login**.
57
+ - **Drive a page in the background**: get a screenshot plus numbered actionable elements in a background Chrome window, then keep clicking, scrolling, typing, pressing keys, and selecting options — without leaving the tab you're on.
58
+ - **Zhihu**: search, questions, answers, articles, answer invitations.
59
+ - **Xiaohongshu**: search, an account's posts, note details, **full comments (resumable)**, like/collect, image & video download.
60
+ - **Douyin**: search, video/image-post details, **full comments (resumable)**, like/collect, image & video download.
61
+ - **Bilibili**: video search, content metadata, multi-part info, video or audio-only download.
62
+ - **X (Twitter)**: post search, post details.
63
+ - **Reddit**: post search, post details, comments.
64
+ - **Search**: Google, Bing, Sogou.
65
+
66
+ Current version is `0.12.0`. See [CHANGELOG.md](CHANGELOG.md) for release notes.
67
+
68
+ Final actions that affect external state — like, collect, publish, send, buy, delete — should be
69
+ confirmed with the user before executing. The extension uses the current Chrome Profile's login
70
+ state but **never** returns or persists cookies over MCP.
71
+
72
+ ---
73
+
74
+ ## Quick start
75
+
76
+ ### 1. Requirements
77
+
78
+ - Python 3.12+
79
+ - [uv](https://docs.astral.sh/uv/)
80
+ - Google Chrome
81
+
82
+ ### 2. Install
83
+
84
+ ```bash
85
+ git clone https://github.com/ywleeo/browser-mcp.git
86
+ cd browser-mcp
87
+ uv sync
88
+ ```
89
+
90
+ (`/path/to/browser-mcp` below is the absolute path to this repo on your machine.)
91
+
92
+ ### 3. Load the Chrome extension
93
+
94
+ 1. Call `browser_status` and take the `extension_dir` from the result.
95
+ 2. Open `chrome://extensions`.
96
+ 3. Turn on **Developer mode**.
97
+ 4. Click **Load unpacked**.
98
+ 5. Pick the directory returned by `extension_dir`.
99
+ 6. Call `browser_status` again.
100
+
101
+ A successful connection returns:
102
+
103
+ ```json
104
+ {
105
+ "state": "connected",
106
+ "connected": true,
107
+ "bridge_port": 17880,
108
+ "server_version": "0.12.0",
109
+ "install_mode": "source",
110
+ "project_root": "/path/to/browser-mcp",
111
+ "source_commit": "<git-commit>",
112
+ "upgrade_check_command": "uv --directory /path/to/browser-mcp run browser-mcp upgrade --check --json",
113
+ "upgrade_apply_command": "uv --directory /path/to/browser-mcp run browser-mcp upgrade --apply --json"
114
+ }
115
+ ```
116
+
117
+ `bridge_port` may also fall in `17880..17889`. The extension normally only needs to be loaded once
118
+ and reconnects automatically when the MCP server starts. After updating the extension, click
119
+ **Reload** in `chrome://extensions` if it doesn't pick up automatically.
120
+
121
+ ### 4. Connect your AI client
122
+
123
+ **Codex**
124
+
125
+ ```bash
126
+ codex mcp add browser_mcp -- \
127
+ uv --directory /path/to/browser-mcp run browser-mcp
128
+ ```
129
+
130
+ Or write the Codex MCP config directly:
131
+
132
+ ```toml
133
+ [mcp_servers.browser_mcp]
134
+ command = "uv"
135
+ args = [
136
+ "--directory",
137
+ "/path/to/browser-mcp",
138
+ "run",
139
+ "browser-mcp",
140
+ ]
141
+ ```
142
+
143
+ Restart Codex after adding or editing the config. Once connected, use the [MCP tools](#mcp-tools).
144
+ To remove it: `codex mcp remove browser_mcp`.
145
+
146
+ **Claude Desktop**
147
+
148
+ Add this to your Claude Desktop config file:
149
+
150
+ ```json
151
+ {
152
+ "mcpServers": {
153
+ "browser-mcp": {
154
+ "command": "uv",
155
+ "args": [
156
+ "--directory",
157
+ "/path/to/browser-mcp",
158
+ "run",
159
+ "browser-mcp"
160
+ ]
161
+ }
162
+ }
163
+ }
164
+ ```
165
+
166
+ Save and restart Claude Desktop.
167
+
168
+ ### 5. Start using it
169
+
170
+ Once connected, just describe what you want in plain language:
171
+
172
+ > “Search Zhihu for answers about MCP.”
173
+ > “Download all images from this Xiaohongshu note to `/abs/path`.”
174
+ > “Read this Douyin video's content and comments.”
175
+ > “Open this page, fill the search box from the screenshot, and click search.”
176
+
177
+ More copy-paste examples are in [Examples](#examples).
178
+
179
+ ---
180
+
181
+ ## Reference
182
+
183
+ ### Upgrading
184
+
185
+ Source installs provide a safe upgrade command an agent can run directly. First check the version
186
+ and repo state:
187
+
188
+ ```bash
189
+ uv --directory /path/to/browser-mcp run browser-mcp upgrade --check --json
190
+ ```
191
+
192
+ Then apply if it's safe:
193
+
194
+ ```bash
195
+ uv --directory /path/to/browser-mcp run browser-mcp upgrade --apply --json
196
+ ```
197
+
198
+ The updater only accepts Git branches that have an `upstream`, and follows these safety rules:
199
+
200
+ - Refuses to upgrade if the worktree has untracked or uncommitted files.
201
+ - Refuses to upgrade if local and remote have diverged; no implicit merge commit is created.
202
+ - Only updates source with `git pull --ff-only`.
203
+ - Syncs locked dependencies with `uv sync --frozen`, never touching `uv.lock`.
204
+
205
+ After `--apply` succeeds and returns `restart_required: true`, have the client reconnect the MCP
206
+ server — start a new task or reconnect it in Codex; only restart the client if it can't reconnect on
207
+ its own. A fresh MCP server refreshes the extension bundle, and the already-loaded Chrome extension
208
+ auto-reloads by build ID.
209
+
210
+ Agents don't need to guess the project path: call `browser_status` and use the returned
211
+ `upgrade_check_command` / `upgrade_apply_command`. If installed from a wheel or another package
212
+ manager, `install_mode` is `"package"`, so upgrade with your original installer instead of mutating
213
+ an arbitrary Git repo.
214
+
215
+ ### Extension permissions
216
+
217
+ - `<all_urls>`: to open public HTTP(S) pages explicitly requested by the caller, and to support the
218
+ per-site adapters. It never crawls browsing history on its own.
219
+ - `debugger`: to capture page request responses and to send trusted browser input events in comment
220
+ streams and visual interaction.
221
+ - `tabs`, `scripting`: to manage isolated background tabs and run the bundled, fixed extraction scripts.
222
+ - `storage`, `alarms`: for local pairing config and MV3 service-worker keep-alive.
223
+
224
+ The extension does **not** request the `cookies` permission. Login state is only used normally by
225
+ the target page inside the current Chrome Profile; cookies are never returned through MCP tool
226
+ results.
227
+
228
+ Zhihu, Xiaohongshu, Douyin, X and Reddit tools check the platform login state of the current Chrome
229
+ Profile before doing anything:
230
+
231
+ - Logged in: proceed.
232
+ - Not logged in: stop and return the platform's login URL, and the client prompts the user to log in.
233
+ - Can't tell: stop, to avoid touching protected content when the login state is ambiguous.
234
+
235
+ Login state isn't cached. Once you finish logging in in Chrome, just retry the original request.
236
+
237
+ ### Media download
238
+
239
+ `xhs_download` and `douyin_download` accept these common parameters:
240
+
241
+ - `media`: choose `images`, `video`, or `all`.
242
+ - `output_dir`: optional absolute dir; defaults to `downloads` under the Browser MCP data dir.
243
+ - `overwrite`: default `false`; a same-named file gets a new name unless explicitly set.
244
+ - `max_file_mb`: per-file size cap, default 1024 MiB.
245
+
246
+ Before downloading, the tool reads the post detail through the current Chrome login state, then
247
+ validates the page-derived media URL against platform CDN allowlists, public addresses, hop-by-hop
248
+ redirects, and the response media type. Files stream to a `.part` temp file and atomically land on
249
+ disk; the result includes the final path, byte count, Content-Type, and SHA-256.
250
+
251
+ `bilibili_download_video` and `bilibili_download_audio` share the same absolute-dir, overwrite, and
252
+ size rules. Bilibili usually returns separate DASH video/audio tracks: the video tool losslessly
253
+ muxes them into MP4 via `ffmpeg` when available, and returns the two tracks as separate files when
254
+ it isn't, rather than pretending to have a complete video. The audio-only tool keeps the most
255
+ compatible track. For multipart videos, pass `?p=N` in the URL.
256
+
257
+ ### Comment completeness & resumable collection
258
+
259
+ `xhs_comments` and `douyin_comments` scroll the comment stream, expand replies, and watch the page's
260
+ own signed pagination requests. In the result, `complete` means every discovered comment stream
261
+ reached its terminal page; `limit_reached` means `max_comments` truncated the collection;
262
+ `pages_fetched` and `scrolls` help diagnose the run.
263
+
264
+ A popular post's comment stream can take minutes to scroll — longer than any MCP client is willing
265
+ to wait in one call — so a single call doesn't try to finish: when `time_budget_seconds` (default
266
+ 40s) elapses the collection **suspends instead of failing**, returning the newly gathered comments and
267
+ a `session_id`. Call again with the same `url` plus that `session_id` to resume from where it
268
+ stopped, without re-gathering already-collected comments:
269
+
270
+ - `budget_exhausted` means the run wrapped up at the budget; the data is complete and usable, just not
271
+ fully collected yet.
272
+ - A non-empty `session_id` means you can resume; an empty one means it's done (finished, hit the cap,
273
+ or the stream bottomed out).
274
+ - `collected_total` is the session's cumulative count; compare with `total` to gauge progress.
275
+ - Each call's `items` holds only the **newly** collected comments; merge them yourself.
276
+
277
+ A suspended session keeps a background collection window that closes after 5 minutes idle, after
278
+ which an old `session_id` is invalid and you must start over. Raise `time_budget_seconds` to resume
279
+ fewer times, but make sure the MCP client's single-call timeout (commonly 60s) leaves room.
280
+
281
+ ### Like & collect
282
+
283
+ `xhs_like`, `xhs_collect`, `douyin_like`, and `douyin_collect` take a post `url` and the desired
284
+ `enabled` state (default `true`). The tool reads the current state, clicks once only if it's out of
285
+ sync, then polls to verify; passing the same state again doesn't toggle it back. Pass `enabled=false`
286
+ to unlike or uncollect.
287
+
288
+ These four tools modify external account state for the current Chrome Profile. The MCP client must
289
+ get the user's explicit confirmation immediately before each call; the tool won't auto-retry a click
290
+ that didn't verify.
291
+
292
+ ### Extension troubleshooting
293
+
294
+ - `state: disconnected`: confirm the extension is enabled, then click **Reload** on its detail page.
295
+ - Port in use: the service auto-tries `17880..17889`; trust the `bridge_port` in the status result.
296
+ It monitors the real MCP host behind `uv` and frees the listening port automatically when the host
297
+ exits, so the agent doesn't have to guess at and kill other processes.
298
+ - Changed extension dir: rely on the latest `extension_dir` from `browser_status`.
299
+ - Never share `pairing.json` or `pairing-token`; they contain local connection credentials.
300
+
301
+ A special process supervisor can take the host PID via `BROWSER_MCP_OWNER_PID`; set it to `0` only to
302
+ disable host-liveness monitoring. Normal Codex, Claude, or CLI configs don't need this variable.
303
+
304
+ ### MCP tools
305
+
306
+ These tools are called automatically by any MCP-capable client. In daily use just describe your goal;
307
+ you don't need to fill in parameters by hand.
308
+
309
+ | Tool | Scope | What it does |
310
+ | --- | --- | --- |
311
+ | `browser_status` | connection | Checks whether the MCP server and Chrome extension are connected, and returns server version, install mode, source commit, and runnable upgrade commands. |
312
+ | `browser_read` | general | Opens a page in real Chrome and reads article body, visible text, JS-rendered content, and data returned by page requests; can use the current Chrome login state. |
313
+ | `browser_read_page` | general | Continues reading paginated content while keeping the same page snapshot as the first read. |
314
+ | `browser_snapshot` | interact | Opens a page in a background Chrome window that shares the current login state, without leaving the user's page; returns a viewport screenshot, visible text, and numbered actionable elements (buttons, links, inputs). With no URL, observes the current page. |
315
+ | `browser_click` | interact | Sends one trusted Chrome click by the element's number in the current screenshot first; falls back to click by pixel coordinate when no element is identifiable. Screenshot coords are mapped to the CSS viewport; pass `coordinate_space=viewport` explicitly if needed. Returns a new screenshot and element numbers. |
316
+ | `browser_scroll` | interact | Scrolls the page up/down/left/right, or brings a given element into view. |
317
+ | `browser_type` | interact | Fills, appends, or replaces text in an input or editable area and returns the resulting state; passwords never appear in the element info. |
318
+ | `browser_press` | interact | Sends common keyboard actions: Enter, Escape, Tab, arrow keys, PageUp/Down, Home, End. |
319
+ | `browser_select` | interact | Picks an option in a native dropdown and returns the resulting state. |
320
+ | `site_login_status` | login | Checks whether the current Chrome Profile is logged into Zhihu, Xiaohongshu, Douyin, X, or Reddit; only checks session state, runs no platform task, and returns no cookies. |
321
+ | `zhihu_search` | Zhihu | Searches Zhihu's combined content, answers, articles, or questions; gets titles, authors, summaries, engagement data, and original links. |
322
+ | `zhihu_content` | Zhihu | Reads the body of a Zhihu question, answer, or article — good for summarizing, extracting opinions, or deeper analysis. |
323
+ | `zhihu_invitations` | Zhihu | Lists answer invitations for the logged-in account, including inviter, related question, time, and source. |
324
+ | `xhs_search` | Xiaohongshu | Searches Xiaohongshu notes, sorted by general / newest / hottest; returns titles, authors, covers, and engagement info. |
325
+ | `xhs_note` | Xiaohongshu | Reads a single image/video note's title, body, author, publish time, engagement data, and image/video URLs. |
326
+ | `xhs_like` | Xiaohongshu | Sets a note to the desired liked/unliked state; confirm before calling; re-sending the same state doesn't toggle it back. |
327
+ | `xhs_collect` | Xiaohongshu | Sets a note to the desired collected/uncollected state; confirm before calling; re-sending the same state doesn't toggle it back. |
328
+ | `xhs_download` | Xiaohongshu | Streams a note's images, video, or all media to a local dir; defaults to `downloads` under the Browser MCP data dir, or takes an absolute path. |
329
+ | `xhs_comments` | Xiaohongshu | Scrolls the note's own comment stream and expands replies, deduping by comment ID; returns incremental comments within a time budget, with a resumable `session_id` if incomplete. |
330
+ | `xhs_user_notes` | Xiaohongshu | Lists notes published by the logged-in account or a given account; can page through and dedupe, showing title, publish time, cover, likes, pinned status, and note link. |
331
+ | `douyin_search` | Douyin | Searches Douyin video/image posts; gets ID, description, author, publish time, cover, and engagement counts. |
332
+ | `douyin_video` | Douyin | Reads a single Douyin video/image post's author, body, publish time, engagement data, media URLs, and music. |
333
+ | `douyin_like` | Douyin | Sets a post to the desired liked/unliked state; confirm before calling; re-sending the same state doesn't toggle it back. |
334
+ | `douyin_collect` | Douyin | Sets a post to the desired collected/uncollected state; confirm before calling; re-sending the same state doesn't toggle it back. |
335
+ | `douyin_download` | Douyin | Streams a video/image post's media to a local dir; supports images only, video only, or all. |
336
+ | `douyin_comments` | Douyin | Scrolls the post's actual comment stream and expands replies, deduping by comment ID; returns incremental comments within a time budget, with a resumable `session_id` if incomplete. |
337
+ | `bilibili_search` | Bilibili | Searches videos sorted by general, plays, newest, danmaku, or favorites; returns title, author, duration, tags, stats, and canonical BV links. |
338
+ | `bilibili_video` | Bilibili | Reads a BV/AV video's title, intro, author, publish time, engagement stats, tags, and all multipart info; supports `?p=N`. |
339
+ | `bilibili_download_video` | Bilibili | Downloads the best compatible video+audio for a video or part; muxes losslessly to MP4 with FFmpeg, otherwise returns two tracks explicitly. |
340
+ | `bilibili_download_audio` | Bilibili | Downloads only the best compatible audio track of a video or part, saved as a directly playable audio file. |
341
+ | `x_search` | X | Searches X posts (top or latest) and gets author, body, publish time, engagement, media, and links; uses the current Chrome X login state. |
342
+ | `x_post` | X | Reads a single X post's body, author, publish time, replies, retweets, likes, views, media, and external links. |
343
+ | `reddit_search` | Reddit | Searches Reddit posts by relevance, hot, top, new, or comments; gets subreddit, author, score, comment count, and post link. |
344
+ | `reddit_post` | Reddit | Reads a Reddit post's body/media and the comments already loaded on the page, with author, time, score, and depth. |
345
+ | `google_search` | Google | Searches the web with Google; gets titles, URLs, sites, and snippets. |
346
+ | `bing_search` | Bing | Searches the web with Bing; gets titles, target URLs, sites, and snippets. |
347
+ | `sogou_search` | Sogou | Searches the web with Sogou; returns original site links, titles, sites, and snippets; excludes Sogou's own site nav and clearly-advertised results. |
348
+ | `site_read_page` | Zhihu etc. | Continues reading paginated content for long answers/articles without revisiting the page. |
349
+
350
+ ### Examples
351
+
352
+ Just ask an MCP-capable client in plain language:
353
+
354
+ - “Read this page and summarize the key points.”
355
+ - “Open this page, fill the search box from the screenshot, and click search.”
356
+ - “Scroll down and click the contact-us button.”
357
+ - “Download all images from this Xiaohongshu note to `/abs/path`.”
358
+ - “Download this Douyin video and return the path and SHA-256.”
359
+ - “Check whether I'm logged into Xiaohongshu.”
360
+ - “Search Zhihu for answers about MCP.”
361
+ - “Read today's Zhihu answer invitations.”
362
+ - “Search Xiaohongshu for recent camping notes.”
363
+ - “Read this Xiaohongshu note's body and images.”
364
+ - “Get all comments and replies on this Xiaohongshu note.”
365
+ - “Like and collect this Xiaohongshu note.” (the client confirms before calling)
366
+ - “List every post my Xiaohongshu account has published.”
367
+ - “Search Douyin for posts about '牵手 APP'.”
368
+ - “Read this Douyin video's content and comments.”
369
+ - “Unlike this Douyin post.” (the client confirms before calling)
370
+ - “Search Bilibili for videos about OpenAI, and read the first video's meta.”
371
+ - “Download this Bilibili video, and also extract an audio-only track.”
372
+ - “Search X for the latest posts about OpenAI.”
373
+ - “Read this X post's body and engagement data.”
374
+ - “Search Reddit for high-scoring posts about MCP.”
375
+ - “Read this Reddit post and the first 20 comments.”
376
+ - “Search for Browser MCP on Google, Bing, and Sogou.”
377
+
378
+ Each page change produces a fresh set of element numbers; the agent should use the numbers from the
379
+ latest screenshot.
380
+
381
+ ### Running directly
382
+
383
+ To start the stdio server manually:
384
+
385
+ ```bash
386
+ uv run browser-mcp
387
+ ```
388
+
389
+ The process waits for MCP JSON-RPC on stdin; no output in the terminal is normal.
390
+
391
+ ## License
392
+
393
+ [MIT](LICENSE)