tiktok-mcp-server 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tiktok_mcp_server-1.0.0/.gitignore +18 -0
- tiktok_mcp_server-1.0.0/LICENSE +20 -0
- tiktok_mcp_server-1.0.0/PKG-INFO +233 -0
- tiktok_mcp_server-1.0.0/README.md +214 -0
- tiktok_mcp_server-1.0.0/pyproject.toml +50 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/__init__.py +3 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/__main__.py +6 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/app.py +52 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/browser.py +97 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/config.py +46 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/errors.py +48 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/models.py +85 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/py.typed +0 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/scraper.py +221 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/server.py +61 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/tools/__init__.py +12 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/tools/_params.py +46 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/tools/comments.py +20 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/tools/discover.py +34 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/tools/profile.py +20 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/tools/search.py +30 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/tools/transcript.py +67 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/tools/videos.py +20 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/transcribe.py +179 -0
- tiktok_mcp_server-1.0.0/src/tiktokmcp/validation.py +63 -0
- tiktok_mcp_server-1.0.0/tests/__init__.py +0 -0
- tiktok_mcp_server-1.0.0/tests/test_server.py +84 -0
- tiktok_mcp_server-1.0.0/tests/test_transcribe.py +101 -0
- tiktok_mcp_server-1.0.0/tests/test_validation.py +41 -0
- tiktok_mcp_server-1.0.0/uv.lock +1084 -0
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""MIT License
|
|
2
|
+
|
|
3
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
4
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
5
|
+
in the Software without restriction, including without limitation the rights
|
|
6
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
7
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
8
|
+
furnished to do so, subject to the following conditions:
|
|
9
|
+
|
|
10
|
+
The above copyright notice and this permission notice shall be included in all
|
|
11
|
+
copies or substantial portions of the Software.
|
|
12
|
+
|
|
13
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
14
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
15
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
16
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
17
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
18
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
19
|
+
SOFTWARE.
|
|
20
|
+
"""
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: tiktok-mcp-server
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: MCP server for TikTok data extraction, discovery, search, and transcription
|
|
5
|
+
License: MIT
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Keywords: mcp,model-context-protocol,playwright,tiktok,whisper
|
|
8
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
9
|
+
Classifier: Programming Language :: Python :: 3
|
|
10
|
+
Classifier: Typing :: Typed
|
|
11
|
+
Requires-Python: >=3.11
|
|
12
|
+
Requires-Dist: httpx>=0.27
|
|
13
|
+
Requires-Dist: imageio-ffmpeg>=0.5
|
|
14
|
+
Requires-Dist: mcp[cli]<3,>=2.2
|
|
15
|
+
Requires-Dist: playwright>=1.48
|
|
16
|
+
Requires-Dist: pydantic>=2.8
|
|
17
|
+
Requires-Dist: yt-dlp>=2024.10
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
|
|
20
|
+
# TikTok MCP Server
|
|
21
|
+
|
|
22
|
+
A [Model Context Protocol](https://modelcontextprotocol.io) server that lets an AI agent read public TikTok data: **search**, **profiles**, **videos**, **discovery**, **comments**, and **transcription**.
|
|
23
|
+
|
|
24
|
+
There is no TikTok developer account to apply for and no OAuth flow. Everything comes off public pages, so the only credential you might enter is a key for your own transcription endpoint.
|
|
25
|
+
|
|
26
|
+
It runs on any MCP client (Claude, Cursor, opencode, Codex) and returns structured output (`structuredContent` + `outputSchema`) from every tool. Transcription works against any OpenAI-compatible speech-to-text endpoint: Groq, OpenAI, or a Whisper server you host yourself.
|
|
27
|
+
|
|
28
|
+
## Quick start
|
|
29
|
+
|
|
30
|
+
Paste this into your MCP client config. There is nothing to install first: [`uvx`](https://docs.astral.sh/uv/) fetches the package and runs it on the initial launch.
|
|
31
|
+
|
|
32
|
+
```json
|
|
33
|
+
{
|
|
34
|
+
"mcpServers": {
|
|
35
|
+
"tiktok": {
|
|
36
|
+
"type": "stdio",
|
|
37
|
+
"command": "uvx",
|
|
38
|
+
"args": ["tiktok-mcp-server"],
|
|
39
|
+
"env": {
|
|
40
|
+
"TRANSCRIBE_API_URL": "https://api.groq.com/openai/v1",
|
|
41
|
+
"TRANSCRIBE_API_KEY": "your-key"
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Things worth knowing before the first call:
|
|
49
|
+
|
|
50
|
+
- The `env` block exists only for `transcribe_video`. Write `"env": {}` if you want the other five tools and nothing else. The endpoint must serve an OpenAI-compatible `POST /audio/transcriptions` route backed by a speech-to-text model such as `whisper-large-v3`.
|
|
51
|
+
- The first tool call downloads Playwright Chromium once, about 150 MB. ffmpeg ships with the package (`imageio-ffmpeg`); set `FFMPEG_PATH` if you'd rather use your own binary.
|
|
52
|
+
- You need [uv](https://docs.astral.sh/uv/getting-started/installation/) on the machine, since it provides `uvx`. Python 3.11+ comes along with it; uv installs that itself.
|
|
53
|
+
- **Linux only:** headless Chromium needs system libraries that `uvx` can't install for you. On a fresh machine or Docker image, run this once (it may prompt for `sudo`):
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
uvx --from playwright playwright install-deps chromium
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Windows and macOS don't need this step.
|
|
60
|
+
|
|
61
|
+
### Running from a checkout (before PyPI)
|
|
62
|
+
|
|
63
|
+
Not on PyPI yet? Point `uvx` at a local checkout or at the git repo (use whichever fits):
|
|
64
|
+
|
|
65
|
+
```json
|
|
66
|
+
"args": ["--from", "C:\\path\\to\\tiktokmcp", "tiktok-mcp-server"]
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
```json
|
|
70
|
+
"args": ["--from", "git+https://github.com/<you>/tiktokmcp", "tiktok-mcp-server"]
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
## MCP tools
|
|
74
|
+
|
|
75
|
+
| Tool | Description | Parameters |
|
|
76
|
+
|------|-------------|------------|
|
|
77
|
+
| `get_profile` | Profile info: bio, follower/following/like/video counts, verified status, avatar | `username` |
|
|
78
|
+
| `get_videos` | A user's recent videos with id, caption, URL, view count | `username`, `count` (default 10) |
|
|
79
|
+
| `get_comments` | Top-level comments: author, text, likes | `video_id`, `count` (default 20) |
|
|
80
|
+
| `search_videos` | Search by keyword or hashtag (`#booktok`) | `query`, `count` (default 10) |
|
|
81
|
+
| `discover_creators` | Creators posting about a topic/hashtag | `topic`, `count` (default 10) |
|
|
82
|
+
| `transcribe_video` | Download, extract audio, and transcribe via your speech-to-text API (reports progress) | `video_url`, `language` (optional) |
|
|
83
|
+
|
|
84
|
+
What the tools have in common:
|
|
85
|
+
|
|
86
|
+
- `count` is validated to 1–50 by the input schema; `username` accepts handles with or without `@`
|
|
87
|
+
- Video tools accept a full URL (including `vm.tiktok.com` / `vt.tiktok.com` short links), `@user/video/<id>`, or a bare numeric id
|
|
88
|
+
- Every tool is annotated `readOnlyHint`, `idempotentHint`, `openWorldHint`, `destructiveHint: false`
|
|
89
|
+
- Anticipated failures (bad input, user not found, missing `TRANSCRIBE_API_URL`, TikTok timeouts) come back as `isError: true` with a readable message
|
|
90
|
+
- When TikTok blocks or hides data, tools return an empty list plus a `note` instead of failing
|
|
91
|
+
- Each browser call takes several seconds; `transcribe_video` can take up to a minute. Counts like views/likes are TikTok's display strings (e.g. `1.2M`)
|
|
92
|
+
|
|
93
|
+
## Configuration
|
|
94
|
+
|
|
95
|
+
Your MCP client injects configuration as environment variables. The server never reads a `.env` file of its own.
|
|
96
|
+
|
|
97
|
+
| Variable | Required | Description |
|
|
98
|
+
|----------|----------|-------------|
|
|
99
|
+
| `TRANSCRIBE_API_URL` | For transcription | OpenAI-compatible base URL (`https://api.groq.com/openai/v1`, `https://api.openai.com/v1`, `http://localhost:8000/v1`) or the full `.../audio/transcriptions` URL |
|
|
100
|
+
| `TRANSCRIBE_API_KEY` | For transcription | API key for that endpoint (omit for keyless local servers) |
|
|
101
|
+
| `TRANSCRIBE_MODEL` | No | Model name (default `whisper-large-v3`; use `whisper-1` for OpenAI) |
|
|
102
|
+
| `FFMPEG_PATH` | No | Path to an ffmpeg binary (default: `ffmpeg` on PATH, else the bundled one) |
|
|
103
|
+
| `TIKTOK_MCP_HEADLESS` | No | `false` shows the browser while debugging (default `true`) |
|
|
104
|
+
| `TIKTOK_MCP_LOG_LEVEL` | No | `DEBUG`, `INFO` (default), `WARNING`, `ERROR` |
|
|
105
|
+
|
|
106
|
+
Built-in safety limits: 100 MB download cap, 120 s ffmpeg timeout, 600 s transcription timeout, 30 s navigation timeout.
|
|
107
|
+
|
|
108
|
+
## Client setup
|
|
109
|
+
|
|
110
|
+
Every client launches the same command, `uvx tiktok-mcp-server`. Only the config format differs.
|
|
111
|
+
|
|
112
|
+
### Claude Desktop / Claude Code / Cursor
|
|
113
|
+
|
|
114
|
+
```json
|
|
115
|
+
{
|
|
116
|
+
"mcpServers": {
|
|
117
|
+
"tiktok": {
|
|
118
|
+
"type": "stdio",
|
|
119
|
+
"command": "uvx",
|
|
120
|
+
"args": ["tiktok-mcp-server"],
|
|
121
|
+
"env": {
|
|
122
|
+
"TRANSCRIBE_API_URL": "https://api.groq.com/openai/v1",
|
|
123
|
+
"TRANSCRIBE_API_KEY": "your-key"
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
Claude Code one-liner:
|
|
131
|
+
|
|
132
|
+
```bash
|
|
133
|
+
claude mcp add tiktok -e TRANSCRIBE_API_URL=... -e TRANSCRIBE_API_KEY=... -- uvx tiktok-mcp-server
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
### opencode (`~/.config/opencode/opencode.json` → `mcp`)
|
|
137
|
+
|
|
138
|
+
```json
|
|
139
|
+
"tiktok": {
|
|
140
|
+
"type": "local",
|
|
141
|
+
"command": ["uvx", "tiktok-mcp-server"],
|
|
142
|
+
"environment": {
|
|
143
|
+
"TRANSCRIBE_API_URL": "https://api.groq.com/openai/v1",
|
|
144
|
+
"TRANSCRIBE_API_KEY": "your-key"
|
|
145
|
+
},
|
|
146
|
+
"enabled": true,
|
|
147
|
+
"timeout": 120000
|
|
148
|
+
}
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
### Codex (`~/.codex/config.toml`)
|
|
152
|
+
|
|
153
|
+
```toml
|
|
154
|
+
[mcp_servers.tiktok]
|
|
155
|
+
command = "uvx"
|
|
156
|
+
args = ["tiktok-mcp-server"]
|
|
157
|
+
|
|
158
|
+
[mcp_servers.tiktok.env]
|
|
159
|
+
TRANSCRIBE_API_URL = "https://api.groq.com/openai/v1"
|
|
160
|
+
TRANSCRIBE_API_KEY = "your-key"
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
## Running manually
|
|
164
|
+
|
|
165
|
+
```bash
|
|
166
|
+
uvx tiktok-mcp-server # stdio (what MCP clients launch)
|
|
167
|
+
uvx tiktok-mcp-server --transport streamable-http --host 127.0.0.1 --port 8000
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
Inspect it interactively with the MCP Inspector:
|
|
171
|
+
|
|
172
|
+
```bash
|
|
173
|
+
npx @modelcontextprotocol/inspector uvx tiktok-mcp-server
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
## How it works
|
|
177
|
+
|
|
178
|
+
- A real browser (**Playwright Chromium**) opens TikTok's public pages, just like you would — no API keys or developer accounts
|
|
179
|
+
- Profiles are read from the page's built-in data; search, discovery, and comments are read after the page finishes loading
|
|
180
|
+
- **Transcription**: download the video → trim it to a small audio file → send it to your speech-to-text API → return the text
|
|
181
|
+
- Logs go to stderr so the connection to your AI client stays clean
|
|
182
|
+
|
|
183
|
+
## Architecture
|
|
184
|
+
|
|
185
|
+
```
|
|
186
|
+
src/tiktokmcp/
|
|
187
|
+
├── __main__.py # python -m tiktokmcp
|
|
188
|
+
├── server.py # create_server() factory, instructions, CLI (--transport/--host/--port)
|
|
189
|
+
├── app.py # lifespan + AppContext (browser, scraper, transcriber)
|
|
190
|
+
├── config.py # Settings.from_env()
|
|
191
|
+
├── models.py # Pydantic result models -> outputSchema / structuredContent
|
|
192
|
+
├── errors.py # domain errors -> ToolError translation
|
|
193
|
+
├── validation.py # username / video-reference normalization
|
|
194
|
+
├── browser.py # BrowserManager: lazy Playwright Chromium, owned by the lifespan
|
|
195
|
+
├── scraper.py # TikTokScraper: profile JSON, video grids, search, comments
|
|
196
|
+
├── transcribe.py # Transcriber: yt-dlp -> ffmpeg -> speech-to-text API
|
|
197
|
+
└── tools/ # one module per tool, each exposing register(mcp)
|
|
198
|
+
├── _params.py # shared Annotated parameter types + ToolAnnotations
|
|
199
|
+
├── profile.py videos.py comments.py
|
|
200
|
+
└── search.py discover.py transcript.py
|
|
201
|
+
tests/ # pytest, in-memory MCP client (no network)
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
Tool functions are thin: they validate input, pull shared services from the lifespan context, and return a typed model. Scraping and transcription logic lives in services that know nothing about MCP.
|
|
205
|
+
|
|
206
|
+
## Development
|
|
207
|
+
|
|
208
|
+
```bash
|
|
209
|
+
uv sync # install deps
|
|
210
|
+
uv run playwright install chromium # browser for the scraper
|
|
211
|
+
uv run pytest # offline test suite
|
|
212
|
+
uv run ruff check src tests && uv run ruff format src tests
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
Tests use an in-memory MCP client and mock HTTP transport, so they never touch the network.
|
|
216
|
+
|
|
217
|
+
## Publishing
|
|
218
|
+
|
|
219
|
+
```bash
|
|
220
|
+
uv build
|
|
221
|
+
uv publish # needs a PyPI token; after this, `uvx tiktok-mcp-server` works anywhere
|
|
222
|
+
```
|
|
223
|
+
|
|
224
|
+
## Limitations
|
|
225
|
+
|
|
226
|
+
- Read-only: no posting, liking, or any other write actions
|
|
227
|
+
- TikTok may rate-limit or block scraping from some IPs; tools respond with an empty result + `note` rather than an error
|
|
228
|
+
- `get_comments` can return nothing when TikTok hides comments from logged-out browsers
|
|
229
|
+
- Transcription requires your own speech-to-text endpoint; nothing is relayed through third parties
|
|
230
|
+
|
|
231
|
+
## License
|
|
232
|
+
|
|
233
|
+
MIT
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
# TikTok MCP Server
|
|
2
|
+
|
|
3
|
+
A [Model Context Protocol](https://modelcontextprotocol.io) server that lets an AI agent read public TikTok data: **search**, **profiles**, **videos**, **discovery**, **comments**, and **transcription**.
|
|
4
|
+
|
|
5
|
+
There is no TikTok developer account to apply for and no OAuth flow. Everything comes off public pages, so the only credential you might enter is a key for your own transcription endpoint.
|
|
6
|
+
|
|
7
|
+
It runs on any MCP client (Claude, Cursor, opencode, Codex) and returns structured output (`structuredContent` + `outputSchema`) from every tool. Transcription works against any OpenAI-compatible speech-to-text endpoint: Groq, OpenAI, or a Whisper server you host yourself.
|
|
8
|
+
|
|
9
|
+
## Quick start
|
|
10
|
+
|
|
11
|
+
Paste this into your MCP client config. There is nothing to install first: [`uvx`](https://docs.astral.sh/uv/) fetches the package and runs it on the initial launch.
|
|
12
|
+
|
|
13
|
+
```json
|
|
14
|
+
{
|
|
15
|
+
"mcpServers": {
|
|
16
|
+
"tiktok": {
|
|
17
|
+
"type": "stdio",
|
|
18
|
+
"command": "uvx",
|
|
19
|
+
"args": ["tiktok-mcp-server"],
|
|
20
|
+
"env": {
|
|
21
|
+
"TRANSCRIBE_API_URL": "https://api.groq.com/openai/v1",
|
|
22
|
+
"TRANSCRIBE_API_KEY": "your-key"
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
Things worth knowing before the first call:
|
|
30
|
+
|
|
31
|
+
- The `env` block exists only for `transcribe_video`. Write `"env": {}` if you want the other five tools and nothing else. The endpoint must serve an OpenAI-compatible `POST /audio/transcriptions` route backed by a speech-to-text model such as `whisper-large-v3`.
|
|
32
|
+
- The first tool call downloads Playwright Chromium once, about 150 MB. ffmpeg ships with the package (`imageio-ffmpeg`); set `FFMPEG_PATH` if you'd rather use your own binary.
|
|
33
|
+
- You need [uv](https://docs.astral.sh/uv/getting-started/installation/) on the machine, since it provides `uvx`. Python 3.11+ comes along with it; uv installs that itself.
|
|
34
|
+
- **Linux only:** headless Chromium needs system libraries that `uvx` can't install for you. On a fresh machine or Docker image, run this once (it may prompt for `sudo`):
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
uvx --from playwright playwright install-deps chromium
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Windows and macOS don't need this step.
|
|
41
|
+
|
|
42
|
+
### Running from a checkout (before PyPI)
|
|
43
|
+
|
|
44
|
+
Not on PyPI yet? Point `uvx` at a local checkout or at the git repo (use whichever fits):
|
|
45
|
+
|
|
46
|
+
```json
|
|
47
|
+
"args": ["--from", "C:\\path\\to\\tiktokmcp", "tiktok-mcp-server"]
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
```json
|
|
51
|
+
"args": ["--from", "git+https://github.com/<you>/tiktokmcp", "tiktok-mcp-server"]
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## MCP tools
|
|
55
|
+
|
|
56
|
+
| Tool | Description | Parameters |
|
|
57
|
+
|------|-------------|------------|
|
|
58
|
+
| `get_profile` | Profile info: bio, follower/following/like/video counts, verified status, avatar | `username` |
|
|
59
|
+
| `get_videos` | A user's recent videos with id, caption, URL, view count | `username`, `count` (default 10) |
|
|
60
|
+
| `get_comments` | Top-level comments: author, text, likes | `video_id`, `count` (default 20) |
|
|
61
|
+
| `search_videos` | Search by keyword or hashtag (`#booktok`) | `query`, `count` (default 10) |
|
|
62
|
+
| `discover_creators` | Creators posting about a topic/hashtag | `topic`, `count` (default 10) |
|
|
63
|
+
| `transcribe_video` | Download, extract audio, and transcribe via your speech-to-text API (reports progress) | `video_url`, `language` (optional) |
|
|
64
|
+
|
|
65
|
+
What the tools have in common:
|
|
66
|
+
|
|
67
|
+
- `count` is validated to 1–50 by the input schema; `username` accepts handles with or without `@`
|
|
68
|
+
- Video tools accept a full URL (including `vm.tiktok.com` / `vt.tiktok.com` short links), `@user/video/<id>`, or a bare numeric id
|
|
69
|
+
- Every tool is annotated `readOnlyHint`, `idempotentHint`, `openWorldHint`, `destructiveHint: false`
|
|
70
|
+
- Anticipated failures (bad input, user not found, missing `TRANSCRIBE_API_URL`, TikTok timeouts) come back as `isError: true` with a readable message
|
|
71
|
+
- When TikTok blocks or hides data, tools return an empty list plus a `note` instead of failing
|
|
72
|
+
- Each browser call takes several seconds; `transcribe_video` can take up to a minute. Counts like views/likes are TikTok's display strings (e.g. `1.2M`)
|
|
73
|
+
|
|
74
|
+
## Configuration
|
|
75
|
+
|
|
76
|
+
Your MCP client injects configuration as environment variables. The server never reads a `.env` file of its own.
|
|
77
|
+
|
|
78
|
+
| Variable | Required | Description |
|
|
79
|
+
|----------|----------|-------------|
|
|
80
|
+
| `TRANSCRIBE_API_URL` | For transcription | OpenAI-compatible base URL (`https://api.groq.com/openai/v1`, `https://api.openai.com/v1`, `http://localhost:8000/v1`) or the full `.../audio/transcriptions` URL |
|
|
81
|
+
| `TRANSCRIBE_API_KEY` | For transcription | API key for that endpoint (omit for keyless local servers) |
|
|
82
|
+
| `TRANSCRIBE_MODEL` | No | Model name (default `whisper-large-v3`; use `whisper-1` for OpenAI) |
|
|
83
|
+
| `FFMPEG_PATH` | No | Path to an ffmpeg binary (default: `ffmpeg` on PATH, else the bundled one) |
|
|
84
|
+
| `TIKTOK_MCP_HEADLESS` | No | `false` shows the browser while debugging (default `true`) |
|
|
85
|
+
| `TIKTOK_MCP_LOG_LEVEL` | No | `DEBUG`, `INFO` (default), `WARNING`, `ERROR` |
|
|
86
|
+
|
|
87
|
+
Built-in safety limits: 100 MB download cap, 120 s ffmpeg timeout, 600 s transcription timeout, 30 s navigation timeout.
|
|
88
|
+
|
|
89
|
+
## Client setup
|
|
90
|
+
|
|
91
|
+
Every client launches the same command, `uvx tiktok-mcp-server`. Only the config format differs.
|
|
92
|
+
|
|
93
|
+
### Claude Desktop / Claude Code / Cursor
|
|
94
|
+
|
|
95
|
+
```json
|
|
96
|
+
{
|
|
97
|
+
"mcpServers": {
|
|
98
|
+
"tiktok": {
|
|
99
|
+
"type": "stdio",
|
|
100
|
+
"command": "uvx",
|
|
101
|
+
"args": ["tiktok-mcp-server"],
|
|
102
|
+
"env": {
|
|
103
|
+
"TRANSCRIBE_API_URL": "https://api.groq.com/openai/v1",
|
|
104
|
+
"TRANSCRIBE_API_KEY": "your-key"
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
Claude Code one-liner:
|
|
112
|
+
|
|
113
|
+
```bash
|
|
114
|
+
claude mcp add tiktok -e TRANSCRIBE_API_URL=... -e TRANSCRIBE_API_KEY=... -- uvx tiktok-mcp-server
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
### opencode (`~/.config/opencode/opencode.json` → `mcp`)
|
|
118
|
+
|
|
119
|
+
```json
|
|
120
|
+
"tiktok": {
|
|
121
|
+
"type": "local",
|
|
122
|
+
"command": ["uvx", "tiktok-mcp-server"],
|
|
123
|
+
"environment": {
|
|
124
|
+
"TRANSCRIBE_API_URL": "https://api.groq.com/openai/v1",
|
|
125
|
+
"TRANSCRIBE_API_KEY": "your-key"
|
|
126
|
+
},
|
|
127
|
+
"enabled": true,
|
|
128
|
+
"timeout": 120000
|
|
129
|
+
}
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
### Codex (`~/.codex/config.toml`)
|
|
133
|
+
|
|
134
|
+
```toml
|
|
135
|
+
[mcp_servers.tiktok]
|
|
136
|
+
command = "uvx"
|
|
137
|
+
args = ["tiktok-mcp-server"]
|
|
138
|
+
|
|
139
|
+
[mcp_servers.tiktok.env]
|
|
140
|
+
TRANSCRIBE_API_URL = "https://api.groq.com/openai/v1"
|
|
141
|
+
TRANSCRIBE_API_KEY = "your-key"
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
## Running manually
|
|
145
|
+
|
|
146
|
+
```bash
|
|
147
|
+
uvx tiktok-mcp-server # stdio (what MCP clients launch)
|
|
148
|
+
uvx tiktok-mcp-server --transport streamable-http --host 127.0.0.1 --port 8000
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
Inspect it interactively with the MCP Inspector:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
npx @modelcontextprotocol/inspector uvx tiktok-mcp-server
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
## How it works
|
|
158
|
+
|
|
159
|
+
- A real browser (**Playwright Chromium**) opens TikTok's public pages, just like you would — no API keys or developer accounts
|
|
160
|
+
- Profiles are read from the page's built-in data; search, discovery, and comments are read after the page finishes loading
|
|
161
|
+
- **Transcription**: download the video → trim it to a small audio file → send it to your speech-to-text API → return the text
|
|
162
|
+
- Logs go to stderr so the connection to your AI client stays clean
|
|
163
|
+
|
|
164
|
+
## Architecture
|
|
165
|
+
|
|
166
|
+
```
|
|
167
|
+
src/tiktokmcp/
|
|
168
|
+
├── __main__.py # python -m tiktokmcp
|
|
169
|
+
├── server.py # create_server() factory, instructions, CLI (--transport/--host/--port)
|
|
170
|
+
├── app.py # lifespan + AppContext (browser, scraper, transcriber)
|
|
171
|
+
├── config.py # Settings.from_env()
|
|
172
|
+
├── models.py # Pydantic result models -> outputSchema / structuredContent
|
|
173
|
+
├── errors.py # domain errors -> ToolError translation
|
|
174
|
+
├── validation.py # username / video-reference normalization
|
|
175
|
+
├── browser.py # BrowserManager: lazy Playwright Chromium, owned by the lifespan
|
|
176
|
+
├── scraper.py # TikTokScraper: profile JSON, video grids, search, comments
|
|
177
|
+
├── transcribe.py # Transcriber: yt-dlp -> ffmpeg -> speech-to-text API
|
|
178
|
+
└── tools/ # one module per tool, each exposing register(mcp)
|
|
179
|
+
├── _params.py # shared Annotated parameter types + ToolAnnotations
|
|
180
|
+
├── profile.py videos.py comments.py
|
|
181
|
+
└── search.py discover.py transcript.py
|
|
182
|
+
tests/ # pytest, in-memory MCP client (no network)
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
Tool functions are thin: they validate input, pull shared services from the lifespan context, and return a typed model. Scraping and transcription logic lives in services that know nothing about MCP.
|
|
186
|
+
|
|
187
|
+
## Development
|
|
188
|
+
|
|
189
|
+
```bash
|
|
190
|
+
uv sync # install deps
|
|
191
|
+
uv run playwright install chromium # browser for the scraper
|
|
192
|
+
uv run pytest # offline test suite
|
|
193
|
+
uv run ruff check src tests && uv run ruff format src tests
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
Tests use an in-memory MCP client and mock HTTP transport, so they never touch the network.
|
|
197
|
+
|
|
198
|
+
## Publishing
|
|
199
|
+
|
|
200
|
+
```bash
|
|
201
|
+
uv build
|
|
202
|
+
uv publish # needs a PyPI token; after this, `uvx tiktok-mcp-server` works anywhere
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
## Limitations
|
|
206
|
+
|
|
207
|
+
- Read-only: no posting, liking, or any other write actions
|
|
208
|
+
- TikTok may rate-limit or block scraping from some IPs; tools respond with an empty result + `note` rather than an error
|
|
209
|
+
- `get_comments` can return nothing when TikTok hides comments from logged-out browsers
|
|
210
|
+
- Transcription requires your own speech-to-text endpoint; nothing is relayed through third parties
|
|
211
|
+
|
|
212
|
+
## License
|
|
213
|
+
|
|
214
|
+
MIT
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "tiktok-mcp-server"
|
|
3
|
+
version = "1.0.0"
|
|
4
|
+
description = "MCP server for TikTok data extraction, discovery, search, and transcription"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.11"
|
|
7
|
+
license = { text = "MIT" }
|
|
8
|
+
keywords = ["mcp", "model-context-protocol", "tiktok", "playwright", "whisper"]
|
|
9
|
+
classifiers = [
|
|
10
|
+
"Programming Language :: Python :: 3",
|
|
11
|
+
"License :: OSI Approved :: MIT License",
|
|
12
|
+
"Typing :: Typed",
|
|
13
|
+
]
|
|
14
|
+
dependencies = [
|
|
15
|
+
"mcp[cli]>=2.2,<3",
|
|
16
|
+
"playwright>=1.48",
|
|
17
|
+
"pydantic>=2.8",
|
|
18
|
+
"yt-dlp>=2024.10",
|
|
19
|
+
"httpx>=0.27",
|
|
20
|
+
"imageio-ffmpeg>=0.5",
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
[project.scripts]
|
|
24
|
+
# `uvx tiktok-mcp-server` runs the script named after the package.
|
|
25
|
+
tiktok-mcp-server = "tiktokmcp.server:main"
|
|
26
|
+
tiktok-mcp = "tiktokmcp.server:main"
|
|
27
|
+
|
|
28
|
+
[build-system]
|
|
29
|
+
requires = ["hatchling"]
|
|
30
|
+
build-backend = "hatchling.build"
|
|
31
|
+
|
|
32
|
+
[tool.hatch.build.targets.wheel]
|
|
33
|
+
packages = ["src/tiktokmcp"]
|
|
34
|
+
|
|
35
|
+
[dependency-groups]
|
|
36
|
+
dev = [
|
|
37
|
+
"pytest>=8",
|
|
38
|
+
"ruff>=0.6",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
[tool.pytest.ini_options]
|
|
42
|
+
testpaths = ["tests"]
|
|
43
|
+
|
|
44
|
+
[tool.ruff]
|
|
45
|
+
line-length = 110
|
|
46
|
+
target-version = "py311"
|
|
47
|
+
src = ["src", "tests"]
|
|
48
|
+
|
|
49
|
+
[tool.ruff.lint]
|
|
50
|
+
select = ["E", "F", "I", "B", "UP", "ASYNC"]
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""Server-wide state created by the lifespan and handed to every tool call."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import AsyncIterator
|
|
6
|
+
from contextlib import asynccontextmanager
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from typing import TYPE_CHECKING, Any
|
|
9
|
+
|
|
10
|
+
from mcp.server.mcpserver import Context
|
|
11
|
+
|
|
12
|
+
from tiktokmcp.browser import BrowserManager
|
|
13
|
+
from tiktokmcp.config import Settings
|
|
14
|
+
from tiktokmcp.scraper import TikTokScraper
|
|
15
|
+
from tiktokmcp.transcribe import Transcriber
|
|
16
|
+
|
|
17
|
+
if TYPE_CHECKING:
|
|
18
|
+
from mcp.server.mcpserver import MCPServer
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True, slots=True)
|
|
22
|
+
class AppContext:
|
|
23
|
+
settings: Settings
|
|
24
|
+
browser: BrowserManager
|
|
25
|
+
scraper: TikTokScraper
|
|
26
|
+
transcriber: Transcriber
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def build_app_context(settings: Settings) -> AppContext:
|
|
30
|
+
browser = BrowserManager(headless=settings.headless)
|
|
31
|
+
return AppContext(
|
|
32
|
+
settings=settings,
|
|
33
|
+
browser=browser,
|
|
34
|
+
scraper=TikTokScraper(browser, settings),
|
|
35
|
+
transcriber=Transcriber(settings),
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def make_lifespan(settings: Settings):
|
|
40
|
+
@asynccontextmanager
|
|
41
|
+
async def lifespan(_: MCPServer[Any]) -> AsyncIterator[AppContext]:
|
|
42
|
+
app = build_app_context(settings)
|
|
43
|
+
try:
|
|
44
|
+
yield app
|
|
45
|
+
finally:
|
|
46
|
+
await app.browser.close()
|
|
47
|
+
|
|
48
|
+
return lifespan
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def get_app(ctx: Context) -> AppContext:
|
|
52
|
+
return ctx.request_context.lifespan_context
|