babelscribe 0.3.0__tar.gz → 0.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {babelscribe-0.3.0/babelscribe.egg-info → babelscribe-0.3.2}/PKG-INFO +88 -10
- babelscribe-0.3.0/PKG-INFO → babelscribe-0.3.2/README.md +68 -33
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/__init__.py +1 -1
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/cli.py +56 -52
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/mcp_server.py +139 -136
- babelscribe-0.3.0/README.md → babelscribe-0.3.2/babelscribe.egg-info/PKG-INFO +110 -8
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe.egg-info/requires.txt +4 -0
- babelscribe-0.3.2/pyproject.toml +48 -0
- babelscribe-0.3.0/pyproject.toml +0 -29
- {babelscribe-0.3.0 → babelscribe-0.3.2}/LICENSE +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/__main__.py +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/align.py +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/api.py +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/backend.py +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/hybrid.py +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/models.py +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/transcribe.py +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/writers.py +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe.egg-info/SOURCES.txt +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe.egg-info/dependency_links.txt +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe.egg-info/entry_points.txt +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe.egg-info/top_level.txt +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/setup.cfg +0 -0
- {babelscribe-0.3.0 → babelscribe-0.3.2}/tests/test_core.py +0 -0
|
@@ -1,15 +1,34 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: babelscribe
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.2
|
|
4
4
|
Summary: Transcribe any audio/video in 99 languages on any GPU (AMD, NVIDIA, Intel via Vulkan; NVIDIA via CUDA; Apple via Metal) or CPU — whisper.cpp with batteries included.
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/phonology024/babelscribe
|
|
7
|
+
Project-URL: Documentation, https://phonology024.github.io/babelscribe/
|
|
8
|
+
Project-URL: Repository, https://github.com/phonology024/babelscribe
|
|
7
9
|
Project-URL: Issues, https://github.com/phonology024/babelscribe/issues
|
|
8
|
-
Keywords: whisper,speech-to-text,transcription,subtitles,vulkan,amd,gpu,thai,mcp,claude,codex
|
|
10
|
+
Keywords: whisper,speech-to-text,transcription,subtitles,vulkan,amd,gpu,thai,mcp,mcp-server,claude,codex,srt,speech-recognition,offline,radeon,intel-arc,openai-whisper
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Environment :: GPU
|
|
14
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
18
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
19
|
+
Classifier: Operating System :: MacOS
|
|
20
|
+
Classifier: Programming Language :: Python :: 3
|
|
21
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
22
|
+
Classifier: Topic :: Multimedia :: Video
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Classifier: Natural Language :: Thai
|
|
25
|
+
Classifier: Natural Language :: English
|
|
9
26
|
Requires-Python: >=3.9
|
|
10
27
|
Description-Content-Type: text/markdown
|
|
11
28
|
License-File: LICENSE
|
|
12
29
|
Requires-Dist: imageio-ffmpeg>=0.5
|
|
30
|
+
Requires-Dist: mcp<3,>=2.3; python_version >= "3.10"
|
|
31
|
+
Requires-Dist: anyio>=4; python_version >= "3.10"
|
|
13
32
|
Provides-Extra: finetune
|
|
14
33
|
Requires-Dist: torch; extra == "finetune"
|
|
15
34
|
Requires-Dist: transformers; extra == "finetune"
|
|
@@ -22,10 +41,18 @@ Requires-Dist: mcp<3,>=2.3; python_version >= "3.10" and extra == "mcp"
|
|
|
22
41
|
Requires-Dist: anyio>=4; extra == "mcp"
|
|
23
42
|
Dynamic: license-file
|
|
24
43
|
|
|
25
|
-
# babelscribe
|
|
44
|
+
# babelscribe — free, offline speech-to-text and subtitles on any GPU (AMD, NVIDIA, Intel, Apple)
|
|
26
45
|
|
|
27
|
-
|
|
28
|
-
|
|
46
|
+
<!-- mcp-name: io.github.phonology024/babelscribe -->
|
|
47
|
+
|
|
48
|
+
[](https://pypi.org/project/babelscribe/)
|
|
49
|
+
[](LICENSE)
|
|
50
|
+
[](#use-it-from-an-ai-app-mcp--no-terminal-needed)
|
|
51
|
+
|
|
52
|
+
**babelscribe turns any audio or video file into subtitles (SRT, VTT) and text, in any of Whisper's 99 languages, on your
|
|
53
|
+
own graphics card — AMD Radeon, NVIDIA, Intel Arc or Apple Silicon — or the CPU. No CUDA needed, no cloud, free (MIT).**
|
|
54
|
+
It runs OpenAI Whisper through [whisper.cpp](https://github.com/ggml-org/whisper.cpp) with Vulkan, CUDA or Metal, and works
|
|
55
|
+
from the command line, from Python, or from AI apps (Claude, Codex, Antigravity, Gemini CLI, Cursor) as an MCP server.
|
|
29
56
|
|
|
30
57
|
```bash
|
|
31
58
|
pip install babelscribe # Python 3.9+
|
|
@@ -60,12 +87,12 @@ Tools: `transcribe` (file → subtitles + text), `find_media` (newest audio/vide
|
|
|
60
87
|
and double-click it (or *Settings → Extensions → Install extension*).
|
|
61
88
|
|
|
62
89
|
**Everything else** runs the same command — [uv](https://docs.astral.sh/uv/) fetches babelscribe for you:
|
|
63
|
-
`uvx
|
|
90
|
+
`uvx babelscribe mcp`
|
|
64
91
|
|
|
65
92
|
| App | How to add it |
|
|
66
93
|
|---|---|
|
|
67
|
-
| Claude Code | `claude mcp add babelscribe -- uvx
|
|
68
|
-
| OpenAI Codex CLI | `codex mcp add babelscribe -- uvx
|
|
94
|
+
| Claude Code | `claude mcp add babelscribe -- uvx babelscribe mcp` |
|
|
95
|
+
| OpenAI Codex CLI | `codex mcp add babelscribe -- uvx babelscribe mcp`, then in `~/.codex/config.toml` under `[mcp_servers.babelscribe]` set `tool_timeout_sec = 3600` and `startup_timeout_sec = 120` (defaults are 60 s / 10 s — too short for a long video or the first model download) |
|
|
69
96
|
| Google Antigravity | agent panel → ⋯ → *MCP Servers* → *Manage MCP Servers* → *View raw config*, add the JSON below |
|
|
70
97
|
| Gemini CLI | add the JSON below to `~/.gemini/settings.json` |
|
|
71
98
|
| Cursor | add the JSON below to `~/.cursor/mcp.json` |
|
|
@@ -74,11 +101,11 @@ and double-click it (or *Settings → Extensions → Install extension*).
|
|
|
74
101
|
```json
|
|
75
102
|
{
|
|
76
103
|
"mcpServers": {
|
|
77
|
-
"babelscribe": { "command": "uvx", "args": ["
|
|
104
|
+
"babelscribe": { "command": "uvx", "args": ["babelscribe", "mcp"], "timeout": 3600000 }
|
|
78
105
|
}
|
|
79
106
|
}
|
|
80
107
|
```
|
|
81
|
-
(`timeout` is in milliseconds and only some apps read it.) Already installed with pip? Use `"command": "babelscribe
|
|
108
|
+
(`timeout` is in milliseconds and only some apps read it.) Already installed with pip? Use `"command": "babelscribe", "args": ["mcp"]`.
|
|
82
109
|
Web-only chat apps (e.g. grok.com, chatgpt.com) can only reach servers on the internet, not your computer, so they can't use your GPU or files.
|
|
83
110
|
|
|
84
111
|
## Benchmark: real talks, human captions as the answer key
|
|
@@ -142,6 +169,32 @@ babelscribe does **not** replace those projects — it stands on whisper.cpp and
|
|
|
142
169
|
compiling for your GPU, converting media, picking the right device, avoiding the long-file repeat bug, and combining
|
|
143
170
|
a language-specific fine-tune with accurate timestamps.
|
|
144
171
|
|
|
172
|
+
## FAQ
|
|
173
|
+
**How do I run Whisper on an AMD GPU on Windows?**
|
|
174
|
+
`pip install babelscribe`, then `babelscribe video.mp4`. It downloads a Vulkan build of whisper.cpp that runs on AMD Radeon
|
|
175
|
+
(and Intel Arc / NVIDIA) cards on Windows and Linux — no ROCm, no CUDA, no compiling.
|
|
176
|
+
|
|
177
|
+
**How do I make subtitles (SRT) from a video for free, offline?**
|
|
178
|
+
`babelscribe video.mp4 -f srt` writes `video.srt` next to the video. Nothing is uploaded; it runs on your own computer.
|
|
179
|
+
|
|
180
|
+
**What is the most accurate free transcription for Thai?**
|
|
181
|
+
`babelscribe video.mp4 -l th --accurate` — Pathumma Whisper (NECTEC) for the text plus Whisper turbo for timing:
|
|
182
|
+
CER 8.9% on Google FLEURS vs 15.9% for plain Whisper turbo. Thonburian Whisper is available too (`--text-model thai-thonburian`).
|
|
183
|
+
|
|
184
|
+
**Can Claude / ChatGPT Codex / Gemini transcribe a video on my computer?**
|
|
185
|
+
Yes — add babelscribe as an MCP server (see *Use it from an AI app*). Claude Desktop installs it with one click from
|
|
186
|
+
`babelscribe.mcpb`. The AI app can then find a file, transcribe it on your GPU, and proofread, translate or summarise the text.
|
|
187
|
+
|
|
188
|
+
**How fast is it?**
|
|
189
|
+
About 20x real time with Whisper large-v3-turbo on an AMD Radeon RX 9070 XT: a 19-minute talk in 48 seconds.
|
|
190
|
+
|
|
191
|
+
**Which languages are supported?**
|
|
192
|
+
All 99 Whisper languages, with automatic language detection. FLEURS error rates for 16 of them are in the benchmark table.
|
|
193
|
+
|
|
194
|
+
**Is it better than faster-whisper or WhisperX?**
|
|
195
|
+
Those are excellent on NVIDIA GPUs; on AMD / Intel GPUs they fall back to the CPU. babelscribe's niche is any GPU, zero setup,
|
|
196
|
+
and better Thai / Hindi through community fine-tunes. If you have an NVIDIA card and like Python, faster-whisper is a fine choice.
|
|
197
|
+
|
|
145
198
|
## Languages
|
|
146
199
|
All 99 languages Whisper was trained on, auto-detected or forced with `-l`:
|
|
147
200
|
af am ar as az ba be bg bn bo br bs ca cs cy da de el en es et eu fa fi fo fr gl gu ha haw he hi hr ht hu hy id is it ja jw ka kk km kn ko la lb ln lo lt lv mg mi mk ml mn mr ms mt my ne nl nn no oc pa pl ps pt ro ru sa sd si sk sl sn so sq sr su sv sw ta te tg th tk tl tr tt uk ur uz vi yi yo yue zh
|
|
@@ -175,6 +228,12 @@ babelscribe devices # GPUs whisper.cpp can see
|
|
|
175
228
|
babelscribe models # models and fine-tunes
|
|
176
229
|
babelscribe talk.mp4 --bin /path/to/whisper-cli # use your own whisper.cpp build
|
|
177
230
|
```
|
|
231
|
+
From Python:
|
|
232
|
+
```python
|
|
233
|
+
from babelscribe.api import transcribe_file
|
|
234
|
+
r = transcribe_file("talk.mp4", lang="auto", formats=["srt", "txt"]) # or accurate=True
|
|
235
|
+
print(r["lang"], r["device"], r["files"]); print(r["text"][:200])
|
|
236
|
+
```
|
|
178
237
|
Models download on first use to `~/.babelscribe/models` (`BABELSCRIBE_MODELS` to change). Hybrid fine-tunes are converted on your
|
|
179
238
|
machine from their original Hugging Face repo — install `pip install "babelscribe[finetune]"` once; converted weights are never
|
|
180
239
|
redistributed, so each fine-tune keeps its own licence. Thai word boundaries: `pip install "babelscribe[thai]"`.
|
|
@@ -192,6 +251,25 @@ them to each `v*` release. Point `BABELSCRIBE_RELEASES` at another URL to self-h
|
|
|
192
251
|
ภาษาไทยแนะนำ `babelscribe ไฟล์.mp4 -l th --accurate` — ข้อความจาก Pathumma Whisper (NECTEC) ที่ผิดน้อยที่สุดใน FLEURS (CER 8.9% เทียบ turbo 15.9%) + เวลาจาก large-v3-turbo
|
|
193
252
|
หรือเลือก Thonburian Whisper เอง: `--text-model thai-thonburian`
|
|
194
253
|
|
|
254
|
+
## Privacy Policy
|
|
255
|
+
babelscribe runs entirely on your own computer. Last updated 2026-10-06.
|
|
256
|
+
|
|
257
|
+
- **Data collection:** none. babelscribe has no telemetry, analytics, accounts or crash reporting, and never uploads your
|
|
258
|
+
audio, video, transcripts or file names anywhere.
|
|
259
|
+
- **What it processes and where:** the media file you choose is converted and transcribed locally; subtitles and text are
|
|
260
|
+
written to your disk (next to the file, or the folder you choose). Through MCP, the transcript text is returned to the AI
|
|
261
|
+
app that called the tool — what that app does with it is governed by that app's own privacy policy.
|
|
262
|
+
- **Network access (downloads only):** on first use it downloads the `whisper-cli` program from this project's GitHub
|
|
263
|
+
releases, speech models from Hugging Face (`huggingface.co/ggerganov/whisper.cpp`, and for `--accurate` Thai/Hindi the
|
|
264
|
+
fine-tune's own repository), and Python packages from PyPI when installed with pip/uv. These are plain downloads;
|
|
265
|
+
no personal data is sent. GitHub, Hugging Face and PyPI see a normal download request (IP address, user agent) under
|
|
266
|
+
their own privacy policies.
|
|
267
|
+
- **Storage and retention:** downloaded programs and models are cached in `~/.babelscribe` (or `BABELSCRIBE_HOME` /
|
|
268
|
+
`BABELSCRIBE_MODELS`) until you delete that folder. Outputs stay wherever they were written until you delete them.
|
|
269
|
+
babelscribe keeps no other data.
|
|
270
|
+
- **Third-party sharing:** none.
|
|
271
|
+
- **Contact:** open an issue at https://github.com/phonology024/babelscribe/issues
|
|
272
|
+
|
|
195
273
|
## Credits & licence
|
|
196
274
|
MIT. Built on [whisper.cpp](https://github.com/ggml-org/whisper.cpp) (MIT) and OpenAI Whisper models (MIT).
|
|
197
275
|
Fine-tunes belong to their authors: [Pathumma Whisper](https://huggingface.co/nectec/Pathumma-whisper-th-large-v3) by NECTEC,
|
|
@@ -1,31 +1,15 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
Provides-Extra: finetune
|
|
14
|
-
Requires-Dist: torch; extra == "finetune"
|
|
15
|
-
Requires-Dist: transformers; extra == "finetune"
|
|
16
|
-
Requires-Dist: huggingface_hub; extra == "finetune"
|
|
17
|
-
Requires-Dist: numpy; extra == "finetune"
|
|
18
|
-
Provides-Extra: thai
|
|
19
|
-
Requires-Dist: pythainlp; extra == "thai"
|
|
20
|
-
Provides-Extra: mcp
|
|
21
|
-
Requires-Dist: mcp<3,>=2.3; python_version >= "3.10" and extra == "mcp"
|
|
22
|
-
Requires-Dist: anyio>=4; extra == "mcp"
|
|
23
|
-
Dynamic: license-file
|
|
24
|
-
|
|
25
|
-
# babelscribe
|
|
26
|
-
|
|
27
|
-
**Transcribe any audio or video, in any of Whisper's 99 languages, on any GPU — AMD, NVIDIA, Intel or Apple — or just the CPU.**
|
|
28
|
-
One command, no CUDA required, subtitles out.
|
|
1
|
+
# babelscribe — free, offline speech-to-text and subtitles on any GPU (AMD, NVIDIA, Intel, Apple)
|
|
2
|
+
|
|
3
|
+
<!-- mcp-name: io.github.phonology024/babelscribe -->
|
|
4
|
+
|
|
5
|
+
[](https://pypi.org/project/babelscribe/)
|
|
6
|
+
[](LICENSE)
|
|
7
|
+
[](#use-it-from-an-ai-app-mcp--no-terminal-needed)
|
|
8
|
+
|
|
9
|
+
**babelscribe turns any audio or video file into subtitles (SRT, VTT) and text, in any of Whisper's 99 languages, on your
|
|
10
|
+
own graphics card — AMD Radeon, NVIDIA, Intel Arc or Apple Silicon — or the CPU. No CUDA needed, no cloud, free (MIT).**
|
|
11
|
+
It runs OpenAI Whisper through [whisper.cpp](https://github.com/ggml-org/whisper.cpp) with Vulkan, CUDA or Metal, and works
|
|
12
|
+
from the command line, from Python, or from AI apps (Claude, Codex, Antigravity, Gemini CLI, Cursor) as an MCP server.
|
|
29
13
|
|
|
30
14
|
```bash
|
|
31
15
|
pip install babelscribe # Python 3.9+
|
|
@@ -60,12 +44,12 @@ Tools: `transcribe` (file → subtitles + text), `find_media` (newest audio/vide
|
|
|
60
44
|
and double-click it (or *Settings → Extensions → Install extension*).
|
|
61
45
|
|
|
62
46
|
**Everything else** runs the same command — [uv](https://docs.astral.sh/uv/) fetches babelscribe for you:
|
|
63
|
-
`uvx
|
|
47
|
+
`uvx babelscribe mcp`
|
|
64
48
|
|
|
65
49
|
| App | How to add it |
|
|
66
50
|
|---|---|
|
|
67
|
-
| Claude Code | `claude mcp add babelscribe -- uvx
|
|
68
|
-
| OpenAI Codex CLI | `codex mcp add babelscribe -- uvx
|
|
51
|
+
| Claude Code | `claude mcp add babelscribe -- uvx babelscribe mcp` |
|
|
52
|
+
| OpenAI Codex CLI | `codex mcp add babelscribe -- uvx babelscribe mcp`, then in `~/.codex/config.toml` under `[mcp_servers.babelscribe]` set `tool_timeout_sec = 3600` and `startup_timeout_sec = 120` (defaults are 60 s / 10 s — too short for a long video or the first model download) |
|
|
69
53
|
| Google Antigravity | agent panel → ⋯ → *MCP Servers* → *Manage MCP Servers* → *View raw config*, add the JSON below |
|
|
70
54
|
| Gemini CLI | add the JSON below to `~/.gemini/settings.json` |
|
|
71
55
|
| Cursor | add the JSON below to `~/.cursor/mcp.json` |
|
|
@@ -74,11 +58,11 @@ and double-click it (or *Settings → Extensions → Install extension*).
|
|
|
74
58
|
```json
|
|
75
59
|
{
|
|
76
60
|
"mcpServers": {
|
|
77
|
-
"babelscribe": { "command": "uvx", "args": ["
|
|
61
|
+
"babelscribe": { "command": "uvx", "args": ["babelscribe", "mcp"], "timeout": 3600000 }
|
|
78
62
|
}
|
|
79
63
|
}
|
|
80
64
|
```
|
|
81
|
-
(`timeout` is in milliseconds and only some apps read it.) Already installed with pip? Use `"command": "babelscribe
|
|
65
|
+
(`timeout` is in milliseconds and only some apps read it.) Already installed with pip? Use `"command": "babelscribe", "args": ["mcp"]`.
|
|
82
66
|
Web-only chat apps (e.g. grok.com, chatgpt.com) can only reach servers on the internet, not your computer, so they can't use your GPU or files.
|
|
83
67
|
|
|
84
68
|
## Benchmark: real talks, human captions as the answer key
|
|
@@ -142,6 +126,32 @@ babelscribe does **not** replace those projects — it stands on whisper.cpp and
|
|
|
142
126
|
compiling for your GPU, converting media, picking the right device, avoiding the long-file repeat bug, and combining
|
|
143
127
|
a language-specific fine-tune with accurate timestamps.
|
|
144
128
|
|
|
129
|
+
## FAQ
|
|
130
|
+
**How do I run Whisper on an AMD GPU on Windows?**
|
|
131
|
+
`pip install babelscribe`, then `babelscribe video.mp4`. It downloads a Vulkan build of whisper.cpp that runs on AMD Radeon
|
|
132
|
+
(and Intel Arc / NVIDIA) cards on Windows and Linux — no ROCm, no CUDA, no compiling.
|
|
133
|
+
|
|
134
|
+
**How do I make subtitles (SRT) from a video for free, offline?**
|
|
135
|
+
`babelscribe video.mp4 -f srt` writes `video.srt` next to the video. Nothing is uploaded; it runs on your own computer.
|
|
136
|
+
|
|
137
|
+
**What is the most accurate free transcription for Thai?**
|
|
138
|
+
`babelscribe video.mp4 -l th --accurate` — Pathumma Whisper (NECTEC) for the text plus Whisper turbo for timing:
|
|
139
|
+
CER 8.9% on Google FLEURS vs 15.9% for plain Whisper turbo. Thonburian Whisper is available too (`--text-model thai-thonburian`).
|
|
140
|
+
|
|
141
|
+
**Can Claude / ChatGPT Codex / Gemini transcribe a video on my computer?**
|
|
142
|
+
Yes — add babelscribe as an MCP server (see *Use it from an AI app*). Claude Desktop installs it with one click from
|
|
143
|
+
`babelscribe.mcpb`. The AI app can then find a file, transcribe it on your GPU, and proofread, translate or summarise the text.
|
|
144
|
+
|
|
145
|
+
**How fast is it?**
|
|
146
|
+
About 20x real time with Whisper large-v3-turbo on an AMD Radeon RX 9070 XT: a 19-minute talk in 48 seconds.
|
|
147
|
+
|
|
148
|
+
**Which languages are supported?**
|
|
149
|
+
All 99 Whisper languages, with automatic language detection. FLEURS error rates for 16 of them are in the benchmark table.
|
|
150
|
+
|
|
151
|
+
**Is it better than faster-whisper or WhisperX?**
|
|
152
|
+
Those are excellent on NVIDIA GPUs; on AMD / Intel GPUs they fall back to the CPU. babelscribe's niche is any GPU, zero setup,
|
|
153
|
+
and better Thai / Hindi through community fine-tunes. If you have an NVIDIA card and like Python, faster-whisper is a fine choice.
|
|
154
|
+
|
|
145
155
|
## Languages
|
|
146
156
|
All 99 languages Whisper was trained on, auto-detected or forced with `-l`:
|
|
147
157
|
af am ar as az ba be bg bn bo br bs ca cs cy da de el en es et eu fa fi fo fr gl gu ha haw he hi hr ht hu hy id is it ja jw ka kk km kn ko la lb ln lo lt lv mg mi mk ml mn mr ms mt my ne nl nn no oc pa pl ps pt ro ru sa sd si sk sl sn so sq sr su sv sw ta te tg th tk tl tr tt uk ur uz vi yi yo yue zh
|
|
@@ -175,6 +185,12 @@ babelscribe devices # GPUs whisper.cpp can see
|
|
|
175
185
|
babelscribe models # models and fine-tunes
|
|
176
186
|
babelscribe talk.mp4 --bin /path/to/whisper-cli # use your own whisper.cpp build
|
|
177
187
|
```
|
|
188
|
+
From Python:
|
|
189
|
+
```python
|
|
190
|
+
from babelscribe.api import transcribe_file
|
|
191
|
+
r = transcribe_file("talk.mp4", lang="auto", formats=["srt", "txt"]) # or accurate=True
|
|
192
|
+
print(r["lang"], r["device"], r["files"]); print(r["text"][:200])
|
|
193
|
+
```
|
|
178
194
|
Models download on first use to `~/.babelscribe/models` (`BABELSCRIBE_MODELS` to change). Hybrid fine-tunes are converted on your
|
|
179
195
|
machine from their original Hugging Face repo — install `pip install "babelscribe[finetune]"` once; converted weights are never
|
|
180
196
|
redistributed, so each fine-tune keeps its own licence. Thai word boundaries: `pip install "babelscribe[thai]"`.
|
|
@@ -192,6 +208,25 @@ them to each `v*` release. Point `BABELSCRIBE_RELEASES` at another URL to self-h
|
|
|
192
208
|
ภาษาไทยแนะนำ `babelscribe ไฟล์.mp4 -l th --accurate` — ข้อความจาก Pathumma Whisper (NECTEC) ที่ผิดน้อยที่สุดใน FLEURS (CER 8.9% เทียบ turbo 15.9%) + เวลาจาก large-v3-turbo
|
|
193
209
|
หรือเลือก Thonburian Whisper เอง: `--text-model thai-thonburian`
|
|
194
210
|
|
|
211
|
+
## Privacy Policy
|
|
212
|
+
babelscribe runs entirely on your own computer. Last updated 2026-10-06.
|
|
213
|
+
|
|
214
|
+
- **Data collection:** none. babelscribe has no telemetry, analytics, accounts or crash reporting, and never uploads your
|
|
215
|
+
audio, video, transcripts or file names anywhere.
|
|
216
|
+
- **What it processes and where:** the media file you choose is converted and transcribed locally; subtitles and text are
|
|
217
|
+
written to your disk (next to the file, or the folder you choose). Through MCP, the transcript text is returned to the AI
|
|
218
|
+
app that called the tool — what that app does with it is governed by that app's own privacy policy.
|
|
219
|
+
- **Network access (downloads only):** on first use it downloads the `whisper-cli` program from this project's GitHub
|
|
220
|
+
releases, speech models from Hugging Face (`huggingface.co/ggerganov/whisper.cpp`, and for `--accurate` Thai/Hindi the
|
|
221
|
+
fine-tune's own repository), and Python packages from PyPI when installed with pip/uv. These are plain downloads;
|
|
222
|
+
no personal data is sent. GitHub, Hugging Face and PyPI see a normal download request (IP address, user agent) under
|
|
223
|
+
their own privacy policies.
|
|
224
|
+
- **Storage and retention:** downloaded programs and models are cached in `~/.babelscribe` (or `BABELSCRIBE_HOME` /
|
|
225
|
+
`BABELSCRIBE_MODELS`) until you delete that folder. Outputs stay wherever they were written until you delete them.
|
|
226
|
+
babelscribe keeps no other data.
|
|
227
|
+
- **Third-party sharing:** none.
|
|
228
|
+
- **Contact:** open an issue at https://github.com/phonology024/babelscribe/issues
|
|
229
|
+
|
|
195
230
|
## Credits & licence
|
|
196
231
|
MIT. Built on [whisper.cpp](https://github.com/ggml-org/whisper.cpp) (MIT) and OpenAI Whisper models (MIT).
|
|
197
232
|
Fine-tunes belong to their authors: [Pathumma Whisper](https://huggingface.co/nectec/Pathumma-whisper-th-large-v3) by NECTEC,
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
"""babelscribe — any language, any GPU speech-to-text built on whisper.cpp."""
|
|
2
|
-
__version__ = "0.3.
|
|
2
|
+
__version__ = "0.3.2"
|
|
@@ -1,52 +1,56 @@
|
|
|
1
|
-
"""babelscribe — transcribe any audio/video, in any language Whisper knows, on any GPU (AMD / NVIDIA / Intel via
|
|
2
|
-
Vulkan, NVIDIA via CUDA, Apple via Metal) or the CPU.
|
|
3
|
-
|
|
4
|
-
babelscribe talk.mp4 # auto language, turbo model, best GPU -> talk.srt + talk.json
|
|
5
|
-
babelscribe vo.wav -l th --text-model thai-thonburian # hybrid: Thai fine-tune text + turbo timing
|
|
6
|
-
babelscribe talk.mp4 --accurate # slower, fewest errors: large-v3 + beam search, or the best fine-tune
|
|
7
|
-
babelscribe
|
|
8
|
-
babelscribe
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
import
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
ap.
|
|
29
|
-
ap.add_argument("
|
|
30
|
-
|
|
31
|
-
ap.add_argument("-
|
|
32
|
-
ap.add_argument("-
|
|
33
|
-
ap.add_argument("--
|
|
34
|
-
|
|
35
|
-
ap.add_argument("--
|
|
36
|
-
ap.add_argument("-
|
|
37
|
-
ap.add_argument("--
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
1
|
+
"""babelscribe — transcribe any audio/video, in any language Whisper knows, on any GPU (AMD / NVIDIA / Intel via
|
|
2
|
+
Vulkan, NVIDIA via CUDA, Apple via Metal) or the CPU.
|
|
3
|
+
|
|
4
|
+
babelscribe talk.mp4 # auto language, turbo model, best GPU -> talk.srt + talk.json
|
|
5
|
+
babelscribe vo.wav -l th --text-model thai-thonburian # hybrid: Thai fine-tune text + turbo timing
|
|
6
|
+
babelscribe talk.mp4 --accurate # slower, fewest errors: large-v3 + beam search, or the best fine-tune
|
|
7
|
+
babelscribe mcp # MCP server for AI apps (Claude, Codex, Gemini CLI, ...)
|
|
8
|
+
babelscribe devices # list GPUs whisper.cpp can use
|
|
9
|
+
babelscribe models # list models and fine-tunes"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import sys
|
|
14
|
+
|
|
15
|
+
from . import __version__, api, backend, models
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def main(argv: list[str] | None = None) -> None:
|
|
19
|
+
argv = sys.argv[1:] if argv is None else argv
|
|
20
|
+
if argv[:1] == ["mcp"]: # MCP server for AI apps (stdio)
|
|
21
|
+
from .mcp_server import main as serve
|
|
22
|
+
return serve()
|
|
23
|
+
if argv[:1] == ["models"]:
|
|
24
|
+
print("general (99 languages):"); [print(f" {k:16} {v}") for k, v in models.GENERAL.items()]
|
|
25
|
+
print("fine-tunes (text quality for one language, use with --text-model):")
|
|
26
|
+
[print(f" {k:16} [{v['lang']}] {v['note']}") for k, v in models.FINETUNES.items()]
|
|
27
|
+
return
|
|
28
|
+
ap = argparse.ArgumentParser(prog="babelscribe", description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
29
|
+
ap.add_argument("input", help="audio or video file, or 'devices'")
|
|
30
|
+
ap.add_argument("-l", "--lang", default="auto", help="language code (th, en, ja, ...) or auto")
|
|
31
|
+
ap.add_argument("-m", "--model", default="turbo", help="timing/general model (default turbo = large-v3-turbo)")
|
|
32
|
+
ap.add_argument("--text-model", help="language fine-tune for the text (hybrid mode), e.g. thai-thonburian")
|
|
33
|
+
ap.add_argument("--accurate", action="store_true",
|
|
34
|
+
help="slower but fewest errors: large-v3 with beam search, or the language's best fine-tune (hybrid)")
|
|
35
|
+
ap.add_argument("-f", "--formats", default="srt,json", help="comma list: srt,vtt,txt,json")
|
|
36
|
+
ap.add_argument("-o", "--out", help="output base path (default: next to the input)")
|
|
37
|
+
ap.add_argument("--device", default="auto", help="GPU id from `babelscribe devices`, or auto")
|
|
38
|
+
ap.add_argument("--bin", help="path to a whisper-cli you built yourself")
|
|
39
|
+
ap.add_argument("--flavor", help="prebuilt flavour to download: vulkan | cuda | metal | cpu")
|
|
40
|
+
ap.add_argument("-v", "--verbose", action="store_true")
|
|
41
|
+
ap.add_argument("--version", action="version", version=__version__)
|
|
42
|
+
a = ap.parse_args(argv)
|
|
43
|
+
|
|
44
|
+
if a.input == "devices":
|
|
45
|
+
_, _, found = api.setup(a.model, a.bin, a.flavor)
|
|
46
|
+
for d in found:
|
|
47
|
+
print(f" [{d['id']}] {d['backend']:6} {d['name']} {d.get('detail', '')}")
|
|
48
|
+
print(f"auto picks device {backend.pick_device(found)}"); return
|
|
49
|
+
try:
|
|
50
|
+
api.transcribe_file(a.input, a.lang, a.model, a.text_model, a.accurate,
|
|
51
|
+
[f.strip() for f in a.formats.split(",") if f.strip()], a.out, a.device, a.bin, a.flavor, a.verbose)
|
|
52
|
+
except FileNotFoundError as e:
|
|
53
|
+
raise SystemExit(str(e))
|
|
54
|
+
|
|
55
|
+
if __name__ == "__main__":
|
|
56
|
+
main()
|
|
@@ -1,136 +1,139 @@
|
|
|
1
|
-
"""babelscribe as an MCP server: any MCP client (Claude Desktop / Code, Codex, Antigravity, Gemini CLI, Cursor,
|
|
2
|
-
VS Code, ...) can transcribe local audio/video on the user's own GPU.
|
|
3
|
-
|
|
4
|
-
babelscribe
|
|
5
|
-
|
|
6
|
-
Needs
|
|
7
|
-
|
|
8
|
-
import io
|
|
9
|
-
import os
|
|
10
|
-
import sys
|
|
11
|
-
import time
|
|
12
|
-
from pathlib import Path
|
|
13
|
-
|
|
14
|
-
MEDIA = {".mp4", ".mkv", ".mov", ".avi", ".webm", ".m4v", ".mp3", ".wav", ".m4a", ".aac", ".flac", ".ogg", ".opus", ".wma"}
|
|
15
|
-
TEXT_LIMIT = 30000 # characters of transcript returned to the model; the full text is always in the files
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
def _guard_stdout():
|
|
19
|
-
"""stdout is the JSON-RPC channel. Keep a private handle to it for the protocol and point everything else
|
|
20
|
-
(print, logging, stray library output) at stderr so it can never corrupt a message."""
|
|
21
|
-
proto = io.TextIOWrapper(os.fdopen(os.dup(sys.stdout.fileno()), "wb"), encoding="utf-8", newline="\n",
|
|
22
|
-
write_through=True)
|
|
23
|
-
sys.stdout = sys.stderr
|
|
24
|
-
return proto
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
def build():
|
|
28
|
-
import anyio
|
|
29
|
-
from mcp.server.mcpserver import Context, MCPServer
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
"
|
|
36
|
-
"
|
|
37
|
-
"
|
|
38
|
-
"(
|
|
39
|
-
"
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
1
|
+
"""babelscribe as an MCP server: any MCP client (Claude Desktop / Code, Codex, Antigravity, Gemini CLI, Cursor,
|
|
2
|
+
VS Code, ...) can transcribe local audio/video on the user's own GPU.
|
|
3
|
+
|
|
4
|
+
babelscribe mcp # stdio server; AI apps launch it themselves (see README "Use it from an AI app")
|
|
5
|
+
|
|
6
|
+
Needs Python 3.10+ (the mcp SDK)."""
|
|
7
|
+
|
|
8
|
+
import io
|
|
9
|
+
import os
|
|
10
|
+
import sys
|
|
11
|
+
import time
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
MEDIA = {".mp4", ".mkv", ".mov", ".avi", ".webm", ".m4v", ".mp3", ".wav", ".m4a", ".aac", ".flac", ".ogg", ".opus", ".wma"}
|
|
15
|
+
TEXT_LIMIT = 30000 # characters of transcript returned to the model; the full text is always in the files
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _guard_stdout():
|
|
19
|
+
"""stdout is the JSON-RPC channel. Keep a private handle to it for the protocol and point everything else
|
|
20
|
+
(print, logging, stray library output) at stderr so it can never corrupt a message."""
|
|
21
|
+
proto = io.TextIOWrapper(os.fdopen(os.dup(sys.stdout.fileno()), "wb"), encoding="utf-8", newline="\n",
|
|
22
|
+
write_through=True)
|
|
23
|
+
sys.stdout = sys.stderr
|
|
24
|
+
return proto
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def build():
|
|
28
|
+
import anyio
|
|
29
|
+
from mcp.server.mcpserver import Context, MCPServer
|
|
30
|
+
from mcp.types import ToolAnnotations
|
|
31
|
+
|
|
32
|
+
from . import __version__, api, backend, models
|
|
33
|
+
|
|
34
|
+
srv = MCPServer("babelscribe", version=__version__, instructions=(
|
|
35
|
+
"Local speech-to-text on the user's own GPU (AMD/NVIDIA/Intel/Apple) for 99 languages. Needs a file path on "
|
|
36
|
+
"this computer: if the user only names or describes a file, call find_media first. Subtitles (.srt) and text "
|
|
37
|
+
"(.txt) are written next to the media file unless output_dir is given. Use accurate=true when the user wants "
|
|
38
|
+
"the fewest mistakes (slower); it is much better for Thai and Hindi. The first run downloads the model "
|
|
39
|
+
"(~1.6 GB), so it can take a few minutes. After transcribing you can proofread names, translate, or "
|
|
40
|
+
"summarise from the returned text."))
|
|
41
|
+
|
|
42
|
+
# transcribe writes new .srt/.txt files (re-running rewrites its own output); first use downloads whisper-cli and the model
|
|
43
|
+
@srv.tool(title="Transcribe audio/video to subtitles", annotations=ToolAnnotations(
|
|
44
|
+
read_only_hint=False, destructive_hint=False, idempotent_hint=True, open_world_hint=True))
|
|
45
|
+
async def transcribe(file_path: str, language: str = "auto", accurate: bool = False, formats: str = "srt,txt",
|
|
46
|
+
output_dir: str = "", ctx: Context | None = None) -> dict:
|
|
47
|
+
"""Transcribe a local audio or video file into subtitles / text.
|
|
48
|
+
|
|
49
|
+
file_path: absolute path to the media file (mp4, mkv, mov, mp3, wav, m4a, ...).
|
|
50
|
+
language: ISO code such as en, th, ja, es, hi — or "auto" to detect.
|
|
51
|
+
accurate: slower, fewest errors (large-v3 with beam search, or the best fine-tune for Thai / Hindi).
|
|
52
|
+
formats: comma list from srt, vtt, txt, json.
|
|
53
|
+
output_dir: folder for the output files; empty = next to the media file.
|
|
54
|
+
Returns the files written, detected language, device used, and the transcript text."""
|
|
55
|
+
src = Path(file_path).expanduser()
|
|
56
|
+
out = Path(output_dir).expanduser() / src.stem if output_dir else None
|
|
57
|
+
if out:
|
|
58
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
59
|
+
fmts = [f.strip() for f in formats.split(",") if f.strip()]
|
|
60
|
+
lines: list[str] = []
|
|
61
|
+
done = anyio.Event()
|
|
62
|
+
|
|
63
|
+
async def heartbeat(): # progress keeps clients from timing out on long files
|
|
64
|
+
t0 = time.time()
|
|
65
|
+
while not done.is_set():
|
|
66
|
+
with anyio.move_on_after(5):
|
|
67
|
+
await done.wait()
|
|
68
|
+
if ctx and not done.is_set():
|
|
69
|
+
await ctx.report_progress(time.time() - t0, None, lines[-1] if lines else "starting ...")
|
|
70
|
+
|
|
71
|
+
result: dict = {}
|
|
72
|
+
async with anyio.create_task_group() as tg:
|
|
73
|
+
tg.start_soon(heartbeat)
|
|
74
|
+
try:
|
|
75
|
+
result = await anyio.to_thread.run_sync(lambda: api.transcribe_file(
|
|
76
|
+
src, language, accurate=accurate, formats=fmts, out=out, log=lambda m: (lines.append(m), print(m))))
|
|
77
|
+
finally:
|
|
78
|
+
done.set()
|
|
79
|
+
text = result.pop("text", "")
|
|
80
|
+
result["text"] = text[:TEXT_LIMIT]
|
|
81
|
+
if len(text) > TEXT_LIMIT:
|
|
82
|
+
result["note"] = f"transcript truncated to {TEXT_LIMIT} characters; the full text is in the files listed"
|
|
83
|
+
return result
|
|
84
|
+
|
|
85
|
+
@srv.tool(title="Find audio/video files", annotations=ToolAnnotations(read_only_hint=True, open_world_hint=False))
|
|
86
|
+
def find_media(name_contains: str = "", folder: str = "", limit: int = 15) -> list[dict]:
|
|
87
|
+
"""Find audio/video files on this computer, newest first. Searches Downloads, Videos, Desktop, Music and
|
|
88
|
+
Documents (two levels deep) unless folder is given. Use it when the user names a file without a full path."""
|
|
89
|
+
home = Path.home()
|
|
90
|
+
roots = [Path(folder).expanduser()] if folder else [home / d for d in ("Downloads", "Videos", "Desktop", "Music", "Documents", "Movies")]
|
|
91
|
+
hits = []
|
|
92
|
+
for r in roots:
|
|
93
|
+
if not r.is_dir():
|
|
94
|
+
continue
|
|
95
|
+
for p in [*r.glob("*"), *r.glob("*/*"), *(r.glob("*/*/*") if folder else [])]:
|
|
96
|
+
if p.suffix.lower() in MEDIA and name_contains.lower() in p.name.lower():
|
|
97
|
+
try:
|
|
98
|
+
st = p.stat()
|
|
99
|
+
except OSError:
|
|
100
|
+
continue
|
|
101
|
+
hits.append((st.st_mtime, p, st.st_size))
|
|
102
|
+
hits.sort(reverse=True)
|
|
103
|
+
return [{"path": str(p), "size_mb": round(sz / 1e6, 1), "modified": time.strftime("%Y-%m-%d %H:%M", time.localtime(t))}
|
|
104
|
+
for t, p, sz in hits[:limit]]
|
|
105
|
+
|
|
106
|
+
@srv.tool(title="List GPUs", annotations=ToolAnnotations(read_only_hint=True, open_world_hint=True))
|
|
107
|
+
def list_devices() -> dict:
|
|
108
|
+
"""GPUs whisper.cpp can use on this computer and which one babelscribe picks automatically."""
|
|
109
|
+
_, _, found = api.setup()
|
|
110
|
+
return {"devices": found, "auto_pick": backend.pick_device(found)}
|
|
111
|
+
|
|
112
|
+
@srv.tool(title="List languages and models", annotations=ToolAnnotations(read_only_hint=True, open_world_hint=False))
|
|
113
|
+
def list_languages_and_models() -> dict:
|
|
114
|
+
"""Models available, and which model --accurate uses per language."""
|
|
115
|
+
return {"general": models.GENERAL, "fine_tunes": {k: v["note"] for k, v in models.FINETUNES.items()},
|
|
116
|
+
"accurate_per_language": {**models.ACCURATE, "*": "large-v3 beam 5"}}
|
|
117
|
+
|
|
118
|
+
return srv
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def main() -> None:
|
|
122
|
+
try:
|
|
123
|
+
import anyio
|
|
124
|
+
from mcp.server.stdio import stdio_server
|
|
125
|
+
except ImportError:
|
|
126
|
+
raise SystemExit('the MCP server needs Python 3.10+ (the mcp SDK)')
|
|
127
|
+
proto = _guard_stdout()
|
|
128
|
+
srv = build()
|
|
129
|
+
|
|
130
|
+
async def run():
|
|
131
|
+
async with stdio_server(stdout=anyio.wrap_file(proto)) as (r, w):
|
|
132
|
+
low = srv._lowlevel_server
|
|
133
|
+
await low.run(r, w, low.create_initialization_options())
|
|
134
|
+
|
|
135
|
+
anyio.run(run)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
if __name__ == "__main__":
|
|
139
|
+
main()
|
|
@@ -1,7 +1,58 @@
|
|
|
1
|
-
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: babelscribe
|
|
3
|
+
Version: 0.3.2
|
|
4
|
+
Summary: Transcribe any audio/video in 99 languages on any GPU (AMD, NVIDIA, Intel via Vulkan; NVIDIA via CUDA; Apple via Metal) or CPU — whisper.cpp with batteries included.
|
|
5
|
+
License: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/phonology024/babelscribe
|
|
7
|
+
Project-URL: Documentation, https://phonology024.github.io/babelscribe/
|
|
8
|
+
Project-URL: Repository, https://github.com/phonology024/babelscribe
|
|
9
|
+
Project-URL: Issues, https://github.com/phonology024/babelscribe/issues
|
|
10
|
+
Keywords: whisper,speech-to-text,transcription,subtitles,vulkan,amd,gpu,thai,mcp,mcp-server,claude,codex,srt,speech-recognition,offline,radeon,intel-arc,openai-whisper
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Environment :: GPU
|
|
14
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
18
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
19
|
+
Classifier: Operating System :: MacOS
|
|
20
|
+
Classifier: Programming Language :: Python :: 3
|
|
21
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
22
|
+
Classifier: Topic :: Multimedia :: Video
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Classifier: Natural Language :: Thai
|
|
25
|
+
Classifier: Natural Language :: English
|
|
26
|
+
Requires-Python: >=3.9
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
License-File: LICENSE
|
|
29
|
+
Requires-Dist: imageio-ffmpeg>=0.5
|
|
30
|
+
Requires-Dist: mcp<3,>=2.3; python_version >= "3.10"
|
|
31
|
+
Requires-Dist: anyio>=4; python_version >= "3.10"
|
|
32
|
+
Provides-Extra: finetune
|
|
33
|
+
Requires-Dist: torch; extra == "finetune"
|
|
34
|
+
Requires-Dist: transformers; extra == "finetune"
|
|
35
|
+
Requires-Dist: huggingface_hub; extra == "finetune"
|
|
36
|
+
Requires-Dist: numpy; extra == "finetune"
|
|
37
|
+
Provides-Extra: thai
|
|
38
|
+
Requires-Dist: pythainlp; extra == "thai"
|
|
39
|
+
Provides-Extra: mcp
|
|
40
|
+
Requires-Dist: mcp<3,>=2.3; python_version >= "3.10" and extra == "mcp"
|
|
41
|
+
Requires-Dist: anyio>=4; extra == "mcp"
|
|
42
|
+
Dynamic: license-file
|
|
2
43
|
|
|
3
|
-
|
|
4
|
-
|
|
44
|
+
# babelscribe — free, offline speech-to-text and subtitles on any GPU (AMD, NVIDIA, Intel, Apple)
|
|
45
|
+
|
|
46
|
+
<!-- mcp-name: io.github.phonology024/babelscribe -->
|
|
47
|
+
|
|
48
|
+
[](https://pypi.org/project/babelscribe/)
|
|
49
|
+
[](LICENSE)
|
|
50
|
+
[](#use-it-from-an-ai-app-mcp--no-terminal-needed)
|
|
51
|
+
|
|
52
|
+
**babelscribe turns any audio or video file into subtitles (SRT, VTT) and text, in any of Whisper's 99 languages, on your
|
|
53
|
+
own graphics card — AMD Radeon, NVIDIA, Intel Arc or Apple Silicon — or the CPU. No CUDA needed, no cloud, free (MIT).**
|
|
54
|
+
It runs OpenAI Whisper through [whisper.cpp](https://github.com/ggml-org/whisper.cpp) with Vulkan, CUDA or Metal, and works
|
|
55
|
+
from the command line, from Python, or from AI apps (Claude, Codex, Antigravity, Gemini CLI, Cursor) as an MCP server.
|
|
5
56
|
|
|
6
57
|
```bash
|
|
7
58
|
pip install babelscribe # Python 3.9+
|
|
@@ -36,12 +87,12 @@ Tools: `transcribe` (file → subtitles + text), `find_media` (newest audio/vide
|
|
|
36
87
|
and double-click it (or *Settings → Extensions → Install extension*).
|
|
37
88
|
|
|
38
89
|
**Everything else** runs the same command — [uv](https://docs.astral.sh/uv/) fetches babelscribe for you:
|
|
39
|
-
`uvx
|
|
90
|
+
`uvx babelscribe mcp`
|
|
40
91
|
|
|
41
92
|
| App | How to add it |
|
|
42
93
|
|---|---|
|
|
43
|
-
| Claude Code | `claude mcp add babelscribe -- uvx
|
|
44
|
-
| OpenAI Codex CLI | `codex mcp add babelscribe -- uvx
|
|
94
|
+
| Claude Code | `claude mcp add babelscribe -- uvx babelscribe mcp` |
|
|
95
|
+
| OpenAI Codex CLI | `codex mcp add babelscribe -- uvx babelscribe mcp`, then in `~/.codex/config.toml` under `[mcp_servers.babelscribe]` set `tool_timeout_sec = 3600` and `startup_timeout_sec = 120` (defaults are 60 s / 10 s — too short for a long video or the first model download) |
|
|
45
96
|
| Google Antigravity | agent panel → ⋯ → *MCP Servers* → *Manage MCP Servers* → *View raw config*, add the JSON below |
|
|
46
97
|
| Gemini CLI | add the JSON below to `~/.gemini/settings.json` |
|
|
47
98
|
| Cursor | add the JSON below to `~/.cursor/mcp.json` |
|
|
@@ -50,11 +101,11 @@ and double-click it (or *Settings → Extensions → Install extension*).
|
|
|
50
101
|
```json
|
|
51
102
|
{
|
|
52
103
|
"mcpServers": {
|
|
53
|
-
"babelscribe": { "command": "uvx", "args": ["
|
|
104
|
+
"babelscribe": { "command": "uvx", "args": ["babelscribe", "mcp"], "timeout": 3600000 }
|
|
54
105
|
}
|
|
55
106
|
}
|
|
56
107
|
```
|
|
57
|
-
(`timeout` is in milliseconds and only some apps read it.) Already installed with pip? Use `"command": "babelscribe
|
|
108
|
+
(`timeout` is in milliseconds and only some apps read it.) Already installed with pip? Use `"command": "babelscribe", "args": ["mcp"]`.
|
|
58
109
|
Web-only chat apps (e.g. grok.com, chatgpt.com) can only reach servers on the internet, not your computer, so they can't use your GPU or files.
|
|
59
110
|
|
|
60
111
|
## Benchmark: real talks, human captions as the answer key
|
|
@@ -118,6 +169,32 @@ babelscribe does **not** replace those projects — it stands on whisper.cpp and
|
|
|
118
169
|
compiling for your GPU, converting media, picking the right device, avoiding the long-file repeat bug, and combining
|
|
119
170
|
a language-specific fine-tune with accurate timestamps.
|
|
120
171
|
|
|
172
|
+
## FAQ
|
|
173
|
+
**How do I run Whisper on an AMD GPU on Windows?**
|
|
174
|
+
`pip install babelscribe`, then `babelscribe video.mp4`. It downloads a Vulkan build of whisper.cpp that runs on AMD Radeon
|
|
175
|
+
(and Intel Arc / NVIDIA) cards on Windows and Linux — no ROCm, no CUDA, no compiling.
|
|
176
|
+
|
|
177
|
+
**How do I make subtitles (SRT) from a video for free, offline?**
|
|
178
|
+
`babelscribe video.mp4 -f srt` writes `video.srt` next to the video. Nothing is uploaded; it runs on your own computer.
|
|
179
|
+
|
|
180
|
+
**What is the most accurate free transcription for Thai?**
|
|
181
|
+
`babelscribe video.mp4 -l th --accurate` — Pathumma Whisper (NECTEC) for the text plus Whisper turbo for timing:
|
|
182
|
+
CER 8.9% on Google FLEURS vs 15.9% for plain Whisper turbo. Thonburian Whisper is available too (`--text-model thai-thonburian`).
|
|
183
|
+
|
|
184
|
+
**Can Claude / ChatGPT Codex / Gemini transcribe a video on my computer?**
|
|
185
|
+
Yes — add babelscribe as an MCP server (see *Use it from an AI app*). Claude Desktop installs it with one click from
|
|
186
|
+
`babelscribe.mcpb`. The AI app can then find a file, transcribe it on your GPU, and proofread, translate or summarise the text.
|
|
187
|
+
|
|
188
|
+
**How fast is it?**
|
|
189
|
+
About 20x real time with Whisper large-v3-turbo on an AMD Radeon RX 9070 XT: a 19-minute talk in 48 seconds.
|
|
190
|
+
|
|
191
|
+
**Which languages are supported?**
|
|
192
|
+
All 99 Whisper languages, with automatic language detection. FLEURS error rates for 16 of them are in the benchmark table.
|
|
193
|
+
|
|
194
|
+
**Is it better than faster-whisper or WhisperX?**
|
|
195
|
+
Those are excellent on NVIDIA GPUs; on AMD / Intel GPUs they fall back to the CPU. babelscribe's niche is any GPU, zero setup,
|
|
196
|
+
and better Thai / Hindi through community fine-tunes. If you have an NVIDIA card and like Python, faster-whisper is a fine choice.
|
|
197
|
+
|
|
121
198
|
## Languages
|
|
122
199
|
All 99 languages Whisper was trained on, auto-detected or forced with `-l`:
|
|
123
200
|
af am ar as az ba be bg bn bo br bs ca cs cy da de el en es et eu fa fi fo fr gl gu ha haw he hi hr ht hu hy id is it ja jw ka kk km kn ko la lb ln lo lt lv mg mi mk ml mn mr ms mt my ne nl nn no oc pa pl ps pt ro ru sa sd si sk sl sn so sq sr su sv sw ta te tg th tk tl tr tt uk ur uz vi yi yo yue zh
|
|
@@ -151,6 +228,12 @@ babelscribe devices # GPUs whisper.cpp can see
|
|
|
151
228
|
babelscribe models # models and fine-tunes
|
|
152
229
|
babelscribe talk.mp4 --bin /path/to/whisper-cli # use your own whisper.cpp build
|
|
153
230
|
```
|
|
231
|
+
From Python:
|
|
232
|
+
```python
|
|
233
|
+
from babelscribe.api import transcribe_file
|
|
234
|
+
r = transcribe_file("talk.mp4", lang="auto", formats=["srt", "txt"]) # or accurate=True
|
|
235
|
+
print(r["lang"], r["device"], r["files"]); print(r["text"][:200])
|
|
236
|
+
```
|
|
154
237
|
Models download on first use to `~/.babelscribe/models` (`BABELSCRIBE_MODELS` to change). Hybrid fine-tunes are converted on your
|
|
155
238
|
machine from their original Hugging Face repo — install `pip install "babelscribe[finetune]"` once; converted weights are never
|
|
156
239
|
redistributed, so each fine-tune keeps its own licence. Thai word boundaries: `pip install "babelscribe[thai]"`.
|
|
@@ -168,6 +251,25 @@ them to each `v*` release. Point `BABELSCRIBE_RELEASES` at another URL to self-h
|
|
|
168
251
|
ภาษาไทยแนะนำ `babelscribe ไฟล์.mp4 -l th --accurate` — ข้อความจาก Pathumma Whisper (NECTEC) ที่ผิดน้อยที่สุดใน FLEURS (CER 8.9% เทียบ turbo 15.9%) + เวลาจาก large-v3-turbo
|
|
169
252
|
หรือเลือก Thonburian Whisper เอง: `--text-model thai-thonburian`
|
|
170
253
|
|
|
254
|
+
## Privacy Policy
|
|
255
|
+
babelscribe runs entirely on your own computer. Last updated 2026-10-06.
|
|
256
|
+
|
|
257
|
+
- **Data collection:** none. babelscribe has no telemetry, analytics, accounts or crash reporting, and never uploads your
|
|
258
|
+
audio, video, transcripts or file names anywhere.
|
|
259
|
+
- **What it processes and where:** the media file you choose is converted and transcribed locally; subtitles and text are
|
|
260
|
+
written to your disk (next to the file, or the folder you choose). Through MCP, the transcript text is returned to the AI
|
|
261
|
+
app that called the tool — what that app does with it is governed by that app's own privacy policy.
|
|
262
|
+
- **Network access (downloads only):** on first use it downloads the `whisper-cli` program from this project's GitHub
|
|
263
|
+
releases, speech models from Hugging Face (`huggingface.co/ggerganov/whisper.cpp`, and for `--accurate` Thai/Hindi the
|
|
264
|
+
fine-tune's own repository), and Python packages from PyPI when installed with pip/uv. These are plain downloads;
|
|
265
|
+
no personal data is sent. GitHub, Hugging Face and PyPI see a normal download request (IP address, user agent) under
|
|
266
|
+
their own privacy policies.
|
|
267
|
+
- **Storage and retention:** downloaded programs and models are cached in `~/.babelscribe` (or `BABELSCRIBE_HOME` /
|
|
268
|
+
`BABELSCRIBE_MODELS`) until you delete that folder. Outputs stay wherever they were written until you delete them.
|
|
269
|
+
babelscribe keeps no other data.
|
|
270
|
+
- **Third-party sharing:** none.
|
|
271
|
+
- **Contact:** open an issue at https://github.com/phonology024/babelscribe/issues
|
|
272
|
+
|
|
171
273
|
## Credits & licence
|
|
172
274
|
MIT. Built on [whisper.cpp](https://github.com/ggml-org/whisper.cpp) (MIT) and OpenAI Whisper models (MIT).
|
|
173
275
|
Fine-tunes belong to their authors: [Pathumma Whisper](https://huggingface.co/nectec/Pathumma-whisper-th-large-v3) by NECTEC,
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "babelscribe"
|
|
7
|
+
version = "0.3.2"
|
|
8
|
+
description = "Transcribe any audio/video in 99 languages on any GPU (AMD, NVIDIA, Intel via Vulkan; NVIDIA via CUDA; Apple via Metal) or CPU — whisper.cpp with batteries included."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { text = "MIT" }
|
|
11
|
+
requires-python = ">=3.9"
|
|
12
|
+
dependencies = ["imageio-ffmpeg>=0.5", "mcp>=2.3,<3; python_version>='3.10'", "anyio>=4; python_version>='3.10'"]
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Development Status :: 4 - Beta",
|
|
15
|
+
"Environment :: Console",
|
|
16
|
+
"Environment :: GPU",
|
|
17
|
+
"Intended Audience :: End Users/Desktop",
|
|
18
|
+
"Intended Audience :: Developers",
|
|
19
|
+
"License :: OSI Approved :: MIT License",
|
|
20
|
+
"Operating System :: Microsoft :: Windows",
|
|
21
|
+
"Operating System :: POSIX :: Linux",
|
|
22
|
+
"Operating System :: MacOS",
|
|
23
|
+
"Programming Language :: Python :: 3",
|
|
24
|
+
"Topic :: Multimedia :: Sound/Audio :: Speech",
|
|
25
|
+
"Topic :: Multimedia :: Video",
|
|
26
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
27
|
+
"Natural Language :: Thai",
|
|
28
|
+
"Natural Language :: English",
|
|
29
|
+
]
|
|
30
|
+
keywords = ["whisper", "speech-to-text", "transcription", "subtitles", "vulkan", "amd", "gpu", "thai", "mcp", "mcp-server", "claude", "codex", "srt", "speech-recognition", "offline", "radeon", "intel-arc", "openai-whisper"]
|
|
31
|
+
|
|
32
|
+
[project.urls]
|
|
33
|
+
Homepage = "https://github.com/phonology024/babelscribe"
|
|
34
|
+
Documentation = "https://phonology024.github.io/babelscribe/"
|
|
35
|
+
Repository = "https://github.com/phonology024/babelscribe"
|
|
36
|
+
Issues = "https://github.com/phonology024/babelscribe/issues"
|
|
37
|
+
|
|
38
|
+
[project.optional-dependencies]
|
|
39
|
+
finetune = ["torch", "transformers", "huggingface_hub", "numpy"]
|
|
40
|
+
thai = ["pythainlp"]
|
|
41
|
+
mcp = ["mcp>=2.3,<3; python_version>='3.10'", "anyio>=4"]
|
|
42
|
+
|
|
43
|
+
[project.scripts]
|
|
44
|
+
babelscribe = "babelscribe.cli:main"
|
|
45
|
+
babelscribe-mcp = "babelscribe.mcp_server:main"
|
|
46
|
+
|
|
47
|
+
[tool.setuptools]
|
|
48
|
+
packages = ["babelscribe"]
|
babelscribe-0.3.0/pyproject.toml
DELETED
|
@@ -1,29 +0,0 @@
|
|
|
1
|
-
[build-system]
|
|
2
|
-
requires = ["setuptools>=68"]
|
|
3
|
-
build-backend = "setuptools.build_meta"
|
|
4
|
-
|
|
5
|
-
[project]
|
|
6
|
-
name = "babelscribe"
|
|
7
|
-
version = "0.3.0"
|
|
8
|
-
description = "Transcribe any audio/video in 99 languages on any GPU (AMD, NVIDIA, Intel via Vulkan; NVIDIA via CUDA; Apple via Metal) or CPU — whisper.cpp with batteries included."
|
|
9
|
-
readme = "README.md"
|
|
10
|
-
license = { text = "MIT" }
|
|
11
|
-
requires-python = ">=3.9"
|
|
12
|
-
dependencies = ["imageio-ffmpeg>=0.5"]
|
|
13
|
-
keywords = ["whisper", "speech-to-text", "transcription", "subtitles", "vulkan", "amd", "gpu", "thai", "mcp", "claude", "codex"]
|
|
14
|
-
|
|
15
|
-
[project.urls]
|
|
16
|
-
Homepage = "https://github.com/phonology024/babelscribe"
|
|
17
|
-
Issues = "https://github.com/phonology024/babelscribe/issues"
|
|
18
|
-
|
|
19
|
-
[project.optional-dependencies]
|
|
20
|
-
finetune = ["torch", "transformers", "huggingface_hub", "numpy"]
|
|
21
|
-
thai = ["pythainlp"]
|
|
22
|
-
mcp = ["mcp>=2.3,<3; python_version>='3.10'", "anyio>=4"]
|
|
23
|
-
|
|
24
|
-
[project.scripts]
|
|
25
|
-
babelscribe = "babelscribe.cli:main"
|
|
26
|
-
babelscribe-mcp = "babelscribe.mcp_server:main"
|
|
27
|
-
|
|
28
|
-
[tool.setuptools]
|
|
29
|
-
packages = ["babelscribe"]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|