babelscribe 0.3.0__tar.gz → 0.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (24) hide show
  1. {babelscribe-0.3.0/babelscribe.egg-info → babelscribe-0.3.2}/PKG-INFO +88 -10
  2. babelscribe-0.3.0/PKG-INFO → babelscribe-0.3.2/README.md +68 -33
  3. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/__init__.py +1 -1
  4. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/cli.py +56 -52
  5. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/mcp_server.py +139 -136
  6. babelscribe-0.3.0/README.md → babelscribe-0.3.2/babelscribe.egg-info/PKG-INFO +110 -8
  7. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe.egg-info/requires.txt +4 -0
  8. babelscribe-0.3.2/pyproject.toml +48 -0
  9. babelscribe-0.3.0/pyproject.toml +0 -29
  10. {babelscribe-0.3.0 → babelscribe-0.3.2}/LICENSE +0 -0
  11. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/__main__.py +0 -0
  12. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/align.py +0 -0
  13. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/api.py +0 -0
  14. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/backend.py +0 -0
  15. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/hybrid.py +0 -0
  16. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/models.py +0 -0
  17. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/transcribe.py +0 -0
  18. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe/writers.py +0 -0
  19. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe.egg-info/SOURCES.txt +0 -0
  20. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe.egg-info/dependency_links.txt +0 -0
  21. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe.egg-info/entry_points.txt +0 -0
  22. {babelscribe-0.3.0 → babelscribe-0.3.2}/babelscribe.egg-info/top_level.txt +0 -0
  23. {babelscribe-0.3.0 → babelscribe-0.3.2}/setup.cfg +0 -0
  24. {babelscribe-0.3.0 → babelscribe-0.3.2}/tests/test_core.py +0 -0
@@ -1,15 +1,34 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: babelscribe
3
- Version: 0.3.0
3
+ Version: 0.3.2
4
4
  Summary: Transcribe any audio/video in 99 languages on any GPU (AMD, NVIDIA, Intel via Vulkan; NVIDIA via CUDA; Apple via Metal) or CPU — whisper.cpp with batteries included.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/phonology024/babelscribe
7
+ Project-URL: Documentation, https://phonology024.github.io/babelscribe/
8
+ Project-URL: Repository, https://github.com/phonology024/babelscribe
7
9
  Project-URL: Issues, https://github.com/phonology024/babelscribe/issues
8
- Keywords: whisper,speech-to-text,transcription,subtitles,vulkan,amd,gpu,thai,mcp,claude,codex
10
+ Keywords: whisper,speech-to-text,transcription,subtitles,vulkan,amd,gpu,thai,mcp,mcp-server,claude,codex,srt,speech-recognition,offline,radeon,intel-arc,openai-whisper
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Environment :: Console
13
+ Classifier: Environment :: GPU
14
+ Classifier: Intended Audience :: End Users/Desktop
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: License :: OSI Approved :: MIT License
17
+ Classifier: Operating System :: Microsoft :: Windows
18
+ Classifier: Operating System :: POSIX :: Linux
19
+ Classifier: Operating System :: MacOS
20
+ Classifier: Programming Language :: Python :: 3
21
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
22
+ Classifier: Topic :: Multimedia :: Video
23
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
24
+ Classifier: Natural Language :: Thai
25
+ Classifier: Natural Language :: English
9
26
  Requires-Python: >=3.9
10
27
  Description-Content-Type: text/markdown
11
28
  License-File: LICENSE
12
29
  Requires-Dist: imageio-ffmpeg>=0.5
30
+ Requires-Dist: mcp<3,>=2.3; python_version >= "3.10"
31
+ Requires-Dist: anyio>=4; python_version >= "3.10"
13
32
  Provides-Extra: finetune
14
33
  Requires-Dist: torch; extra == "finetune"
15
34
  Requires-Dist: transformers; extra == "finetune"
@@ -22,10 +41,18 @@ Requires-Dist: mcp<3,>=2.3; python_version >= "3.10" and extra == "mcp"
22
41
  Requires-Dist: anyio>=4; extra == "mcp"
23
42
  Dynamic: license-file
24
43
 
25
- # babelscribe
44
+ # babelscribe — free, offline speech-to-text and subtitles on any GPU (AMD, NVIDIA, Intel, Apple)
26
45
 
27
- **Transcribe any audio or video, in any of Whisper's 99 languages, on any GPU — AMD, NVIDIA, Intel or Apple — or just the CPU.**
28
- One command, no CUDA required, subtitles out.
46
+ <!-- mcp-name: io.github.phonology024/babelscribe -->
47
+
48
+ [![PyPI](https://img.shields.io/pypi/v/babelscribe)](https://pypi.org/project/babelscribe/)
49
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
50
+ [![MCP server](https://img.shields.io/badge/MCP-server-blue)](#use-it-from-an-ai-app-mcp--no-terminal-needed)
51
+
52
+ **babelscribe turns any audio or video file into subtitles (SRT, VTT) and text, in any of Whisper's 99 languages, on your
53
+ own graphics card — AMD Radeon, NVIDIA, Intel Arc or Apple Silicon — or the CPU. No CUDA needed, no cloud, free (MIT).**
54
+ It runs OpenAI Whisper through [whisper.cpp](https://github.com/ggml-org/whisper.cpp) with Vulkan, CUDA or Metal, and works
55
+ from the command line, from Python, or from AI apps (Claude, Codex, Antigravity, Gemini CLI, Cursor) as an MCP server.
29
56
 
30
57
  ```bash
31
58
  pip install babelscribe # Python 3.9+
@@ -60,12 +87,12 @@ Tools: `transcribe` (file → subtitles + text), `find_media` (newest audio/vide
60
87
  and double-click it (or *Settings → Extensions → Install extension*).
61
88
 
62
89
  **Everything else** runs the same command — [uv](https://docs.astral.sh/uv/) fetches babelscribe for you:
63
- `uvx --from "babelscribe[mcp]" babelscribe-mcp`
90
+ `uvx babelscribe mcp`
64
91
 
65
92
  | App | How to add it |
66
93
  |---|---|
67
- | Claude Code | `claude mcp add babelscribe -- uvx --from "babelscribe[mcp]" babelscribe-mcp` |
68
- | OpenAI Codex CLI | `codex mcp add babelscribe -- uvx --from "babelscribe[mcp]" babelscribe-mcp`, then in `~/.codex/config.toml` under `[mcp_servers.babelscribe]` set `tool_timeout_sec = 3600` and `startup_timeout_sec = 120` (defaults are 60 s / 10 s — too short for a long video or the first model download) |
94
+ | Claude Code | `claude mcp add babelscribe -- uvx babelscribe mcp` |
95
+ | OpenAI Codex CLI | `codex mcp add babelscribe -- uvx babelscribe mcp`, then in `~/.codex/config.toml` under `[mcp_servers.babelscribe]` set `tool_timeout_sec = 3600` and `startup_timeout_sec = 120` (defaults are 60 s / 10 s — too short for a long video or the first model download) |
69
96
  | Google Antigravity | agent panel → ⋯ → *MCP Servers* → *Manage MCP Servers* → *View raw config*, add the JSON below |
70
97
  | Gemini CLI | add the JSON below to `~/.gemini/settings.json` |
71
98
  | Cursor | add the JSON below to `~/.cursor/mcp.json` |
@@ -74,11 +101,11 @@ and double-click it (or *Settings → Extensions → Install extension*).
74
101
  ```json
75
102
  {
76
103
  "mcpServers": {
77
- "babelscribe": { "command": "uvx", "args": ["--from", "babelscribe[mcp]", "babelscribe-mcp"], "timeout": 3600000 }
104
+ "babelscribe": { "command": "uvx", "args": ["babelscribe", "mcp"], "timeout": 3600000 }
78
105
  }
79
106
  }
80
107
  ```
81
- (`timeout` is in milliseconds and only some apps read it.) Already installed with pip? Use `"command": "babelscribe-mcp"` with no args.
108
+ (`timeout` is in milliseconds and only some apps read it.) Already installed with pip? Use `"command": "babelscribe", "args": ["mcp"]`.
82
109
  Web-only chat apps (e.g. grok.com, chatgpt.com) can only reach servers on the internet, not your computer, so they can't use your GPU or files.
83
110
 
84
111
  ## Benchmark: real talks, human captions as the answer key
@@ -142,6 +169,32 @@ babelscribe does **not** replace those projects — it stands on whisper.cpp and
142
169
  compiling for your GPU, converting media, picking the right device, avoiding the long-file repeat bug, and combining
143
170
  a language-specific fine-tune with accurate timestamps.
144
171
 
172
+ ## FAQ
173
+ **How do I run Whisper on an AMD GPU on Windows?**
174
+ `pip install babelscribe`, then `babelscribe video.mp4`. It downloads a Vulkan build of whisper.cpp that runs on AMD Radeon
175
+ (and Intel Arc / NVIDIA) cards on Windows and Linux — no ROCm, no CUDA, no compiling.
176
+
177
+ **How do I make subtitles (SRT) from a video for free, offline?**
178
+ `babelscribe video.mp4 -f srt` writes `video.srt` next to the video. Nothing is uploaded; it runs on your own computer.
179
+
180
+ **What is the most accurate free transcription for Thai?**
181
+ `babelscribe video.mp4 -l th --accurate` — Pathumma Whisper (NECTEC) for the text plus Whisper turbo for timing:
182
+ CER 8.9% on Google FLEURS vs 15.9% for plain Whisper turbo. Thonburian Whisper is available too (`--text-model thai-thonburian`).
183
+
184
+ **Can Claude / ChatGPT Codex / Gemini transcribe a video on my computer?**
185
+ Yes — add babelscribe as an MCP server (see *Use it from an AI app*). Claude Desktop installs it with one click from
186
+ `babelscribe.mcpb`. The AI app can then find a file, transcribe it on your GPU, and proofread, translate or summarise the text.
187
+
188
+ **How fast is it?**
189
+ About 20x real time with Whisper large-v3-turbo on an AMD Radeon RX 9070 XT: a 19-minute talk in 48 seconds.
190
+
191
+ **Which languages are supported?**
192
+ All 99 Whisper languages, with automatic language detection. FLEURS error rates for 16 of them are in the benchmark table.
193
+
194
+ **Is it better than faster-whisper or WhisperX?**
195
+ Those are excellent on NVIDIA GPUs; on AMD / Intel GPUs they fall back to the CPU. babelscribe's niche is any GPU, zero setup,
196
+ and better Thai / Hindi through community fine-tunes. If you have an NVIDIA card and like Python, faster-whisper is a fine choice.
197
+
145
198
  ## Languages
146
199
  All 99 languages Whisper was trained on, auto-detected or forced with `-l`:
147
200
  af am ar as az ba be bg bn bo br bs ca cs cy da de el en es et eu fa fi fo fr gl gu ha haw he hi hr ht hu hy id is it ja jw ka kk km kn ko la lb ln lo lt lv mg mi mk ml mn mr ms mt my ne nl nn no oc pa pl ps pt ro ru sa sd si sk sl sn so sq sr su sv sw ta te tg th tk tl tr tt uk ur uz vi yi yo yue zh
@@ -175,6 +228,12 @@ babelscribe devices # GPUs whisper.cpp can see
175
228
  babelscribe models # models and fine-tunes
176
229
  babelscribe talk.mp4 --bin /path/to/whisper-cli # use your own whisper.cpp build
177
230
  ```
231
+ From Python:
232
+ ```python
233
+ from babelscribe.api import transcribe_file
234
+ r = transcribe_file("talk.mp4", lang="auto", formats=["srt", "txt"]) # or accurate=True
235
+ print(r["lang"], r["device"], r["files"]); print(r["text"][:200])
236
+ ```
178
237
  Models download on first use to `~/.babelscribe/models` (`BABELSCRIBE_MODELS` to change). Hybrid fine-tunes are converted on your
179
238
  machine from their original Hugging Face repo — install `pip install "babelscribe[finetune]"` once; converted weights are never
180
239
  redistributed, so each fine-tune keeps its own licence. Thai word boundaries: `pip install "babelscribe[thai]"`.
@@ -192,6 +251,25 @@ them to each `v*` release. Point `BABELSCRIBE_RELEASES` at another URL to self-h
192
251
  ภาษาไทยแนะนำ `babelscribe ไฟล์.mp4 -l th --accurate` — ข้อความจาก Pathumma Whisper (NECTEC) ที่ผิดน้อยที่สุดใน FLEURS (CER 8.9% เทียบ turbo 15.9%) + เวลาจาก large-v3-turbo
193
252
  หรือเลือก Thonburian Whisper เอง: `--text-model thai-thonburian`
194
253
 
254
+ ## Privacy Policy
255
+ babelscribe runs entirely on your own computer. Last updated 2026-10-06.
256
+
257
+ - **Data collection:** none. babelscribe has no telemetry, analytics, accounts or crash reporting, and never uploads your
258
+ audio, video, transcripts or file names anywhere.
259
+ - **What it processes and where:** the media file you choose is converted and transcribed locally; subtitles and text are
260
+ written to your disk (next to the file, or the folder you choose). Through MCP, the transcript text is returned to the AI
261
+ app that called the tool — what that app does with it is governed by that app's own privacy policy.
262
+ - **Network access (downloads only):** on first use it downloads the `whisper-cli` program from this project's GitHub
263
+ releases, speech models from Hugging Face (`huggingface.co/ggerganov/whisper.cpp`, and for `--accurate` Thai/Hindi the
264
+ fine-tune's own repository), and Python packages from PyPI when installed with pip/uv. These are plain downloads;
265
+ no personal data is sent. GitHub, Hugging Face and PyPI see a normal download request (IP address, user agent) under
266
+ their own privacy policies.
267
+ - **Storage and retention:** downloaded programs and models are cached in `~/.babelscribe` (or `BABELSCRIBE_HOME` /
268
+ `BABELSCRIBE_MODELS`) until you delete that folder. Outputs stay wherever they were written until you delete them.
269
+ babelscribe keeps no other data.
270
+ - **Third-party sharing:** none.
271
+ - **Contact:** open an issue at https://github.com/phonology024/babelscribe/issues
272
+
195
273
  ## Credits & licence
196
274
  MIT. Built on [whisper.cpp](https://github.com/ggml-org/whisper.cpp) (MIT) and OpenAI Whisper models (MIT).
197
275
  Fine-tunes belong to their authors: [Pathumma Whisper](https://huggingface.co/nectec/Pathumma-whisper-th-large-v3) by NECTEC,
@@ -1,31 +1,15 @@
1
- Metadata-Version: 2.4
2
- Name: babelscribe
3
- Version: 0.3.0
4
- Summary: Transcribe any audio/video in 99 languages on any GPU (AMD, NVIDIA, Intel via Vulkan; NVIDIA via CUDA; Apple via Metal) or CPU — whisper.cpp with batteries included.
5
- License: MIT
6
- Project-URL: Homepage, https://github.com/phonology024/babelscribe
7
- Project-URL: Issues, https://github.com/phonology024/babelscribe/issues
8
- Keywords: whisper,speech-to-text,transcription,subtitles,vulkan,amd,gpu,thai,mcp,claude,codex
9
- Requires-Python: >=3.9
10
- Description-Content-Type: text/markdown
11
- License-File: LICENSE
12
- Requires-Dist: imageio-ffmpeg>=0.5
13
- Provides-Extra: finetune
14
- Requires-Dist: torch; extra == "finetune"
15
- Requires-Dist: transformers; extra == "finetune"
16
- Requires-Dist: huggingface_hub; extra == "finetune"
17
- Requires-Dist: numpy; extra == "finetune"
18
- Provides-Extra: thai
19
- Requires-Dist: pythainlp; extra == "thai"
20
- Provides-Extra: mcp
21
- Requires-Dist: mcp<3,>=2.3; python_version >= "3.10" and extra == "mcp"
22
- Requires-Dist: anyio>=4; extra == "mcp"
23
- Dynamic: license-file
24
-
25
- # babelscribe
26
-
27
- **Transcribe any audio or video, in any of Whisper's 99 languages, on any GPU — AMD, NVIDIA, Intel or Apple — or just the CPU.**
28
- One command, no CUDA required, subtitles out.
1
+ # babelscribe — free, offline speech-to-text and subtitles on any GPU (AMD, NVIDIA, Intel, Apple)
2
+
3
+ <!-- mcp-name: io.github.phonology024/babelscribe -->
4
+
5
+ [![PyPI](https://img.shields.io/pypi/v/babelscribe)](https://pypi.org/project/babelscribe/)
6
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
7
+ [![MCP server](https://img.shields.io/badge/MCP-server-blue)](#use-it-from-an-ai-app-mcp--no-terminal-needed)
8
+
9
+ **babelscribe turns any audio or video file into subtitles (SRT, VTT) and text, in any of Whisper's 99 languages, on your
10
+ own graphics card — AMD Radeon, NVIDIA, Intel Arc or Apple Silicon — or the CPU. No CUDA needed, no cloud, free (MIT).**
11
+ It runs OpenAI Whisper through [whisper.cpp](https://github.com/ggml-org/whisper.cpp) with Vulkan, CUDA or Metal, and works
12
+ from the command line, from Python, or from AI apps (Claude, Codex, Antigravity, Gemini CLI, Cursor) as an MCP server.
29
13
 
30
14
  ```bash
31
15
  pip install babelscribe # Python 3.9+
@@ -60,12 +44,12 @@ Tools: `transcribe` (file → subtitles + text), `find_media` (newest audio/vide
60
44
  and double-click it (or *Settings → Extensions → Install extension*).
61
45
 
62
46
  **Everything else** runs the same command — [uv](https://docs.astral.sh/uv/) fetches babelscribe for you:
63
- `uvx --from "babelscribe[mcp]" babelscribe-mcp`
47
+ `uvx babelscribe mcp`
64
48
 
65
49
  | App | How to add it |
66
50
  |---|---|
67
- | Claude Code | `claude mcp add babelscribe -- uvx --from "babelscribe[mcp]" babelscribe-mcp` |
68
- | OpenAI Codex CLI | `codex mcp add babelscribe -- uvx --from "babelscribe[mcp]" babelscribe-mcp`, then in `~/.codex/config.toml` under `[mcp_servers.babelscribe]` set `tool_timeout_sec = 3600` and `startup_timeout_sec = 120` (defaults are 60 s / 10 s — too short for a long video or the first model download) |
51
+ | Claude Code | `claude mcp add babelscribe -- uvx babelscribe mcp` |
52
+ | OpenAI Codex CLI | `codex mcp add babelscribe -- uvx babelscribe mcp`, then in `~/.codex/config.toml` under `[mcp_servers.babelscribe]` set `tool_timeout_sec = 3600` and `startup_timeout_sec = 120` (defaults are 60 s / 10 s — too short for a long video or the first model download) |
69
53
  | Google Antigravity | agent panel → ⋯ → *MCP Servers* → *Manage MCP Servers* → *View raw config*, add the JSON below |
70
54
  | Gemini CLI | add the JSON below to `~/.gemini/settings.json` |
71
55
  | Cursor | add the JSON below to `~/.cursor/mcp.json` |
@@ -74,11 +58,11 @@ and double-click it (or *Settings → Extensions → Install extension*).
74
58
  ```json
75
59
  {
76
60
  "mcpServers": {
77
- "babelscribe": { "command": "uvx", "args": ["--from", "babelscribe[mcp]", "babelscribe-mcp"], "timeout": 3600000 }
61
+ "babelscribe": { "command": "uvx", "args": ["babelscribe", "mcp"], "timeout": 3600000 }
78
62
  }
79
63
  }
80
64
  ```
81
- (`timeout` is in milliseconds and only some apps read it.) Already installed with pip? Use `"command": "babelscribe-mcp"` with no args.
65
+ (`timeout` is in milliseconds and only some apps read it.) Already installed with pip? Use `"command": "babelscribe", "args": ["mcp"]`.
82
66
  Web-only chat apps (e.g. grok.com, chatgpt.com) can only reach servers on the internet, not your computer, so they can't use your GPU or files.
83
67
 
84
68
  ## Benchmark: real talks, human captions as the answer key
@@ -142,6 +126,32 @@ babelscribe does **not** replace those projects — it stands on whisper.cpp and
142
126
  compiling for your GPU, converting media, picking the right device, avoiding the long-file repeat bug, and combining
143
127
  a language-specific fine-tune with accurate timestamps.
144
128
 
129
+ ## FAQ
130
+ **How do I run Whisper on an AMD GPU on Windows?**
131
+ `pip install babelscribe`, then `babelscribe video.mp4`. It downloads a Vulkan build of whisper.cpp that runs on AMD Radeon
132
+ (and Intel Arc / NVIDIA) cards on Windows and Linux — no ROCm, no CUDA, no compiling.
133
+
134
+ **How do I make subtitles (SRT) from a video for free, offline?**
135
+ `babelscribe video.mp4 -f srt` writes `video.srt` next to the video. Nothing is uploaded; it runs on your own computer.
136
+
137
+ **What is the most accurate free transcription for Thai?**
138
+ `babelscribe video.mp4 -l th --accurate` — Pathumma Whisper (NECTEC) for the text plus Whisper turbo for timing:
139
+ CER 8.9% on Google FLEURS vs 15.9% for plain Whisper turbo. Thonburian Whisper is available too (`--text-model thai-thonburian`).
140
+
141
+ **Can Claude / ChatGPT Codex / Gemini transcribe a video on my computer?**
142
+ Yes — add babelscribe as an MCP server (see *Use it from an AI app*). Claude Desktop installs it with one click from
143
+ `babelscribe.mcpb`. The AI app can then find a file, transcribe it on your GPU, and proofread, translate or summarise the text.
144
+
145
+ **How fast is it?**
146
+ About 20x real time with Whisper large-v3-turbo on an AMD Radeon RX 9070 XT: a 19-minute talk in 48 seconds.
147
+
148
+ **Which languages are supported?**
149
+ All 99 Whisper languages, with automatic language detection. FLEURS error rates for 16 of them are in the benchmark table.
150
+
151
+ **Is it better than faster-whisper or WhisperX?**
152
+ Those are excellent on NVIDIA GPUs; on AMD / Intel GPUs they fall back to the CPU. babelscribe's niche is any GPU, zero setup,
153
+ and better Thai / Hindi through community fine-tunes. If you have an NVIDIA card and like Python, faster-whisper is a fine choice.
154
+
145
155
  ## Languages
146
156
  All 99 languages Whisper was trained on, auto-detected or forced with `-l`:
147
157
  af am ar as az ba be bg bn bo br bs ca cs cy da de el en es et eu fa fi fo fr gl gu ha haw he hi hr ht hu hy id is it ja jw ka kk km kn ko la lb ln lo lt lv mg mi mk ml mn mr ms mt my ne nl nn no oc pa pl ps pt ro ru sa sd si sk sl sn so sq sr su sv sw ta te tg th tk tl tr tt uk ur uz vi yi yo yue zh
@@ -175,6 +185,12 @@ babelscribe devices # GPUs whisper.cpp can see
175
185
  babelscribe models # models and fine-tunes
176
186
  babelscribe talk.mp4 --bin /path/to/whisper-cli # use your own whisper.cpp build
177
187
  ```
188
+ From Python:
189
+ ```python
190
+ from babelscribe.api import transcribe_file
191
+ r = transcribe_file("talk.mp4", lang="auto", formats=["srt", "txt"]) # or accurate=True
192
+ print(r["lang"], r["device"], r["files"]); print(r["text"][:200])
193
+ ```
178
194
  Models download on first use to `~/.babelscribe/models` (`BABELSCRIBE_MODELS` to change). Hybrid fine-tunes are converted on your
179
195
  machine from their original Hugging Face repo — install `pip install "babelscribe[finetune]"` once; converted weights are never
180
196
  redistributed, so each fine-tune keeps its own licence. Thai word boundaries: `pip install "babelscribe[thai]"`.
@@ -192,6 +208,25 @@ them to each `v*` release. Point `BABELSCRIBE_RELEASES` at another URL to self-h
192
208
  ภาษาไทยแนะนำ `babelscribe ไฟล์.mp4 -l th --accurate` — ข้อความจาก Pathumma Whisper (NECTEC) ที่ผิดน้อยที่สุดใน FLEURS (CER 8.9% เทียบ turbo 15.9%) + เวลาจาก large-v3-turbo
193
209
  หรือเลือก Thonburian Whisper เอง: `--text-model thai-thonburian`
194
210
 
211
+ ## Privacy Policy
212
+ babelscribe runs entirely on your own computer. Last updated 2026-10-06.
213
+
214
+ - **Data collection:** none. babelscribe has no telemetry, analytics, accounts or crash reporting, and never uploads your
215
+ audio, video, transcripts or file names anywhere.
216
+ - **What it processes and where:** the media file you choose is converted and transcribed locally; subtitles and text are
217
+ written to your disk (next to the file, or the folder you choose). Through MCP, the transcript text is returned to the AI
218
+ app that called the tool — what that app does with it is governed by that app's own privacy policy.
219
+ - **Network access (downloads only):** on first use it downloads the `whisper-cli` program from this project's GitHub
220
+ releases, speech models from Hugging Face (`huggingface.co/ggerganov/whisper.cpp`, and for `--accurate` Thai/Hindi the
221
+ fine-tune's own repository), and Python packages from PyPI when installed with pip/uv. These are plain downloads;
222
+ no personal data is sent. GitHub, Hugging Face and PyPI see a normal download request (IP address, user agent) under
223
+ their own privacy policies.
224
+ - **Storage and retention:** downloaded programs and models are cached in `~/.babelscribe` (or `BABELSCRIBE_HOME` /
225
+ `BABELSCRIBE_MODELS`) until you delete that folder. Outputs stay wherever they were written until you delete them.
226
+ babelscribe keeps no other data.
227
+ - **Third-party sharing:** none.
228
+ - **Contact:** open an issue at https://github.com/phonology024/babelscribe/issues
229
+
195
230
  ## Credits & licence
196
231
  MIT. Built on [whisper.cpp](https://github.com/ggml-org/whisper.cpp) (MIT) and OpenAI Whisper models (MIT).
197
232
  Fine-tunes belong to their authors: [Pathumma Whisper](https://huggingface.co/nectec/Pathumma-whisper-th-large-v3) by NECTEC,
@@ -1,2 +1,2 @@
1
1
  """babelscribe — any language, any GPU speech-to-text built on whisper.cpp."""
2
- __version__ = "0.3.0"
2
+ __version__ = "0.3.2"
@@ -1,52 +1,56 @@
1
- """babelscribe — transcribe any audio/video, in any language Whisper knows, on any GPU (AMD / NVIDIA / Intel via
2
- Vulkan, NVIDIA via CUDA, Apple via Metal) or the CPU.
3
-
4
- babelscribe talk.mp4 # auto language, turbo model, best GPU -> talk.srt + talk.json
5
- babelscribe vo.wav -l th --text-model thai-thonburian # hybrid: Thai fine-tune text + turbo timing
6
- babelscribe talk.mp4 --accurate # slower, fewest errors: large-v3 + beam search, or the best fine-tune
7
- babelscribe devices # list GPUs whisper.cpp can use
8
- babelscribe models # list models and fine-tunes"""
9
- from __future__ import annotations
10
-
11
- import argparse
12
- import sys
13
-
14
- from . import __version__, api, backend, models
15
-
16
-
17
- def main(argv: list[str] | None = None) -> None:
18
- argv = sys.argv[1:] if argv is None else argv
19
- if argv[:1] == ["models"]:
20
- print("general (99 languages):"); [print(f" {k:16} {v}") for k, v in models.GENERAL.items()]
21
- print("fine-tunes (text quality for one language, use with --text-model):")
22
- [print(f" {k:16} [{v['lang']}] {v['note']}") for k, v in models.FINETUNES.items()]
23
- return
24
- ap = argparse.ArgumentParser(prog="babelscribe", description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
25
- ap.add_argument("input", help="audio or video file, or 'devices'")
26
- ap.add_argument("-l", "--lang", default="auto", help="language code (th, en, ja, ...) or auto")
27
- ap.add_argument("-m", "--model", default="turbo", help="timing/general model (default turbo = large-v3-turbo)")
28
- ap.add_argument("--text-model", help="language fine-tune for the text (hybrid mode), e.g. thai-thonburian")
29
- ap.add_argument("--accurate", action="store_true",
30
- help="slower but fewest errors: large-v3 with beam search, or the language's best fine-tune (hybrid)")
31
- ap.add_argument("-f", "--formats", default="srt,json", help="comma list: srt,vtt,txt,json")
32
- ap.add_argument("-o", "--out", help="output base path (default: next to the input)")
33
- ap.add_argument("--device", default="auto", help="GPU id from `babelscribe devices`, or auto")
34
- ap.add_argument("--bin", help="path to a whisper-cli you built yourself")
35
- ap.add_argument("--flavor", help="prebuilt flavour to download: vulkan | cuda | metal | cpu")
36
- ap.add_argument("-v", "--verbose", action="store_true")
37
- ap.add_argument("--version", action="version", version=__version__)
38
- a = ap.parse_args(argv)
39
-
40
- if a.input == "devices":
41
- _, _, found = api.setup(a.model, a.bin, a.flavor)
42
- for d in found:
43
- print(f" [{d['id']}] {d['backend']:6} {d['name']} {d.get('detail', '')}")
44
- print(f"auto picks device {backend.pick_device(found)}"); return
45
- try:
46
- api.transcribe_file(a.input, a.lang, a.model, a.text_model, a.accurate,
47
- [f.strip() for f in a.formats.split(",") if f.strip()], a.out, a.device, a.bin, a.flavor, a.verbose)
48
- except FileNotFoundError as e:
49
- raise SystemExit(str(e))
50
-
51
- if __name__ == "__main__":
52
- main()
1
+ """babelscribe — transcribe any audio/video, in any language Whisper knows, on any GPU (AMD / NVIDIA / Intel via
2
+ Vulkan, NVIDIA via CUDA, Apple via Metal) or the CPU.
3
+
4
+ babelscribe talk.mp4 # auto language, turbo model, best GPU -> talk.srt + talk.json
5
+ babelscribe vo.wav -l th --text-model thai-thonburian # hybrid: Thai fine-tune text + turbo timing
6
+ babelscribe talk.mp4 --accurate # slower, fewest errors: large-v3 + beam search, or the best fine-tune
7
+ babelscribe mcp # MCP server for AI apps (Claude, Codex, Gemini CLI, ...)
8
+ babelscribe devices # list GPUs whisper.cpp can use
9
+ babelscribe models # list models and fine-tunes"""
10
+ from __future__ import annotations
11
+
12
+ import argparse
13
+ import sys
14
+
15
+ from . import __version__, api, backend, models
16
+
17
+
18
+ def main(argv: list[str] | None = None) -> None:
19
+ argv = sys.argv[1:] if argv is None else argv
20
+ if argv[:1] == ["mcp"]: # MCP server for AI apps (stdio)
21
+ from .mcp_server import main as serve
22
+ return serve()
23
+ if argv[:1] == ["models"]:
24
+ print("general (99 languages):"); [print(f" {k:16} {v}") for k, v in models.GENERAL.items()]
25
+ print("fine-tunes (text quality for one language, use with --text-model):")
26
+ [print(f" {k:16} [{v['lang']}] {v['note']}") for k, v in models.FINETUNES.items()]
27
+ return
28
+ ap = argparse.ArgumentParser(prog="babelscribe", description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
29
+ ap.add_argument("input", help="audio or video file, or 'devices'")
30
+ ap.add_argument("-l", "--lang", default="auto", help="language code (th, en, ja, ...) or auto")
31
+ ap.add_argument("-m", "--model", default="turbo", help="timing/general model (default turbo = large-v3-turbo)")
32
+ ap.add_argument("--text-model", help="language fine-tune for the text (hybrid mode), e.g. thai-thonburian")
33
+ ap.add_argument("--accurate", action="store_true",
34
+ help="slower but fewest errors: large-v3 with beam search, or the language's best fine-tune (hybrid)")
35
+ ap.add_argument("-f", "--formats", default="srt,json", help="comma list: srt,vtt,txt,json")
36
+ ap.add_argument("-o", "--out", help="output base path (default: next to the input)")
37
+ ap.add_argument("--device", default="auto", help="GPU id from `babelscribe devices`, or auto")
38
+ ap.add_argument("--bin", help="path to a whisper-cli you built yourself")
39
+ ap.add_argument("--flavor", help="prebuilt flavour to download: vulkan | cuda | metal | cpu")
40
+ ap.add_argument("-v", "--verbose", action="store_true")
41
+ ap.add_argument("--version", action="version", version=__version__)
42
+ a = ap.parse_args(argv)
43
+
44
+ if a.input == "devices":
45
+ _, _, found = api.setup(a.model, a.bin, a.flavor)
46
+ for d in found:
47
+ print(f" [{d['id']}] {d['backend']:6} {d['name']} {d.get('detail', '')}")
48
+ print(f"auto picks device {backend.pick_device(found)}"); return
49
+ try:
50
+ api.transcribe_file(a.input, a.lang, a.model, a.text_model, a.accurate,
51
+ [f.strip() for f in a.formats.split(",") if f.strip()], a.out, a.device, a.bin, a.flavor, a.verbose)
52
+ except FileNotFoundError as e:
53
+ raise SystemExit(str(e))
54
+
55
+ if __name__ == "__main__":
56
+ main()
@@ -1,136 +1,139 @@
1
- """babelscribe as an MCP server: any MCP client (Claude Desktop / Code, Codex, Antigravity, Gemini CLI, Cursor,
2
- VS Code, ...) can transcribe local audio/video on the user's own GPU.
3
-
4
- babelscribe-mcp # stdio server; clients launch it themselves (see README "Use from an AI app")
5
-
6
- Needs: pip install "babelscribe[mcp]" (Python 3.10+)."""
7
-
8
- import io
9
- import os
10
- import sys
11
- import time
12
- from pathlib import Path
13
-
14
- MEDIA = {".mp4", ".mkv", ".mov", ".avi", ".webm", ".m4v", ".mp3", ".wav", ".m4a", ".aac", ".flac", ".ogg", ".opus", ".wma"}
15
- TEXT_LIMIT = 30000 # characters of transcript returned to the model; the full text is always in the files
16
-
17
-
18
- def _guard_stdout():
19
- """stdout is the JSON-RPC channel. Keep a private handle to it for the protocol and point everything else
20
- (print, logging, stray library output) at stderr so it can never corrupt a message."""
21
- proto = io.TextIOWrapper(os.fdopen(os.dup(sys.stdout.fileno()), "wb"), encoding="utf-8", newline="\n",
22
- write_through=True)
23
- sys.stdout = sys.stderr
24
- return proto
25
-
26
-
27
- def build():
28
- import anyio
29
- from mcp.server.mcpserver import Context, MCPServer
30
-
31
- from . import __version__, api, backend, models
32
-
33
- srv = MCPServer("babelscribe", version=__version__, instructions=(
34
- "Local speech-to-text on the user's own GPU (AMD/NVIDIA/Intel/Apple) for 99 languages. Needs a file path on "
35
- "this computer: if the user only names or describes a file, call find_media first. Subtitles (.srt) and text "
36
- "(.txt) are written next to the media file unless output_dir is given. Use accurate=true when the user wants "
37
- "the fewest mistakes (slower); it is much better for Thai and Hindi. The first run downloads the model "
38
- "(~1.6 GB), so it can take a few minutes. After transcribing you can proofread names, translate, or "
39
- "summarise from the returned text."))
40
-
41
- @srv.tool()
42
- async def transcribe(file_path: str, language: str = "auto", accurate: bool = False, formats: str = "srt,txt",
43
- output_dir: str = "", ctx: Context | None = None) -> dict:
44
- """Transcribe a local audio or video file into subtitles / text.
45
-
46
- file_path: absolute path to the media file (mp4, mkv, mov, mp3, wav, m4a, ...).
47
- language: ISO code such as en, th, ja, es, hi — or "auto" to detect.
48
- accurate: slower, fewest errors (large-v3 with beam search, or the best fine-tune for Thai / Hindi).
49
- formats: comma list from srt, vtt, txt, json.
50
- output_dir: folder for the output files; empty = next to the media file.
51
- Returns the files written, detected language, device used, and the transcript text."""
52
- src = Path(file_path).expanduser()
53
- out = Path(output_dir).expanduser() / src.stem if output_dir else None
54
- if out:
55
- out.parent.mkdir(parents=True, exist_ok=True)
56
- fmts = [f.strip() for f in formats.split(",") if f.strip()]
57
- lines: list[str] = []
58
- done = anyio.Event()
59
-
60
- async def heartbeat(): # progress keeps clients from timing out on long files
61
- t0 = time.time()
62
- while not done.is_set():
63
- with anyio.move_on_after(5):
64
- await done.wait()
65
- if ctx and not done.is_set():
66
- await ctx.report_progress(time.time() - t0, None, lines[-1] if lines else "starting ...")
67
-
68
- result: dict = {}
69
- async with anyio.create_task_group() as tg:
70
- tg.start_soon(heartbeat)
71
- try:
72
- result = await anyio.to_thread.run_sync(lambda: api.transcribe_file(
73
- src, language, accurate=accurate, formats=fmts, out=out, log=lambda m: (lines.append(m), print(m))))
74
- finally:
75
- done.set()
76
- text = result.pop("text", "")
77
- result["text"] = text[:TEXT_LIMIT]
78
- if len(text) > TEXT_LIMIT:
79
- result["note"] = f"transcript truncated to {TEXT_LIMIT} characters; the full text is in the files listed"
80
- return result
81
-
82
- @srv.tool()
83
- def find_media(name_contains: str = "", folder: str = "", limit: int = 15) -> list[dict]:
84
- """Find audio/video files on this computer, newest first. Searches Downloads, Videos, Desktop, Music and
85
- Documents (two levels deep) unless folder is given. Use it when the user names a file without a full path."""
86
- home = Path.home()
87
- roots = [Path(folder).expanduser()] if folder else [home / d for d in ("Downloads", "Videos", "Desktop", "Music", "Documents", "Movies")]
88
- hits = []
89
- for r in roots:
90
- if not r.is_dir():
91
- continue
92
- for p in [*r.glob("*"), *r.glob("*/*"), *(r.glob("*/*/*") if folder else [])]:
93
- if p.suffix.lower() in MEDIA and name_contains.lower() in p.name.lower():
94
- try:
95
- st = p.stat()
96
- except OSError:
97
- continue
98
- hits.append((st.st_mtime, p, st.st_size))
99
- hits.sort(reverse=True)
100
- return [{"path": str(p), "size_mb": round(sz / 1e6, 1), "modified": time.strftime("%Y-%m-%d %H:%M", time.localtime(t))}
101
- for t, p, sz in hits[:limit]]
102
-
103
- @srv.tool()
104
- def list_devices() -> dict:
105
- """GPUs whisper.cpp can use on this computer and which one babelscribe picks automatically."""
106
- _, _, found = api.setup()
107
- return {"devices": found, "auto_pick": backend.pick_device(found)}
108
-
109
- @srv.tool()
110
- def list_languages_and_models() -> dict:
111
- """Models available, and which model --accurate uses per language."""
112
- return {"general": models.GENERAL, "fine_tunes": {k: v["note"] for k, v in models.FINETUNES.items()},
113
- "accurate_per_language": {**models.ACCURATE, "*": "large-v3 beam 5"}}
114
-
115
- return srv
116
-
117
-
118
- def main() -> None:
119
- try:
120
- import anyio
121
- from mcp.server.stdio import stdio_server
122
- except ImportError:
123
- raise SystemExit('the MCP server needs: pip install "babelscribe[mcp]" (Python 3.10+)')
124
- proto = _guard_stdout()
125
- srv = build()
126
-
127
- async def run():
128
- async with stdio_server(stdout=anyio.wrap_file(proto)) as (r, w):
129
- low = srv._lowlevel_server
130
- await low.run(r, w, low.create_initialization_options())
131
-
132
- anyio.run(run)
133
-
134
-
135
- if __name__ == "__main__":
136
- main()
1
+ """babelscribe as an MCP server: any MCP client (Claude Desktop / Code, Codex, Antigravity, Gemini CLI, Cursor,
2
+ VS Code, ...) can transcribe local audio/video on the user's own GPU.
3
+
4
+ babelscribe mcp # stdio server; AI apps launch it themselves (see README "Use it from an AI app")
5
+
6
+ Needs Python 3.10+ (the mcp SDK)."""
7
+
8
+ import io
9
+ import os
10
+ import sys
11
+ import time
12
+ from pathlib import Path
13
+
14
+ MEDIA = {".mp4", ".mkv", ".mov", ".avi", ".webm", ".m4v", ".mp3", ".wav", ".m4a", ".aac", ".flac", ".ogg", ".opus", ".wma"}
15
+ TEXT_LIMIT = 30000 # characters of transcript returned to the model; the full text is always in the files
16
+
17
+
18
+ def _guard_stdout():
19
+ """stdout is the JSON-RPC channel. Keep a private handle to it for the protocol and point everything else
20
+ (print, logging, stray library output) at stderr so it can never corrupt a message."""
21
+ proto = io.TextIOWrapper(os.fdopen(os.dup(sys.stdout.fileno()), "wb"), encoding="utf-8", newline="\n",
22
+ write_through=True)
23
+ sys.stdout = sys.stderr
24
+ return proto
25
+
26
+
27
+ def build():
28
+ import anyio
29
+ from mcp.server.mcpserver import Context, MCPServer
30
+ from mcp.types import ToolAnnotations
31
+
32
+ from . import __version__, api, backend, models
33
+
34
+ srv = MCPServer("babelscribe", version=__version__, instructions=(
35
+ "Local speech-to-text on the user's own GPU (AMD/NVIDIA/Intel/Apple) for 99 languages. Needs a file path on "
36
+ "this computer: if the user only names or describes a file, call find_media first. Subtitles (.srt) and text "
37
+ "(.txt) are written next to the media file unless output_dir is given. Use accurate=true when the user wants "
38
+ "the fewest mistakes (slower); it is much better for Thai and Hindi. The first run downloads the model "
39
+ "(~1.6 GB), so it can take a few minutes. After transcribing you can proofread names, translate, or "
40
+ "summarise from the returned text."))
41
+
42
+ # transcribe writes new .srt/.txt files (re-running rewrites its own output); first use downloads whisper-cli and the model
43
+ @srv.tool(title="Transcribe audio/video to subtitles", annotations=ToolAnnotations(
44
+ read_only_hint=False, destructive_hint=False, idempotent_hint=True, open_world_hint=True))
45
+ async def transcribe(file_path: str, language: str = "auto", accurate: bool = False, formats: str = "srt,txt",
46
+ output_dir: str = "", ctx: Context | None = None) -> dict:
47
+ """Transcribe a local audio or video file into subtitles / text.
48
+
49
+ file_path: absolute path to the media file (mp4, mkv, mov, mp3, wav, m4a, ...).
50
+ language: ISO code such as en, th, ja, es, hi — or "auto" to detect.
51
+ accurate: slower, fewest errors (large-v3 with beam search, or the best fine-tune for Thai / Hindi).
52
+ formats: comma list from srt, vtt, txt, json.
53
+ output_dir: folder for the output files; empty = next to the media file.
54
+ Returns the files written, detected language, device used, and the transcript text."""
55
+ src = Path(file_path).expanduser()
56
+ out = Path(output_dir).expanduser() / src.stem if output_dir else None
57
+ if out:
58
+ out.parent.mkdir(parents=True, exist_ok=True)
59
+ fmts = [f.strip() for f in formats.split(",") if f.strip()]
60
+ lines: list[str] = []
61
+ done = anyio.Event()
62
+
63
+ async def heartbeat(): # progress keeps clients from timing out on long files
64
+ t0 = time.time()
65
+ while not done.is_set():
66
+ with anyio.move_on_after(5):
67
+ await done.wait()
68
+ if ctx and not done.is_set():
69
+ await ctx.report_progress(time.time() - t0, None, lines[-1] if lines else "starting ...")
70
+
71
+ result: dict = {}
72
+ async with anyio.create_task_group() as tg:
73
+ tg.start_soon(heartbeat)
74
+ try:
75
+ result = await anyio.to_thread.run_sync(lambda: api.transcribe_file(
76
+ src, language, accurate=accurate, formats=fmts, out=out, log=lambda m: (lines.append(m), print(m))))
77
+ finally:
78
+ done.set()
79
+ text = result.pop("text", "")
80
+ result["text"] = text[:TEXT_LIMIT]
81
+ if len(text) > TEXT_LIMIT:
82
+ result["note"] = f"transcript truncated to {TEXT_LIMIT} characters; the full text is in the files listed"
83
+ return result
84
+
85
+ @srv.tool(title="Find audio/video files", annotations=ToolAnnotations(read_only_hint=True, open_world_hint=False))
86
+ def find_media(name_contains: str = "", folder: str = "", limit: int = 15) -> list[dict]:
87
+ """Find audio/video files on this computer, newest first. Searches Downloads, Videos, Desktop, Music and
88
+ Documents (two levels deep) unless folder is given. Use it when the user names a file without a full path."""
89
+ home = Path.home()
90
+ roots = [Path(folder).expanduser()] if folder else [home / d for d in ("Downloads", "Videos", "Desktop", "Music", "Documents", "Movies")]
91
+ hits = []
92
+ for r in roots:
93
+ if not r.is_dir():
94
+ continue
95
+ for p in [*r.glob("*"), *r.glob("*/*"), *(r.glob("*/*/*") if folder else [])]:
96
+ if p.suffix.lower() in MEDIA and name_contains.lower() in p.name.lower():
97
+ try:
98
+ st = p.stat()
99
+ except OSError:
100
+ continue
101
+ hits.append((st.st_mtime, p, st.st_size))
102
+ hits.sort(reverse=True)
103
+ return [{"path": str(p), "size_mb": round(sz / 1e6, 1), "modified": time.strftime("%Y-%m-%d %H:%M", time.localtime(t))}
104
+ for t, p, sz in hits[:limit]]
105
+
106
+ @srv.tool(title="List GPUs", annotations=ToolAnnotations(read_only_hint=True, open_world_hint=True))
107
+ def list_devices() -> dict:
108
+ """GPUs whisper.cpp can use on this computer and which one babelscribe picks automatically."""
109
+ _, _, found = api.setup()
110
+ return {"devices": found, "auto_pick": backend.pick_device(found)}
111
+
112
+ @srv.tool(title="List languages and models", annotations=ToolAnnotations(read_only_hint=True, open_world_hint=False))
113
+ def list_languages_and_models() -> dict:
114
+ """Models available, and which model --accurate uses per language."""
115
+ return {"general": models.GENERAL, "fine_tunes": {k: v["note"] for k, v in models.FINETUNES.items()},
116
+ "accurate_per_language": {**models.ACCURATE, "*": "large-v3 beam 5"}}
117
+
118
+ return srv
119
+
120
+
121
+ def main() -> None:
122
+ try:
123
+ import anyio
124
+ from mcp.server.stdio import stdio_server
125
+ except ImportError:
126
+ raise SystemExit('the MCP server needs Python 3.10+ (the mcp SDK)')
127
+ proto = _guard_stdout()
128
+ srv = build()
129
+
130
+ async def run():
131
+ async with stdio_server(stdout=anyio.wrap_file(proto)) as (r, w):
132
+ low = srv._lowlevel_server
133
+ await low.run(r, w, low.create_initialization_options())
134
+
135
+ anyio.run(run)
136
+
137
+
138
+ if __name__ == "__main__":
139
+ main()
@@ -1,7 +1,58 @@
1
- # babelscribe
1
+ Metadata-Version: 2.4
2
+ Name: babelscribe
3
+ Version: 0.3.2
4
+ Summary: Transcribe any audio/video in 99 languages on any GPU (AMD, NVIDIA, Intel via Vulkan; NVIDIA via CUDA; Apple via Metal) or CPU — whisper.cpp with batteries included.
5
+ License: MIT
6
+ Project-URL: Homepage, https://github.com/phonology024/babelscribe
7
+ Project-URL: Documentation, https://phonology024.github.io/babelscribe/
8
+ Project-URL: Repository, https://github.com/phonology024/babelscribe
9
+ Project-URL: Issues, https://github.com/phonology024/babelscribe/issues
10
+ Keywords: whisper,speech-to-text,transcription,subtitles,vulkan,amd,gpu,thai,mcp,mcp-server,claude,codex,srt,speech-recognition,offline,radeon,intel-arc,openai-whisper
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Environment :: Console
13
+ Classifier: Environment :: GPU
14
+ Classifier: Intended Audience :: End Users/Desktop
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: License :: OSI Approved :: MIT License
17
+ Classifier: Operating System :: Microsoft :: Windows
18
+ Classifier: Operating System :: POSIX :: Linux
19
+ Classifier: Operating System :: MacOS
20
+ Classifier: Programming Language :: Python :: 3
21
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
22
+ Classifier: Topic :: Multimedia :: Video
23
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
24
+ Classifier: Natural Language :: Thai
25
+ Classifier: Natural Language :: English
26
+ Requires-Python: >=3.9
27
+ Description-Content-Type: text/markdown
28
+ License-File: LICENSE
29
+ Requires-Dist: imageio-ffmpeg>=0.5
30
+ Requires-Dist: mcp<3,>=2.3; python_version >= "3.10"
31
+ Requires-Dist: anyio>=4; python_version >= "3.10"
32
+ Provides-Extra: finetune
33
+ Requires-Dist: torch; extra == "finetune"
34
+ Requires-Dist: transformers; extra == "finetune"
35
+ Requires-Dist: huggingface_hub; extra == "finetune"
36
+ Requires-Dist: numpy; extra == "finetune"
37
+ Provides-Extra: thai
38
+ Requires-Dist: pythainlp; extra == "thai"
39
+ Provides-Extra: mcp
40
+ Requires-Dist: mcp<3,>=2.3; python_version >= "3.10" and extra == "mcp"
41
+ Requires-Dist: anyio>=4; extra == "mcp"
42
+ Dynamic: license-file
2
43
 
3
- **Transcribe any audio or video, in any of Whisper's 99 languages, on any GPU — AMD, NVIDIA, Intel or Apple — or just the CPU.**
4
- One command, no CUDA required, subtitles out.
44
+ # babelscribe — free, offline speech-to-text and subtitles on any GPU (AMD, NVIDIA, Intel, Apple)
45
+
46
+ <!-- mcp-name: io.github.phonology024/babelscribe -->
47
+
48
+ [![PyPI](https://img.shields.io/pypi/v/babelscribe)](https://pypi.org/project/babelscribe/)
49
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
50
+ [![MCP server](https://img.shields.io/badge/MCP-server-blue)](#use-it-from-an-ai-app-mcp--no-terminal-needed)
51
+
52
+ **babelscribe turns any audio or video file into subtitles (SRT, VTT) and text, in any of Whisper's 99 languages, on your
53
+ own graphics card — AMD Radeon, NVIDIA, Intel Arc or Apple Silicon — or the CPU. No CUDA needed, no cloud, free (MIT).**
54
+ It runs OpenAI Whisper through [whisper.cpp](https://github.com/ggml-org/whisper.cpp) with Vulkan, CUDA or Metal, and works
55
+ from the command line, from Python, or from AI apps (Claude, Codex, Antigravity, Gemini CLI, Cursor) as an MCP server.
5
56
 
6
57
  ```bash
7
58
  pip install babelscribe # Python 3.9+
@@ -36,12 +87,12 @@ Tools: `transcribe` (file → subtitles + text), `find_media` (newest audio/vide
36
87
  and double-click it (or *Settings → Extensions → Install extension*).
37
88
 
38
89
  **Everything else** runs the same command — [uv](https://docs.astral.sh/uv/) fetches babelscribe for you:
39
- `uvx --from "babelscribe[mcp]" babelscribe-mcp`
90
+ `uvx babelscribe mcp`
40
91
 
41
92
  | App | How to add it |
42
93
  |---|---|
43
- | Claude Code | `claude mcp add babelscribe -- uvx --from "babelscribe[mcp]" babelscribe-mcp` |
44
- | OpenAI Codex CLI | `codex mcp add babelscribe -- uvx --from "babelscribe[mcp]" babelscribe-mcp`, then in `~/.codex/config.toml` under `[mcp_servers.babelscribe]` set `tool_timeout_sec = 3600` and `startup_timeout_sec = 120` (defaults are 60 s / 10 s — too short for a long video or the first model download) |
94
+ | Claude Code | `claude mcp add babelscribe -- uvx babelscribe mcp` |
95
+ | OpenAI Codex CLI | `codex mcp add babelscribe -- uvx babelscribe mcp`, then in `~/.codex/config.toml` under `[mcp_servers.babelscribe]` set `tool_timeout_sec = 3600` and `startup_timeout_sec = 120` (defaults are 60 s / 10 s — too short for a long video or the first model download) |
45
96
  | Google Antigravity | agent panel → ⋯ → *MCP Servers* → *Manage MCP Servers* → *View raw config*, add the JSON below |
46
97
  | Gemini CLI | add the JSON below to `~/.gemini/settings.json` |
47
98
  | Cursor | add the JSON below to `~/.cursor/mcp.json` |
@@ -50,11 +101,11 @@ and double-click it (or *Settings → Extensions → Install extension*).
50
101
  ```json
51
102
  {
52
103
  "mcpServers": {
53
- "babelscribe": { "command": "uvx", "args": ["--from", "babelscribe[mcp]", "babelscribe-mcp"], "timeout": 3600000 }
104
+ "babelscribe": { "command": "uvx", "args": ["babelscribe", "mcp"], "timeout": 3600000 }
54
105
  }
55
106
  }
56
107
  ```
57
- (`timeout` is in milliseconds and only some apps read it.) Already installed with pip? Use `"command": "babelscribe-mcp"` with no args.
108
+ (`timeout` is in milliseconds and only some apps read it.) Already installed with pip? Use `"command": "babelscribe", "args": ["mcp"]`.
58
109
  Web-only chat apps (e.g. grok.com, chatgpt.com) can only reach servers on the internet, not your computer, so they can't use your GPU or files.
59
110
 
60
111
  ## Benchmark: real talks, human captions as the answer key
@@ -118,6 +169,32 @@ babelscribe does **not** replace those projects — it stands on whisper.cpp and
118
169
  compiling for your GPU, converting media, picking the right device, avoiding the long-file repeat bug, and combining
119
170
  a language-specific fine-tune with accurate timestamps.
120
171
 
172
+ ## FAQ
173
+ **How do I run Whisper on an AMD GPU on Windows?**
174
+ `pip install babelscribe`, then `babelscribe video.mp4`. It downloads a Vulkan build of whisper.cpp that runs on AMD Radeon
175
+ (and Intel Arc / NVIDIA) cards on Windows and Linux — no ROCm, no CUDA, no compiling.
176
+
177
+ **How do I make subtitles (SRT) from a video for free, offline?**
178
+ `babelscribe video.mp4 -f srt` writes `video.srt` next to the video. Nothing is uploaded; it runs on your own computer.
179
+
180
+ **What is the most accurate free transcription for Thai?**
181
+ `babelscribe video.mp4 -l th --accurate` — Pathumma Whisper (NECTEC) for the text plus Whisper turbo for timing:
182
+ CER 8.9% on Google FLEURS vs 15.9% for plain Whisper turbo. Thonburian Whisper is available too (`--text-model thai-thonburian`).
183
+
184
+ **Can Claude / ChatGPT Codex / Gemini transcribe a video on my computer?**
185
+ Yes — add babelscribe as an MCP server (see *Use it from an AI app*). Claude Desktop installs it with one click from
186
+ `babelscribe.mcpb`. The AI app can then find a file, transcribe it on your GPU, and proofread, translate or summarise the text.
187
+
188
+ **How fast is it?**
189
+ About 20x real time with Whisper large-v3-turbo on an AMD Radeon RX 9070 XT: a 19-minute talk in 48 seconds.
190
+
191
+ **Which languages are supported?**
192
+ All 99 Whisper languages, with automatic language detection. FLEURS error rates for 16 of them are in the benchmark table.
193
+
194
+ **Is it better than faster-whisper or WhisperX?**
195
+ Those are excellent on NVIDIA GPUs; on AMD / Intel GPUs they fall back to the CPU. babelscribe's niche is any GPU, zero setup,
196
+ and better Thai / Hindi through community fine-tunes. If you have an NVIDIA card and like Python, faster-whisper is a fine choice.
197
+
121
198
  ## Languages
122
199
  All 99 languages Whisper was trained on, auto-detected or forced with `-l`:
123
200
  af am ar as az ba be bg bn bo br bs ca cs cy da de el en es et eu fa fi fo fr gl gu ha haw he hi hr ht hu hy id is it ja jw ka kk km kn ko la lb ln lo lt lv mg mi mk ml mn mr ms mt my ne nl nn no oc pa pl ps pt ro ru sa sd si sk sl sn so sq sr su sv sw ta te tg th tk tl tr tt uk ur uz vi yi yo yue zh
@@ -151,6 +228,12 @@ babelscribe devices # GPUs whisper.cpp can see
151
228
  babelscribe models # models and fine-tunes
152
229
  babelscribe talk.mp4 --bin /path/to/whisper-cli # use your own whisper.cpp build
153
230
  ```
231
+ From Python:
232
+ ```python
233
+ from babelscribe.api import transcribe_file
234
+ r = transcribe_file("talk.mp4", lang="auto", formats=["srt", "txt"]) # or accurate=True
235
+ print(r["lang"], r["device"], r["files"]); print(r["text"][:200])
236
+ ```
154
237
  Models download on first use to `~/.babelscribe/models` (`BABELSCRIBE_MODELS` to change). Hybrid fine-tunes are converted on your
155
238
  machine from their original Hugging Face repo — install `pip install "babelscribe[finetune]"` once; converted weights are never
156
239
  redistributed, so each fine-tune keeps its own licence. Thai word boundaries: `pip install "babelscribe[thai]"`.
@@ -168,6 +251,25 @@ them to each `v*` release. Point `BABELSCRIBE_RELEASES` at another URL to self-h
168
251
  ภาษาไทยแนะนำ `babelscribe ไฟล์.mp4 -l th --accurate` — ข้อความจาก Pathumma Whisper (NECTEC) ที่ผิดน้อยที่สุดใน FLEURS (CER 8.9% เทียบ turbo 15.9%) + เวลาจาก large-v3-turbo
169
252
  หรือเลือก Thonburian Whisper เอง: `--text-model thai-thonburian`
170
253
 
254
+ ## Privacy Policy
255
+ babelscribe runs entirely on your own computer. Last updated 2026-10-06.
256
+
257
+ - **Data collection:** none. babelscribe has no telemetry, analytics, accounts or crash reporting, and never uploads your
258
+ audio, video, transcripts or file names anywhere.
259
+ - **What it processes and where:** the media file you choose is converted and transcribed locally; subtitles and text are
260
+ written to your disk (next to the file, or the folder you choose). Through MCP, the transcript text is returned to the AI
261
+ app that called the tool — what that app does with it is governed by that app's own privacy policy.
262
+ - **Network access (downloads only):** on first use it downloads the `whisper-cli` program from this project's GitHub
263
+ releases, speech models from Hugging Face (`huggingface.co/ggerganov/whisper.cpp`, and for `--accurate` Thai/Hindi the
264
+ fine-tune's own repository), and Python packages from PyPI when installed with pip/uv. These are plain downloads;
265
+ no personal data is sent. GitHub, Hugging Face and PyPI see a normal download request (IP address, user agent) under
266
+ their own privacy policies.
267
+ - **Storage and retention:** downloaded programs and models are cached in `~/.babelscribe` (or `BABELSCRIBE_HOME` /
268
+ `BABELSCRIBE_MODELS`) until you delete that folder. Outputs stay wherever they were written until you delete them.
269
+ babelscribe keeps no other data.
270
+ - **Third-party sharing:** none.
271
+ - **Contact:** open an issue at https://github.com/phonology024/babelscribe/issues
272
+
171
273
  ## Credits & licence
172
274
  MIT. Built on [whisper.cpp](https://github.com/ggml-org/whisper.cpp) (MIT) and OpenAI Whisper models (MIT).
173
275
  Fine-tunes belong to their authors: [Pathumma Whisper](https://huggingface.co/nectec/Pathumma-whisper-th-large-v3) by NECTEC,
@@ -1,5 +1,9 @@
1
1
  imageio-ffmpeg>=0.5
2
2
 
3
+ [:python_version >= "3.10"]
4
+ mcp<3,>=2.3
5
+ anyio>=4
6
+
3
7
  [finetune]
4
8
  torch
5
9
  transformers
@@ -0,0 +1,48 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "babelscribe"
7
+ version = "0.3.2"
8
+ description = "Transcribe any audio/video in 99 languages on any GPU (AMD, NVIDIA, Intel via Vulkan; NVIDIA via CUDA; Apple via Metal) or CPU — whisper.cpp with batteries included."
9
+ readme = "README.md"
10
+ license = { text = "MIT" }
11
+ requires-python = ">=3.9"
12
+ dependencies = ["imageio-ffmpeg>=0.5", "mcp>=2.3,<3; python_version>='3.10'", "anyio>=4; python_version>='3.10'"]
13
+ classifiers = [
14
+ "Development Status :: 4 - Beta",
15
+ "Environment :: Console",
16
+ "Environment :: GPU",
17
+ "Intended Audience :: End Users/Desktop",
18
+ "Intended Audience :: Developers",
19
+ "License :: OSI Approved :: MIT License",
20
+ "Operating System :: Microsoft :: Windows",
21
+ "Operating System :: POSIX :: Linux",
22
+ "Operating System :: MacOS",
23
+ "Programming Language :: Python :: 3",
24
+ "Topic :: Multimedia :: Sound/Audio :: Speech",
25
+ "Topic :: Multimedia :: Video",
26
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
27
+ "Natural Language :: Thai",
28
+ "Natural Language :: English",
29
+ ]
30
+ keywords = ["whisper", "speech-to-text", "transcription", "subtitles", "vulkan", "amd", "gpu", "thai", "mcp", "mcp-server", "claude", "codex", "srt", "speech-recognition", "offline", "radeon", "intel-arc", "openai-whisper"]
31
+
32
+ [project.urls]
33
+ Homepage = "https://github.com/phonology024/babelscribe"
34
+ Documentation = "https://phonology024.github.io/babelscribe/"
35
+ Repository = "https://github.com/phonology024/babelscribe"
36
+ Issues = "https://github.com/phonology024/babelscribe/issues"
37
+
38
+ [project.optional-dependencies]
39
+ finetune = ["torch", "transformers", "huggingface_hub", "numpy"]
40
+ thai = ["pythainlp"]
41
+ mcp = ["mcp>=2.3,<3; python_version>='3.10'", "anyio>=4"]
42
+
43
+ [project.scripts]
44
+ babelscribe = "babelscribe.cli:main"
45
+ babelscribe-mcp = "babelscribe.mcp_server:main"
46
+
47
+ [tool.setuptools]
48
+ packages = ["babelscribe"]
@@ -1,29 +0,0 @@
1
- [build-system]
2
- requires = ["setuptools>=68"]
3
- build-backend = "setuptools.build_meta"
4
-
5
- [project]
6
- name = "babelscribe"
7
- version = "0.3.0"
8
- description = "Transcribe any audio/video in 99 languages on any GPU (AMD, NVIDIA, Intel via Vulkan; NVIDIA via CUDA; Apple via Metal) or CPU — whisper.cpp with batteries included."
9
- readme = "README.md"
10
- license = { text = "MIT" }
11
- requires-python = ">=3.9"
12
- dependencies = ["imageio-ffmpeg>=0.5"]
13
- keywords = ["whisper", "speech-to-text", "transcription", "subtitles", "vulkan", "amd", "gpu", "thai", "mcp", "claude", "codex"]
14
-
15
- [project.urls]
16
- Homepage = "https://github.com/phonology024/babelscribe"
17
- Issues = "https://github.com/phonology024/babelscribe/issues"
18
-
19
- [project.optional-dependencies]
20
- finetune = ["torch", "transformers", "huggingface_hub", "numpy"]
21
- thai = ["pythainlp"]
22
- mcp = ["mcp>=2.3,<3; python_version>='3.10'", "anyio>=4"]
23
-
24
- [project.scripts]
25
- babelscribe = "babelscribe.cli:main"
26
- babelscribe-mcp = "babelscribe.mcp_server:main"
27
-
28
- [tool.setuptools]
29
- packages = ["babelscribe"]
File without changes
File without changes