babelscribe 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- babelscribe/__init__.py +2 -0
- babelscribe/__main__.py +3 -0
- babelscribe/align.py +111 -0
- babelscribe/backend.py +94 -0
- babelscribe/cli.py +80 -0
- babelscribe/hybrid.py +51 -0
- babelscribe/models.py +94 -0
- babelscribe/transcribe.py +69 -0
- babelscribe/writers.py +29 -0
- babelscribe-0.2.0.dist-info/METADATA +161 -0
- babelscribe-0.2.0.dist-info/RECORD +15 -0
- babelscribe-0.2.0.dist-info/WHEEL +5 -0
- babelscribe-0.2.0.dist-info/entry_points.txt +2 -0
- babelscribe-0.2.0.dist-info/licenses/LICENSE +21 -0
- babelscribe-0.2.0.dist-info/top_level.txt +1 -0
babelscribe/__init__.py
ADDED
babelscribe/__main__.py
ADDED
babelscribe/align.py
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""Hybrid mode: accurate TEXT from a language fine-tune (often without usable timestamps) + accurate TIMING from a
|
|
2
|
+
general model. Both transcripts are aligned character by character (spaces/punctuation ignored, works for languages
|
|
3
|
+
without spaces such as Thai, Chinese, Japanese); every character of the accurate text gets a time, and the text is
|
|
4
|
+
re-segmented on the timing model's segment boundaries."""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import difflib
|
|
8
|
+
import re
|
|
9
|
+
import unicodedata
|
|
10
|
+
|
|
11
|
+
_SKIP = re.compile(r"[\s\W_]", re.UNICODE)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _chars_with_times(segs: list[dict]) -> tuple[str, list[float]]:
|
|
15
|
+
chars, times = [], []
|
|
16
|
+
for s in segs:
|
|
17
|
+
toks = s.get("words") or [[s["text"], s["start"], s["end"]]]
|
|
18
|
+
for text, t0, t1 in toks:
|
|
19
|
+
clean = [c for c in unicodedata.normalize("NFC", text) if not _SKIP.match(c)]
|
|
20
|
+
for i, c in enumerate(clean):
|
|
21
|
+
chars.append(c)
|
|
22
|
+
times.append(t0 + (t1 - t0) * (i + .5) / max(1, len(clean)))
|
|
23
|
+
return "".join(chars), times
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _word_breaks(text: str) -> set[int]:
|
|
27
|
+
"""Positions where a new word starts. Spaces everywhere; for Thai/Lao/Khmer/Burmese use pythainlp if installed."""
|
|
28
|
+
br = {i for i, c in enumerate(text) if i and (text[i - 1].isspace() or c.isspace())}
|
|
29
|
+
if any('' <= c <= '' for c in text):
|
|
30
|
+
try:
|
|
31
|
+
from pythainlp.tokenize import word_tokenize
|
|
32
|
+
pos = 0
|
|
33
|
+
for w in word_tokenize(text, keep_whitespace=True):
|
|
34
|
+
br.add(pos); pos += len(w)
|
|
35
|
+
except ImportError:
|
|
36
|
+
pass
|
|
37
|
+
return br
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def hybrid(text_segs: list[dict], time_segs: list[dict]) -> list[dict]:
|
|
41
|
+
full = " ".join(s["text"] for s in text_segs)
|
|
42
|
+
full = unicodedata.normalize("NFC", full)
|
|
43
|
+
ref, ref_t = _chars_with_times(time_segs)
|
|
44
|
+
keep = [(i, c) for i, c in enumerate(full) if not _SKIP.match(c)]
|
|
45
|
+
hyp = "".join(c for _, c in keep)
|
|
46
|
+
sm = difflib.SequenceMatcher(None, hyp, ref, autojunk=False)
|
|
47
|
+
t_of = [None] * len(hyp)
|
|
48
|
+
for a, b, n in sm.get_matching_blocks():
|
|
49
|
+
for k in range(n):
|
|
50
|
+
t_of[a + k] = ref_t[b + k]
|
|
51
|
+
# fill gaps by linear interpolation between matched neighbours
|
|
52
|
+
known = [i for i, t in enumerate(t_of) if t is not None]
|
|
53
|
+
if not known:
|
|
54
|
+
return time_segs
|
|
55
|
+
for i in range(len(t_of)):
|
|
56
|
+
if t_of[i] is None:
|
|
57
|
+
lo = max((k for k in known if k < i), default=None)
|
|
58
|
+
hi = min((k for k in known if k > i), default=None)
|
|
59
|
+
if lo is None:
|
|
60
|
+
t_of[i] = t_of[hi]
|
|
61
|
+
elif hi is None:
|
|
62
|
+
t_of[i] = t_of[lo]
|
|
63
|
+
else:
|
|
64
|
+
t_of[i] = t_of[lo] + (t_of[hi] - t_of[lo]) * (i - lo) / (hi - lo)
|
|
65
|
+
char_time = {}
|
|
66
|
+
for (pos, _), t in zip(keep, t_of):
|
|
67
|
+
char_time[pos] = t
|
|
68
|
+
# cut the accurate text at the timing model's segment ends
|
|
69
|
+
bounds = [s["end"] for s in time_segs]
|
|
70
|
+
out, cur, cur_t0, bi = [], [], None, 0
|
|
71
|
+
last_t = 0.0
|
|
72
|
+
breaks = _word_breaks(full)
|
|
73
|
+
hold = False
|
|
74
|
+
for pos, ch in enumerate(full):
|
|
75
|
+
t = char_time.get(pos, last_t)
|
|
76
|
+
last_t = t
|
|
77
|
+
# never cut inside a word: if we are mid-word, keep filling the current segment until the next word break
|
|
78
|
+
hold = bool(cur) and pos not in breaks and pos - (max((b for b in breaks if b <= pos), default=0)) < 14
|
|
79
|
+
while not hold and bi < len(bounds) - 1 and t > bounds[bi]:
|
|
80
|
+
if cur and "".join(cur).strip():
|
|
81
|
+
out.append({"start": cur_t0, "end": bounds[bi], "text": "".join(cur).strip(), "words": []})
|
|
82
|
+
cur, cur_t0 = [], None
|
|
83
|
+
bi += 1
|
|
84
|
+
if cur_t0 is None and not ch.isspace():
|
|
85
|
+
cur_t0 = t
|
|
86
|
+
cur.append(ch)
|
|
87
|
+
if cur and "".join(cur).strip():
|
|
88
|
+
out.append({"start": cur_t0, "end": time_segs[-1]["end"], "text": "".join(cur).strip(), "words": []})
|
|
89
|
+
return _fill_uncovered(out, time_segs, sm, ref, ref_t)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _fill_uncovered(out: list[dict], time_segs: list[dict], sm: difflib.SequenceMatcher, ref: str, ref_t: list[float]) -> list[dict]:
|
|
93
|
+
"""Text models often skip the first/last words of a clip. Where the timing transcript has speech that the
|
|
94
|
+
accurate text never matched (head or tail), keep the timing model's words there instead of losing them."""
|
|
95
|
+
blocks = [b for b in sm.get_matching_blocks() if b.size >= 3] # ignore stray 1–2 char matches
|
|
96
|
+
if not blocks or not ref:
|
|
97
|
+
return out
|
|
98
|
+
first_ref, last_ref = blocks[0].b, blocks[-1].b + blocks[-1].size - 1
|
|
99
|
+
def words_between(t0: float, t1: float) -> str:
|
|
100
|
+
ws = [w for s in time_segs for w in (s.get("words") or [[s["text"], s["start"], s["end"]]]) if t0 < (w[1] + w[2]) / 2 <= t1]
|
|
101
|
+
return "".join(w[0] for w in ws).strip()
|
|
102
|
+
if len(ref) - 1 - last_ref > 2:
|
|
103
|
+
tail = words_between(ref_t[last_ref], time_segs[-1]["end"] + 1)
|
|
104
|
+
tail = tail.lstrip("".join(chr(c) for c in range(0x0e31, 0x0e4f) if unicodedata.category(chr(c)) == "Mn")) # no orphan vowel/tone marks
|
|
105
|
+
if tail and out:
|
|
106
|
+
out[-1]["text"] = (out[-1]["text"] + " " + tail).strip(); out[-1]["end"] = time_segs[-1]["end"]
|
|
107
|
+
if first_ref > 2:
|
|
108
|
+
head = words_between(time_segs[0]["start"] - 1, ref_t[first_ref] - .01)
|
|
109
|
+
if head and out:
|
|
110
|
+
out[0]["text"] = (head + " " + out[0]["text"]).strip(); out[0]["start"] = time_segs[0]["start"]
|
|
111
|
+
return out
|
babelscribe/backend.py
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""Find or fetch a whisper.cpp `whisper-cli` build for this machine and list its GPU devices.
|
|
2
|
+
|
|
3
|
+
Search order: --bin / BABELSCRIBE_WHISPER_BIN -> cached download -> `whisper-cli` on PATH -> download a prebuilt
|
|
4
|
+
release asset for this OS (Vulkan on Windows/Linux = AMD + NVIDIA + Intel, Metal on macOS, CPU fallback)."""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import io
|
|
8
|
+
import json
|
|
9
|
+
import os
|
|
10
|
+
import platform
|
|
11
|
+
import re
|
|
12
|
+
import shutil
|
|
13
|
+
import subprocess
|
|
14
|
+
import urllib.request
|
|
15
|
+
import zipfile
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
CACHE = Path(os.environ.get("BABELSCRIBE_HOME", Path.home() / ".babelscribe"))
|
|
19
|
+
# Prebuilt binaries are produced by .github/workflows/build-binaries.yml and attached to a GitHub release.
|
|
20
|
+
RELEASES = os.environ.get("BABELSCRIBE_RELEASES", "https://github.com/phonology024/babelscribe/releases/latest/download")
|
|
21
|
+
EXE = "whisper-cli.exe" if os.name == "nt" else "whisper-cli"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def asset_name(flavor: str | None = None) -> str:
|
|
25
|
+
system = {"Windows": "windows", "Linux": "linux", "Darwin": "macos"}[platform.system()]
|
|
26
|
+
arch = "arm64" if platform.machine().lower() in ("arm64", "aarch64") else "x64"
|
|
27
|
+
flavor = flavor or ("metal" if system == "macos" else "vulkan")
|
|
28
|
+
return f"whisper-cli-{system}-{arch}-{flavor}.zip"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def find_binary(explicit: str | None = None, flavor: str | None = None, download: bool = True) -> Path:
|
|
32
|
+
for cand in (explicit, os.environ.get("BABELSCRIBE_WHISPER_BIN")):
|
|
33
|
+
if cand and Path(cand).is_file():
|
|
34
|
+
return Path(cand)
|
|
35
|
+
cached = next(iter(sorted((CACHE / "bin").rglob(EXE))), None) if (CACHE / "bin").exists() else None
|
|
36
|
+
if cached:
|
|
37
|
+
return cached
|
|
38
|
+
on_path = shutil.which("whisper-cli")
|
|
39
|
+
if on_path:
|
|
40
|
+
return Path(on_path)
|
|
41
|
+
if not download:
|
|
42
|
+
raise FileNotFoundError("whisper-cli not found; pass --bin or allow download")
|
|
43
|
+
return fetch_binary(flavor)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def fetch_binary(flavor: str | None = None) -> Path:
|
|
47
|
+
name = asset_name(flavor)
|
|
48
|
+
dest = CACHE / "bin" / name[:-4]
|
|
49
|
+
print(f"downloading {name} ...")
|
|
50
|
+
data = urllib.request.urlopen(f"{RELEASES}/{name}").read()
|
|
51
|
+
zipfile.ZipFile(io.BytesIO(data)).extractall(dest)
|
|
52
|
+
exe = next(dest.rglob(EXE))
|
|
53
|
+
if os.name != "nt":
|
|
54
|
+
exe.chmod(0o755)
|
|
55
|
+
return exe
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def devices(binary: Path) -> list[dict]:
|
|
59
|
+
"""Ask whisper.cpp which GPUs it can see (it prints them while loading a model)."""
|
|
60
|
+
out = subprocess.run([str(binary), "--help"], capture_output=True, text=True, errors="replace")
|
|
61
|
+
text = out.stdout + out.stderr
|
|
62
|
+
found = []
|
|
63
|
+
for m in re.finditer(r"ggml_(vulkan|cuda|metal): *(\d+) = ([^|\n]+)", text):
|
|
64
|
+
found.append({"backend": m.group(1), "id": int(m.group(2)), "name": m.group(3).strip()})
|
|
65
|
+
return found
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def probe_devices(binary: Path, model: Path) -> list[dict]:
|
|
69
|
+
"""Load a model on an empty clip to see the GPU list and which device whisper.cpp picks."""
|
|
70
|
+
import tempfile, wave
|
|
71
|
+
with tempfile.TemporaryDirectory() as d:
|
|
72
|
+
wav = Path(d) / "s.wav"
|
|
73
|
+
with wave.open(str(wav), "wb") as w:
|
|
74
|
+
w.setnchannels(1); w.setsampwidth(2); w.setframerate(16000); w.writeframes(b"\0\0" * 16000)
|
|
75
|
+
out = subprocess.run([str(binary), "-m", str(model), "-f", str(wav), "-np"], capture_output=True, text=True, errors="replace")
|
|
76
|
+
text = out.stdout + out.stderr
|
|
77
|
+
found = [{"backend": m.group(1), "id": int(m.group(2)), "name": m.group(3).strip(), "detail": m.group(4).strip()}
|
|
78
|
+
for m in re.finditer(r"ggml_(vulkan|cuda|metal): *(\d+) = ([^|\n]+)\|?([^\n]*)", text)]
|
|
79
|
+
if not found and "Metal" in text:
|
|
80
|
+
found.append({"backend": "metal", "id": 0, "name": "Apple GPU", "detail": ""})
|
|
81
|
+
return found or [{"backend": "cpu", "id": 0, "name": platform.processor() or "CPU", "detail": ""}]
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def pick_device(found: list[dict]) -> int:
|
|
85
|
+
"""Prefer a discrete GPU: NVIDIA/AMD over integrated Intel/AMD APU (uma: 1)."""
|
|
86
|
+
gpus = [d for d in found if d["backend"] != "cpu"]
|
|
87
|
+
if not gpus:
|
|
88
|
+
return 0
|
|
89
|
+
discrete = [d for d in gpus if "uma: 0" in d.get("detail", "")] or gpus
|
|
90
|
+
return discrete[0]["id"]
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
if __name__ == "__main__":
|
|
94
|
+
print(json.dumps({"asset": asset_name(), "binary": str(find_binary(download=False))}, indent=1))
|
babelscribe/cli.py
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""babelscribe — transcribe any audio/video, in any language Whisper knows, on any GPU (AMD / NVIDIA / Intel via
|
|
2
|
+
Vulkan, NVIDIA via CUDA, Apple via Metal) or the CPU.
|
|
3
|
+
|
|
4
|
+
babelscribe talk.mp4 # auto language, turbo model, best GPU -> talk.srt + talk.json
|
|
5
|
+
babelscribe vo.wav -l th --text-model thai-thonburian # hybrid: Thai fine-tune text + turbo timing
|
|
6
|
+
babelscribe talk.mp4 --accurate # slower, fewest errors: large-v3 + beam search, or the best fine-tune
|
|
7
|
+
babelscribe devices # list GPUs whisper.cpp can use
|
|
8
|
+
babelscribe models # list models and fine-tunes"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import sys
|
|
13
|
+
import time
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
from . import __version__, backend, hybrid, models, transcribe, writers
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def main(argv: list[str] | None = None) -> None:
|
|
20
|
+
argv = sys.argv[1:] if argv is None else argv
|
|
21
|
+
if argv[:1] == ["models"]:
|
|
22
|
+
print("general (99 languages):"); [print(f" {k:16} {v}") for k, v in models.GENERAL.items()]
|
|
23
|
+
print("fine-tunes (text quality for one language, use with --text-model):")
|
|
24
|
+
[print(f" {k:16} [{v['lang']}] {v['note']}") for k, v in models.FINETUNES.items()]
|
|
25
|
+
return
|
|
26
|
+
ap = argparse.ArgumentParser(prog="babelscribe", description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
27
|
+
ap.add_argument("input", help="audio or video file, or 'devices'")
|
|
28
|
+
ap.add_argument("-l", "--lang", default="auto", help="language code (th, en, ja, ...) or auto")
|
|
29
|
+
ap.add_argument("-m", "--model", default="turbo", help="timing/general model (default turbo = large-v3-turbo)")
|
|
30
|
+
ap.add_argument("--text-model", help="language fine-tune for the text (hybrid mode), e.g. thai-thonburian")
|
|
31
|
+
ap.add_argument("--accurate", action="store_true",
|
|
32
|
+
help="slower but fewest errors: large-v3 with beam search, or the language's best fine-tune (hybrid)")
|
|
33
|
+
ap.add_argument("-f", "--formats", default="srt,json", help="comma list: srt,vtt,txt,json")
|
|
34
|
+
ap.add_argument("-o", "--out", help="output base path (default: next to the input)")
|
|
35
|
+
ap.add_argument("--device", default="auto", help="GPU id from `babelscribe devices`, or auto")
|
|
36
|
+
ap.add_argument("--bin", help="path to a whisper-cli you built yourself")
|
|
37
|
+
ap.add_argument("--flavor", help="prebuilt flavour to download: vulkan | cuda | metal | cpu")
|
|
38
|
+
ap.add_argument("-v", "--verbose", action="store_true")
|
|
39
|
+
ap.add_argument("--version", action="version", version=__version__)
|
|
40
|
+
a = ap.parse_args(argv)
|
|
41
|
+
|
|
42
|
+
binary = backend.find_binary(a.bin, a.flavor)
|
|
43
|
+
timing = models.ensure(a.model, binary.parent / ("whisper-quantize.exe" if binary.suffix == ".exe" else "whisper-quantize"))
|
|
44
|
+
found = backend.probe_devices(binary, timing)
|
|
45
|
+
if a.input == "devices":
|
|
46
|
+
for d in found:
|
|
47
|
+
print(f" [{d['id']}] {d['backend']:6} {d['name']} {d.get('detail', '')}")
|
|
48
|
+
print(f"auto picks device {backend.pick_device(found)}"); return
|
|
49
|
+
dev = backend.pick_device(found) if a.device == "auto" else int(a.device)
|
|
50
|
+
gpu = next((d for d in found if d["id"] == dev), found[0])
|
|
51
|
+
src = Path(a.input)
|
|
52
|
+
if not src.exists():
|
|
53
|
+
raise SystemExit(f"no such file: {src}")
|
|
54
|
+
lang = a.lang
|
|
55
|
+
if lang == "auto":
|
|
56
|
+
lang = transcribe.detected_language(binary, timing, src, dev); print(f"language: {lang}")
|
|
57
|
+
beam = None
|
|
58
|
+
if a.accurate:
|
|
59
|
+
if not a.text_model and lang in models.ACCURATE:
|
|
60
|
+
a.text_model = models.ACCURATE[lang] # fine-tune text + turbo timing
|
|
61
|
+
elif not a.text_model and a.model == "turbo":
|
|
62
|
+
a.model, beam = "large-v3", 5
|
|
63
|
+
timing = models.ensure(a.model)
|
|
64
|
+
t0 = time.time()
|
|
65
|
+
print(f"transcribing with {a.model} on {gpu['backend']} {gpu['name']} ...")
|
|
66
|
+
segs = transcribe.run(binary, timing, src, lang, dev, beam=beam, verbose=a.verbose)
|
|
67
|
+
meta = {"tool": f"babelscribe {__version__}", "model": a.model, "lang": lang, "device": f"{gpu['backend']} {gpu['name']}"}
|
|
68
|
+
if a.text_model:
|
|
69
|
+
_, ft = models.resolve(a.text_model)
|
|
70
|
+
tm = models.ensure(a.text_model, binary.parent / ("whisper-quantize.exe" if binary.suffix == ".exe" else "whisper-quantize"))
|
|
71
|
+
print(f"text pass with {a.text_model} ...")
|
|
72
|
+
segs = hybrid.run(binary, tm, src, segs, ft["lang"] if ft else lang, dev, (ft or {}).get("beam"), a.verbose)
|
|
73
|
+
meta["text_model"] = a.text_model
|
|
74
|
+
base = Path(a.out) if a.out else src.with_suffix("")
|
|
75
|
+
files = writers.write(segs, base, [f.strip() for f in a.formats.split(",") if f.strip()], meta)
|
|
76
|
+
print(f"{len(segs)} segments in {time.time() - t0:.1f}s -> " + ", ".join(str(f) for f in files))
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
if __name__ == "__main__":
|
|
80
|
+
main()
|
babelscribe/hybrid.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Hybrid pass done safely: language fine-tunes without timestamps drop words at whisper's 30 s window seams.
|
|
2
|
+
So we cut the audio into short chunks (<= max_len s) at the timing model's segment boundaries (natural pauses),
|
|
3
|
+
transcribe every chunk with the text model in ONE whisper-cli run (model loaded once), and align each chunk's
|
|
4
|
+
accurate text to the timing segments inside that chunk."""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
import subprocess
|
|
9
|
+
import tempfile
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from . import align
|
|
13
|
+
from .transcribe import to_wav16k
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def chunks(time_segs: list[dict], max_len: float = 24.0) -> list[tuple[float, float, list[dict]]]:
|
|
17
|
+
out, cur = [], []
|
|
18
|
+
for s in time_segs:
|
|
19
|
+
if cur and s["end"] - cur[0]["start"] > max_len:
|
|
20
|
+
out.append(cur); cur = []
|
|
21
|
+
cur.append(s)
|
|
22
|
+
if cur:
|
|
23
|
+
out.append(cur)
|
|
24
|
+
return [(c[0]["start"], c[-1]["end"], c) for c in out]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def run(binary: Path, text_model: Path, media: Path, time_segs: list[dict], lang: str, device: int, beam: int | None,
|
|
28
|
+
verbose: bool = False, pad: float = .25) -> list[dict]:
|
|
29
|
+
parts = chunks(time_segs)
|
|
30
|
+
with tempfile.TemporaryDirectory() as d:
|
|
31
|
+
wavs = []
|
|
32
|
+
for i, (a, b, _) in enumerate(parts):
|
|
33
|
+
w = Path(d) / f"c{i:04d}.wav"; to_wav16k(media, w, max(0.0, a - pad), b + pad, tail_silence=1.5); wavs.append(w)
|
|
34
|
+
cmd = [str(binary), "-m", str(text_model), "-l", lang, "-oj", "-mc", "0", "-dev", str(device), "-np"]
|
|
35
|
+
if beam:
|
|
36
|
+
cmd += ["-bs", str(beam)]
|
|
37
|
+
for w in wavs:
|
|
38
|
+
cmd += ["-f", str(w)]
|
|
39
|
+
p = subprocess.run(cmd, capture_output=not verbose, text=True, errors="replace")
|
|
40
|
+
if p.returncode:
|
|
41
|
+
raise RuntimeError(f"whisper-cli failed ({p.returncode}):\n{(p.stderr or '')[-2000:]}")
|
|
42
|
+
texts = []
|
|
43
|
+
for w in wavs:
|
|
44
|
+
j = json.loads(Path(str(w) + ".json").read_text(encoding="utf-8", errors="replace"))
|
|
45
|
+
texts.append(" ".join(s["text"].strip() for s in j.get("transcription", [])))
|
|
46
|
+
out = []
|
|
47
|
+
for (a, b, segs), text in zip(parts, texts):
|
|
48
|
+
if not text.strip():
|
|
49
|
+
out.extend(segs); continue
|
|
50
|
+
out.extend(align.hybrid([{"start": a, "end": b, "text": text, "words": []}], segs))
|
|
51
|
+
return out
|
babelscribe/models.py
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""Model registry + download.
|
|
2
|
+
|
|
3
|
+
General models are the official whisper.cpp ggml files (99 languages). Language fine-tunes are listed by their
|
|
4
|
+
Hugging Face repo and converted on the user's machine (we never redistribute converted weights — each fine-tune
|
|
5
|
+
keeps its own licence). Fine-tunes often lose timestamp prediction, so they are used for TEXT and a general model
|
|
6
|
+
gives the TIMING (see align.py)."""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import os
|
|
10
|
+
import subprocess
|
|
11
|
+
import sys
|
|
12
|
+
import urllib.request
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from .backend import CACHE
|
|
16
|
+
|
|
17
|
+
GGML = "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-{name}.bin"
|
|
18
|
+
GENERAL = { # name -> size (approx) ; 'turbo' is the best speed/quality default
|
|
19
|
+
"tiny": "75 MB", "base": "142 MB", "small": "466 MB", "medium": "1.5 GB",
|
|
20
|
+
"large-v3": "3.1 GB", "large-v3-turbo": "1.6 GB",
|
|
21
|
+
}
|
|
22
|
+
ALIASES = {"turbo": "large-v3-turbo", "large": "large-v3"}
|
|
23
|
+
# Community fine-tunes: add a line to support a new language better. 'beam' = needs beam search (greedy loops).
|
|
24
|
+
FINETUNES = {
|
|
25
|
+
"thai-thonburian": {"repo": "biodatlab/whisper-th-large-v3-combined", "lang": "th", "beam": 5, "timestamps": False,
|
|
26
|
+
"note": "Thonburian Whisper (Thai) — strong Thai spelling; pair with turbo for timing"},
|
|
27
|
+
"thai-pathumma": {"repo": "nectec/Pathumma-whisper-th-large-v3", "lang": "th", "beam": 5, "timestamps": False,
|
|
28
|
+
"note": "Pathumma Whisper by NECTEC (Thai) — lowest Thai CER on FLEURS"},
|
|
29
|
+
"hindi-vasista": {"repo": "vasista22/whisper-hindi-large-v2", "lang": "hi", "beam": 5, "timestamps": False,
|
|
30
|
+
"note": "Hindi fine-tune of large-v2 (Speech Lab, IIT Madras) — halves Hindi WER on FLEURS"},
|
|
31
|
+
}
|
|
32
|
+
# --accurate: per language, the model that scored best on FLEURS (bench/fleurs.py). Languages not listed use large-v3
|
|
33
|
+
# with beam search, which beat every public fine-tune we tried for vi / ar.
|
|
34
|
+
ACCURATE = {"th": "thai-pathumma", "hi": "hindi-vasista"}
|
|
35
|
+
MODELS = Path(os.environ.get("BABELSCRIBE_MODELS", CACHE / "models"))
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def resolve(name: str) -> tuple[str, dict | None]:
|
|
39
|
+
name = ALIASES.get(name, name)
|
|
40
|
+
if name in GENERAL:
|
|
41
|
+
return name, None
|
|
42
|
+
if name in FINETUNES:
|
|
43
|
+
return name, FINETUNES[name]
|
|
44
|
+
raise SystemExit(f"unknown model '{name}'. General: {', '.join(GENERAL)}; fine-tunes: {', '.join(FINETUNES)}")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def path_for(name: str) -> Path:
|
|
48
|
+
name, ft = resolve(name)
|
|
49
|
+
return MODELS / (f"ggml-{name}-q8_0.bin" if ft else f"ggml-{name}.bin")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def ensure(name: str, quantizer: Path | None = None) -> Path:
|
|
53
|
+
name, ft = resolve(name)
|
|
54
|
+
p = path_for(name)
|
|
55
|
+
if p.exists():
|
|
56
|
+
return p
|
|
57
|
+
MODELS.mkdir(parents=True, exist_ok=True)
|
|
58
|
+
if not ft:
|
|
59
|
+
print(f"downloading model {name} ({GENERAL[name]}) ...")
|
|
60
|
+
tmp = p.with_suffix(".part")
|
|
61
|
+
urllib.request.urlretrieve(GGML.format(name=name), tmp)
|
|
62
|
+
tmp.replace(p)
|
|
63
|
+
return p
|
|
64
|
+
return convert_finetune(name, ft, p, quantizer)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def convert_finetune(name: str, ft: dict, out: Path, quantizer: Path | None) -> Path:
|
|
68
|
+
"""HF transformers checkpoint -> ggml (whisper.cpp convert-h5-to-ggml.py) -> q8_0. Needs: pip install babelscribe[finetune]."""
|
|
69
|
+
try:
|
|
70
|
+
import torch, transformers # noqa: F401
|
|
71
|
+
from huggingface_hub import snapshot_download
|
|
72
|
+
except ImportError:
|
|
73
|
+
raise SystemExit("fine-tune conversion needs extras: pip install \"babelscribe[finetune]\"")
|
|
74
|
+
work = MODELS / f"_{name}"
|
|
75
|
+
work.mkdir(parents=True, exist_ok=True)
|
|
76
|
+
hf = snapshot_download(ft["repo"], local_dir=work / "hf")
|
|
77
|
+
oa = work / "openai-whisper"
|
|
78
|
+
if not oa.exists():
|
|
79
|
+
subprocess.run(["git", "clone", "-q", "--depth", "1", "https://github.com/openai/whisper.git", str(oa)], check=True)
|
|
80
|
+
script = work / "convert-h5-to-ggml.py"
|
|
81
|
+
if not script.exists():
|
|
82
|
+
urllib.request.urlretrieve("https://raw.githubusercontent.com/ggml-org/whisper.cpp/master/models/convert-h5-to-ggml.py", script)
|
|
83
|
+
# some fine-tunes ship bf16 weights, which the converter cannot export -> load them as float32
|
|
84
|
+
src = script.read_text(encoding="utf-8").replace("WhisperForConditionalGeneration.from_pretrained(dir_model)\n",
|
|
85
|
+
"WhisperForConditionalGeneration.from_pretrained(dir_model).float()\n")
|
|
86
|
+
script.write_text(src, encoding="utf-8")
|
|
87
|
+
subprocess.run([sys.executable, str(script), hf, str(oa), str(work)], check=True)
|
|
88
|
+
f32 = work / "ggml-model.bin"
|
|
89
|
+
if quantizer and quantizer.exists():
|
|
90
|
+
subprocess.run([str(quantizer), str(f32), str(out), "q8_0"], check=True)
|
|
91
|
+
f32.unlink()
|
|
92
|
+
else:
|
|
93
|
+
f32.replace(out)
|
|
94
|
+
return out
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Run whisper.cpp on any audio/video file and return segments [{start, end, text, words: [[text, t0, t1], ...]}]."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import shutil
|
|
7
|
+
import subprocess
|
|
8
|
+
import tempfile
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def ffmpeg() -> str:
|
|
13
|
+
exe = shutil.which("ffmpeg")
|
|
14
|
+
if exe:
|
|
15
|
+
return exe
|
|
16
|
+
try:
|
|
17
|
+
import imageio_ffmpeg
|
|
18
|
+
return imageio_ffmpeg.get_ffmpeg_exe()
|
|
19
|
+
except ImportError:
|
|
20
|
+
raise SystemExit("ffmpeg not found: install it or `pip install imageio-ffmpeg`")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def to_wav16k(src: Path, dst: Path, start: float | None = None, end: float | None = None, tail_silence: float = 0) -> None:
|
|
24
|
+
cmd = [ffmpeg(), "-v", "error", "-y"]
|
|
25
|
+
if start is not None:
|
|
26
|
+
cmd += ["-ss", str(start)]
|
|
27
|
+
if end is not None:
|
|
28
|
+
cmd += ["-to", str(end)]
|
|
29
|
+
cmd += ["-i", str(src), "-vn", "-ac", "1", "-ar", "16000"]
|
|
30
|
+
if tail_silence: # whisper tends to drop the last words when audio stops abruptly
|
|
31
|
+
cmd += ["-af", f"apad=pad_dur={tail_silence}"]
|
|
32
|
+
cmd += ["-c:a", "pcm_s16le", str(dst)]
|
|
33
|
+
subprocess.run(cmd, check=True)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def run(binary: Path, model: Path, media: Path, lang: str = "auto", device: int = 0, beam: int | None = None,
|
|
37
|
+
threads: int | None = None, extra: list[str] | None = None, verbose: bool = False) -> list[dict]:
|
|
38
|
+
with tempfile.TemporaryDirectory() as d:
|
|
39
|
+
wav = Path(d) / "in.wav"
|
|
40
|
+
to_wav16k(media, wav)
|
|
41
|
+
out = Path(d) / "out"
|
|
42
|
+
# -mc 0: never feed previous text back as a prompt — stops the repeat-loop hallucination on long files
|
|
43
|
+
cmd = [str(binary), "-m", str(model), "-f", str(wav), "-l", lang, "-ojf", "-of", str(out), "-mc", "0", "-dev", str(device)]
|
|
44
|
+
if beam:
|
|
45
|
+
cmd += ["-bs", str(beam)]
|
|
46
|
+
if threads:
|
|
47
|
+
cmd += ["-t", str(threads)]
|
|
48
|
+
cmd += extra or []
|
|
49
|
+
p = subprocess.run(cmd, capture_output=not verbose, text=True, errors="replace")
|
|
50
|
+
if p.returncode:
|
|
51
|
+
raise RuntimeError(f"whisper-cli failed ({p.returncode}):\n{(p.stderr or '')[-2000:]}")
|
|
52
|
+
data = json.loads(Path(str(out) + ".json").read_text(encoding="utf-8", errors="replace"))
|
|
53
|
+
segs = []
|
|
54
|
+
for s in data.get("transcription", []):
|
|
55
|
+
words = [[t["text"], t["offsets"]["from"] / 1000, t["offsets"]["to"] / 1000] for t in s.get("tokens", [])
|
|
56
|
+
if not t["text"].startswith("[_") and t["text"].strip()]
|
|
57
|
+
segs.append({"start": s["offsets"]["from"] / 1000, "end": s["offsets"]["to"] / 1000, "text": s["text"].strip(), "words": words})
|
|
58
|
+
return [s for s in segs if s["text"]]
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def detected_language(binary: Path, model: Path, media: Path, device: int = 0) -> str:
|
|
62
|
+
with tempfile.TemporaryDirectory() as d:
|
|
63
|
+
wav = Path(d) / "in.wav"
|
|
64
|
+
to_wav16k(media, wav, 0, 30)
|
|
65
|
+
p = subprocess.run([str(binary), "-m", str(model), "-f", str(wav), "-l", "auto", "-dl", "-dev", str(device)],
|
|
66
|
+
capture_output=True, text=True, errors="replace")
|
|
67
|
+
import re
|
|
68
|
+
m = re.search(r"auto-detected language: (\w+)", p.stdout + p.stderr)
|
|
69
|
+
return m.group(1) if m else "auto"
|
babelscribe/writers.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Output formats: srt, vtt, txt, json."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def _ts(t: float, sep: str) -> str:
|
|
9
|
+
ms = int(round(max(0.0, t) * 1000))
|
|
10
|
+
h, ms = divmod(ms, 3600000); m, ms = divmod(ms, 60000); s, ms = divmod(ms, 1000)
|
|
11
|
+
return f"{h:02d}:{m:02d}:{s:02d}{sep}{ms:03d}"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def write(segs: list[dict], base: Path, formats: list[str], meta: dict) -> list[Path]:
|
|
15
|
+
out = []
|
|
16
|
+
for f in formats:
|
|
17
|
+
p = base.with_suffix("." + f)
|
|
18
|
+
if f == "srt":
|
|
19
|
+
p.write_text("".join(f"{i}\n{_ts(s['start'], ',')} --> {_ts(s['end'], ',')}\n{s['text']}\n\n" for i, s in enumerate(segs, 1)), encoding="utf-8")
|
|
20
|
+
elif f == "vtt":
|
|
21
|
+
p.write_text("WEBVTT\n\n" + "".join(f"{_ts(s['start'], '.')} --> {_ts(s['end'], '.')}\n{s['text']}\n\n" for s in segs), encoding="utf-8")
|
|
22
|
+
elif f == "txt":
|
|
23
|
+
p.write_text("\n".join(s["text"] for s in segs) + "\n", encoding="utf-8")
|
|
24
|
+
elif f == "json":
|
|
25
|
+
p.write_text(json.dumps({"meta": meta, "segments": segs}, ensure_ascii=False, indent=1), encoding="utf-8")
|
|
26
|
+
else:
|
|
27
|
+
raise SystemExit(f"unknown format {f}")
|
|
28
|
+
out.append(p)
|
|
29
|
+
return out
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: babelscribe
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Transcribe any audio/video in 99 languages on any GPU (AMD, NVIDIA, Intel via Vulkan; NVIDIA via CUDA; Apple via Metal) or CPU — whisper.cpp with batteries included.
|
|
5
|
+
License: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/phonology024/babelscribe
|
|
7
|
+
Project-URL: Issues, https://github.com/phonology024/babelscribe/issues
|
|
8
|
+
Keywords: whisper,speech-to-text,transcription,subtitles,vulkan,amd,gpu,thai
|
|
9
|
+
Requires-Python: >=3.9
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Requires-Dist: imageio-ffmpeg>=0.5
|
|
13
|
+
Provides-Extra: finetune
|
|
14
|
+
Requires-Dist: torch; extra == "finetune"
|
|
15
|
+
Requires-Dist: transformers; extra == "finetune"
|
|
16
|
+
Requires-Dist: huggingface_hub; extra == "finetune"
|
|
17
|
+
Requires-Dist: numpy; extra == "finetune"
|
|
18
|
+
Provides-Extra: thai
|
|
19
|
+
Requires-Dist: pythainlp; extra == "thai"
|
|
20
|
+
Dynamic: license-file
|
|
21
|
+
|
|
22
|
+
# babelscribe
|
|
23
|
+
|
|
24
|
+
**Transcribe any audio or video, in any of Whisper's 99 languages, on any GPU — AMD, NVIDIA, Intel or Apple — or just the CPU.**
|
|
25
|
+
One command, no CUDA required, subtitles out.
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
pip install babelscribe # Python 3.9+
|
|
29
|
+
babelscribe interview.mp4 # auto-detects language, picks your best GPU -> interview.srt + interview.json
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
babelscribe is a thin, friendly layer on top of [whisper.cpp](https://github.com/ggml-org/whisper.cpp). It exists because
|
|
33
|
+
getting fast Whisper on a non-NVIDIA card (e.g. an AMD Radeon on Windows) still means building whisper.cpp with the
|
|
34
|
+
Vulkan SDK yourself. babelscribe downloads a prebuilt `whisper-cli` for your system and handles the rest.
|
|
35
|
+
|
|
36
|
+
| Your hardware | Backend used | Prebuilt asset |
|
|
37
|
+
|---|---|---|
|
|
38
|
+
| AMD / NVIDIA / Intel GPU on Windows | Vulkan | `windows-x64-vulkan` |
|
|
39
|
+
| NVIDIA on Windows | Vulkan (same build) | `windows-x64-vulkan` |
|
|
40
|
+
| NVIDIA on Linux (alternative) | CUDA | `linux-x64-cuda` (`--flavor cuda`) |
|
|
41
|
+
| AMD / NVIDIA / Intel GPU on Linux | Vulkan | `linux-x64-vulkan` |
|
|
42
|
+
| Apple Silicon | Metal | `macos-arm64-metal` |
|
|
43
|
+
| No usable GPU | CPU | `*-cpu` (`--flavor cpu`) |
|
|
44
|
+
|
|
45
|
+
A 19-minute English TED talk transcribes in 48 seconds on an AMD Radeon RX 9070 XT with 1.5% word error — see the benchmark below.
|
|
46
|
+
|
|
47
|
+
## Benchmark: real talks, human captions as the answer key
|
|
48
|
+
Four TED / TEDx talks, scored against the **human-made captions in the spoken language** (`bench/bench.py`, reproducible).
|
|
49
|
+
Error = word error rate (WER) for space-separated languages, character error rate (CER) for Japanese and Thai.
|
|
50
|
+
GPU: AMD Radeon RX 9070 XT via Vulkan. *default* = `large-v3-turbo`; *accurate* = `--accurate` (see the FLEURS section).
|
|
51
|
+
|
|
52
|
+
| Language | Talk | Length | default: time / error | `--accurate`: time / error |
|
|
53
|
+
|---|---|---|---|---|
|
|
54
|
+
| English | [Matt Walker — Sleep Is Your Superpower (TED)](https://youtu.be/5MuIMqhT8DM) | 19.3 min | 48 s (24x) · WER **1.5%** | 107 s (11x) · WER **1.5%** |
|
|
55
|
+
| Japanese | [Kazunari Taguchi (TEDxHimi)](https://youtu.be/cjtmDEG-B7U) | 16.2 min | 47 s (20x) · CER 4.7% | 115 s (8x) · CER **4.2%** |
|
|
56
|
+
| Spanish | [Adrià Solà Pastor — Cómo hablar (TEDxESIC University)](https://youtu.be/XUqrvbsTfck) | 19.9 min | 56 s (21x) · WER 13.8% | 134 s (9x) · WER 13.5% |
|
|
57
|
+
| Thai | [นิติ ชัยชิตาทร — โปรดเรียกฉันด้วยนามอันแท้จริง (TEDxBangkok)](https://youtu.be/48A9SU6_bQ8) | 14.1 min | 74 s (12x) · CER 22.7% | 565 s (2x) · CER **16.4%** |
|
|
58
|
+
|
|
59
|
+
How to read it: TED captions are edited for reading (fillers dropped, light rewording), so these numbers are an upper bound —
|
|
60
|
+
most of the Spanish "errors" are the speaker's actual words versus the tidied caption. Thai is genuinely harder: fast,
|
|
61
|
+
casual speech with slang; `--accurate` (Pathumma Whisper text + turbo timing) cuts its error by more than a quarter.
|
|
62
|
+
Talks are used only to measure accuracy; their transcripts are not redistributed (TED content is CC BY-NC-ND).
|
|
63
|
+
|
|
64
|
+
## Benchmark: 16 languages on FLEURS
|
|
65
|
+
[Google FLEURS](https://huggingface.co/datasets/google/fleurs) test set, 50 utterances per language, human-verified
|
|
66
|
+
verbatim transcripts (`bench/fleurs.py`, reproducible). Same normaliser for every language (Whisper's rule: lower-case,
|
|
67
|
+
drop punctuation and non-spacing marks, numbers spelled out). WER for space-separated languages, CER for ja / zh / ko / th.
|
|
68
|
+
|
|
69
|
+
| Language | default (turbo) | `--accurate` | model `--accurate` uses | `--accurate`, edge-trimmed* |
|
|
70
|
+
|---|---|---|---|---|
|
|
71
|
+
| English | 6.1% | 5.7% | large-v3, beam 5 | 5.6% |
|
|
72
|
+
| Spanish | 4.1% | 4.2% | large-v3, beam 5 | **2.6%** |
|
|
73
|
+
| French | 7.4% | 7.3% | large-v3, beam 5 | 7.2% |
|
|
74
|
+
| German | 4.3% | 4.0% | large-v3, beam 5 | **3.3%** |
|
|
75
|
+
| Portuguese | 8.5% | 7.7% | large-v3, beam 5 | 5.3% |
|
|
76
|
+
| Italian | 6.7% | 6.5% | large-v3, beam 5 | **3.8%** |
|
|
77
|
+
| Russian | 6.8% | 5.8% | large-v3, beam 5 | 5.6% |
|
|
78
|
+
| Arabic | 11.0% | 10.6% | large-v3, beam 5 | 9.7% |
|
|
79
|
+
| Hindi | 28.4% | **12.6%** | [vasista22/whisper-hindi-large-v2](https://huggingface.co/vasista22/whisper-hindi-large-v2) + turbo timing | 11.1% |
|
|
80
|
+
| Indonesian | 9.8% | 7.8% | large-v3, beam 5 | 5.6% |
|
|
81
|
+
| Vietnamese | 10.8% | 9.1% | large-v3, beam 5 | 9.1% |
|
|
82
|
+
| Turkish | 6.2% | 6.7% | large-v3, beam 5 | 5.8% |
|
|
83
|
+
| Japanese (CER) | 6.4% | 5.7% | large-v3, beam 5 | **4.6%** |
|
|
84
|
+
| Chinese (CER) | 6.3% | 5.3% | large-v3, beam 5 | **4.8%** |
|
|
85
|
+
| Korean (CER) | 4.1% | 3.9% | large-v3, beam 5 | **3.1%** |
|
|
86
|
+
| Thai (CER) | 15.9% | **8.9%** | [Pathumma Whisper](https://huggingface.co/nectec/Pathumma-whisper-th-large-v3) (NECTEC) + turbo timing | 8.8% |
|
|
87
|
+
|
|
88
|
+
\* Some FLEURS clips contain more speech than their reference transcript, so a correct model is charged for words the
|
|
89
|
+
reference leaves out. *Edge-trimmed* ignores extra words before the first / after the last reference word; both numbers
|
|
90
|
+
are stored by `bench/fleurs.py`. 50 utterances per language means differences under ~0.5 points are noise.
|
|
91
|
+
|
|
92
|
+
Also measured and **not** used, because large-v3 was as good or better: large-v2 (all languages),
|
|
93
|
+
PhoWhisper-large (vi 19.0%), whisper-large-v3 dialectal / code-switching Arabic fine-tunes (12.2% / 15.2%),
|
|
94
|
+
Typhoon Whisper (th 11.9%), Thonburian Whisper (th 9.1% — kept as an option), Vaani Hindi (16.5%).
|
|
95
|
+
|
|
96
|
+
## Why babelscribe (vs. what already exists)
|
|
97
|
+
| | GPU on AMD / Intel | Windows, no build step | Video in, subtitles out | Long files don't loop | Better text for your language |
|
|
98
|
+
|---|---|---|---|---|---|
|
|
99
|
+
| **babelscribe** | ✅ Vulkan | ✅ prebuilt `whisper-cli` downloaded for you | ✅ ffmpeg bundled | ✅ `--max-context 0` by default | ✅ hybrid fine-tune text + turbo timing |
|
|
100
|
+
| whisper.cpp (raw) | ✅ Vulkan — if you compile it | ❌ official releases ship no Windows Vulkan build | ❌ WAV 16 kHz only | ⚠️ you must know the flag | ❌ |
|
|
101
|
+
| faster-whisper / WhisperX | ❌ GPU = NVIDIA CUDA only | ✅ pip | ✅ | ⚠️ | ⚠️ manual |
|
|
102
|
+
| Cloud APIs | n/a (cloud) | ✅ | ✅ | ✅ | ❌ — and your audio leaves your machine, paid per minute |
|
|
103
|
+
|
|
104
|
+
babelscribe does **not** replace those projects — it stands on whisper.cpp and simply removes the hard parts:
|
|
105
|
+
compiling for your GPU, converting media, picking the right device, avoiding the long-file repeat bug, and combining
|
|
106
|
+
a language-specific fine-tune with accurate timestamps.
|
|
107
|
+
|
|
108
|
+
## Languages
|
|
109
|
+
All 99 languages Whisper was trained on, auto-detected or forced with `-l`:
|
|
110
|
+
af am ar as az ba be bg bn bo br bs ca cs cy da de el en es et eu fa fi fo fr gl gu ha haw he hi hr ht hu hy id is it ja jw ka kk km kn ko la lb ln lo lt lv mg mi mk ml mn mr ms mt my ne nl nn no oc pa pl ps pt ro ru sa sd si sk sl sn so sq sr su sv sw ta te tg th tk tl tr tt uk ur uz vi yi yo yue zh
|
|
111
|
+
|
|
112
|
+
Accuracy follows Whisper's own training data: excellent for high-resource languages (English, Spanish, Japanese, …),
|
|
113
|
+
weaker for low-resource ones. That is what hybrid mode is for — a community fine-tune for one language can be plugged in
|
|
114
|
+
with one line in `babelscribe/models.py` (Thai and Hindi so far). PRs adding fine-tunes for other
|
|
115
|
+
languages are the most valuable contribution.
|
|
116
|
+
|
|
117
|
+
## Limitations (honest)
|
|
118
|
+
- Tested end to end so far on an AMD Radeon RX 9070 XT (Windows, Vulkan). CUDA, Linux and macOS builds are produced by CI; reports from those machines are welcome.
|
|
119
|
+
- Hybrid mode runs two models, so it is slower (≈1 min per minute of audio with a large fine-tune on that GPU).
|
|
120
|
+
- Proper nouns can still be misspelled — check names before publishing subtitles.
|
|
121
|
+
|
|
122
|
+
## Features
|
|
123
|
+
- **Any input** — mp4, mkv, mov, mp3, wav, m4a… (ffmpeg is bundled through `imageio-ffmpeg`).
|
|
124
|
+
- **Any language** — `-l auto` detects it; or pass `-l th`, `-l ja`, `-l es`…
|
|
125
|
+
- **Picks the right GPU** — prefers a discrete card over an integrated one; `babelscribe devices` lists them, `--device N` overrides.
|
|
126
|
+
- **Long files that don't loop** — runs whisper with `--max-context 0`, which stops the classic "same sentence repeated forever" hallucination on long recordings.
|
|
127
|
+
- **Hybrid mode for better spelling in your language** — community fine-tunes (e.g. Thai *Thonburian Whisper*) spell far better but often lose timestamps. `--text-model` takes the text from the fine-tune and the timing from `large-v3-turbo`, aligned character by character (works for languages without spaces), cut only at word boundaries, with the timing model filling any words the fine-tune skipped.
|
|
128
|
+
- **`--accurate`** — slower, fewest errors: large-v3 with beam search, or for Thai and Hindi the best community fine-tune (hybrid). Chosen per language from the FLEURS benchmark above.
|
|
129
|
+
- **Outputs** — `srt`, `vtt`, `txt`, `json` (segments with token timings).
|
|
130
|
+
|
|
131
|
+
## Usage
|
|
132
|
+
```bash
|
|
133
|
+
babelscribe talk.mp4 -f srt,vtt,txt,json # all formats
|
|
134
|
+
babelscribe podcast.mp3 -l en -m large-v3 # pick language and model
|
|
135
|
+
babelscribe talk.mp4 -l hi --accurate # slower, fewest errors (best model per language)
|
|
136
|
+
babelscribe vo.wav -l th --text-model thai-thonburian # hybrid: pick a Thai fine-tune yourself
|
|
137
|
+
babelscribe devices # GPUs whisper.cpp can see
|
|
138
|
+
babelscribe models # models and fine-tunes
|
|
139
|
+
babelscribe talk.mp4 --bin /path/to/whisper-cli # use your own whisper.cpp build
|
|
140
|
+
```
|
|
141
|
+
Models download on first use to `~/.babelscribe/models` (`BABELSCRIBE_MODELS` to change). Hybrid fine-tunes are converted on your
|
|
142
|
+
machine from their original Hugging Face repo — install `pip install "babelscribe[finetune]"` once; converted weights are never
|
|
143
|
+
redistributed, so each fine-tune keeps its own licence. Thai word boundaries: `pip install "babelscribe[thai]"`.
|
|
144
|
+
|
|
145
|
+
## Add a language fine-tune
|
|
146
|
+
Add one entry to `FINETUNES` in `babelscribe/models.py` (Hugging Face repo, language code, whether it needs beam search) and open a PR.
|
|
147
|
+
|
|
148
|
+
## Prebuilt binaries
|
|
149
|
+
`.github/workflows/build-binaries.yml` builds `whisper-cli` + `whisper-quantize` for every row of the table above and attaches
|
|
150
|
+
them to each `v*` release. Point `BABELSCRIBE_RELEASES` at another URL to self-host.
|
|
151
|
+
|
|
152
|
+
## ภาษาไทย
|
|
153
|
+
ถอดเสียงจากไฟล์เสียงหรือวิดีโอได้ทุกภาษา บนการ์ดจอทุกยี่ห้อ (AMD / NVIDIA / Intel ผ่าน Vulkan, Apple ผ่าน Metal) หรือ CPU
|
|
154
|
+
ภาษาไทยแนะนำ `babelscribe ไฟล์.mp4 -l th --accurate` — ข้อความจาก Pathumma Whisper (NECTEC) ที่ผิดน้อยที่สุดใน FLEURS (CER 8.9% เทียบ turbo 15.9%) + เวลาจาก large-v3-turbo
|
|
155
|
+
หรือเลือก Thonburian Whisper เอง: `--text-model thai-thonburian`
|
|
156
|
+
|
|
157
|
+
## Credits & licence
|
|
158
|
+
MIT. Built on [whisper.cpp](https://github.com/ggml-org/whisper.cpp) (MIT) and OpenAI Whisper models (MIT).
|
|
159
|
+
Fine-tunes belong to their authors: [Pathumma Whisper](https://huggingface.co/nectec/Pathumma-whisper-th-large-v3) by NECTEC,
|
|
160
|
+
[Thonburian Whisper](https://huggingface.co/biodatlab/whisper-th-large-v3-combined) by biodatlab,
|
|
161
|
+
[whisper-hindi-large-v2](https://huggingface.co/vasista22/whisper-hindi-large-v2) by vasista22 (Speech Lab, IIT Madras).
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
babelscribe/__init__.py,sha256=VGezPdaYYtL55ILTeVGmlhe-iqS4YFLXY7nraKeCdbY,103
|
|
2
|
+
babelscribe/__main__.py,sha256=bYt9eEaoRQWdejEHFD8REx9jxVEdZptECFsV7F49Ink,30
|
|
3
|
+
babelscribe/align.py,sha256=ih6Z51NhNb56ZdgRVvHalV61YQhczZSPZJ5L2arF-IY,5448
|
|
4
|
+
babelscribe/backend.py,sha256=nK_XfB1ks-rzxV7NsoguuZGaeavLh9tHlTet7UvXEFQ,4173
|
|
5
|
+
babelscribe/cli.py,sha256=-rPXUO1LYZpOLah14LHo4cHFdcZRfzKUzrCyIC-NPbA,4831
|
|
6
|
+
babelscribe/hybrid.py,sha256=p4_8Tff7TX44SA2nuk7Fep6oUy0DaZAe9kuFTVs6uwY,2313
|
|
7
|
+
babelscribe/models.py,sha256=ZJlP8FmFSACXAK9BQqmGtHDiJo_c69zzipdLgOK3UTs,4748
|
|
8
|
+
babelscribe/transcribe.py,sha256=P6PHvlgXV4qQoXlxQgx0NOzqZVRkdBWD9EDxWg80CW0,3180
|
|
9
|
+
babelscribe/writers.py,sha256=KgBgU8ffo4ZivfKB1nYUoLvyKFN0_KS8URGq73fnE9g,1206
|
|
10
|
+
babelscribe-0.2.0.dist-info/licenses/LICENSE,sha256=3ZK_oMT5HxgxqJMijLqNWomKlKvxrPgyQfMOHpg-X-s,1081
|
|
11
|
+
babelscribe-0.2.0.dist-info/METADATA,sha256=_RpYsw7J7ac9hLfZ2lRyLuu8jQolMjjYC_OCyW37cEM,12200
|
|
12
|
+
babelscribe-0.2.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
13
|
+
babelscribe-0.2.0.dist-info/entry_points.txt,sha256=y2Qve_Z8e-VqQI3YfjnTfT8gXaQ5xKFAJk3KyVBBRh4,53
|
|
14
|
+
babelscribe-0.2.0.dist-info/top_level.txt,sha256=hv5KCl8pcQP8PT6r1d63GdGfprcubJlOZBZipJYP-fc,12
|
|
15
|
+
babelscribe-0.2.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 babelscribe contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
babelscribe
|