ttgrep 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ttgrep/__init__.py +1 -0
- ttgrep/__main__.py +4 -0
- ttgrep/asr.py +120 -0
- ttgrep/cli.py +796 -0
- ttgrep/cost.py +91 -0
- ttgrep/fetch.py +360 -0
- ttgrep/mcp_server.py +128 -0
- ttgrep/store.py +125 -0
- ttgrep-0.5.0.dist-info/METADATA +241 -0
- ttgrep-0.5.0.dist-info/RECORD +13 -0
- ttgrep-0.5.0.dist-info/WHEEL +4 -0
- ttgrep-0.5.0.dist-info/entry_points.txt +2 -0
- ttgrep-0.5.0.dist-info/licenses/LICENSE +21 -0
ttgrep/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.5.0"
|
ttgrep/__main__.py
ADDED
ttgrep/asr.py
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"""Local speech-to-text for videos that have no TikTok captions.
|
|
2
|
+
|
|
3
|
+
Backend preference: mlx-whisper (Apple Silicon; shells out to the ffmpeg CLI)
|
|
4
|
+
then faster-whisper (portable; decodes audio itself, no ffmpeg needed).
|
|
5
|
+
Models download from Hugging Face on first use (~500MB for `small`) into the
|
|
6
|
+
HF cache and are reused forever after.
|
|
7
|
+
|
|
8
|
+
Everything runs on-device: no API, no cost, no audio leaves the machine.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import os
|
|
14
|
+
import shutil
|
|
15
|
+
|
|
16
|
+
os.environ.setdefault("HF_HUB_DISABLE_PROGRESS_BARS", "1")
|
|
17
|
+
|
|
18
|
+
MODELS = ("tiny", "base", "small", "medium", "large-v3")
|
|
19
|
+
DEFAULT_MODEL = "small" # smallest model that handles non-English speech well
|
|
20
|
+
|
|
21
|
+
_state: dict = {"checked": False, "kind": None, "why": None}
|
|
22
|
+
_faster_models: dict = {}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def backend() -> str | None:
|
|
26
|
+
"""'mlx' | 'faster' | None, decided once per process."""
|
|
27
|
+
if not _state["checked"]:
|
|
28
|
+
_state["checked"] = True
|
|
29
|
+
try:
|
|
30
|
+
import mlx_whisper # noqa: F401
|
|
31
|
+
|
|
32
|
+
if shutil.which("ffmpeg"):
|
|
33
|
+
_state["kind"] = "mlx"
|
|
34
|
+
else:
|
|
35
|
+
_state["why"] = "mlx-whisper installed but ffmpeg not on PATH"
|
|
36
|
+
except ImportError:
|
|
37
|
+
pass
|
|
38
|
+
if _state["kind"] is None:
|
|
39
|
+
try:
|
|
40
|
+
import faster_whisper # noqa: F401
|
|
41
|
+
|
|
42
|
+
_state["kind"] = "faster"
|
|
43
|
+
except ImportError:
|
|
44
|
+
_state["why"] = _state["why"] or "no whisper backend installed"
|
|
45
|
+
return _state["kind"]
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def unavailable_reason() -> str:
|
|
49
|
+
backend()
|
|
50
|
+
return _state["why"] or "unknown"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _mlx_repo(model: str) -> str:
|
|
54
|
+
if "/" in model: # full HF repo id passed through
|
|
55
|
+
return model
|
|
56
|
+
return f"mlx-community/whisper-{model}-mlx"
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def transcribe(path, model: str = DEFAULT_MODEL) -> tuple[str | None, list[dict]]:
|
|
60
|
+
"""Transcribe an audio/video file. Returns (language_code, segments).
|
|
61
|
+
|
|
62
|
+
Segments use the same {start, end, text} shape as caption cues. The
|
|
63
|
+
language code is whisper-style ('fr', 'en'), distinct from TikTok's
|
|
64
|
+
caption codes ('fra-FR'), which keeps the two kinds of track apart.
|
|
65
|
+
"""
|
|
66
|
+
kind = backend()
|
|
67
|
+
if kind is None:
|
|
68
|
+
raise RuntimeError(f"no ASR backend: {unavailable_reason()}")
|
|
69
|
+
|
|
70
|
+
if kind == "mlx":
|
|
71
|
+
import mlx_whisper
|
|
72
|
+
|
|
73
|
+
result = mlx_whisper.transcribe(str(path), path_or_hf_repo=_mlx_repo(model), verbose=None)
|
|
74
|
+
lang = result.get("language")
|
|
75
|
+
raw = [(float(s["start"]), float(s["end"]), str(s.get("text", "")),
|
|
76
|
+
s.get("no_speech_prob"), s.get("avg_logprob"), s.get("compression_ratio"))
|
|
77
|
+
for s in result.get("segments", [])]
|
|
78
|
+
else:
|
|
79
|
+
from faster_whisper import WhisperModel
|
|
80
|
+
|
|
81
|
+
m = _faster_models.get(model)
|
|
82
|
+
if m is None:
|
|
83
|
+
m = _faster_models[model] = WhisperModel(model, compute_type="int8")
|
|
84
|
+
segs, info = m.transcribe(str(path), vad_filter=True)
|
|
85
|
+
lang = info.language
|
|
86
|
+
raw = [(s.start, s.end, s.text, s.no_speech_prob, s.avg_logprob, s.compression_ratio)
|
|
87
|
+
for s in segs]
|
|
88
|
+
|
|
89
|
+
segments: list[dict] = []
|
|
90
|
+
for start, end, text, nsp, alp, cr in raw:
|
|
91
|
+
text = text.strip()
|
|
92
|
+
if not text or not _looks_like_speech(nsp, alp, cr):
|
|
93
|
+
continue
|
|
94
|
+
if segments and segments[-1]["text"] == text: # whisper stutter
|
|
95
|
+
segments[-1]["end"] = round(end, 3)
|
|
96
|
+
else:
|
|
97
|
+
segments.append({"start": round(start, 3), "end": round(end, 3), "text": text})
|
|
98
|
+
return lang, segments
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _looks_like_speech(no_speech_prob, avg_logprob, compression_ratio) -> bool:
|
|
102
|
+
"""Reject whisper hallucinations on music/ambient audio.
|
|
103
|
+
|
|
104
|
+
Calibrated on TikTok audio: real speech (even over loud music) scores
|
|
105
|
+
avg_logprob ≳ -0.9; hallucinated noise scores ≲ -3. compression_ratio
|
|
106
|
+
> 2.4 (openai-whisper's own threshold) kills repetition loops ("a little
|
|
107
|
+
bit of a little bit of ..."). The classic silence rule (no_speech > 0.6
|
|
108
|
+
and logprob < -1) is kept as a final net. Accurately-heard song lyrics
|
|
109
|
+
pass all three on purpose: they are real audio content — the caller can
|
|
110
|
+
judge them by the track's whisper provenance.
|
|
111
|
+
"""
|
|
112
|
+
if compression_ratio is not None and compression_ratio > 2.4:
|
|
113
|
+
return False
|
|
114
|
+
if avg_logprob is None:
|
|
115
|
+
return True
|
|
116
|
+
if avg_logprob < -2.0:
|
|
117
|
+
return False
|
|
118
|
+
if no_speech_prob is not None and no_speech_prob > 0.6 and avg_logprob < -1.0:
|
|
119
|
+
return False
|
|
120
|
+
return True
|