ttgrep 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ttgrep/__init__.py ADDED
@@ -0,0 +1 @@
1
+ __version__ = "0.5.0"
ttgrep/__main__.py ADDED
@@ -0,0 +1,4 @@
1
+ from ttgrep.cli import main
2
+
3
+ if __name__ == "__main__":
4
+ main()
ttgrep/asr.py ADDED
@@ -0,0 +1,120 @@
1
+ """Local speech-to-text for videos that have no TikTok captions.
2
+
3
+ Backend preference: mlx-whisper (Apple Silicon; shells out to the ffmpeg CLI)
4
+ then faster-whisper (portable; decodes audio itself, no ffmpeg needed).
5
+ Models download from Hugging Face on first use (~500MB for `small`) into the
6
+ HF cache and are reused forever after.
7
+
8
+ Everything runs on-device: no API, no cost, no audio leaves the machine.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import os
14
+ import shutil
15
+
16
+ os.environ.setdefault("HF_HUB_DISABLE_PROGRESS_BARS", "1")
17
+
18
+ MODELS = ("tiny", "base", "small", "medium", "large-v3")
19
+ DEFAULT_MODEL = "small" # smallest model that handles non-English speech well
20
+
21
+ _state: dict = {"checked": False, "kind": None, "why": None}
22
+ _faster_models: dict = {}
23
+
24
+
25
+ def backend() -> str | None:
26
+ """'mlx' | 'faster' | None, decided once per process."""
27
+ if not _state["checked"]:
28
+ _state["checked"] = True
29
+ try:
30
+ import mlx_whisper # noqa: F401
31
+
32
+ if shutil.which("ffmpeg"):
33
+ _state["kind"] = "mlx"
34
+ else:
35
+ _state["why"] = "mlx-whisper installed but ffmpeg not on PATH"
36
+ except ImportError:
37
+ pass
38
+ if _state["kind"] is None:
39
+ try:
40
+ import faster_whisper # noqa: F401
41
+
42
+ _state["kind"] = "faster"
43
+ except ImportError:
44
+ _state["why"] = _state["why"] or "no whisper backend installed"
45
+ return _state["kind"]
46
+
47
+
48
+ def unavailable_reason() -> str:
49
+ backend()
50
+ return _state["why"] or "unknown"
51
+
52
+
53
+ def _mlx_repo(model: str) -> str:
54
+ if "/" in model: # full HF repo id passed through
55
+ return model
56
+ return f"mlx-community/whisper-{model}-mlx"
57
+
58
+
59
+ def transcribe(path, model: str = DEFAULT_MODEL) -> tuple[str | None, list[dict]]:
60
+ """Transcribe an audio/video file. Returns (language_code, segments).
61
+
62
+ Segments use the same {start, end, text} shape as caption cues. The
63
+ language code is whisper-style ('fr', 'en'), distinct from TikTok's
64
+ caption codes ('fra-FR'), which keeps the two kinds of track apart.
65
+ """
66
+ kind = backend()
67
+ if kind is None:
68
+ raise RuntimeError(f"no ASR backend: {unavailable_reason()}")
69
+
70
+ if kind == "mlx":
71
+ import mlx_whisper
72
+
73
+ result = mlx_whisper.transcribe(str(path), path_or_hf_repo=_mlx_repo(model), verbose=None)
74
+ lang = result.get("language")
75
+ raw = [(float(s["start"]), float(s["end"]), str(s.get("text", "")),
76
+ s.get("no_speech_prob"), s.get("avg_logprob"), s.get("compression_ratio"))
77
+ for s in result.get("segments", [])]
78
+ else:
79
+ from faster_whisper import WhisperModel
80
+
81
+ m = _faster_models.get(model)
82
+ if m is None:
83
+ m = _faster_models[model] = WhisperModel(model, compute_type="int8")
84
+ segs, info = m.transcribe(str(path), vad_filter=True)
85
+ lang = info.language
86
+ raw = [(s.start, s.end, s.text, s.no_speech_prob, s.avg_logprob, s.compression_ratio)
87
+ for s in segs]
88
+
89
+ segments: list[dict] = []
90
+ for start, end, text, nsp, alp, cr in raw:
91
+ text = text.strip()
92
+ if not text or not _looks_like_speech(nsp, alp, cr):
93
+ continue
94
+ if segments and segments[-1]["text"] == text: # whisper stutter
95
+ segments[-1]["end"] = round(end, 3)
96
+ else:
97
+ segments.append({"start": round(start, 3), "end": round(end, 3), "text": text})
98
+ return lang, segments
99
+
100
+
101
+ def _looks_like_speech(no_speech_prob, avg_logprob, compression_ratio) -> bool:
102
+ """Reject whisper hallucinations on music/ambient audio.
103
+
104
+ Calibrated on TikTok audio: real speech (even over loud music) scores
105
+ avg_logprob ≳ -0.9; hallucinated noise scores ≲ -3. compression_ratio
106
+ > 2.4 (openai-whisper's own threshold) kills repetition loops ("a little
107
+ bit of a little bit of ..."). The classic silence rule (no_speech > 0.6
108
+ and logprob < -1) is kept as a final net. Accurately-heard song lyrics
109
+ pass all three on purpose: they are real audio content — the caller can
110
+ judge them by the track's whisper provenance.
111
+ """
112
+ if compression_ratio is not None and compression_ratio > 2.4:
113
+ return False
114
+ if avg_logprob is None:
115
+ return True
116
+ if avg_logprob < -2.0:
117
+ return False
118
+ if no_speech_prob is not None and no_speech_prob > 0.6 and avg_logprob < -1.0:
119
+ return False
120
+ return True