clipscribe 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- clipscribe/__init__.py +52 -0
- clipscribe/extractor.py +295 -0
- clipscribe/py.typed +0 -0
- clipscribe/utils/__init__.py +0 -0
- clipscribe/utils/extras.py +10 -0
- clipscribe/utils/instagram_utils.py +164 -0
- clipscribe/utils/tiktok_utils.py +142 -0
- clipscribe/utils/transcript_utils.py +217 -0
- clipscribe/utils/whisper_backend.py +44 -0
- clipscribe/utils/ydl_audio.py +131 -0
- clipscribe/utils/youtube_utils.py +173 -0
- clipscribe-0.1.0.dist-info/METADATA +164 -0
- clipscribe-0.1.0.dist-info/RECORD +16 -0
- clipscribe-0.1.0.dist-info/WHEEL +5 -0
- clipscribe-0.1.0.dist-info/licenses/LICENSE +21 -0
- clipscribe-0.1.0.dist-info/top_level.txt +1 -0
clipscribe/__init__.py
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
2
|
+
|
|
3
|
+
from .extractor import ExtractionResponse, TranscriptExtractor
|
|
4
|
+
from .utils.extras import missing_extra
|
|
5
|
+
from .utils.transcript_utils import (
|
|
6
|
+
TranscriptResult,
|
|
7
|
+
TranscriptSaver,
|
|
8
|
+
TranscriptSegment,
|
|
9
|
+
TranscriptSource,
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
try:
|
|
13
|
+
__version__ = version("clipscribe")
|
|
14
|
+
except PackageNotFoundError:
|
|
15
|
+
__version__ = "0.1.0"
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"ExtractionResponse",
|
|
19
|
+
"GenericProxyConfig",
|
|
20
|
+
"InstagramTranscriptExtractor",
|
|
21
|
+
"TikTokTranscriptExtractor",
|
|
22
|
+
"TranscriptExtractor",
|
|
23
|
+
"TranscriptResult",
|
|
24
|
+
"TranscriptSaver",
|
|
25
|
+
"TranscriptSegment",
|
|
26
|
+
"TranscriptSource",
|
|
27
|
+
"YouTubeTranscriptExtractor",
|
|
28
|
+
"__version__",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def __dir__():
|
|
33
|
+
return sorted(set(globals()) | set(__all__))
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def __getattr__(name: str):
|
|
37
|
+
if name == "GenericProxyConfig":
|
|
38
|
+
try:
|
|
39
|
+
from youtube_transcript_api.proxies import GenericProxyConfig
|
|
40
|
+
except ImportError as exc:
|
|
41
|
+
raise missing_extra("youtube", "YouTube proxy config") from exc
|
|
42
|
+
return GenericProxyConfig
|
|
43
|
+
if name == "YouTubeTranscriptExtractor":
|
|
44
|
+
from .utils.youtube_utils import YouTubeTranscriptExtractor
|
|
45
|
+
return YouTubeTranscriptExtractor
|
|
46
|
+
if name == "TikTokTranscriptExtractor":
|
|
47
|
+
from .utils.tiktok_utils import TikTokTranscriptExtractor
|
|
48
|
+
return TikTokTranscriptExtractor
|
|
49
|
+
if name == "InstagramTranscriptExtractor":
|
|
50
|
+
from .utils.instagram_utils import InstagramTranscriptExtractor
|
|
51
|
+
return InstagramTranscriptExtractor
|
|
52
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
clipscribe/extractor.py
ADDED
|
@@ -0,0 +1,295 @@
|
|
|
1
|
+
"""Unified transcript extraction API for YouTube, TikTok, and Instagram."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
import re
|
|
7
|
+
import time
|
|
8
|
+
from typing import Callable
|
|
9
|
+
|
|
10
|
+
from .utils.transcript_utils import PlatformExtractor, TranscriptResult, TranscriptSaver
|
|
11
|
+
|
|
12
|
+
log = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
_NON_RETRYABLE = (ValueError, FileNotFoundError, ImportError)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class ExtractionResponse:
|
|
18
|
+
"""
|
|
19
|
+
Wraps a TranscriptResult alongside optional save paths.
|
|
20
|
+
This is what TranscriptExtractor.extract() always returns.
|
|
21
|
+
|
|
22
|
+
Attributes:
|
|
23
|
+
result: The raw TranscriptResult (segments, metadata, full_text).
|
|
24
|
+
saved: Dict of format -> Path if save=True, else None.
|
|
25
|
+
url: The original URL that was extracted.
|
|
26
|
+
platform: Detected TranscriptSource.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
def __init__(self, result: TranscriptResult, saved: dict | None = None):
|
|
30
|
+
self.result = result
|
|
31
|
+
self.saved = saved
|
|
32
|
+
self.url = result.url
|
|
33
|
+
self.platform = result.source
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def full_text(self) -> str:
|
|
37
|
+
return self.result.full_text
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def segments(self):
|
|
41
|
+
return self.result.segments
|
|
42
|
+
|
|
43
|
+
@property
|
|
44
|
+
def metadata(self):
|
|
45
|
+
return self.result.metadata
|
|
46
|
+
|
|
47
|
+
def to_dict(self) -> dict:
|
|
48
|
+
data = self.result.to_dict()
|
|
49
|
+
if self.saved:
|
|
50
|
+
data["saved_files"] = {k: str(v) for k, v in self.saved.items()}
|
|
51
|
+
return data
|
|
52
|
+
|
|
53
|
+
def __repr__(self) -> str:
|
|
54
|
+
return (
|
|
55
|
+
f"<ExtractionResponse platform={self.platform.value!r} "
|
|
56
|
+
f"segments={len(self.segments)} url={self.url!r}>"
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class TranscriptExtractor:
|
|
61
|
+
"""
|
|
62
|
+
Unified transcript extraction API.
|
|
63
|
+
|
|
64
|
+
Built-in extractors and extra platforms added via ``register()`` are
|
|
65
|
+
per-instance.
|
|
66
|
+
|
|
67
|
+
Args:
|
|
68
|
+
language: Preferred transcript language for YouTube (default: "en").
|
|
69
|
+
whisper_model: faster-whisper model size for TikTok/Instagram (default: "tiny").
|
|
70
|
+
whisper_device: "cpu" or "cuda" (default: "cpu").
|
|
71
|
+
whisper_compute: Quantization type for faster-whisper (default: "int8").
|
|
72
|
+
output_dir: Directory for saved transcripts (default: "output", relative to CWD).
|
|
73
|
+
allowed_outputs: File formats to write when save=True.
|
|
74
|
+
Choose from "json", "txt", "srt" (default: json).
|
|
75
|
+
on_error: Optional callback(url, exc) called on per-URL failures
|
|
76
|
+
in extract_many(). Defaults to logging the error.
|
|
77
|
+
youtube_proxy_config: Optional youtube-transcript-api ProxyConfig.
|
|
78
|
+
tiktok_cookies_from_browser: Browser name for TikTok cookies.
|
|
79
|
+
tiktok_cookies: Netscape cookies.txt path for TikTok.
|
|
80
|
+
instagram_cookies_from_browser: Browser name for Instagram cookies.
|
|
81
|
+
instagram_cookies: Netscape cookies.txt path for Instagram.
|
|
82
|
+
max_duration_s: Reject TikTok/Instagram videos longer than this
|
|
83
|
+
(default: 900). Pass None for no limit.
|
|
84
|
+
"""
|
|
85
|
+
|
|
86
|
+
def __init__(
|
|
87
|
+
self,
|
|
88
|
+
language: str = "en",
|
|
89
|
+
whisper_model: str = "tiny",
|
|
90
|
+
whisper_device: str = "cpu",
|
|
91
|
+
whisper_compute: str = "int8",
|
|
92
|
+
output_dir: str = "output",
|
|
93
|
+
allowed_outputs: list[str] | None = None,
|
|
94
|
+
on_error: Callable[[str, Exception], None] | None = None,
|
|
95
|
+
instagram_cookies_from_browser: str | None = None,
|
|
96
|
+
instagram_cookies: str | None = None,
|
|
97
|
+
tiktok_cookies_from_browser: str | None = None,
|
|
98
|
+
tiktok_cookies: str | None = None,
|
|
99
|
+
youtube_proxy_config=None,
|
|
100
|
+
max_duration_s: int | None = 900,
|
|
101
|
+
):
|
|
102
|
+
self._saver = TranscriptSaver(output_dir=output_dir, allowed_outputs=allowed_outputs)
|
|
103
|
+
self._on_error = on_error or self._default_error_handler
|
|
104
|
+
self._registry: list[tuple[re.Pattern, PlatformExtractor]] = []
|
|
105
|
+
self._extra: list[tuple[re.Pattern, PlatformExtractor]] = []
|
|
106
|
+
|
|
107
|
+
self._register_defaults(
|
|
108
|
+
language=language,
|
|
109
|
+
whisper_model=whisper_model,
|
|
110
|
+
whisper_device=whisper_device,
|
|
111
|
+
whisper_compute=whisper_compute,
|
|
112
|
+
instagram_cookies_from_browser=instagram_cookies_from_browser,
|
|
113
|
+
instagram_cookies=instagram_cookies,
|
|
114
|
+
tiktok_cookies_from_browser=tiktok_cookies_from_browser,
|
|
115
|
+
tiktok_cookies=tiktok_cookies,
|
|
116
|
+
youtube_proxy_config=youtube_proxy_config,
|
|
117
|
+
max_duration_s=max_duration_s,
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
def extract(
|
|
121
|
+
self,
|
|
122
|
+
url: str,
|
|
123
|
+
save: bool = False,
|
|
124
|
+
max_retries: int = 0,
|
|
125
|
+
allowed_outputs: list[str] | None = None,
|
|
126
|
+
) -> ExtractionResponse:
|
|
127
|
+
"""
|
|
128
|
+
Extract transcript from a single URL.
|
|
129
|
+
|
|
130
|
+
Args:
|
|
131
|
+
url: A YouTube, TikTok, or Instagram URL (or any registered platform).
|
|
132
|
+
save: If True, saves files to the configured output_dir.
|
|
133
|
+
max_retries: Extra attempts after a failed extract (default: 0).
|
|
134
|
+
allowed_outputs: Formats for this call only. Choose from
|
|
135
|
+
"json", "txt", "srt". Defaults to the instance list.
|
|
136
|
+
|
|
137
|
+
Returns:
|
|
138
|
+
ExtractionResponse with .full_text, .segments, .metadata, .to_dict()
|
|
139
|
+
|
|
140
|
+
Raises:
|
|
141
|
+
ValueError: If the URL doesn't match any registered platform.
|
|
142
|
+
RuntimeError: If extraction fails.
|
|
143
|
+
ImportError: If the matching platform extra is not installed.
|
|
144
|
+
"""
|
|
145
|
+
extractor = self._resolve(url)
|
|
146
|
+
if allowed_outputs is not None and not save:
|
|
147
|
+
log.warning("allowed_outputs is ignored because save=False")
|
|
148
|
+
max_retries = max(0, max_retries)
|
|
149
|
+
result: TranscriptResult | None = None
|
|
150
|
+
for attempt in range(max_retries + 1):
|
|
151
|
+
try:
|
|
152
|
+
result = extractor.extract(url)
|
|
153
|
+
break
|
|
154
|
+
except _NON_RETRYABLE:
|
|
155
|
+
raise
|
|
156
|
+
except Exception:
|
|
157
|
+
if attempt >= max_retries:
|
|
158
|
+
raise
|
|
159
|
+
time.sleep(min(8, 2 ** attempt))
|
|
160
|
+
if result is None:
|
|
161
|
+
raise RuntimeError(f"Extraction failed for {url}")
|
|
162
|
+
saved = (
|
|
163
|
+
self._saver.save(result, self._slug(result), allowed_outputs=allowed_outputs)
|
|
164
|
+
if save
|
|
165
|
+
else None
|
|
166
|
+
)
|
|
167
|
+
return ExtractionResponse(result=result, saved=saved)
|
|
168
|
+
|
|
169
|
+
def extract_many(
|
|
170
|
+
self,
|
|
171
|
+
urls: list[str],
|
|
172
|
+
save: bool = False,
|
|
173
|
+
max_retries: int = 0,
|
|
174
|
+
allowed_outputs: list[str] | None = None,
|
|
175
|
+
) -> list[ExtractionResponse]:
|
|
176
|
+
"""
|
|
177
|
+
Extract transcripts from multiple URLs.
|
|
178
|
+
Failures are caught, passed to on_error, and skipped — processing continues.
|
|
179
|
+
"""
|
|
180
|
+
if allowed_outputs is not None and not save:
|
|
181
|
+
log.warning("allowed_outputs is ignored because save=False")
|
|
182
|
+
allowed_outputs = None
|
|
183
|
+
responses = []
|
|
184
|
+
for url in urls:
|
|
185
|
+
try:
|
|
186
|
+
responses.append(
|
|
187
|
+
self.extract(
|
|
188
|
+
url,
|
|
189
|
+
save=save,
|
|
190
|
+
max_retries=max_retries,
|
|
191
|
+
allowed_outputs=allowed_outputs,
|
|
192
|
+
)
|
|
193
|
+
)
|
|
194
|
+
except Exception as exc:
|
|
195
|
+
self._on_error(url, exc)
|
|
196
|
+
return responses
|
|
197
|
+
|
|
198
|
+
def supports(self, url: str) -> bool:
|
|
199
|
+
"""Return True if the URL matches a registered platform."""
|
|
200
|
+
try:
|
|
201
|
+
self._resolve(url)
|
|
202
|
+
return True
|
|
203
|
+
except ValueError:
|
|
204
|
+
return False
|
|
205
|
+
|
|
206
|
+
def register(self, url_pattern: str, extractor: PlatformExtractor) -> None:
|
|
207
|
+
"""
|
|
208
|
+
Register a new platform extractor on this instance.
|
|
209
|
+
|
|
210
|
+
Args:
|
|
211
|
+
url_pattern: A regex string matched against the full URL.
|
|
212
|
+
extractor: Any object with an .extract(url) -> TranscriptResult method.
|
|
213
|
+
"""
|
|
214
|
+
self._extra.append((re.compile(url_pattern, re.IGNORECASE), extractor))
|
|
215
|
+
|
|
216
|
+
def _register_defaults(
|
|
217
|
+
self,
|
|
218
|
+
language: str,
|
|
219
|
+
whisper_model: str,
|
|
220
|
+
whisper_device: str,
|
|
221
|
+
whisper_compute: str,
|
|
222
|
+
instagram_cookies_from_browser: str | None = None,
|
|
223
|
+
instagram_cookies: str | None = None,
|
|
224
|
+
tiktok_cookies_from_browser: str | None = None,
|
|
225
|
+
tiktok_cookies: str | None = None,
|
|
226
|
+
youtube_proxy_config=None,
|
|
227
|
+
max_duration_s: int | None = 900,
|
|
228
|
+
) -> None:
|
|
229
|
+
from .utils.youtube_utils import YouTubeTranscriptExtractor
|
|
230
|
+
from .utils.tiktok_utils import TikTokTranscriptExtractor
|
|
231
|
+
|
|
232
|
+
self._registry = [
|
|
233
|
+
(
|
|
234
|
+
re.compile(r"youtube\.com|youtu\.be", re.IGNORECASE),
|
|
235
|
+
YouTubeTranscriptExtractor(
|
|
236
|
+
language=language,
|
|
237
|
+
proxy_config=youtube_proxy_config,
|
|
238
|
+
),
|
|
239
|
+
),
|
|
240
|
+
(
|
|
241
|
+
re.compile(r"tiktok\.com", re.IGNORECASE),
|
|
242
|
+
TikTokTranscriptExtractor(
|
|
243
|
+
whisper_model=whisper_model,
|
|
244
|
+
device=whisper_device,
|
|
245
|
+
compute_type=whisper_compute,
|
|
246
|
+
cookies_from_browser=tiktok_cookies_from_browser,
|
|
247
|
+
cookies_file=tiktok_cookies,
|
|
248
|
+
max_duration_s=max_duration_s,
|
|
249
|
+
),
|
|
250
|
+
),
|
|
251
|
+
]
|
|
252
|
+
|
|
253
|
+
if instagram_cookies_from_browser or instagram_cookies:
|
|
254
|
+
from .utils.instagram_utils import InstagramTranscriptExtractor
|
|
255
|
+
|
|
256
|
+
self._registry.append((
|
|
257
|
+
re.compile(r"instagram\.com", re.IGNORECASE),
|
|
258
|
+
InstagramTranscriptExtractor(
|
|
259
|
+
cookies_from_browser=instagram_cookies_from_browser,
|
|
260
|
+
cookies_file=instagram_cookies,
|
|
261
|
+
whisper_model=whisper_model,
|
|
262
|
+
device=whisper_device,
|
|
263
|
+
compute_type=whisper_compute,
|
|
264
|
+
max_duration_s=max_duration_s,
|
|
265
|
+
),
|
|
266
|
+
))
|
|
267
|
+
|
|
268
|
+
def _resolve(self, url: str) -> PlatformExtractor:
|
|
269
|
+
for pattern, extractor in (*self._registry, *self._extra):
|
|
270
|
+
if pattern.search(url):
|
|
271
|
+
return extractor
|
|
272
|
+
if "instagram.com" in url.lower():
|
|
273
|
+
raise ValueError(
|
|
274
|
+
"Instagram URLs require authentication. "
|
|
275
|
+
"Pass instagram_cookies_from_browser or instagram_cookies."
|
|
276
|
+
)
|
|
277
|
+
raise ValueError(
|
|
278
|
+
f"No extractor registered for URL: {url}\n"
|
|
279
|
+
f"Registered platforms: {self._registered_platforms()}"
|
|
280
|
+
)
|
|
281
|
+
|
|
282
|
+
def _slug(self, result: TranscriptResult) -> str:
|
|
283
|
+
media_id = (
|
|
284
|
+
result.metadata.get("video_id")
|
|
285
|
+
or result.metadata.get("post_id")
|
|
286
|
+
or "unknown"
|
|
287
|
+
)
|
|
288
|
+
return re.sub(r'[\\/*?:"<>|]', "_", f"{result.source.value}_{media_id}").strip()
|
|
289
|
+
|
|
290
|
+
def _registered_platforms(self) -> list[str]:
|
|
291
|
+
return [pattern.pattern for pattern, _ in (*self._registry, *self._extra)]
|
|
292
|
+
|
|
293
|
+
@staticmethod
|
|
294
|
+
def _default_error_handler(url: str, exc: Exception) -> None:
|
|
295
|
+
log.error("Failed (%s): %s", url, exc)
|
clipscribe/py.typed
ADDED
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""Install-extra hints for optional platform dependencies."""
|
|
2
|
+
|
|
3
|
+
def extra_install(platform: str) -> str:
|
|
4
|
+
return f'pip install "clipscribe[{platform}]" or "clipscribe[all]"'
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def missing_extra(platform: str, what: str) -> ImportError:
|
|
8
|
+
return ImportError(
|
|
9
|
+
f"{what} requires extra dependencies. Install with: {extra_install(platform)}"
|
|
10
|
+
)
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""
|
|
2
|
+
instagram_utils.py
|
|
3
|
+
───────────────────
|
|
4
|
+
Instagram transcript extraction via yt-dlp (audio download) + faster-whisper.
|
|
5
|
+
|
|
6
|
+
Instagram requires authentication. Two options, in recommended order:
|
|
7
|
+
|
|
8
|
+
1. cookies_from_browser (BEST — always fresh, no manual export needed)
|
|
9
|
+
2. cookies_file (OK — but expires fast)
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import os
|
|
15
|
+
import re
|
|
16
|
+
import tempfile
|
|
17
|
+
from urllib.parse import urlparse
|
|
18
|
+
|
|
19
|
+
from .transcript_utils import (
|
|
20
|
+
BaseTranscriptExtractor,
|
|
21
|
+
TranscriptResult,
|
|
22
|
+
TranscriptSource,
|
|
23
|
+
)
|
|
24
|
+
from .whisper_backend import load_whisper_model, transcribe_to_segments
|
|
25
|
+
from .ydl_audio import (
|
|
26
|
+
SUPPORTED_BROWSERS,
|
|
27
|
+
build_audio_ydl_opts,
|
|
28
|
+
download_audio,
|
|
29
|
+
validate_browser,
|
|
30
|
+
validate_cookies_file,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
_URL_PATTERNS: list[tuple[re.Pattern[str], str]] = [
|
|
34
|
+
(re.compile(r"instagram\.com/share/reel/([A-Za-z0-9_-]+)", re.IGNORECASE), "reel"),
|
|
35
|
+
(re.compile(r"instagram\.com/share/p/([A-Za-z0-9_-]+)", re.IGNORECASE), "p"),
|
|
36
|
+
(re.compile(r"instagram\.com/reels/([A-Za-z0-9_-]+)", re.IGNORECASE), "reel"),
|
|
37
|
+
(re.compile(r"instagram\.com/reel/([A-Za-z0-9_-]+)", re.IGNORECASE), "reel"),
|
|
38
|
+
(re.compile(r"instagram\.com/p/([A-Za-z0-9_-]+)", re.IGNORECASE), "p"),
|
|
39
|
+
(re.compile(r"instagram\.com/tv/([A-Za-z0-9_-]+)", re.IGNORECASE), "tv"),
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def parse_instagram_post(url: str) -> tuple[str | None, str]:
|
|
44
|
+
for pattern, kind in _URL_PATTERNS:
|
|
45
|
+
m = pattern.search(url)
|
|
46
|
+
if m:
|
|
47
|
+
return m.group(1), kind
|
|
48
|
+
return None, ""
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def canonical_instagram_media_url(url: str, post_id: str, kind: str) -> str:
|
|
52
|
+
host = (urlparse(url.strip()).hostname or "").lower()
|
|
53
|
+
if host not in ("www.instagram.com", "instagram.com", "m.instagram.com"):
|
|
54
|
+
raise ValueError(f"Instagram transcript URL host not allowed: {host!r}")
|
|
55
|
+
base = "https://www.instagram.com"
|
|
56
|
+
if kind == "p":
|
|
57
|
+
return f"{base}/p/{post_id}/"
|
|
58
|
+
if kind == "tv":
|
|
59
|
+
return f"{base}/tv/{post_id}/"
|
|
60
|
+
return f"{base}/reel/{post_id}/"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class InstagramTranscriptExtractor(BaseTranscriptExtractor):
|
|
64
|
+
"""
|
|
65
|
+
Downloads Instagram audio with yt-dlp and transcribes it with faster-whisper.
|
|
66
|
+
|
|
67
|
+
Requires: pip install "clipscribe[instagram]"
|
|
68
|
+
|
|
69
|
+
Instagram requires authentication — provide cookies_from_browser or cookies_file.
|
|
70
|
+
|
|
71
|
+
Args:
|
|
72
|
+
cookies_from_browser: Browser name to read live cookies from.
|
|
73
|
+
cookies_file: Path to a Netscape-format cookies.txt file.
|
|
74
|
+
whisper_model: Whisper model size (default: "tiny").
|
|
75
|
+
device: "cpu" or "cuda" (default: "cpu").
|
|
76
|
+
compute_type: Quantization — "int8", "float16", "float32".
|
|
77
|
+
ydl_opts_extra: Optional dict merged into yt-dlp options.
|
|
78
|
+
max_duration_s: Reject videos longer than this (default: 900).
|
|
79
|
+
Pass None for no limit.
|
|
80
|
+
socket_timeout: yt-dlp socket timeout in seconds (default: 30).
|
|
81
|
+
"""
|
|
82
|
+
|
|
83
|
+
SUPPORTED_BROWSERS = SUPPORTED_BROWSERS
|
|
84
|
+
|
|
85
|
+
def __init__(
|
|
86
|
+
self,
|
|
87
|
+
cookies_from_browser: str | None = None,
|
|
88
|
+
cookies_file: str | None = None,
|
|
89
|
+
whisper_model: str = "tiny",
|
|
90
|
+
device: str = "cpu",
|
|
91
|
+
compute_type: str = "int8",
|
|
92
|
+
ydl_opts_extra: dict | None = None,
|
|
93
|
+
max_duration_s: int | None = 900,
|
|
94
|
+
socket_timeout: int = 30,
|
|
95
|
+
):
|
|
96
|
+
if not any([cookies_from_browser, cookies_file]):
|
|
97
|
+
raise ValueError(
|
|
98
|
+
"Instagram requires authentication. Provide one of:\n"
|
|
99
|
+
" - cookies_from_browser: 'chrome', 'firefox', 'edge', etc. (recommended)\n"
|
|
100
|
+
" - cookies_file: path to a Netscape-format cookies.txt"
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
validate_browser(cookies_from_browser)
|
|
104
|
+
validate_cookies_file(cookies_file)
|
|
105
|
+
|
|
106
|
+
self.cookies_from_browser = cookies_from_browser
|
|
107
|
+
self.cookies_file = cookies_file
|
|
108
|
+
self.whisper_model = whisper_model
|
|
109
|
+
self.device = device
|
|
110
|
+
self.compute_type = compute_type
|
|
111
|
+
self.ydl_opts_extra = ydl_opts_extra or {}
|
|
112
|
+
self.max_duration_s = max_duration_s
|
|
113
|
+
self.socket_timeout = socket_timeout
|
|
114
|
+
self._model = None
|
|
115
|
+
|
|
116
|
+
def extract(self, url: str) -> TranscriptResult:
|
|
117
|
+
post_id, kind = parse_instagram_post(url)
|
|
118
|
+
if not post_id or not kind:
|
|
119
|
+
raise ValueError(
|
|
120
|
+
f"Unsupported Instagram URL: {url}\n"
|
|
121
|
+
"Supported formats: /reel/<id>/, /p/<id>/, /tv/<id>/"
|
|
122
|
+
)
|
|
123
|
+
fetch_url = canonical_instagram_media_url(url, post_id, kind)
|
|
124
|
+
|
|
125
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
126
|
+
audio_path, ydl_info = self._download_audio(fetch_url, tmp)
|
|
127
|
+
self.load_model()
|
|
128
|
+
segments, transcription_info = transcribe_to_segments(self._model, audio_path)
|
|
129
|
+
|
|
130
|
+
result = self._make_result(
|
|
131
|
+
fetch_url,
|
|
132
|
+
source=TranscriptSource.INSTAGRAM,
|
|
133
|
+
post_id=post_id,
|
|
134
|
+
title=ydl_info.get("title", ""),
|
|
135
|
+
author=ydl_info.get("uploader", ""),
|
|
136
|
+
language_detected=transcription_info.language,
|
|
137
|
+
language_probability=round(transcription_info.language_probability, 3),
|
|
138
|
+
whisper_model=self.whisper_model,
|
|
139
|
+
)
|
|
140
|
+
result.segments = segments
|
|
141
|
+
result.duration_s = ydl_info.get("duration")
|
|
142
|
+
return result
|
|
143
|
+
|
|
144
|
+
def load_model(self) -> None:
|
|
145
|
+
if self._model is None:
|
|
146
|
+
self._model = load_whisper_model(
|
|
147
|
+
self.whisper_model,
|
|
148
|
+
self.device,
|
|
149
|
+
self.compute_type,
|
|
150
|
+
extra="instagram",
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
def _download_audio(self, url: str, tmp_dir: str) -> tuple[str, dict]:
|
|
154
|
+
template = os.path.join(tmp_dir, "audio.%(ext)s")
|
|
155
|
+
opts = build_audio_ydl_opts(
|
|
156
|
+
template,
|
|
157
|
+
cookies_from_browser=self.cookies_from_browser,
|
|
158
|
+
cookies_file=self.cookies_file,
|
|
159
|
+
ydl_opts_extra=self.ydl_opts_extra,
|
|
160
|
+
socket_timeout=self.socket_timeout,
|
|
161
|
+
max_duration_s=self.max_duration_s,
|
|
162
|
+
extra="instagram",
|
|
163
|
+
)
|
|
164
|
+
return download_audio(url, tmp_dir, opts, platform="instagram")
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
"""
|
|
2
|
+
tiktok_utils.py
|
|
3
|
+
────────────────
|
|
4
|
+
TikTok transcript extraction via yt-dlp (audio download) + faster-whisper.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import os
|
|
10
|
+
import re
|
|
11
|
+
import tempfile
|
|
12
|
+
from urllib.parse import urlparse
|
|
13
|
+
|
|
14
|
+
from .transcript_utils import (
|
|
15
|
+
BaseTranscriptExtractor,
|
|
16
|
+
TranscriptResult,
|
|
17
|
+
TranscriptSource,
|
|
18
|
+
)
|
|
19
|
+
from .whisper_backend import load_whisper_model, transcribe_to_segments
|
|
20
|
+
from .ydl_audio import (
|
|
21
|
+
build_audio_ydl_opts,
|
|
22
|
+
download_audio,
|
|
23
|
+
validate_browser,
|
|
24
|
+
validate_cookies_file,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
_SHORT_HOSTS = {"vm.tiktok.com", "vt.tiktok.com"}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def assert_allowed_tiktok_url(url: str) -> None:
|
|
31
|
+
u = urlparse((url or "").strip())
|
|
32
|
+
host = (u.hostname or "").lower()
|
|
33
|
+
if not host:
|
|
34
|
+
raise ValueError("TikTok URL is missing a host")
|
|
35
|
+
if not (
|
|
36
|
+
host in ("tiktok.com", "www.tiktok.com", "m.tiktok.com", "vm.tiktok.com", "vt.tiktok.com")
|
|
37
|
+
or host.endswith(".tiktok.com")
|
|
38
|
+
):
|
|
39
|
+
raise ValueError(f"TikTok transcript URL host not allowed: {host!r}")
|
|
40
|
+
|
|
41
|
+
path = u.path or ""
|
|
42
|
+
if host in _SHORT_HOSTS:
|
|
43
|
+
if path.strip("/"):
|
|
44
|
+
return
|
|
45
|
+
raise ValueError("TikTok short URL is missing a path")
|
|
46
|
+
if re.search(r"/video/\d+", path):
|
|
47
|
+
return
|
|
48
|
+
if re.search(r"/v/\d+", path):
|
|
49
|
+
return
|
|
50
|
+
if re.search(r"/t/[A-Za-z0-9]+", path):
|
|
51
|
+
return
|
|
52
|
+
raise ValueError(
|
|
53
|
+
"TikTok URL must be a single video (e.g. /@user/video/ID or vm.tiktok.com/...), "
|
|
54
|
+
"not a profile, tag, or playlist."
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class TikTokTranscriptExtractor(BaseTranscriptExtractor):
|
|
59
|
+
"""
|
|
60
|
+
Downloads TikTok audio with yt-dlp and transcribes it with faster-whisper.
|
|
61
|
+
|
|
62
|
+
Requires: pip install "clipscribe[tiktok]"
|
|
63
|
+
|
|
64
|
+
Args:
|
|
65
|
+
whisper_model: Whisper model size — "tiny", "base", "small",
|
|
66
|
+
"medium", or "large" (default: "tiny").
|
|
67
|
+
device: "cpu" or "cuda" (default: "cpu").
|
|
68
|
+
compute_type: Quantization type — e.g. "int8", "float16", "float32".
|
|
69
|
+
cookies_from_browser: Browser name to read cookies from.
|
|
70
|
+
cookies_file: Path to a Netscape-format cookies.txt file.
|
|
71
|
+
ydl_opts_extra: Optional dict merged into yt-dlp options.
|
|
72
|
+
max_duration_s: Reject videos longer than this (default: 900).
|
|
73
|
+
Pass None for no limit.
|
|
74
|
+
socket_timeout: yt-dlp socket timeout in seconds (default: 30).
|
|
75
|
+
"""
|
|
76
|
+
|
|
77
|
+
def __init__(
|
|
78
|
+
self,
|
|
79
|
+
whisper_model: str = "tiny",
|
|
80
|
+
device: str = "cpu",
|
|
81
|
+
compute_type: str = "int8",
|
|
82
|
+
cookies_from_browser: str | None = None,
|
|
83
|
+
cookies_file: str | None = None,
|
|
84
|
+
ydl_opts_extra: dict | None = None,
|
|
85
|
+
max_duration_s: int | None = 900,
|
|
86
|
+
socket_timeout: int = 30,
|
|
87
|
+
):
|
|
88
|
+
validate_browser(cookies_from_browser)
|
|
89
|
+
validate_cookies_file(cookies_file)
|
|
90
|
+
|
|
91
|
+
self.whisper_model = whisper_model
|
|
92
|
+
self.device = device
|
|
93
|
+
self.compute_type = compute_type
|
|
94
|
+
self.cookies_from_browser = cookies_from_browser
|
|
95
|
+
self.cookies_file = cookies_file
|
|
96
|
+
self.ydl_opts_extra = ydl_opts_extra or {}
|
|
97
|
+
self.max_duration_s = max_duration_s
|
|
98
|
+
self.socket_timeout = socket_timeout
|
|
99
|
+
self._model = None
|
|
100
|
+
|
|
101
|
+
def extract(self, url: str) -> TranscriptResult:
|
|
102
|
+
assert_allowed_tiktok_url(url)
|
|
103
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
104
|
+
audio_path, ydl_info = self._download_audio(url, tmp)
|
|
105
|
+
self.load_model()
|
|
106
|
+
segments, transcription_info = transcribe_to_segments(self._model, audio_path)
|
|
107
|
+
|
|
108
|
+
result = self._make_result(
|
|
109
|
+
url,
|
|
110
|
+
source=TranscriptSource.TIKTOK,
|
|
111
|
+
video_id=ydl_info.get("id", "unknown"),
|
|
112
|
+
title=ydl_info.get("title", ""),
|
|
113
|
+
author=ydl_info.get("uploader", ""),
|
|
114
|
+
language_detected=transcription_info.language,
|
|
115
|
+
language_probability=round(transcription_info.language_probability, 3),
|
|
116
|
+
whisper_model=self.whisper_model,
|
|
117
|
+
)
|
|
118
|
+
result.segments = segments
|
|
119
|
+
result.duration_s = ydl_info.get("duration")
|
|
120
|
+
return result
|
|
121
|
+
|
|
122
|
+
def load_model(self) -> None:
|
|
123
|
+
if self._model is None:
|
|
124
|
+
self._model = load_whisper_model(
|
|
125
|
+
self.whisper_model,
|
|
126
|
+
self.device,
|
|
127
|
+
self.compute_type,
|
|
128
|
+
extra="tiktok",
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
def _download_audio(self, url: str, tmp_dir: str) -> tuple[str, dict]:
|
|
132
|
+
template = os.path.join(tmp_dir, "audio.%(ext)s")
|
|
133
|
+
opts = build_audio_ydl_opts(
|
|
134
|
+
template,
|
|
135
|
+
cookies_from_browser=self.cookies_from_browser,
|
|
136
|
+
cookies_file=self.cookies_file,
|
|
137
|
+
ydl_opts_extra=self.ydl_opts_extra,
|
|
138
|
+
socket_timeout=self.socket_timeout,
|
|
139
|
+
max_duration_s=self.max_duration_s,
|
|
140
|
+
extra="tiktok",
|
|
141
|
+
)
|
|
142
|
+
return download_audio(url, tmp_dir, opts, platform="tiktok")
|