clipscribe 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
clipscribe/__init__.py ADDED
@@ -0,0 +1,52 @@
1
+ from importlib.metadata import PackageNotFoundError, version
2
+
3
+ from .extractor import ExtractionResponse, TranscriptExtractor
4
+ from .utils.extras import missing_extra
5
+ from .utils.transcript_utils import (
6
+ TranscriptResult,
7
+ TranscriptSaver,
8
+ TranscriptSegment,
9
+ TranscriptSource,
10
+ )
11
+
12
+ try:
13
+ __version__ = version("clipscribe")
14
+ except PackageNotFoundError:
15
+ __version__ = "0.1.0"
16
+
17
+ __all__ = [
18
+ "ExtractionResponse",
19
+ "GenericProxyConfig",
20
+ "InstagramTranscriptExtractor",
21
+ "TikTokTranscriptExtractor",
22
+ "TranscriptExtractor",
23
+ "TranscriptResult",
24
+ "TranscriptSaver",
25
+ "TranscriptSegment",
26
+ "TranscriptSource",
27
+ "YouTubeTranscriptExtractor",
28
+ "__version__",
29
+ ]
30
+
31
+
32
+ def __dir__():
33
+ return sorted(set(globals()) | set(__all__))
34
+
35
+
36
+ def __getattr__(name: str):
37
+ if name == "GenericProxyConfig":
38
+ try:
39
+ from youtube_transcript_api.proxies import GenericProxyConfig
40
+ except ImportError as exc:
41
+ raise missing_extra("youtube", "YouTube proxy config") from exc
42
+ return GenericProxyConfig
43
+ if name == "YouTubeTranscriptExtractor":
44
+ from .utils.youtube_utils import YouTubeTranscriptExtractor
45
+ return YouTubeTranscriptExtractor
46
+ if name == "TikTokTranscriptExtractor":
47
+ from .utils.tiktok_utils import TikTokTranscriptExtractor
48
+ return TikTokTranscriptExtractor
49
+ if name == "InstagramTranscriptExtractor":
50
+ from .utils.instagram_utils import InstagramTranscriptExtractor
51
+ return InstagramTranscriptExtractor
52
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
@@ -0,0 +1,295 @@
1
+ """Unified transcript extraction API for YouTube, TikTok, and Instagram."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ import re
7
+ import time
8
+ from typing import Callable
9
+
10
+ from .utils.transcript_utils import PlatformExtractor, TranscriptResult, TranscriptSaver
11
+
12
+ log = logging.getLogger(__name__)
13
+
14
+ _NON_RETRYABLE = (ValueError, FileNotFoundError, ImportError)
15
+
16
+
17
+ class ExtractionResponse:
18
+ """
19
+ Wraps a TranscriptResult alongside optional save paths.
20
+ This is what TranscriptExtractor.extract() always returns.
21
+
22
+ Attributes:
23
+ result: The raw TranscriptResult (segments, metadata, full_text).
24
+ saved: Dict of format -> Path if save=True, else None.
25
+ url: The original URL that was extracted.
26
+ platform: Detected TranscriptSource.
27
+ """
28
+
29
+ def __init__(self, result: TranscriptResult, saved: dict | None = None):
30
+ self.result = result
31
+ self.saved = saved
32
+ self.url = result.url
33
+ self.platform = result.source
34
+
35
+ @property
36
+ def full_text(self) -> str:
37
+ return self.result.full_text
38
+
39
+ @property
40
+ def segments(self):
41
+ return self.result.segments
42
+
43
+ @property
44
+ def metadata(self):
45
+ return self.result.metadata
46
+
47
+ def to_dict(self) -> dict:
48
+ data = self.result.to_dict()
49
+ if self.saved:
50
+ data["saved_files"] = {k: str(v) for k, v in self.saved.items()}
51
+ return data
52
+
53
+ def __repr__(self) -> str:
54
+ return (
55
+ f"<ExtractionResponse platform={self.platform.value!r} "
56
+ f"segments={len(self.segments)} url={self.url!r}>"
57
+ )
58
+
59
+
60
+ class TranscriptExtractor:
61
+ """
62
+ Unified transcript extraction API.
63
+
64
+ Built-in extractors and extra platforms added via ``register()`` are
65
+ per-instance.
66
+
67
+ Args:
68
+ language: Preferred transcript language for YouTube (default: "en").
69
+ whisper_model: faster-whisper model size for TikTok/Instagram (default: "tiny").
70
+ whisper_device: "cpu" or "cuda" (default: "cpu").
71
+ whisper_compute: Quantization type for faster-whisper (default: "int8").
72
+ output_dir: Directory for saved transcripts (default: "output", relative to CWD).
73
+ allowed_outputs: File formats to write when save=True.
74
+ Choose from "json", "txt", "srt" (default: json).
75
+ on_error: Optional callback(url, exc) called on per-URL failures
76
+ in extract_many(). Defaults to logging the error.
77
+ youtube_proxy_config: Optional youtube-transcript-api ProxyConfig.
78
+ tiktok_cookies_from_browser: Browser name for TikTok cookies.
79
+ tiktok_cookies: Netscape cookies.txt path for TikTok.
80
+ instagram_cookies_from_browser: Browser name for Instagram cookies.
81
+ instagram_cookies: Netscape cookies.txt path for Instagram.
82
+ max_duration_s: Reject TikTok/Instagram videos longer than this
83
+ (default: 900). Pass None for no limit.
84
+ """
85
+
86
+ def __init__(
87
+ self,
88
+ language: str = "en",
89
+ whisper_model: str = "tiny",
90
+ whisper_device: str = "cpu",
91
+ whisper_compute: str = "int8",
92
+ output_dir: str = "output",
93
+ allowed_outputs: list[str] | None = None,
94
+ on_error: Callable[[str, Exception], None] | None = None,
95
+ instagram_cookies_from_browser: str | None = None,
96
+ instagram_cookies: str | None = None,
97
+ tiktok_cookies_from_browser: str | None = None,
98
+ tiktok_cookies: str | None = None,
99
+ youtube_proxy_config=None,
100
+ max_duration_s: int | None = 900,
101
+ ):
102
+ self._saver = TranscriptSaver(output_dir=output_dir, allowed_outputs=allowed_outputs)
103
+ self._on_error = on_error or self._default_error_handler
104
+ self._registry: list[tuple[re.Pattern, PlatformExtractor]] = []
105
+ self._extra: list[tuple[re.Pattern, PlatformExtractor]] = []
106
+
107
+ self._register_defaults(
108
+ language=language,
109
+ whisper_model=whisper_model,
110
+ whisper_device=whisper_device,
111
+ whisper_compute=whisper_compute,
112
+ instagram_cookies_from_browser=instagram_cookies_from_browser,
113
+ instagram_cookies=instagram_cookies,
114
+ tiktok_cookies_from_browser=tiktok_cookies_from_browser,
115
+ tiktok_cookies=tiktok_cookies,
116
+ youtube_proxy_config=youtube_proxy_config,
117
+ max_duration_s=max_duration_s,
118
+ )
119
+
120
+ def extract(
121
+ self,
122
+ url: str,
123
+ save: bool = False,
124
+ max_retries: int = 0,
125
+ allowed_outputs: list[str] | None = None,
126
+ ) -> ExtractionResponse:
127
+ """
128
+ Extract transcript from a single URL.
129
+
130
+ Args:
131
+ url: A YouTube, TikTok, or Instagram URL (or any registered platform).
132
+ save: If True, saves files to the configured output_dir.
133
+ max_retries: Extra attempts after a failed extract (default: 0).
134
+ allowed_outputs: Formats for this call only. Choose from
135
+ "json", "txt", "srt". Defaults to the instance list.
136
+
137
+ Returns:
138
+ ExtractionResponse with .full_text, .segments, .metadata, .to_dict()
139
+
140
+ Raises:
141
+ ValueError: If the URL doesn't match any registered platform.
142
+ RuntimeError: If extraction fails.
143
+ ImportError: If the matching platform extra is not installed.
144
+ """
145
+ extractor = self._resolve(url)
146
+ if allowed_outputs is not None and not save:
147
+ log.warning("allowed_outputs is ignored because save=False")
148
+ max_retries = max(0, max_retries)
149
+ result: TranscriptResult | None = None
150
+ for attempt in range(max_retries + 1):
151
+ try:
152
+ result = extractor.extract(url)
153
+ break
154
+ except _NON_RETRYABLE:
155
+ raise
156
+ except Exception:
157
+ if attempt >= max_retries:
158
+ raise
159
+ time.sleep(min(8, 2 ** attempt))
160
+ if result is None:
161
+ raise RuntimeError(f"Extraction failed for {url}")
162
+ saved = (
163
+ self._saver.save(result, self._slug(result), allowed_outputs=allowed_outputs)
164
+ if save
165
+ else None
166
+ )
167
+ return ExtractionResponse(result=result, saved=saved)
168
+
169
+ def extract_many(
170
+ self,
171
+ urls: list[str],
172
+ save: bool = False,
173
+ max_retries: int = 0,
174
+ allowed_outputs: list[str] | None = None,
175
+ ) -> list[ExtractionResponse]:
176
+ """
177
+ Extract transcripts from multiple URLs.
178
+ Failures are caught, passed to on_error, and skipped — processing continues.
179
+ """
180
+ if allowed_outputs is not None and not save:
181
+ log.warning("allowed_outputs is ignored because save=False")
182
+ allowed_outputs = None
183
+ responses = []
184
+ for url in urls:
185
+ try:
186
+ responses.append(
187
+ self.extract(
188
+ url,
189
+ save=save,
190
+ max_retries=max_retries,
191
+ allowed_outputs=allowed_outputs,
192
+ )
193
+ )
194
+ except Exception as exc:
195
+ self._on_error(url, exc)
196
+ return responses
197
+
198
+ def supports(self, url: str) -> bool:
199
+ """Return True if the URL matches a registered platform."""
200
+ try:
201
+ self._resolve(url)
202
+ return True
203
+ except ValueError:
204
+ return False
205
+
206
+ def register(self, url_pattern: str, extractor: PlatformExtractor) -> None:
207
+ """
208
+ Register a new platform extractor on this instance.
209
+
210
+ Args:
211
+ url_pattern: A regex string matched against the full URL.
212
+ extractor: Any object with an .extract(url) -> TranscriptResult method.
213
+ """
214
+ self._extra.append((re.compile(url_pattern, re.IGNORECASE), extractor))
215
+
216
+ def _register_defaults(
217
+ self,
218
+ language: str,
219
+ whisper_model: str,
220
+ whisper_device: str,
221
+ whisper_compute: str,
222
+ instagram_cookies_from_browser: str | None = None,
223
+ instagram_cookies: str | None = None,
224
+ tiktok_cookies_from_browser: str | None = None,
225
+ tiktok_cookies: str | None = None,
226
+ youtube_proxy_config=None,
227
+ max_duration_s: int | None = 900,
228
+ ) -> None:
229
+ from .utils.youtube_utils import YouTubeTranscriptExtractor
230
+ from .utils.tiktok_utils import TikTokTranscriptExtractor
231
+
232
+ self._registry = [
233
+ (
234
+ re.compile(r"youtube\.com|youtu\.be", re.IGNORECASE),
235
+ YouTubeTranscriptExtractor(
236
+ language=language,
237
+ proxy_config=youtube_proxy_config,
238
+ ),
239
+ ),
240
+ (
241
+ re.compile(r"tiktok\.com", re.IGNORECASE),
242
+ TikTokTranscriptExtractor(
243
+ whisper_model=whisper_model,
244
+ device=whisper_device,
245
+ compute_type=whisper_compute,
246
+ cookies_from_browser=tiktok_cookies_from_browser,
247
+ cookies_file=tiktok_cookies,
248
+ max_duration_s=max_duration_s,
249
+ ),
250
+ ),
251
+ ]
252
+
253
+ if instagram_cookies_from_browser or instagram_cookies:
254
+ from .utils.instagram_utils import InstagramTranscriptExtractor
255
+
256
+ self._registry.append((
257
+ re.compile(r"instagram\.com", re.IGNORECASE),
258
+ InstagramTranscriptExtractor(
259
+ cookies_from_browser=instagram_cookies_from_browser,
260
+ cookies_file=instagram_cookies,
261
+ whisper_model=whisper_model,
262
+ device=whisper_device,
263
+ compute_type=whisper_compute,
264
+ max_duration_s=max_duration_s,
265
+ ),
266
+ ))
267
+
268
+ def _resolve(self, url: str) -> PlatformExtractor:
269
+ for pattern, extractor in (*self._registry, *self._extra):
270
+ if pattern.search(url):
271
+ return extractor
272
+ if "instagram.com" in url.lower():
273
+ raise ValueError(
274
+ "Instagram URLs require authentication. "
275
+ "Pass instagram_cookies_from_browser or instagram_cookies."
276
+ )
277
+ raise ValueError(
278
+ f"No extractor registered for URL: {url}\n"
279
+ f"Registered platforms: {self._registered_platforms()}"
280
+ )
281
+
282
+ def _slug(self, result: TranscriptResult) -> str:
283
+ media_id = (
284
+ result.metadata.get("video_id")
285
+ or result.metadata.get("post_id")
286
+ or "unknown"
287
+ )
288
+ return re.sub(r'[\\/*?:"<>|]', "_", f"{result.source.value}_{media_id}").strip()
289
+
290
+ def _registered_platforms(self) -> list[str]:
291
+ return [pattern.pattern for pattern, _ in (*self._registry, *self._extra)]
292
+
293
+ @staticmethod
294
+ def _default_error_handler(url: str, exc: Exception) -> None:
295
+ log.error("Failed (%s): %s", url, exc)
clipscribe/py.typed ADDED
File without changes
File without changes
@@ -0,0 +1,10 @@
1
+ """Install-extra hints for optional platform dependencies."""
2
+
3
+ def extra_install(platform: str) -> str:
4
+ return f'pip install "clipscribe[{platform}]" or "clipscribe[all]"'
5
+
6
+
7
+ def missing_extra(platform: str, what: str) -> ImportError:
8
+ return ImportError(
9
+ f"{what} requires extra dependencies. Install with: {extra_install(platform)}"
10
+ )
@@ -0,0 +1,164 @@
1
+ """
2
+ instagram_utils.py
3
+ ───────────────────
4
+ Instagram transcript extraction via yt-dlp (audio download) + faster-whisper.
5
+
6
+ Instagram requires authentication. Two options, in recommended order:
7
+
8
+ 1. cookies_from_browser (BEST — always fresh, no manual export needed)
9
+ 2. cookies_file (OK — but expires fast)
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import os
15
+ import re
16
+ import tempfile
17
+ from urllib.parse import urlparse
18
+
19
+ from .transcript_utils import (
20
+ BaseTranscriptExtractor,
21
+ TranscriptResult,
22
+ TranscriptSource,
23
+ )
24
+ from .whisper_backend import load_whisper_model, transcribe_to_segments
25
+ from .ydl_audio import (
26
+ SUPPORTED_BROWSERS,
27
+ build_audio_ydl_opts,
28
+ download_audio,
29
+ validate_browser,
30
+ validate_cookies_file,
31
+ )
32
+
33
+ _URL_PATTERNS: list[tuple[re.Pattern[str], str]] = [
34
+ (re.compile(r"instagram\.com/share/reel/([A-Za-z0-9_-]+)", re.IGNORECASE), "reel"),
35
+ (re.compile(r"instagram\.com/share/p/([A-Za-z0-9_-]+)", re.IGNORECASE), "p"),
36
+ (re.compile(r"instagram\.com/reels/([A-Za-z0-9_-]+)", re.IGNORECASE), "reel"),
37
+ (re.compile(r"instagram\.com/reel/([A-Za-z0-9_-]+)", re.IGNORECASE), "reel"),
38
+ (re.compile(r"instagram\.com/p/([A-Za-z0-9_-]+)", re.IGNORECASE), "p"),
39
+ (re.compile(r"instagram\.com/tv/([A-Za-z0-9_-]+)", re.IGNORECASE), "tv"),
40
+ ]
41
+
42
+
43
+ def parse_instagram_post(url: str) -> tuple[str | None, str]:
44
+ for pattern, kind in _URL_PATTERNS:
45
+ m = pattern.search(url)
46
+ if m:
47
+ return m.group(1), kind
48
+ return None, ""
49
+
50
+
51
+ def canonical_instagram_media_url(url: str, post_id: str, kind: str) -> str:
52
+ host = (urlparse(url.strip()).hostname or "").lower()
53
+ if host not in ("www.instagram.com", "instagram.com", "m.instagram.com"):
54
+ raise ValueError(f"Instagram transcript URL host not allowed: {host!r}")
55
+ base = "https://www.instagram.com"
56
+ if kind == "p":
57
+ return f"{base}/p/{post_id}/"
58
+ if kind == "tv":
59
+ return f"{base}/tv/{post_id}/"
60
+ return f"{base}/reel/{post_id}/"
61
+
62
+
63
+ class InstagramTranscriptExtractor(BaseTranscriptExtractor):
64
+ """
65
+ Downloads Instagram audio with yt-dlp and transcribes it with faster-whisper.
66
+
67
+ Requires: pip install "clipscribe[instagram]"
68
+
69
+ Instagram requires authentication — provide cookies_from_browser or cookies_file.
70
+
71
+ Args:
72
+ cookies_from_browser: Browser name to read live cookies from.
73
+ cookies_file: Path to a Netscape-format cookies.txt file.
74
+ whisper_model: Whisper model size (default: "tiny").
75
+ device: "cpu" or "cuda" (default: "cpu").
76
+ compute_type: Quantization — "int8", "float16", "float32".
77
+ ydl_opts_extra: Optional dict merged into yt-dlp options.
78
+ max_duration_s: Reject videos longer than this (default: 900).
79
+ Pass None for no limit.
80
+ socket_timeout: yt-dlp socket timeout in seconds (default: 30).
81
+ """
82
+
83
+ SUPPORTED_BROWSERS = SUPPORTED_BROWSERS
84
+
85
+ def __init__(
86
+ self,
87
+ cookies_from_browser: str | None = None,
88
+ cookies_file: str | None = None,
89
+ whisper_model: str = "tiny",
90
+ device: str = "cpu",
91
+ compute_type: str = "int8",
92
+ ydl_opts_extra: dict | None = None,
93
+ max_duration_s: int | None = 900,
94
+ socket_timeout: int = 30,
95
+ ):
96
+ if not any([cookies_from_browser, cookies_file]):
97
+ raise ValueError(
98
+ "Instagram requires authentication. Provide one of:\n"
99
+ " - cookies_from_browser: 'chrome', 'firefox', 'edge', etc. (recommended)\n"
100
+ " - cookies_file: path to a Netscape-format cookies.txt"
101
+ )
102
+
103
+ validate_browser(cookies_from_browser)
104
+ validate_cookies_file(cookies_file)
105
+
106
+ self.cookies_from_browser = cookies_from_browser
107
+ self.cookies_file = cookies_file
108
+ self.whisper_model = whisper_model
109
+ self.device = device
110
+ self.compute_type = compute_type
111
+ self.ydl_opts_extra = ydl_opts_extra or {}
112
+ self.max_duration_s = max_duration_s
113
+ self.socket_timeout = socket_timeout
114
+ self._model = None
115
+
116
+ def extract(self, url: str) -> TranscriptResult:
117
+ post_id, kind = parse_instagram_post(url)
118
+ if not post_id or not kind:
119
+ raise ValueError(
120
+ f"Unsupported Instagram URL: {url}\n"
121
+ "Supported formats: /reel/<id>/, /p/<id>/, /tv/<id>/"
122
+ )
123
+ fetch_url = canonical_instagram_media_url(url, post_id, kind)
124
+
125
+ with tempfile.TemporaryDirectory() as tmp:
126
+ audio_path, ydl_info = self._download_audio(fetch_url, tmp)
127
+ self.load_model()
128
+ segments, transcription_info = transcribe_to_segments(self._model, audio_path)
129
+
130
+ result = self._make_result(
131
+ fetch_url,
132
+ source=TranscriptSource.INSTAGRAM,
133
+ post_id=post_id,
134
+ title=ydl_info.get("title", ""),
135
+ author=ydl_info.get("uploader", ""),
136
+ language_detected=transcription_info.language,
137
+ language_probability=round(transcription_info.language_probability, 3),
138
+ whisper_model=self.whisper_model,
139
+ )
140
+ result.segments = segments
141
+ result.duration_s = ydl_info.get("duration")
142
+ return result
143
+
144
+ def load_model(self) -> None:
145
+ if self._model is None:
146
+ self._model = load_whisper_model(
147
+ self.whisper_model,
148
+ self.device,
149
+ self.compute_type,
150
+ extra="instagram",
151
+ )
152
+
153
+ def _download_audio(self, url: str, tmp_dir: str) -> tuple[str, dict]:
154
+ template = os.path.join(tmp_dir, "audio.%(ext)s")
155
+ opts = build_audio_ydl_opts(
156
+ template,
157
+ cookies_from_browser=self.cookies_from_browser,
158
+ cookies_file=self.cookies_file,
159
+ ydl_opts_extra=self.ydl_opts_extra,
160
+ socket_timeout=self.socket_timeout,
161
+ max_duration_s=self.max_duration_s,
162
+ extra="instagram",
163
+ )
164
+ return download_audio(url, tmp_dir, opts, platform="instagram")
@@ -0,0 +1,142 @@
1
+ """
2
+ tiktok_utils.py
3
+ ────────────────
4
+ TikTok transcript extraction via yt-dlp (audio download) + faster-whisper.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import os
10
+ import re
11
+ import tempfile
12
+ from urllib.parse import urlparse
13
+
14
+ from .transcript_utils import (
15
+ BaseTranscriptExtractor,
16
+ TranscriptResult,
17
+ TranscriptSource,
18
+ )
19
+ from .whisper_backend import load_whisper_model, transcribe_to_segments
20
+ from .ydl_audio import (
21
+ build_audio_ydl_opts,
22
+ download_audio,
23
+ validate_browser,
24
+ validate_cookies_file,
25
+ )
26
+
27
+ _SHORT_HOSTS = {"vm.tiktok.com", "vt.tiktok.com"}
28
+
29
+
30
+ def assert_allowed_tiktok_url(url: str) -> None:
31
+ u = urlparse((url or "").strip())
32
+ host = (u.hostname or "").lower()
33
+ if not host:
34
+ raise ValueError("TikTok URL is missing a host")
35
+ if not (
36
+ host in ("tiktok.com", "www.tiktok.com", "m.tiktok.com", "vm.tiktok.com", "vt.tiktok.com")
37
+ or host.endswith(".tiktok.com")
38
+ ):
39
+ raise ValueError(f"TikTok transcript URL host not allowed: {host!r}")
40
+
41
+ path = u.path or ""
42
+ if host in _SHORT_HOSTS:
43
+ if path.strip("/"):
44
+ return
45
+ raise ValueError("TikTok short URL is missing a path")
46
+ if re.search(r"/video/\d+", path):
47
+ return
48
+ if re.search(r"/v/\d+", path):
49
+ return
50
+ if re.search(r"/t/[A-Za-z0-9]+", path):
51
+ return
52
+ raise ValueError(
53
+ "TikTok URL must be a single video (e.g. /@user/video/ID or vm.tiktok.com/...), "
54
+ "not a profile, tag, or playlist."
55
+ )
56
+
57
+
58
+ class TikTokTranscriptExtractor(BaseTranscriptExtractor):
59
+ """
60
+ Downloads TikTok audio with yt-dlp and transcribes it with faster-whisper.
61
+
62
+ Requires: pip install "clipscribe[tiktok]"
63
+
64
+ Args:
65
+ whisper_model: Whisper model size — "tiny", "base", "small",
66
+ "medium", or "large" (default: "tiny").
67
+ device: "cpu" or "cuda" (default: "cpu").
68
+ compute_type: Quantization type — e.g. "int8", "float16", "float32".
69
+ cookies_from_browser: Browser name to read cookies from.
70
+ cookies_file: Path to a Netscape-format cookies.txt file.
71
+ ydl_opts_extra: Optional dict merged into yt-dlp options.
72
+ max_duration_s: Reject videos longer than this (default: 900).
73
+ Pass None for no limit.
74
+ socket_timeout: yt-dlp socket timeout in seconds (default: 30).
75
+ """
76
+
77
+ def __init__(
78
+ self,
79
+ whisper_model: str = "tiny",
80
+ device: str = "cpu",
81
+ compute_type: str = "int8",
82
+ cookies_from_browser: str | None = None,
83
+ cookies_file: str | None = None,
84
+ ydl_opts_extra: dict | None = None,
85
+ max_duration_s: int | None = 900,
86
+ socket_timeout: int = 30,
87
+ ):
88
+ validate_browser(cookies_from_browser)
89
+ validate_cookies_file(cookies_file)
90
+
91
+ self.whisper_model = whisper_model
92
+ self.device = device
93
+ self.compute_type = compute_type
94
+ self.cookies_from_browser = cookies_from_browser
95
+ self.cookies_file = cookies_file
96
+ self.ydl_opts_extra = ydl_opts_extra or {}
97
+ self.max_duration_s = max_duration_s
98
+ self.socket_timeout = socket_timeout
99
+ self._model = None
100
+
101
+ def extract(self, url: str) -> TranscriptResult:
102
+ assert_allowed_tiktok_url(url)
103
+ with tempfile.TemporaryDirectory() as tmp:
104
+ audio_path, ydl_info = self._download_audio(url, tmp)
105
+ self.load_model()
106
+ segments, transcription_info = transcribe_to_segments(self._model, audio_path)
107
+
108
+ result = self._make_result(
109
+ url,
110
+ source=TranscriptSource.TIKTOK,
111
+ video_id=ydl_info.get("id", "unknown"),
112
+ title=ydl_info.get("title", ""),
113
+ author=ydl_info.get("uploader", ""),
114
+ language_detected=transcription_info.language,
115
+ language_probability=round(transcription_info.language_probability, 3),
116
+ whisper_model=self.whisper_model,
117
+ )
118
+ result.segments = segments
119
+ result.duration_s = ydl_info.get("duration")
120
+ return result
121
+
122
+ def load_model(self) -> None:
123
+ if self._model is None:
124
+ self._model = load_whisper_model(
125
+ self.whisper_model,
126
+ self.device,
127
+ self.compute_type,
128
+ extra="tiktok",
129
+ )
130
+
131
+ def _download_audio(self, url: str, tmp_dir: str) -> tuple[str, dict]:
132
+ template = os.path.join(tmp_dir, "audio.%(ext)s")
133
+ opts = build_audio_ydl_opts(
134
+ template,
135
+ cookies_from_browser=self.cookies_from_browser,
136
+ cookies_file=self.cookies_file,
137
+ ydl_opts_extra=self.ydl_opts_extra,
138
+ socket_timeout=self.socket_timeout,
139
+ max_duration_s=self.max_duration_s,
140
+ extra="tiktok",
141
+ )
142
+ return download_audio(url, tmp_dir, opts, platform="tiktok")