getyoutubetranscript 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- getyoutubetranscript/__init__.py +25 -0
- getyoutubetranscript/client.py +450 -0
- getyoutubetranscript/exceptions.py +50 -0
- getyoutubetranscript/py.typed +0 -0
- getyoutubetranscript/types.py +33 -0
- getyoutubetranscript-0.2.0.dist-info/METADATA +186 -0
- getyoutubetranscript-0.2.0.dist-info/RECORD +9 -0
- getyoutubetranscript-0.2.0.dist-info/WHEEL +4 -0
- getyoutubetranscript-0.2.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Python SDK for the GetYouTubeTranscript REST API.
|
|
2
|
+
|
|
3
|
+
from getyoutubetranscript import Client
|
|
4
|
+
|
|
5
|
+
client = Client(api_key="sk_live_...")
|
|
6
|
+
transcript = client.get_transcript("https://www.youtube.com/watch?v=jNQXAC9IVRw")
|
|
7
|
+
|
|
8
|
+
See https://getyoutubetranscript.com/docs for the full API reference.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from .client import Client, signup, verify_signup
|
|
12
|
+
from .exceptions import GetYouTubeTranscriptError
|
|
13
|
+
from .types import Segment, TranscriptData
|
|
14
|
+
|
|
15
|
+
__version__ = "0.2.0"
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"Client",
|
|
19
|
+
"GetYouTubeTranscriptError",
|
|
20
|
+
"Segment",
|
|
21
|
+
"TranscriptData",
|
|
22
|
+
"signup",
|
|
23
|
+
"verify_signup",
|
|
24
|
+
"__version__",
|
|
25
|
+
]
|
|
@@ -0,0 +1,450 @@
|
|
|
1
|
+
"""HTTP client for the GetYouTubeTranscript REST API.
|
|
2
|
+
|
|
3
|
+
API reference: https://getyoutubetranscript.com/docs
|
|
4
|
+
OpenAPI spec: https://getyoutubetranscript.com/openapi.json
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from typing import Any, Optional
|
|
10
|
+
|
|
11
|
+
import requests
|
|
12
|
+
|
|
13
|
+
from .exceptions import GetYouTubeTranscriptError
|
|
14
|
+
from .types import TranscriptData
|
|
15
|
+
|
|
16
|
+
DEFAULT_BASE_URL = "https://getyoutubetranscript.com/api/v1"
|
|
17
|
+
DEFAULT_TIMEOUT = 30.0
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _clean(params: dict[str, Any]) -> dict[str, Any]:
|
|
21
|
+
"""Drop ``None`` values so optional query params are omitted entirely."""
|
|
22
|
+
return {k: v for k, v in params.items() if v is not None}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _send(
|
|
26
|
+
session: requests.Session,
|
|
27
|
+
method: str,
|
|
28
|
+
url: str,
|
|
29
|
+
*,
|
|
30
|
+
headers: Optional[dict[str, str]] = None,
|
|
31
|
+
params: Optional[dict[str, Any]] = None,
|
|
32
|
+
json_body: Optional[dict[str, Any]] = None,
|
|
33
|
+
timeout: float,
|
|
34
|
+
) -> dict[str, Any]:
|
|
35
|
+
"""Send one HTTP request and return the parsed JSON body.
|
|
36
|
+
|
|
37
|
+
Shared by :class:`Client` (authenticated endpoints) and the module-level
|
|
38
|
+
``signup``/``verify_signup`` helpers (no API key needed), so request
|
|
39
|
+
sending and error parsing live in exactly one place.
|
|
40
|
+
|
|
41
|
+
Raises:
|
|
42
|
+
GetYouTubeTranscriptError: on any network failure, non-2xx status, or
|
|
43
|
+
a 2xx response whose body is ``{"success": false, ...}``.
|
|
44
|
+
"""
|
|
45
|
+
try:
|
|
46
|
+
response = session.request(
|
|
47
|
+
method,
|
|
48
|
+
url,
|
|
49
|
+
headers=headers,
|
|
50
|
+
params=params,
|
|
51
|
+
json=json_body,
|
|
52
|
+
timeout=timeout,
|
|
53
|
+
)
|
|
54
|
+
except requests.RequestException as exc:
|
|
55
|
+
raise GetYouTubeTranscriptError(
|
|
56
|
+
code="NETWORK_ERROR",
|
|
57
|
+
message=str(exc),
|
|
58
|
+
status_code=0,
|
|
59
|
+
) from exc
|
|
60
|
+
|
|
61
|
+
try:
|
|
62
|
+
payload = response.json()
|
|
63
|
+
except ValueError:
|
|
64
|
+
payload = None
|
|
65
|
+
|
|
66
|
+
if not response.ok or not isinstance(payload, dict) or payload.get("success") is False:
|
|
67
|
+
code = "UNKNOWN_ERROR"
|
|
68
|
+
message = f"Request failed with HTTP status {response.status_code}"
|
|
69
|
+
if isinstance(payload, dict):
|
|
70
|
+
code = payload.get("code", code)
|
|
71
|
+
message = payload.get("message", message)
|
|
72
|
+
raise GetYouTubeTranscriptError(
|
|
73
|
+
code=code,
|
|
74
|
+
message=message,
|
|
75
|
+
status_code=response.status_code,
|
|
76
|
+
response_body=payload if isinstance(payload, dict) else None,
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
return payload
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class Client:
|
|
83
|
+
"""Authenticated client for the GetYouTubeTranscript API.
|
|
84
|
+
|
|
85
|
+
Args:
|
|
86
|
+
api_key: Your API key (``sk_live_...``). Get one free (100 credits,
|
|
87
|
+
no card) at https://getyoutubetranscript.com, or via the
|
|
88
|
+
self-serve :func:`signup` / :func:`verify_signup` flow in this
|
|
89
|
+
module, which needs no key at all.
|
|
90
|
+
base_url: Override the API base URL. Defaults to the production
|
|
91
|
+
endpoint; mainly useful for testing against a local/staging copy.
|
|
92
|
+
timeout: Per-request timeout in seconds.
|
|
93
|
+
session: Bring your own ``requests.Session`` (e.g. for connection
|
|
94
|
+
pooling or custom retry/adapter configuration). One is created
|
|
95
|
+
for you otherwise.
|
|
96
|
+
|
|
97
|
+
Every method costs 1 credit unless its docstring says "free" - failed
|
|
98
|
+
and rate-limited requests are never charged. Every method raises
|
|
99
|
+
:class:`~getyoutubetranscript.GetYouTubeTranscriptError` on failure.
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
def __init__(
|
|
103
|
+
self,
|
|
104
|
+
api_key: str,
|
|
105
|
+
*,
|
|
106
|
+
base_url: str = DEFAULT_BASE_URL,
|
|
107
|
+
timeout: float = DEFAULT_TIMEOUT,
|
|
108
|
+
session: Optional[requests.Session] = None,
|
|
109
|
+
) -> None:
|
|
110
|
+
if not api_key:
|
|
111
|
+
raise ValueError("api_key is required")
|
|
112
|
+
self.api_key = api_key
|
|
113
|
+
self.base_url = base_url.rstrip("/")
|
|
114
|
+
self.timeout = timeout
|
|
115
|
+
self._session = session or requests.Session()
|
|
116
|
+
|
|
117
|
+
def _headers(self) -> dict[str, str]:
|
|
118
|
+
return {
|
|
119
|
+
"Authorization": f"Bearer {self.api_key}",
|
|
120
|
+
"Accept": "application/json",
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
def _get(self, path: str, params: dict[str, Any]) -> dict[str, Any]:
|
|
124
|
+
return _send(
|
|
125
|
+
self._session,
|
|
126
|
+
"GET",
|
|
127
|
+
f"{self.base_url}{path}",
|
|
128
|
+
headers=self._headers(),
|
|
129
|
+
params=_clean(params),
|
|
130
|
+
timeout=self.timeout,
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
# -- transcript -----------------------------------------------------
|
|
134
|
+
|
|
135
|
+
def get_transcript(
|
|
136
|
+
self,
|
|
137
|
+
video: str,
|
|
138
|
+
*,
|
|
139
|
+
language: Optional[str] = None,
|
|
140
|
+
timestamps: bool = False,
|
|
141
|
+
) -> TranscriptData:
|
|
142
|
+
"""Get a YouTube video's transcript, plus title/author/thumbnail. 1 credit.
|
|
143
|
+
|
|
144
|
+
Args:
|
|
145
|
+
video: Full or short YouTube video URL, or an 11-character video ID.
|
|
146
|
+
language: Caption language code (e.g. ``"en"``, ``"es"``). Defaults
|
|
147
|
+
to the API's default of ``"en"`` when omitted.
|
|
148
|
+
timestamps: When ``True``, also return per-line timing in
|
|
149
|
+
``segments``. Same 1 credit. The query param is only sent when
|
|
150
|
+
this is ``True``.
|
|
151
|
+
|
|
152
|
+
Returns:
|
|
153
|
+
dict with keys ``video_id``, ``language_code``, ``title``,
|
|
154
|
+
``author_name``, ``author_url``, ``thumbnail_url``,
|
|
155
|
+
``transcript`` (one block of text), and ``word_count``. With
|
|
156
|
+
``timestamps=True`` it also has ``segments``, a list of
|
|
157
|
+
``{"start": float, "duration": float, "text": str}`` (seconds),
|
|
158
|
+
one per caption line.
|
|
159
|
+
|
|
160
|
+
Raises:
|
|
161
|
+
GetYouTubeTranscriptError: e.g. ``code="NOT_FOUND"`` (HTTP 404) if
|
|
162
|
+
the video has no transcript/captions available.
|
|
163
|
+
"""
|
|
164
|
+
payload = self._get(
|
|
165
|
+
"/transcript",
|
|
166
|
+
{"v": video, "language": language, "timestamps": "true" if timestamps else None},
|
|
167
|
+
)
|
|
168
|
+
return payload["data"]
|
|
169
|
+
|
|
170
|
+
# -- search -----------------------------------------------------------
|
|
171
|
+
|
|
172
|
+
def search(
|
|
173
|
+
self,
|
|
174
|
+
query: Optional[str] = None,
|
|
175
|
+
*,
|
|
176
|
+
page_token: Optional[str] = None,
|
|
177
|
+
type: Optional[str] = None,
|
|
178
|
+
country: Optional[str] = None,
|
|
179
|
+
language: Optional[str] = None,
|
|
180
|
+
limit: Optional[int] = None,
|
|
181
|
+
) -> dict[str, Any]:
|
|
182
|
+
"""Search YouTube for videos or channels. 1 credit.
|
|
183
|
+
|
|
184
|
+
Args:
|
|
185
|
+
query: Search query. Required for a first page unless
|
|
186
|
+
``page_token`` is given.
|
|
187
|
+
page_token: Continuation token from a previous response's
|
|
188
|
+
``pagination.next_page_token``, to fetch the next page. Treat
|
|
189
|
+
as opaque - don't construct it yourself.
|
|
190
|
+
type: Restrict results to ``"video"`` (default) or ``"channel"``.
|
|
191
|
+
Never mixes both kinds in one response.
|
|
192
|
+
country: Two-letter region code, e.g. ``"us"``.
|
|
193
|
+
language: Result language hint, e.g. ``"en"``.
|
|
194
|
+
limit: Max results to return for this page.
|
|
195
|
+
|
|
196
|
+
Returns:
|
|
197
|
+
dict with ``query``, and either ``video_results`` or
|
|
198
|
+
``channel_results`` depending on ``type``, plus
|
|
199
|
+
``pagination.next_page_token`` when more results are available.
|
|
200
|
+
|
|
201
|
+
Raises:
|
|
202
|
+
ValueError: if neither ``query`` nor ``page_token`` is given.
|
|
203
|
+
GetYouTubeTranscriptError: on API failure.
|
|
204
|
+
"""
|
|
205
|
+
if not query and not page_token:
|
|
206
|
+
raise ValueError("search() requires either 'query' or 'page_token'")
|
|
207
|
+
payload = self._get(
|
|
208
|
+
"/search",
|
|
209
|
+
{
|
|
210
|
+
"q": query,
|
|
211
|
+
"page_token": page_token,
|
|
212
|
+
"type": type,
|
|
213
|
+
"country": country,
|
|
214
|
+
"language": language,
|
|
215
|
+
"limit": limit,
|
|
216
|
+
},
|
|
217
|
+
)
|
|
218
|
+
return payload["data"]
|
|
219
|
+
|
|
220
|
+
# -- channels -----------------------------------------------------------
|
|
221
|
+
|
|
222
|
+
def resolve_channel(self, handle: str) -> dict[str, Any]:
|
|
223
|
+
"""Resolve a channel @handle, URL, or ``UC...`` id to its channel ID. Free.
|
|
224
|
+
|
|
225
|
+
Args:
|
|
226
|
+
handle: Channel ``@handle``, a channel URL, or an existing
|
|
227
|
+
``UC...`` channel id.
|
|
228
|
+
|
|
229
|
+
Returns:
|
|
230
|
+
dict with ``channel_id``, ``title``, ``handle``, ``resolved_via``.
|
|
231
|
+
|
|
232
|
+
Raises:
|
|
233
|
+
GetYouTubeTranscriptError: e.g. ``code="NOT_FOUND"`` (HTTP 404) if
|
|
234
|
+
no channel matches.
|
|
235
|
+
"""
|
|
236
|
+
payload = self._get("/resolve", {"handle": handle})
|
|
237
|
+
return payload["data"]
|
|
238
|
+
|
|
239
|
+
def get_channel_latest(self, channel: str) -> dict[str, Any]:
|
|
240
|
+
"""Get a channel's metadata plus its home-tab "Latest Videos" shelf. Free.
|
|
241
|
+
|
|
242
|
+
For the complete, paginated upload history use
|
|
243
|
+
:meth:`list_channel_videos` instead.
|
|
244
|
+
|
|
245
|
+
Args:
|
|
246
|
+
channel: Channel ``@handle``, URL, or ``UC...`` id.
|
|
247
|
+
|
|
248
|
+
Returns:
|
|
249
|
+
dict of channel metadata plus a home-tab video shelf. The exact
|
|
250
|
+
field set is passed through from upstream and may grow over
|
|
251
|
+
time - treat it as loosely typed.
|
|
252
|
+
|
|
253
|
+
Raises:
|
|
254
|
+
GetYouTubeTranscriptError: e.g. ``code="NOT_FOUND"`` (HTTP 404) if
|
|
255
|
+
the channel doesn't exist.
|
|
256
|
+
"""
|
|
257
|
+
payload = self._get("/channel/latest", {"channel": channel})
|
|
258
|
+
return payload["data"]
|
|
259
|
+
|
|
260
|
+
def search_channel(
|
|
261
|
+
self,
|
|
262
|
+
channel: Optional[str] = None,
|
|
263
|
+
query: Optional[str] = None,
|
|
264
|
+
*,
|
|
265
|
+
continuation: Optional[str] = None,
|
|
266
|
+
) -> dict[str, Any]:
|
|
267
|
+
"""Search within one channel's videos. 1 credit.
|
|
268
|
+
|
|
269
|
+
Args:
|
|
270
|
+
channel: Channel ``@handle``, URL, or ``UC...`` id. Required for a
|
|
271
|
+
first page unless ``continuation`` is given.
|
|
272
|
+
query: Query to search within the channel. Required for a first
|
|
273
|
+
page unless ``continuation`` is given.
|
|
274
|
+
continuation: Continuation token from a previous response's
|
|
275
|
+
``continuation_token``, to fetch the next page.
|
|
276
|
+
|
|
277
|
+
Returns:
|
|
278
|
+
dict with ``videos`` (list), ``has_more`` (bool), and
|
|
279
|
+
``continuation_token`` (present when ``has_more`` is true).
|
|
280
|
+
|
|
281
|
+
Raises:
|
|
282
|
+
ValueError: if ``continuation`` is not given and either
|
|
283
|
+
``channel`` or ``query`` is missing.
|
|
284
|
+
GetYouTubeTranscriptError: on API failure.
|
|
285
|
+
"""
|
|
286
|
+
if not continuation and (not channel or not query):
|
|
287
|
+
raise ValueError(
|
|
288
|
+
"search_channel() requires either 'continuation', or both "
|
|
289
|
+
"'channel' and 'query'"
|
|
290
|
+
)
|
|
291
|
+
payload = self._get(
|
|
292
|
+
"/channel/search",
|
|
293
|
+
{"channel": channel, "q": query, "continuation": continuation},
|
|
294
|
+
)
|
|
295
|
+
return payload["data"]
|
|
296
|
+
|
|
297
|
+
def list_channel_videos(
|
|
298
|
+
self, channel: Optional[str] = None, *, continuation: Optional[str] = None
|
|
299
|
+
) -> dict[str, Any]:
|
|
300
|
+
"""List every video a channel has ever uploaded (paginated). 1 credit.
|
|
301
|
+
|
|
302
|
+
This is the channel's full ``/videos`` tab - not just the home-tab
|
|
303
|
+
shelf :meth:`get_channel_latest` returns.
|
|
304
|
+
|
|
305
|
+
Args:
|
|
306
|
+
channel: Channel ``@handle``, URL, or ``UC...`` id. Required for a
|
|
307
|
+
first page unless ``continuation`` is given.
|
|
308
|
+
continuation: Continuation token from a previous response's
|
|
309
|
+
``continuation_token``, to fetch the next page.
|
|
310
|
+
|
|
311
|
+
Returns:
|
|
312
|
+
dict with ``videos`` (list), ``has_more`` (bool), and
|
|
313
|
+
``continuation_token`` (present when ``has_more`` is true).
|
|
314
|
+
|
|
315
|
+
Raises:
|
|
316
|
+
ValueError: if neither ``channel`` nor ``continuation`` is given.
|
|
317
|
+
GetYouTubeTranscriptError: e.g. ``code="NOT_FOUND"`` (HTTP 404) if
|
|
318
|
+
the channel doesn't exist.
|
|
319
|
+
"""
|
|
320
|
+
if not channel and not continuation:
|
|
321
|
+
raise ValueError(
|
|
322
|
+
"list_channel_videos() requires either 'channel' or 'continuation'"
|
|
323
|
+
)
|
|
324
|
+
payload = self._get(
|
|
325
|
+
"/channel/videos", {"channel": channel, "continuation": continuation}
|
|
326
|
+
)
|
|
327
|
+
return payload["data"]
|
|
328
|
+
|
|
329
|
+
# -- playlists -----------------------------------------------------------
|
|
330
|
+
|
|
331
|
+
def get_playlist(
|
|
332
|
+
self, list_id: Optional[str] = None, *, continuation: Optional[str] = None
|
|
333
|
+
) -> dict[str, Any]:
|
|
334
|
+
"""List every video in a playlist (paginated). 1 credit.
|
|
335
|
+
|
|
336
|
+
Args:
|
|
337
|
+
list_id: Playlist ID or URL. Required for a first page unless
|
|
338
|
+
``continuation`` is given.
|
|
339
|
+
continuation: Continuation token from a previous response's
|
|
340
|
+
``continuation_token``, to fetch the next page.
|
|
341
|
+
|
|
342
|
+
Returns:
|
|
343
|
+
dict with ``playlist_id``, ``title``, ``videos`` (list),
|
|
344
|
+
``has_more`` (bool), and ``continuation_token`` (present when
|
|
345
|
+
``has_more`` is true).
|
|
346
|
+
|
|
347
|
+
Raises:
|
|
348
|
+
ValueError: if neither ``list_id`` nor ``continuation`` is given.
|
|
349
|
+
GetYouTubeTranscriptError: e.g. ``code="NOT_FOUND"`` (HTTP 404) if
|
|
350
|
+
the playlist is missing, private, or deleted.
|
|
351
|
+
"""
|
|
352
|
+
if not list_id and not continuation:
|
|
353
|
+
raise ValueError("get_playlist() requires either 'list_id' or 'continuation'")
|
|
354
|
+
payload = self._get("/playlist", {"list": list_id, "continuation": continuation})
|
|
355
|
+
return payload["data"]
|
|
356
|
+
|
|
357
|
+
# -- account -----------------------------------------------------------
|
|
358
|
+
|
|
359
|
+
def get_credits(self) -> dict[str, Any]:
|
|
360
|
+
"""Check the remaining credit balance and plan for this API key. Free.
|
|
361
|
+
|
|
362
|
+
Returns:
|
|
363
|
+
dict with ``plan_credits_left``, ``topup_credits_left``, ``plan``
|
|
364
|
+
(one of ``"free"``, ``"monthly"``, ``"yearly"``), and
|
|
365
|
+
``rate_limit_per_minute``.
|
|
366
|
+
|
|
367
|
+
Raises:
|
|
368
|
+
GetYouTubeTranscriptError: e.g. ``code="MISSING_API_KEY"``
|
|
369
|
+
(HTTP 401) if no API key is provided.
|
|
370
|
+
"""
|
|
371
|
+
payload = self._get("/credits", {})
|
|
372
|
+
return payload["data"]
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
# -- self-serve signup, no API key required --------------------------------
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def signup(
|
|
379
|
+
email: str,
|
|
380
|
+
*,
|
|
381
|
+
base_url: str = DEFAULT_BASE_URL,
|
|
382
|
+
timeout: float = DEFAULT_TIMEOUT,
|
|
383
|
+
session: Optional[requests.Session] = None,
|
|
384
|
+
) -> str:
|
|
385
|
+
"""Request a 6-digit email OTP to create or access an API key. Free, no key needed.
|
|
386
|
+
|
|
387
|
+
Step 1 of the self-serve signup flow. Follow with :func:`verify_signup`
|
|
388
|
+
once the caller has the code from their inbox. The code is valid for 10
|
|
389
|
+
minutes.
|
|
390
|
+
|
|
391
|
+
Args:
|
|
392
|
+
email: Address to send the one-time code to.
|
|
393
|
+
base_url: Override the API base URL.
|
|
394
|
+
timeout: Request timeout in seconds.
|
|
395
|
+
session: Optional ``requests.Session`` to reuse.
|
|
396
|
+
|
|
397
|
+
Returns:
|
|
398
|
+
The confirmation message string from the API.
|
|
399
|
+
|
|
400
|
+
Raises:
|
|
401
|
+
GetYouTubeTranscriptError: e.g. ``code="BAD_REQUEST"`` (HTTP 400) if
|
|
402
|
+
the email is missing or malformed.
|
|
403
|
+
"""
|
|
404
|
+
payload = _send(
|
|
405
|
+
session or requests.Session(),
|
|
406
|
+
"POST",
|
|
407
|
+
f"{base_url.rstrip('/')}/signup",
|
|
408
|
+
json_body={"email": email},
|
|
409
|
+
timeout=timeout,
|
|
410
|
+
)
|
|
411
|
+
return payload.get("message", "")
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def verify_signup(
|
|
415
|
+
email: str,
|
|
416
|
+
otp: str,
|
|
417
|
+
*,
|
|
418
|
+
base_url: str = DEFAULT_BASE_URL,
|
|
419
|
+
timeout: float = DEFAULT_TIMEOUT,
|
|
420
|
+
session: Optional[requests.Session] = None,
|
|
421
|
+
) -> str:
|
|
422
|
+
"""Verify the OTP from :func:`signup` and mint a fresh API key. Free, no key needed.
|
|
423
|
+
|
|
424
|
+
Step 2 of the self-serve signup flow. Creates the account if it doesn't
|
|
425
|
+
exist yet, or signs in an existing one, either way returning a freshly
|
|
426
|
+
minted API key.
|
|
427
|
+
|
|
428
|
+
Args:
|
|
429
|
+
email: The same address passed to :func:`signup`.
|
|
430
|
+
otp: The 6-digit code the caller received by email.
|
|
431
|
+
base_url: Override the API base URL.
|
|
432
|
+
timeout: Request timeout in seconds.
|
|
433
|
+
session: Optional ``requests.Session`` to reuse.
|
|
434
|
+
|
|
435
|
+
Returns:
|
|
436
|
+
The raw API key string (``sk_live_...``). It is shown once here and
|
|
437
|
+
cannot be retrieved again - the caller is responsible for storing it.
|
|
438
|
+
|
|
439
|
+
Raises:
|
|
440
|
+
GetYouTubeTranscriptError: e.g. ``code="BAD_REQUEST"`` (HTTP 400) if
|
|
441
|
+
fields are missing or the code is invalid/expired.
|
|
442
|
+
"""
|
|
443
|
+
payload = _send(
|
|
444
|
+
session or requests.Session(),
|
|
445
|
+
"POST",
|
|
446
|
+
f"{base_url.rstrip('/')}/signup/verify",
|
|
447
|
+
json_body={"email": email, "otp": otp},
|
|
448
|
+
timeout=timeout,
|
|
449
|
+
)
|
|
450
|
+
return payload["api_key"]
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Exceptions raised by the GetYouTubeTranscript SDK."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, Optional
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class GetYouTubeTranscriptError(Exception):
|
|
9
|
+
"""Raised whenever the API returns a non-2xx status or ``{"success": false}``.
|
|
10
|
+
|
|
11
|
+
The API always responds to errors with a JSON body shaped like
|
|
12
|
+
``{"success": false, "code": "...", "message": "..."}``. This exception
|
|
13
|
+
parses that shape so callers can branch on ``code`` instead of guessing
|
|
14
|
+
from an HTTP status or a generic exception message.
|
|
15
|
+
|
|
16
|
+
Common codes: ``MISSING_API_KEY``, ``INVALID_API_KEY``, ``RATE_LIMITED``,
|
|
17
|
+
``PAYMENT_REQUIRED``, ``NOT_FOUND``. ``NETWORK_ERROR`` is used locally by
|
|
18
|
+
this SDK (status_code 0) when the request never reached the server at all
|
|
19
|
+
(DNS failure, timeout, connection refused, etc).
|
|
20
|
+
|
|
21
|
+
Attributes:
|
|
22
|
+
code: Machine-readable error code from the API's ``code`` field.
|
|
23
|
+
message: Human-readable message from the API's ``message`` field.
|
|
24
|
+
status_code: HTTP status code (400/401/402/404/429/503), or 0 if the
|
|
25
|
+
request never reached the server.
|
|
26
|
+
response_body: The full parsed JSON error body, if any, for callers
|
|
27
|
+
that need fields beyond ``code``/``message`` (e.g. ``RATE_LIMITED``
|
|
28
|
+
includes ``requestsThisMinute``; ``PAYMENT_REQUIRED`` includes
|
|
29
|
+
``creditsLeft`` and ``topupCreditsLeft``).
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
def __init__(
|
|
33
|
+
self,
|
|
34
|
+
code: str,
|
|
35
|
+
message: str,
|
|
36
|
+
status_code: int,
|
|
37
|
+
*,
|
|
38
|
+
response_body: Optional[dict[str, Any]] = None,
|
|
39
|
+
) -> None:
|
|
40
|
+
self.code = code
|
|
41
|
+
self.message = message
|
|
42
|
+
self.status_code = status_code
|
|
43
|
+
self.response_body = response_body or {}
|
|
44
|
+
super().__init__(f"[{code}] {message} (HTTP {status_code})")
|
|
45
|
+
|
|
46
|
+
def __repr__(self) -> str: # pragma: no cover - cosmetic
|
|
47
|
+
return (
|
|
48
|
+
f"GetYouTubeTranscriptError(code={self.code!r}, "
|
|
49
|
+
f"message={self.message!r}, status_code={self.status_code!r})"
|
|
50
|
+
)
|
|
File without changes
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""Response types for the GetYouTubeTranscript API.
|
|
2
|
+
|
|
3
|
+
These are ``TypedDict`` hints only: the client still returns plain dicts.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from typing import List, TypedDict
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class Segment(TypedDict):
|
|
12
|
+
"""One caption line, returned in ``segments`` when ``timestamps=True``."""
|
|
13
|
+
|
|
14
|
+
start: float # start time in seconds
|
|
15
|
+
duration: float # duration in seconds
|
|
16
|
+
text: str
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class _TranscriptRequired(TypedDict):
|
|
20
|
+
video_id: str
|
|
21
|
+
language_code: str
|
|
22
|
+
title: str
|
|
23
|
+
author_name: str
|
|
24
|
+
author_url: str
|
|
25
|
+
thumbnail_url: str
|
|
26
|
+
transcript: str
|
|
27
|
+
word_count: int
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class TranscriptData(_TranscriptRequired, total=False):
|
|
31
|
+
"""Result of :meth:`Client.get_transcript`."""
|
|
32
|
+
|
|
33
|
+
segments: List[Segment] # present only when timestamps=True
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: getyoutubetranscript
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Python SDK for the GetYouTubeTranscript REST API - transcripts, search, channels, and playlists
|
|
5
|
+
Project-URL: Homepage, https://getyoutubetranscript.com
|
|
6
|
+
Project-URL: Documentation, https://getyoutubetranscript.com/docs
|
|
7
|
+
Project-URL: Repository, https://github.com/tubeagentkit/youtube-transcript-api-python
|
|
8
|
+
Project-URL: Bug Tracker, https://github.com/tubeagentkit/youtube-transcript-api-python/issues
|
|
9
|
+
Author: tubeagentkit
|
|
10
|
+
License: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: api-client,sdk,transcript,youtube,youtube-api
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
24
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
25
|
+
Classifier: Typing :: Typed
|
|
26
|
+
Requires-Python: >=3.9
|
|
27
|
+
Requires-Dist: requests>=2.28
|
|
28
|
+
Provides-Extra: dev
|
|
29
|
+
Requires-Dist: pytest>=7.0; extra == 'dev'
|
|
30
|
+
Requires-Dist: responses>=0.23; extra == 'dev'
|
|
31
|
+
Description-Content-Type: text/markdown
|
|
32
|
+
|
|
33
|
+
# YouTube Transcript API: Python SDK
|
|
34
|
+
|
|
35
|
+
[](./LICENSE)
|
|
36
|
+
[](https://getyoutubetranscript.com)
|
|
37
|
+
[](https://www.python.org)
|
|
38
|
+
|
|
39
|
+
The official Python SDK (`getyoutubetranscript`) for the [GetYouTubeTranscript](https://getyoutubetranscript.com) YouTube Transcript API. Get YouTube video transcripts in Python without a Google API key, yt-dlp, or a headless browser. Get YouTube transcripts, search videos and channels, resolve channel handles, browse a channel's full upload history, search inside a channel, pull playlist contents, and check your credit balance, all with one typed client.
|
|
40
|
+
|
|
41
|
+
Not published to PyPI yet - install straight from this repo.
|
|
42
|
+
|
|
43
|
+
## Install
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
pip install git+https://github.com/tubeagentkit/youtube-transcript-api-python.git
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Requires Python 3.9+.
|
|
50
|
+
|
|
51
|
+
## Quickstart
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
from getyoutubetranscript import Client
|
|
55
|
+
|
|
56
|
+
client = Client(api_key="sk_live_...")
|
|
57
|
+
|
|
58
|
+
transcript = client.get_transcript("https://www.youtube.com/watch?v=jNQXAC9IVRw")
|
|
59
|
+
print(transcript["title"], transcript["word_count"])
|
|
60
|
+
print(transcript["transcript"])
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
## Getting an API key
|
|
64
|
+
|
|
65
|
+
Every request needs an API key. There are two ways to get one:
|
|
66
|
+
|
|
67
|
+
1. **Dashboard** - sign up at [getyoutubetranscript.com](https://getyoutubetranscript.com). Free tier: 100 credits, no card required.
|
|
68
|
+
2. **Self-serve, in code** - use the `signup`/`verify_signup` helpers below. No key required for either call.
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
from getyoutubetranscript import signup, verify_signup
|
|
72
|
+
|
|
73
|
+
signup("you@example.com") # sends a 6-digit code, valid 10 minutes
|
|
74
|
+
# ... read the code from your inbox ...
|
|
75
|
+
api_key = verify_signup("you@example.com", "123456") # -> "sk_live_..."
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
The raw key is returned once by `verify_signup` and can't be retrieved again - store it yourself (env var, secret manager, etc).
|
|
79
|
+
|
|
80
|
+
## Usage
|
|
81
|
+
|
|
82
|
+
Every method costs 1 credit unless noted "free" below. Failed and rate-limited requests are never charged. All methods raise `GetYouTubeTranscriptError` on failure - see [Error handling](#error-handling).
|
|
83
|
+
|
|
84
|
+
### Transcripts
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
client.get_transcript("jNQXAC9IVRw", language="en")
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Pass `timestamps=True` to also get one entry per caption line in `segments` (same 1 credit). Without it, the response has no `segments` key.
|
|
91
|
+
|
|
92
|
+
```python
|
|
93
|
+
result = client.get_transcript("5e37ZT3SQbk", timestamps=True)
|
|
94
|
+
print(result["segments"][0])
|
|
95
|
+
# {"start": 3.96, "duration": 4.56, "text": "So, Reed, education, which a lot of"}
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
Each segment is `{"start", "duration", "text"}` with `start` and `duration` in seconds. The `Segment` and `TranscriptData` typed dicts are importable from `getyoutubetranscript`.
|
|
99
|
+
|
|
100
|
+
### Search
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
client.search("lofi beats", type="video", limit=10)
|
|
104
|
+
|
|
105
|
+
# Pagination
|
|
106
|
+
page2 = client.search(page_token=first_page["pagination"]["next_page_token"])
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
### Channels
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
client.resolve_channel("@mkbhd") # free - handle/URL -> channel ID
|
|
113
|
+
client.get_channel_latest("@mkbhd") # free - metadata + latest uploads
|
|
114
|
+
client.search_channel("@mkbhd", "iphone") # search within a channel
|
|
115
|
+
client.list_channel_videos("@mkbhd") # full paginated upload history
|
|
116
|
+
|
|
117
|
+
# Pagination (search_channel and list_channel_videos both work the same way)
|
|
118
|
+
page = client.list_channel_videos("@mkbhd")
|
|
119
|
+
while page["has_more"]:
|
|
120
|
+
page = client.list_channel_videos(continuation=page["continuation_token"])
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
### Playlists
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
page = client.get_playlist("PLillGF-RfqbYE6Ik_EuXA2iZFcE082B3s")
|
|
127
|
+
while page["has_more"]:
|
|
128
|
+
page = client.get_playlist(continuation=page["continuation_token"])
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
### Account
|
|
132
|
+
|
|
133
|
+
```python
|
|
134
|
+
client.get_credits() # free - plan_credits_left, topup_credits_left, plan, rate_limit_per_minute
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
## Error handling
|
|
138
|
+
|
|
139
|
+
Every non-2xx or `{"success": false}` response raises `GetYouTubeTranscriptError` with the API's parsed error shape:
|
|
140
|
+
|
|
141
|
+
```python
|
|
142
|
+
from getyoutubetranscript import Client, GetYouTubeTranscriptError
|
|
143
|
+
|
|
144
|
+
client = Client(api_key="sk_live_...")
|
|
145
|
+
|
|
146
|
+
try:
|
|
147
|
+
client.get_transcript("no-captions-here")
|
|
148
|
+
except GetYouTubeTranscriptError as e:
|
|
149
|
+
print(e.code) # e.g. "NOT_FOUND"
|
|
150
|
+
print(e.message) # human-readable message from the API
|
|
151
|
+
print(e.status_code) # 400 / 401 / 402 / 404 / 429 / 503, or 0 for a local network failure
|
|
152
|
+
print(e.response_body) # full parsed error body, e.g. {"creditsLeft": 0} on PAYMENT_REQUIRED
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
## Development
|
|
156
|
+
|
|
157
|
+
```bash
|
|
158
|
+
pip install -e ".[dev]"
|
|
159
|
+
|
|
160
|
+
# Unit tests - mocked HTTP, no network or API key needed, always safe to run
|
|
161
|
+
pytest tests -v --ignore=tests/live
|
|
162
|
+
|
|
163
|
+
# Live integration tests - hits the real API, spends credits, needs a key
|
|
164
|
+
GYT_API_KEY=sk_live_... pytest tests/live -v
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
## Links
|
|
168
|
+
|
|
169
|
+
- [Full API docs](https://getyoutubetranscript.com/docs)
|
|
170
|
+
- [OpenAPI spec](https://getyoutubetranscript.com/openapi.json)
|
|
171
|
+
- [MCP server](https://getyoutubetranscript.com/youtube-mcp-server) - if you want an AI agent to call this API directly instead of via Python
|
|
172
|
+
|
|
173
|
+
## Related projects
|
|
174
|
+
|
|
175
|
+
Other ways to use the [GetYouTubeTranscript API](https://getyoutubetranscript.com):
|
|
176
|
+
|
|
177
|
+
- [youtube-transcript-api](https://github.com/tubeagentkit/youtube-transcript-api): YouTube Transcript API docs, endpoint reference, OpenAPI spec and examples in curl, Python, JavaScript, Go and PHP
|
|
178
|
+
- [youtube-transcript-api-node](https://github.com/tubeagentkit/youtube-transcript-api-node): YouTube Transcript API SDK for Node.js / TypeScript
|
|
179
|
+
- [youtube-mcp](https://github.com/tubeagentkit/youtube-mcp): Remote YouTube MCP server for Claude, ChatGPT, Cursor and VS Code
|
|
180
|
+
- [youtube-transcript-skills](https://github.com/tubeagentkit/youtube-transcript-skills): YouTube transcript Agent Skill for Claude Code, Cursor, Codex and OpenClaw
|
|
181
|
+
- [youtube-transcript-cursor-plugin](https://github.com/tubeagentkit/youtube-transcript-cursor-plugin): YouTube Transcript Cursor plugin bundling the MCP server, skills, commands and a research agent
|
|
182
|
+
- [n8n-nodes-getyoutubetranscript](https://github.com/tubeagentkit/n8n-nodes-getyoutubetranscript): YouTube transcript n8n community node, also usable as an AI Agent tool
|
|
183
|
+
|
|
184
|
+
## License
|
|
185
|
+
|
|
186
|
+
MIT - see [LICENSE](./LICENSE).
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
getyoutubetranscript/__init__.py,sha256=d95jT5XHIg5NF9IPxZj1Qc115ywZ_n9INYuaX0dr39Q,623
|
|
2
|
+
getyoutubetranscript/client.py,sha256=VKmIZthsqsbsclbgHYQLz2g8apeaOMgGjnZQHO0wV2c,16262
|
|
3
|
+
getyoutubetranscript/exceptions.py,sha256=1EClkyv0btejIzlyxRbsZ61T3spP57Ev-gFndfkkK14,2023
|
|
4
|
+
getyoutubetranscript/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
5
|
+
getyoutubetranscript/types.py,sha256=hUEx8X5gIqmNTdFNlhbigoGBhNjGvPGpTDW7-t2PxJc,785
|
|
6
|
+
getyoutubetranscript-0.2.0.dist-info/METADATA,sha256=DfVl-5LdEUZw6FYXwDcRRlDpF-pzs4Fn9pGgiK3kUj0,7709
|
|
7
|
+
getyoutubetranscript-0.2.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
8
|
+
getyoutubetranscript-0.2.0.dist-info/licenses/LICENSE,sha256=9BV44fzXqxv9egaqFNr-TvzhlWpNBTVX8TqxF6bIWek,1069
|
|
9
|
+
getyoutubetranscript-0.2.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 tubeagentkit
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|