getyoutubetranscript 0.2.2__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/PKG-INFO +81 -2
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/README.md +79 -0
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/pyproject.toml +5 -1
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/src/getyoutubetranscript/__init__.py +8 -1
- getyoutubetranscript-0.3.0/src/getyoutubetranscript/__main__.py +5 -0
- getyoutubetranscript-0.3.0/src/getyoutubetranscript/cli.py +100 -0
- getyoutubetranscript-0.3.0/src/getyoutubetranscript/formatters.py +110 -0
- getyoutubetranscript-0.3.0/tests/test_cli.py +93 -0
- getyoutubetranscript-0.3.0/tests/test_formatters.py +73 -0
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/.github/workflows/tests.yml +0 -0
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/.gitignore +0 -0
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/LICENSE +0 -0
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/src/getyoutubetranscript/client.py +0 -0
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/src/getyoutubetranscript/exceptions.py +0 -0
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/src/getyoutubetranscript/py.typed +0 -0
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/src/getyoutubetranscript/types.py +0 -0
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/tests/live/__init__.py +0 -0
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/tests/live/test_live_api.py +0 -0
- {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/tests/test_client.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: getyoutubetranscript
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: YouTube transcript API for Python: get YouTube video transcripts, captions and subtitles with timestamps, search YouTube, and list channel and playlist videos. No proxies or headless browser.
|
|
5
5
|
Project-URL: Homepage, https://getyoutubetranscript.com
|
|
6
6
|
Project-URL: Documentation, https://getyoutubetranscript.com/docs
|
|
@@ -11,7 +11,7 @@ Project-URL: Get an API key, https://getyoutubetranscript.com/dashboard
|
|
|
11
11
|
Author: tubeagentkit
|
|
12
12
|
License: MIT
|
|
13
13
|
License-File: LICENSE
|
|
14
|
-
Keywords: ai,api-client,captions,llm,rag,sdk,subtitles,timestamps,transcript,transcripts,youtube,youtube-api,youtube-captions,youtube-search,youtube-subtitles,youtube-transcript,youtube-transcript-api,youtube-transcripts
|
|
14
|
+
Keywords: ai,api-client,captions,cli,llm,rag,sdk,srt,subtitles,timestamps,transcript,transcripts,vtt,webvtt,youtube,youtube-api,youtube-captions,youtube-search,youtube-subtitles,youtube-transcript,youtube-transcript-api,youtube-transcripts
|
|
15
15
|
Classifier: Development Status :: 4 - Beta
|
|
16
16
|
Classifier: Intended Audience :: Developers
|
|
17
17
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -43,6 +43,8 @@ Description-Content-Type: text/markdown
|
|
|
43
43
|
|
|
44
44
|
The official Python SDK (`getyoutubetranscript`) for the [GetYouTubeTranscript](https://getyoutubetranscript.com) YouTube Transcript API. Get YouTube video transcripts, captions and subtitles (optionally with per-line timestamps) in Python without a Google API key, yt-dlp, or a headless browser. Get YouTube transcripts, search videos and channels, resolve channel handles, browse a channel's full upload history, search inside a channel, pull playlist contents, and check your credit balance, all with one typed client.
|
|
45
45
|
|
|
46
|
+
Export transcripts as plain text, timed text, JSON, SRT or WebVTT, from Python or the `getyoutubetranscript` command line. Getting `RequestBlocked` or `IpBlocked` from `youtube-transcript-api` on a cloud server? See [below](#getting-requestblocked-or-ipblocked).
|
|
47
|
+
|
|
46
48
|
[](https://pypi.org/project/getyoutubetranscript/)
|
|
47
49
|
|
|
48
50
|
## Install
|
|
@@ -102,6 +104,24 @@ print(result["segments"][0])
|
|
|
102
104
|
|
|
103
105
|
Each segment is `{"start", "duration", "text"}` with `start` and `duration` in seconds. The `Segment` and `TranscriptData` typed dicts are importable from `getyoutubetranscript`.
|
|
104
106
|
|
|
107
|
+
### Formats: text, timed text, JSON, SRT, WebVTT
|
|
108
|
+
|
|
109
|
+
Turn a transcript into a file format with the formatters. Timed text, SRT and WebVTT need per-line timing, so fetch with `timestamps=True` (same 1 credit).
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
from getyoutubetranscript import Client, to_srt, to_vtt, to_timed_text, to_json, to_text
|
|
113
|
+
|
|
114
|
+
result = client.get_transcript("5e37ZT3SQbk", timestamps=True)
|
|
115
|
+
|
|
116
|
+
open("video.srt", "w", encoding="utf-8").write(to_srt(result)) # SubRip subtitles
|
|
117
|
+
open("video.vtt", "w", encoding="utf-8").write(to_vtt(result)) # WebVTT subtitles
|
|
118
|
+
print(to_timed_text(result)) # "[0:03] So, Reed, education, ..." one line per caption
|
|
119
|
+
print(to_json(result)) # metadata, transcript and segments
|
|
120
|
+
print(to_text(result)) # one block of plain text
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
`format_transcript(result, "srt")` does the same with the format as a string (`"text"`, `"timed"`, `"json"`, `"srt"`, `"vtt"`).
|
|
124
|
+
|
|
105
125
|
### Search
|
|
106
126
|
|
|
107
127
|
```python
|
|
@@ -139,6 +159,22 @@ while page["has_more"]:
|
|
|
139
159
|
client.get_credits() # free - plan_credits_left, topup_credits_left, plan, rate_limit_per_minute
|
|
140
160
|
```
|
|
141
161
|
|
|
162
|
+
## Command line
|
|
163
|
+
|
|
164
|
+
Installing the package also installs a `getyoutubetranscript` command.
|
|
165
|
+
|
|
166
|
+
```bash
|
|
167
|
+
export GETYOUTUBETRANSCRIPT_API_KEY=sk_live_...
|
|
168
|
+
|
|
169
|
+
getyoutubetranscript https://youtu.be/5e37ZT3SQbk # plain text
|
|
170
|
+
getyoutubetranscript 5e37ZT3SQbk --format srt > video.srt # SRT subtitles
|
|
171
|
+
getyoutubetranscript 5e37ZT3SQbk --format timed --language en # [m:ss] lines
|
|
172
|
+
getyoutubetranscript VIDEO_1 VIDEO_2 --format json # one JSON list
|
|
173
|
+
getyoutubetranscript VIDEO_1 VIDEO_2 --format vtt --output-dir subs # subs/<video_id>.vtt
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
Formats: `text` (default), `timed`, `json`, `srt`, `vtt`. Videos can be URLs or IDs. If one video fails, the rest still run and the command exits with status 1. `python -m getyoutubetranscript` works too.
|
|
177
|
+
|
|
142
178
|
## Error handling
|
|
143
179
|
|
|
144
180
|
Every non-2xx or `{"success": false}` response raises `GetYouTubeTranscriptError` with the API's parsed error shape:
|
|
@@ -157,6 +193,49 @@ except GetYouTubeTranscriptError as e:
|
|
|
157
193
|
print(e.response_body) # full parsed error body, e.g. {"creditsLeft": 0} on PAYMENT_REQUIRED
|
|
158
194
|
```
|
|
159
195
|
|
|
196
|
+
## Getting RequestBlocked or IpBlocked?
|
|
197
|
+
|
|
198
|
+
If you use the open source `youtube-transcript-api` library, you have probably seen `RequestBlocked` or `IpBlocked` once your code runs on a server. YouTube blocks most IP addresses that belong to cloud providers (AWS, Google Cloud, Azure and others), and can also block a home IP that makes many requests. That library's own docs recommend rotating residential proxies as the workaround.
|
|
199
|
+
|
|
200
|
+
This SDK calls the GetYouTubeTranscript API instead of YouTube, so YouTube never sees your server's IP. There are no proxies to buy, rotate or debug, and the same code works on your laptop, a VPS, a serverless function or a CI job:
|
|
201
|
+
|
|
202
|
+
```python
|
|
203
|
+
from getyoutubetranscript import Client
|
|
204
|
+
|
|
205
|
+
client = Client(api_key="sk_live_...")
|
|
206
|
+
result = client.get_transcript("https://www.youtube.com/watch?v=jNQXAC9IVRw", timestamps=True)
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
The trade-off: it is a paid API with a free tier (each request uses credits), while `youtube-transcript-api` is free to run if you handle the blocking yourself.
|
|
210
|
+
|
|
211
|
+
## Coming from youtube-transcript-api
|
|
212
|
+
|
|
213
|
+
The segment shape is the same (`text`, `start`, `duration`, in seconds), so most code ports directly.
|
|
214
|
+
|
|
215
|
+
| youtube-transcript-api | getyoutubetranscript |
|
|
216
|
+
| --- | --- |
|
|
217
|
+
| `YouTubeTranscriptApi().fetch(video_id)` | `client.get_transcript(video, timestamps=True)` |
|
|
218
|
+
| `fetched.to_raw_data()` | `result["segments"]` (already a list of dicts) |
|
|
219
|
+
| `fetch(video_id, languages=["de"])` | `get_transcript(video, language="de")` |
|
|
220
|
+
| Video ID only | Video ID or any YouTube URL (watch, youtu.be, Shorts, live) |
|
|
221
|
+
| `SRTFormatter()`, `WebVTTFormatter()`, `TextFormatter()`, `JSONFormatter()` | `to_srt`, `to_vtt`, `to_text`, `to_json` |
|
|
222
|
+
| CLI: `youtube_transcript_api VIDEO_ID --format json` | CLI: `getyoutubetranscript VIDEO --format json` |
|
|
223
|
+
| Proxies for cloud servers | Not needed |
|
|
224
|
+
|
|
225
|
+
```python
|
|
226
|
+
# before
|
|
227
|
+
from youtube_transcript_api import YouTubeTranscriptApi
|
|
228
|
+
segments = YouTubeTranscriptApi().fetch("jNQXAC9IVRw").to_raw_data()
|
|
229
|
+
|
|
230
|
+
# after
|
|
231
|
+
from getyoutubetranscript import Client
|
|
232
|
+
segments = Client(api_key="sk_live_...").get_transcript("jNQXAC9IVRw", timestamps=True)["segments"]
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
Not covered here: a list of preferred fallback languages, listing every available caption track, YouTube's machine translation of captions, and `preserve_formatting`. Request one language at a time with `language=`.
|
|
236
|
+
|
|
237
|
+
The response also includes the video title, channel name, channel URL, thumbnail and word count, which `youtube-transcript-api` does not return.
|
|
238
|
+
|
|
160
239
|
## Development
|
|
161
240
|
|
|
162
241
|
```bash
|
|
@@ -6,6 +6,8 @@
|
|
|
6
6
|
|
|
7
7
|
The official Python SDK (`getyoutubetranscript`) for the [GetYouTubeTranscript](https://getyoutubetranscript.com) YouTube Transcript API. Get YouTube video transcripts, captions and subtitles (optionally with per-line timestamps) in Python without a Google API key, yt-dlp, or a headless browser. Get YouTube transcripts, search videos and channels, resolve channel handles, browse a channel's full upload history, search inside a channel, pull playlist contents, and check your credit balance, all with one typed client.
|
|
8
8
|
|
|
9
|
+
Export transcripts as plain text, timed text, JSON, SRT or WebVTT, from Python or the `getyoutubetranscript` command line. Getting `RequestBlocked` or `IpBlocked` from `youtube-transcript-api` on a cloud server? See [below](#getting-requestblocked-or-ipblocked).
|
|
10
|
+
|
|
9
11
|
[](https://pypi.org/project/getyoutubetranscript/)
|
|
10
12
|
|
|
11
13
|
## Install
|
|
@@ -65,6 +67,24 @@ print(result["segments"][0])
|
|
|
65
67
|
|
|
66
68
|
Each segment is `{"start", "duration", "text"}` with `start` and `duration` in seconds. The `Segment` and `TranscriptData` typed dicts are importable from `getyoutubetranscript`.
|
|
67
69
|
|
|
70
|
+
### Formats: text, timed text, JSON, SRT, WebVTT
|
|
71
|
+
|
|
72
|
+
Turn a transcript into a file format with the formatters. Timed text, SRT and WebVTT need per-line timing, so fetch with `timestamps=True` (same 1 credit).
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
from getyoutubetranscript import Client, to_srt, to_vtt, to_timed_text, to_json, to_text
|
|
76
|
+
|
|
77
|
+
result = client.get_transcript("5e37ZT3SQbk", timestamps=True)
|
|
78
|
+
|
|
79
|
+
open("video.srt", "w", encoding="utf-8").write(to_srt(result)) # SubRip subtitles
|
|
80
|
+
open("video.vtt", "w", encoding="utf-8").write(to_vtt(result)) # WebVTT subtitles
|
|
81
|
+
print(to_timed_text(result)) # "[0:03] So, Reed, education, ..." one line per caption
|
|
82
|
+
print(to_json(result)) # metadata, transcript and segments
|
|
83
|
+
print(to_text(result)) # one block of plain text
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
`format_transcript(result, "srt")` does the same with the format as a string (`"text"`, `"timed"`, `"json"`, `"srt"`, `"vtt"`).
|
|
87
|
+
|
|
68
88
|
### Search
|
|
69
89
|
|
|
70
90
|
```python
|
|
@@ -102,6 +122,22 @@ while page["has_more"]:
|
|
|
102
122
|
client.get_credits() # free - plan_credits_left, topup_credits_left, plan, rate_limit_per_minute
|
|
103
123
|
```
|
|
104
124
|
|
|
125
|
+
## Command line
|
|
126
|
+
|
|
127
|
+
Installing the package also installs a `getyoutubetranscript` command.
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
export GETYOUTUBETRANSCRIPT_API_KEY=sk_live_...
|
|
131
|
+
|
|
132
|
+
getyoutubetranscript https://youtu.be/5e37ZT3SQbk # plain text
|
|
133
|
+
getyoutubetranscript 5e37ZT3SQbk --format srt > video.srt # SRT subtitles
|
|
134
|
+
getyoutubetranscript 5e37ZT3SQbk --format timed --language en # [m:ss] lines
|
|
135
|
+
getyoutubetranscript VIDEO_1 VIDEO_2 --format json # one JSON list
|
|
136
|
+
getyoutubetranscript VIDEO_1 VIDEO_2 --format vtt --output-dir subs # subs/<video_id>.vtt
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
Formats: `text` (default), `timed`, `json`, `srt`, `vtt`. Videos can be URLs or IDs. If one video fails, the rest still run and the command exits with status 1. `python -m getyoutubetranscript` works too.
|
|
140
|
+
|
|
105
141
|
## Error handling
|
|
106
142
|
|
|
107
143
|
Every non-2xx or `{"success": false}` response raises `GetYouTubeTranscriptError` with the API's parsed error shape:
|
|
@@ -120,6 +156,49 @@ except GetYouTubeTranscriptError as e:
|
|
|
120
156
|
print(e.response_body) # full parsed error body, e.g. {"creditsLeft": 0} on PAYMENT_REQUIRED
|
|
121
157
|
```
|
|
122
158
|
|
|
159
|
+
## Getting RequestBlocked or IpBlocked?
|
|
160
|
+
|
|
161
|
+
If you use the open source `youtube-transcript-api` library, you have probably seen `RequestBlocked` or `IpBlocked` once your code runs on a server. YouTube blocks most IP addresses that belong to cloud providers (AWS, Google Cloud, Azure and others), and can also block a home IP that makes many requests. That library's own docs recommend rotating residential proxies as the workaround.
|
|
162
|
+
|
|
163
|
+
This SDK calls the GetYouTubeTranscript API instead of YouTube, so YouTube never sees your server's IP. There are no proxies to buy, rotate or debug, and the same code works on your laptop, a VPS, a serverless function or a CI job:
|
|
164
|
+
|
|
165
|
+
```python
|
|
166
|
+
from getyoutubetranscript import Client
|
|
167
|
+
|
|
168
|
+
client = Client(api_key="sk_live_...")
|
|
169
|
+
result = client.get_transcript("https://www.youtube.com/watch?v=jNQXAC9IVRw", timestamps=True)
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
The trade-off: it is a paid API with a free tier (each request uses credits), while `youtube-transcript-api` is free to run if you handle the blocking yourself.
|
|
173
|
+
|
|
174
|
+
## Coming from youtube-transcript-api
|
|
175
|
+
|
|
176
|
+
The segment shape is the same (`text`, `start`, `duration`, in seconds), so most code ports directly.
|
|
177
|
+
|
|
178
|
+
| youtube-transcript-api | getyoutubetranscript |
|
|
179
|
+
| --- | --- |
|
|
180
|
+
| `YouTubeTranscriptApi().fetch(video_id)` | `client.get_transcript(video, timestamps=True)` |
|
|
181
|
+
| `fetched.to_raw_data()` | `result["segments"]` (already a list of dicts) |
|
|
182
|
+
| `fetch(video_id, languages=["de"])` | `get_transcript(video, language="de")` |
|
|
183
|
+
| Video ID only | Video ID or any YouTube URL (watch, youtu.be, Shorts, live) |
|
|
184
|
+
| `SRTFormatter()`, `WebVTTFormatter()`, `TextFormatter()`, `JSONFormatter()` | `to_srt`, `to_vtt`, `to_text`, `to_json` |
|
|
185
|
+
| CLI: `youtube_transcript_api VIDEO_ID --format json` | CLI: `getyoutubetranscript VIDEO --format json` |
|
|
186
|
+
| Proxies for cloud servers | Not needed |
|
|
187
|
+
|
|
188
|
+
```python
|
|
189
|
+
# before
|
|
190
|
+
from youtube_transcript_api import YouTubeTranscriptApi
|
|
191
|
+
segments = YouTubeTranscriptApi().fetch("jNQXAC9IVRw").to_raw_data()
|
|
192
|
+
|
|
193
|
+
# after
|
|
194
|
+
from getyoutubetranscript import Client
|
|
195
|
+
segments = Client(api_key="sk_live_...").get_transcript("jNQXAC9IVRw", timestamps=True)["segments"]
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
Not covered here: a list of preferred fallback languages, listing every available caption track, YouTube's machine translation of captions, and `preserve_formatting`. Request one language at a time with `language=`.
|
|
199
|
+
|
|
200
|
+
The response also includes the video title, channel name, channel URL, thumbnail and word count, which `youtube-transcript-api` does not return.
|
|
201
|
+
|
|
123
202
|
## Development
|
|
124
203
|
|
|
125
204
|
```bash
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "getyoutubetranscript"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.3.0"
|
|
8
8
|
description = "YouTube transcript API for Python: get YouTube video transcripts, captions and subtitles with timestamps, search YouTube, and list channel and playlist videos. No proxies or headless browser."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.9"
|
|
@@ -13,6 +13,7 @@ authors = [{ name = "tubeagentkit" }]
|
|
|
13
13
|
keywords = [
|
|
14
14
|
"youtube", "youtube-transcript", "youtube-transcripts", "youtube-transcript-api", "transcript", "transcripts",
|
|
15
15
|
"captions", "subtitles", "youtube-captions", "youtube-subtitles", "timestamps", "youtube-api", "youtube-search",
|
|
16
|
+
"srt", "webvtt", "vtt", "cli",
|
|
16
17
|
"llm", "rag", "ai", "api-client", "sdk",
|
|
17
18
|
]
|
|
18
19
|
classifiers = [
|
|
@@ -37,6 +38,9 @@ dependencies = [
|
|
|
37
38
|
"requests>=2.28",
|
|
38
39
|
]
|
|
39
40
|
|
|
41
|
+
[project.scripts]
|
|
42
|
+
getyoutubetranscript = "getyoutubetranscript.cli:main"
|
|
43
|
+
|
|
40
44
|
[project.optional-dependencies]
|
|
41
45
|
dev = [
|
|
42
46
|
"pytest>=7.0",
|
{getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/src/getyoutubetranscript/__init__.py
RENAMED
|
@@ -10,16 +10,23 @@ See https://getyoutubetranscript.com/docs for the full API reference.
|
|
|
10
10
|
|
|
11
11
|
from .client import Client, signup, verify_signup
|
|
12
12
|
from .exceptions import GetYouTubeTranscriptError
|
|
13
|
+
from .formatters import format_transcript, to_json, to_srt, to_text, to_timed_text, to_vtt
|
|
13
14
|
from .types import Segment, TranscriptData
|
|
14
15
|
|
|
15
|
-
__version__ = "0.
|
|
16
|
+
__version__ = "0.3.0"
|
|
16
17
|
|
|
17
18
|
__all__ = [
|
|
18
19
|
"Client",
|
|
19
20
|
"GetYouTubeTranscriptError",
|
|
20
21
|
"Segment",
|
|
21
22
|
"TranscriptData",
|
|
23
|
+
"format_transcript",
|
|
22
24
|
"signup",
|
|
25
|
+
"to_json",
|
|
26
|
+
"to_srt",
|
|
27
|
+
"to_text",
|
|
28
|
+
"to_timed_text",
|
|
29
|
+
"to_vtt",
|
|
23
30
|
"verify_signup",
|
|
24
31
|
"__version__",
|
|
25
32
|
]
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""Command line: ``getyoutubetranscript VIDEO [VIDEO ...] [--format srt]``.
|
|
2
|
+
|
|
3
|
+
Reads the API key from ``--api-key`` or the ``GETYOUTUBETRANSCRIPT_API_KEY``
|
|
4
|
+
environment variable.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import os
|
|
11
|
+
import sys
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import List, Optional
|
|
14
|
+
|
|
15
|
+
from . import __version__
|
|
16
|
+
from .client import Client
|
|
17
|
+
from .exceptions import GetYouTubeTranscriptError
|
|
18
|
+
from .formatters import FORMATS, TIMED_FORMATS, format_transcript, to_json
|
|
19
|
+
|
|
20
|
+
API_KEY_ENV = "GETYOUTUBETRANSCRIPT_API_KEY"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _parser() -> argparse.ArgumentParser:
|
|
24
|
+
parser = argparse.ArgumentParser(
|
|
25
|
+
prog="getyoutubetranscript",
|
|
26
|
+
description="Get YouTube video transcripts as text, timed text, JSON, SRT or WebVTT.",
|
|
27
|
+
)
|
|
28
|
+
parser.add_argument("videos", nargs="+", metavar="VIDEO", help="YouTube video URL or 11-character video ID")
|
|
29
|
+
parser.add_argument("-f", "--format", choices=FORMATS, default="text", help="output format (default: text)")
|
|
30
|
+
parser.add_argument("-l", "--language", help="caption language code, e.g. en or es")
|
|
31
|
+
parser.add_argument(
|
|
32
|
+
"-o",
|
|
33
|
+
"--output-dir",
|
|
34
|
+
type=Path,
|
|
35
|
+
help="write one <video_id>.<format> file per video instead of printing (needed for several videos in a timed, SRT or VTT format)",
|
|
36
|
+
)
|
|
37
|
+
parser.add_argument("--api-key", help=f"API key (default: ${API_KEY_ENV})")
|
|
38
|
+
parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
39
|
+
return parser
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _extension(fmt: str) -> str:
|
|
43
|
+
return "txt" if fmt in ("text", "timed") else fmt
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def main(argv: Optional[List[str]] = None, client: Optional[Client] = None) -> int:
|
|
47
|
+
try:
|
|
48
|
+
return _run(argv, client)
|
|
49
|
+
except BrokenPipeError:
|
|
50
|
+
# Output piped into a command that stopped reading (e.g. `| head`): exit quietly.
|
|
51
|
+
devnull = os.open(os.devnull, os.O_WRONLY)
|
|
52
|
+
os.dup2(devnull, sys.stdout.fileno())
|
|
53
|
+
return 0
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _run(argv: Optional[List[str]], client: Optional[Client]) -> int:
|
|
57
|
+
parser = _parser()
|
|
58
|
+
args = parser.parse_args(argv)
|
|
59
|
+
|
|
60
|
+
if args.format in TIMED_FORMATS and len(args.videos) > 1 and not args.output_dir:
|
|
61
|
+
parser.error(f"--output-dir is required to save {args.format} for more than one video")
|
|
62
|
+
|
|
63
|
+
if client is None:
|
|
64
|
+
api_key = args.api_key or os.environ.get(API_KEY_ENV)
|
|
65
|
+
if not api_key:
|
|
66
|
+
parser.error(f"no API key: pass --api-key or set {API_KEY_ENV} (get one at https://getyoutubetranscript.com)")
|
|
67
|
+
client = Client(api_key=api_key)
|
|
68
|
+
|
|
69
|
+
if args.output_dir:
|
|
70
|
+
args.output_dir.mkdir(parents=True, exist_ok=True)
|
|
71
|
+
|
|
72
|
+
failures = 0
|
|
73
|
+
json_results = []
|
|
74
|
+
for video in args.videos:
|
|
75
|
+
try:
|
|
76
|
+
data = client.get_transcript(
|
|
77
|
+
video, language=args.language, timestamps=args.format in TIMED_FORMATS or args.format == "json"
|
|
78
|
+
)
|
|
79
|
+
except GetYouTubeTranscriptError as e:
|
|
80
|
+
failures += 1
|
|
81
|
+
print(f"{video}: {e.message} ({e.code})", file=sys.stderr)
|
|
82
|
+
continue
|
|
83
|
+
|
|
84
|
+
if args.output_dir:
|
|
85
|
+
path = args.output_dir / f"{data['video_id']}.{_extension(args.format)}"
|
|
86
|
+
path.write_text(format_transcript(data, args.format), encoding="utf-8")
|
|
87
|
+
print(path, file=sys.stderr)
|
|
88
|
+
elif args.format == "json":
|
|
89
|
+
json_results.append(data)
|
|
90
|
+
else:
|
|
91
|
+
print(format_transcript(data, args.format))
|
|
92
|
+
|
|
93
|
+
if json_results:
|
|
94
|
+
print(to_json(json_results[0] if len(args.videos) == 1 else json_results))
|
|
95
|
+
|
|
96
|
+
return 1 if failures else 0
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
if __name__ == "__main__": # pragma: no cover
|
|
100
|
+
sys.exit(main())
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
"""Turn a transcript from :meth:`Client.get_transcript` into text, timed text, JSON, SRT or WebVTT.
|
|
2
|
+
|
|
3
|
+
from getyoutubetranscript import Client
|
|
4
|
+
from getyoutubetranscript.formatters import to_srt
|
|
5
|
+
|
|
6
|
+
result = client.get_transcript("jNQXAC9IVRw", timestamps=True)
|
|
7
|
+
open("video.srt", "w", encoding="utf-8").write(to_srt(result))
|
|
8
|
+
|
|
9
|
+
Timed text, SRT and WebVTT need per-line timing, so fetch with ``timestamps=True``.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
from typing import Any, Callable, Dict, List, Mapping
|
|
16
|
+
|
|
17
|
+
from .types import Segment
|
|
18
|
+
|
|
19
|
+
FORMATS = ("text", "timed", "json", "srt", "vtt")
|
|
20
|
+
TIMED_FORMATS = ("timed", "srt", "vtt")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _segments(data: Mapping[str, Any]) -> List[Segment]:
|
|
24
|
+
segments = data.get("segments")
|
|
25
|
+
if not segments:
|
|
26
|
+
raise ValueError(
|
|
27
|
+
"This format needs per-line timing. Fetch the transcript with timestamps=True."
|
|
28
|
+
)
|
|
29
|
+
return segments
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _clock(seconds: float, decimal_mark: str) -> str:
|
|
33
|
+
millis = max(0, int(round(seconds * 1000)))
|
|
34
|
+
hours, millis = divmod(millis, 3_600_000)
|
|
35
|
+
minutes, millis = divmod(millis, 60_000)
|
|
36
|
+
secs, millis = divmod(millis, 1000)
|
|
37
|
+
return f"{hours:02d}:{minutes:02d}:{secs:02d}{decimal_mark}{millis:03d}"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _cues(data: Mapping[str, Any], decimal_mark: str) -> List[str]:
|
|
41
|
+
cues = []
|
|
42
|
+
for segment in _segments(data):
|
|
43
|
+
start = float(segment["start"])
|
|
44
|
+
end = start + float(segment["duration"])
|
|
45
|
+
timing = f"{_clock(start, decimal_mark)} --> {_clock(end, decimal_mark)}"
|
|
46
|
+
cues.append(f"{timing}\n{segment['text'].strip()}")
|
|
47
|
+
return cues
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def to_text(data: Mapping[str, Any]) -> str:
|
|
51
|
+
"""The transcript as one block of plain text."""
|
|
52
|
+
return data.get("transcript", "")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _player_time(seconds: float) -> str:
|
|
56
|
+
total = max(0, int(seconds))
|
|
57
|
+
hours, rest = divmod(total, 3600)
|
|
58
|
+
minutes, secs = divmod(rest, 60)
|
|
59
|
+
return f"{hours}:{minutes:02d}:{secs:02d}" if hours else f"{minutes}:{secs:02d}"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def to_timed_text(data: Mapping[str, Any]) -> str:
|
|
63
|
+
"""One ``[m:ss] text`` line per caption (``[h:mm:ss]`` past an hour). Needs ``segments``.
|
|
64
|
+
|
|
65
|
+
Easy for people and AI models to read and cite.
|
|
66
|
+
"""
|
|
67
|
+
return "\n".join(f"[{_player_time(float(s['start']))}] {s['text'].strip()}" for s in _segments(data))
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def to_json(data: Mapping[str, Any], **json_kwargs: Any) -> str:
|
|
71
|
+
"""The full result (metadata, transcript and any segments) as a JSON string."""
|
|
72
|
+
json_kwargs.setdefault("ensure_ascii", False)
|
|
73
|
+
json_kwargs.setdefault("indent", 2)
|
|
74
|
+
return json.dumps(data, **json_kwargs)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def to_srt(data: Mapping[str, Any]) -> str:
|
|
78
|
+
"""SubRip (.srt) subtitles. Needs ``segments`` (``timestamps=True``)."""
|
|
79
|
+
cues = _cues(data, ",")
|
|
80
|
+
return "\n\n".join(f"{index}\n{cue}" for index, cue in enumerate(cues, start=1)) + "\n"
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def to_vtt(data: Mapping[str, Any]) -> str:
|
|
84
|
+
"""WebVTT (.vtt) subtitles. Needs ``segments`` (``timestamps=True``)."""
|
|
85
|
+
cues = [_escape_vtt_text(cue) for cue in _cues(data, ".")]
|
|
86
|
+
return "WEBVTT\n\n" + "\n\n".join(cues) + "\n"
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _escape_vtt_text(cue: str) -> str:
|
|
90
|
+
# "-->" inside cue text would be read as a timing line, so escape it.
|
|
91
|
+
timing, _, text = cue.partition("\n")
|
|
92
|
+
return f"{timing}\n{text.replace('-->', '-->')}"
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
_FORMATTERS: Dict[str, Callable[[Mapping[str, Any]], str]] = {
|
|
96
|
+
"text": to_text,
|
|
97
|
+
"timed": to_timed_text,
|
|
98
|
+
"json": to_json,
|
|
99
|
+
"srt": to_srt,
|
|
100
|
+
"vtt": to_vtt,
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def format_transcript(data: Mapping[str, Any], fmt: str) -> str:
|
|
105
|
+
"""Format a transcript as ``"text"``, ``"timed"``, ``"json"``, ``"srt"`` or ``"vtt"``."""
|
|
106
|
+
try:
|
|
107
|
+
formatter = _FORMATTERS[fmt]
|
|
108
|
+
except KeyError:
|
|
109
|
+
raise ValueError(f"Unknown format {fmt!r}. Choose one of: {', '.join(FORMATS)}.") from None
|
|
110
|
+
return formatter(data)
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Unit tests for the command line (client is faked, no network)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
from getyoutubetranscript import GetYouTubeTranscriptError
|
|
10
|
+
from getyoutubetranscript.cli import API_KEY_ENV, main
|
|
11
|
+
|
|
12
|
+
SEGMENTS = [{"start": 0.0, "duration": 1.0, "text": "hi"}]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class FakeClient:
|
|
16
|
+
def __init__(self, fail=()):
|
|
17
|
+
self.calls = []
|
|
18
|
+
self.fail = set(fail)
|
|
19
|
+
|
|
20
|
+
def get_transcript(self, video, *, language=None, timestamps=False):
|
|
21
|
+
self.calls.append({"video": video, "language": language, "timestamps": timestamps})
|
|
22
|
+
if video in self.fail:
|
|
23
|
+
raise GetYouTubeTranscriptError("NOT_FOUND", "No transcript for this video.", 404)
|
|
24
|
+
data = {"video_id": video, "language_code": language or "en", "transcript": f"text of {video}"}
|
|
25
|
+
if timestamps:
|
|
26
|
+
data["segments"] = SEGMENTS
|
|
27
|
+
return data
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_text_to_stdout_without_timestamps(capsys):
|
|
31
|
+
client = FakeClient()
|
|
32
|
+
assert main(["vid1"], client=client) == 0
|
|
33
|
+
assert capsys.readouterr().out == "text of vid1\n"
|
|
34
|
+
assert client.calls == [{"video": "vid1", "language": None, "timestamps": False}]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_srt_requests_timestamps_and_passes_language(capsys):
|
|
38
|
+
client = FakeClient()
|
|
39
|
+
assert main(["vid1", "-f", "srt", "-l", "es"], client=client) == 0
|
|
40
|
+
assert capsys.readouterr().out == "1\n00:00:00,000 --> 00:00:01,000\nhi\n\n"
|
|
41
|
+
assert client.calls[0] == {"video": "vid1", "language": "es", "timestamps": True}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_json_for_several_videos_is_one_list(capsys):
|
|
45
|
+
assert main(["a", "b", "-f", "json"], client=FakeClient()) == 0
|
|
46
|
+
out = json.loads(capsys.readouterr().out)
|
|
47
|
+
assert [item["video_id"] for item in out] == ["a", "b"]
|
|
48
|
+
assert out[0]["segments"] == SEGMENTS
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def test_output_dir_writes_one_file_per_video(tmp_path):
|
|
52
|
+
assert main(["a", "b", "-f", "vtt", "-o", str(tmp_path)], client=FakeClient()) == 0
|
|
53
|
+
assert sorted(p.name for p in tmp_path.iterdir()) == ["a.vtt", "b.vtt"]
|
|
54
|
+
assert (tmp_path / "a.vtt").read_text(encoding="utf-8").startswith("WEBVTT\n\n00:00:00.000")
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def test_timed_files_use_txt_extension(tmp_path):
|
|
58
|
+
assert main(["a", "-f", "timed", "-o", str(tmp_path)], client=FakeClient()) == 0
|
|
59
|
+
assert (tmp_path / "a.txt").read_text(encoding="utf-8") == "[0:00] hi"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def test_several_srt_videos_need_output_dir(capsys):
|
|
63
|
+
with pytest.raises(SystemExit) as exit_info:
|
|
64
|
+
main(["a", "b", "-f", "srt"], client=FakeClient())
|
|
65
|
+
assert exit_info.value.code == 2
|
|
66
|
+
assert "--output-dir" in capsys.readouterr().err
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_keeps_going_after_a_failure_and_exits_1(capsys):
|
|
70
|
+
assert main(["bad", "good"], client=FakeClient(fail={"bad"})) == 1
|
|
71
|
+
captured = capsys.readouterr()
|
|
72
|
+
assert captured.out == "text of good\n"
|
|
73
|
+
assert "bad: No transcript for this video. (NOT_FOUND)" in captured.err
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def test_missing_api_key_is_a_usage_error(monkeypatch, capsys):
|
|
77
|
+
monkeypatch.delenv(API_KEY_ENV, raising=False)
|
|
78
|
+
with pytest.raises(SystemExit) as exit_info:
|
|
79
|
+
main(["vid1"])
|
|
80
|
+
assert exit_info.value.code == 2
|
|
81
|
+
assert API_KEY_ENV in capsys.readouterr().err
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def test_pipe_closed_early_exits_quietly(monkeypatch, tmp_path):
|
|
85
|
+
import getyoutubetranscript.cli as cli
|
|
86
|
+
|
|
87
|
+
def broken_pipe(*args, **kwargs):
|
|
88
|
+
raise BrokenPipeError
|
|
89
|
+
|
|
90
|
+
monkeypatch.setattr(cli, "_run", broken_pipe)
|
|
91
|
+
out = (tmp_path / "out").open("w")
|
|
92
|
+
monkeypatch.setattr("sys.stdout", out)
|
|
93
|
+
assert cli.main(["vid1"]) == 0
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Unit tests for the transcript formatters (no network)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
from getyoutubetranscript import format_transcript, to_json, to_srt, to_text, to_timed_text, to_vtt
|
|
10
|
+
|
|
11
|
+
DATA = {
|
|
12
|
+
"video_id": "abc12345678",
|
|
13
|
+
"language_code": "en",
|
|
14
|
+
"title": "A video",
|
|
15
|
+
"author_name": "A channel",
|
|
16
|
+
"transcript": "hello world. café --> next",
|
|
17
|
+
"word_count": 5,
|
|
18
|
+
"segments": [
|
|
19
|
+
{"start": 0.0, "duration": 2.5, "text": "hello world."},
|
|
20
|
+
{"start": 1.9996, "duration": 1.0, "text": "café --> next"},
|
|
21
|
+
{"start": 3725.25, "duration": 4.0, "text": " an hour in "},
|
|
22
|
+
],
|
|
23
|
+
}
|
|
24
|
+
NO_SEGMENTS = {k: v for k, v in DATA.items() if k != "segments"}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def test_text_is_the_plain_transcript():
|
|
28
|
+
assert to_text(DATA) == "hello world. café --> next"
|
|
29
|
+
assert to_text(NO_SEGMENTS) == "hello world. café --> next"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def test_timed_text_uses_player_clock():
|
|
33
|
+
assert to_timed_text(DATA).splitlines() == [
|
|
34
|
+
"[0:00] hello world.",
|
|
35
|
+
"[0:01] café --> next",
|
|
36
|
+
"[1:02:05] an hour in",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def test_srt_cues_numbering_and_rounding():
|
|
41
|
+
assert to_srt(DATA) == (
|
|
42
|
+
"1\n00:00:00,000 --> 00:00:02,500\nhello world.\n\n"
|
|
43
|
+
# 1.9996s rounds to 2.000s, never "00:00:01,1000"
|
|
44
|
+
"2\n00:00:02,000 --> 00:00:03,000\ncafé --> next\n\n"
|
|
45
|
+
"3\n01:02:05,250 --> 01:02:09,250\nan hour in\n"
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def test_vtt_header_dot_millis_and_escaped_arrow():
|
|
50
|
+
assert to_vtt(DATA) == (
|
|
51
|
+
"WEBVTT\n\n"
|
|
52
|
+
"00:00:00.000 --> 00:00:02.500\nhello world.\n\n"
|
|
53
|
+
"00:00:02.000 --> 00:00:03.000\ncafé --> next\n\n"
|
|
54
|
+
"01:02:05.250 --> 01:02:09.250\nan hour in\n"
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def test_json_round_trips_and_keeps_unicode():
|
|
59
|
+
out = to_json(DATA)
|
|
60
|
+
assert json.loads(out) == DATA
|
|
61
|
+
assert "café" in out
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@pytest.mark.parametrize("fn", [to_srt, to_vtt, to_timed_text])
|
|
65
|
+
def test_timed_formats_explain_missing_segments(fn):
|
|
66
|
+
with pytest.raises(ValueError, match="timestamps=True"):
|
|
67
|
+
fn(NO_SEGMENTS)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def test_format_transcript_dispatches_and_rejects_unknown():
|
|
71
|
+
assert format_transcript(DATA, "srt") == to_srt(DATA)
|
|
72
|
+
with pytest.raises(ValueError, match="Unknown format"):
|
|
73
|
+
format_transcript(DATA, "docx")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/src/getyoutubetranscript/client.py
RENAMED
|
File without changes
|
{getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.0}/src/getyoutubetranscript/exceptions.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|