getyoutubetranscript 0.2.2__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (19) hide show
  1. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/PKG-INFO +85 -5
  2. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/README.md +83 -3
  3. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/pyproject.toml +5 -1
  4. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/src/getyoutubetranscript/__init__.py +8 -1
  5. getyoutubetranscript-0.3.1/src/getyoutubetranscript/__main__.py +5 -0
  6. getyoutubetranscript-0.3.1/src/getyoutubetranscript/cli.py +100 -0
  7. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/src/getyoutubetranscript/client.py +5 -4
  8. getyoutubetranscript-0.3.1/src/getyoutubetranscript/formatters.py +110 -0
  9. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/tests/live/test_live_api.py +12 -0
  10. getyoutubetranscript-0.3.1/tests/test_cli.py +93 -0
  11. getyoutubetranscript-0.3.1/tests/test_formatters.py +73 -0
  12. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/.github/workflows/tests.yml +0 -0
  13. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/.gitignore +0 -0
  14. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/LICENSE +0 -0
  15. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/src/getyoutubetranscript/exceptions.py +0 -0
  16. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/src/getyoutubetranscript/py.typed +0 -0
  17. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/src/getyoutubetranscript/types.py +0 -0
  18. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/tests/live/__init__.py +0 -0
  19. {getyoutubetranscript-0.2.2 → getyoutubetranscript-0.3.1}/tests/test_client.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: getyoutubetranscript
3
- Version: 0.2.2
3
+ Version: 0.3.1
4
4
  Summary: YouTube transcript API for Python: get YouTube video transcripts, captions and subtitles with timestamps, search YouTube, and list channel and playlist videos. No proxies or headless browser.
5
5
  Project-URL: Homepage, https://getyoutubetranscript.com
6
6
  Project-URL: Documentation, https://getyoutubetranscript.com/docs
@@ -11,7 +11,7 @@ Project-URL: Get an API key, https://getyoutubetranscript.com/dashboard
11
11
  Author: tubeagentkit
12
12
  License: MIT
13
13
  License-File: LICENSE
14
- Keywords: ai,api-client,captions,llm,rag,sdk,subtitles,timestamps,transcript,transcripts,youtube,youtube-api,youtube-captions,youtube-search,youtube-subtitles,youtube-transcript,youtube-transcript-api,youtube-transcripts
14
+ Keywords: ai,api-client,captions,cli,llm,rag,sdk,srt,subtitles,timestamps,transcript,transcripts,vtt,webvtt,youtube,youtube-api,youtube-captions,youtube-search,youtube-subtitles,youtube-transcript,youtube-transcript-api,youtube-transcripts
15
15
  Classifier: Development Status :: 4 - Beta
16
16
  Classifier: Intended Audience :: Developers
17
17
  Classifier: License :: OSI Approved :: MIT License
@@ -43,6 +43,8 @@ Description-Content-Type: text/markdown
43
43
 
44
44
  The official Python SDK (`getyoutubetranscript`) for the [GetYouTubeTranscript](https://getyoutubetranscript.com) YouTube Transcript API. Get YouTube video transcripts, captions and subtitles (optionally with per-line timestamps) in Python without a Google API key, yt-dlp, or a headless browser. Get YouTube transcripts, search videos and channels, resolve channel handles, browse a channel's full upload history, search inside a channel, pull playlist contents, and check your credit balance, all with one typed client.
45
45
 
46
+ Export transcripts as plain text, timed text, JSON, SRT or WebVTT, from Python or the `getyoutubetranscript` command line. Getting `RequestBlocked` or `IpBlocked` from `youtube-transcript-api` on a cloud server? See [below](#getting-requestblocked-or-ipblocked).
47
+
46
48
  [![PyPI](https://img.shields.io/pypi/v/getyoutubetranscript)](https://pypi.org/project/getyoutubetranscript/)
47
49
 
48
50
  ## Install
@@ -102,13 +104,32 @@ print(result["segments"][0])
102
104
 
103
105
  Each segment is `{"start", "duration", "text"}` with `start` and `duration` in seconds. The `Segment` and `TranscriptData` typed dicts are importable from `getyoutubetranscript`.
104
106
 
107
+ ### Formats: text, timed text, JSON, SRT, WebVTT
108
+
109
+ Turn a transcript into a file format with the formatters. Timed text, SRT and WebVTT need per-line timing, so fetch with `timestamps=True` (same 1 credit).
110
+
111
+ ```python
112
+ from getyoutubetranscript import Client, to_srt, to_vtt, to_timed_text, to_json, to_text
113
+
114
+ result = client.get_transcript("5e37ZT3SQbk", timestamps=True)
115
+
116
+ open("video.srt", "w", encoding="utf-8").write(to_srt(result)) # SubRip subtitles
117
+ open("video.vtt", "w", encoding="utf-8").write(to_vtt(result)) # WebVTT subtitles
118
+ print(to_timed_text(result)) # "[0:03] So, Reed, education, ..." one line per caption
119
+ print(to_json(result)) # metadata, transcript and segments
120
+ print(to_text(result)) # one block of plain text
121
+ ```
122
+
123
+ `format_transcript(result, "srt")` does the same with the format as a string (`"text"`, `"timed"`, `"json"`, `"srt"`, `"vtt"`).
124
+
105
125
  ### Search
106
126
 
107
127
  ```python
108
- client.search("lofi beats", type="video", limit=10)
128
+ first_page = client.search("lofi beats", type="video", limit=10)
109
129
 
110
- # Pagination
111
- page2 = client.search(page_token=first_page["pagination"]["next_page_token"])
130
+ # Pagination: pass continuation_token back as page_token
131
+ if first_page.get("continuation_token"):
132
+ page2 = client.search(page_token=first_page["continuation_token"])
112
133
  ```
113
134
 
114
135
  ### Channels
@@ -139,6 +160,22 @@ while page["has_more"]:
139
160
  client.get_credits() # free - plan_credits_left, topup_credits_left, plan, rate_limit_per_minute
140
161
  ```
141
162
 
163
+ ## Command line
164
+
165
+ Installing the package also installs a `getyoutubetranscript` command.
166
+
167
+ ```bash
168
+ export GETYOUTUBETRANSCRIPT_API_KEY=sk_live_...
169
+
170
+ getyoutubetranscript https://youtu.be/5e37ZT3SQbk # plain text
171
+ getyoutubetranscript 5e37ZT3SQbk --format srt > video.srt # SRT subtitles
172
+ getyoutubetranscript 5e37ZT3SQbk --format timed --language en # [m:ss] lines
173
+ getyoutubetranscript VIDEO_1 VIDEO_2 --format json # one JSON list
174
+ getyoutubetranscript VIDEO_1 VIDEO_2 --format vtt --output-dir subs # subs/<video_id>.vtt
175
+ ```
176
+
177
+ Formats: `text` (default), `timed`, `json`, `srt`, `vtt`. Videos can be URLs or IDs. If one video fails, the rest still run and the command exits with status 1. `python -m getyoutubetranscript` works too.
178
+
142
179
  ## Error handling
143
180
 
144
181
  Every non-2xx or `{"success": false}` response raises `GetYouTubeTranscriptError` with the API's parsed error shape:
@@ -157,6 +194,49 @@ except GetYouTubeTranscriptError as e:
157
194
  print(e.response_body) # full parsed error body, e.g. {"creditsLeft": 0} on PAYMENT_REQUIRED
158
195
  ```
159
196
 
197
+ ## Getting RequestBlocked or IpBlocked?
198
+
199
+ If you use the open source `youtube-transcript-api` library, you have probably seen `RequestBlocked` or `IpBlocked` once your code runs on a server. YouTube blocks most IP addresses that belong to cloud providers (AWS, Google Cloud, Azure and others), and can also block a home IP that makes many requests. That library's own docs recommend rotating residential proxies as the workaround.
200
+
201
+ This SDK calls the GetYouTubeTranscript API instead of YouTube, so YouTube never sees your server's IP. There are no proxies to buy, rotate or debug, and the same code works on your laptop, a VPS, a serverless function or a CI job:
202
+
203
+ ```python
204
+ from getyoutubetranscript import Client
205
+
206
+ client = Client(api_key="sk_live_...")
207
+ result = client.get_transcript("https://www.youtube.com/watch?v=jNQXAC9IVRw", timestamps=True)
208
+ ```
209
+
210
+ The trade-off: it is a paid API with a free tier (each request uses credits), while `youtube-transcript-api` is free to run if you handle the blocking yourself.
211
+
212
+ ## Coming from youtube-transcript-api
213
+
214
+ The segment shape is the same (`text`, `start`, `duration`, in seconds), so most code ports directly.
215
+
216
+ | youtube-transcript-api | getyoutubetranscript |
217
+ | --- | --- |
218
+ | `YouTubeTranscriptApi().fetch(video_id)` | `client.get_transcript(video, timestamps=True)` |
219
+ | `fetched.to_raw_data()` | `result["segments"]` (already a list of dicts) |
220
+ | `fetch(video_id, languages=["de"])` | `get_transcript(video, language="de")` |
221
+ | Video ID only | Video ID or any YouTube URL (watch, youtu.be, Shorts, live) |
222
+ | `SRTFormatter()`, `WebVTTFormatter()`, `TextFormatter()`, `JSONFormatter()` | `to_srt`, `to_vtt`, `to_text`, `to_json` |
223
+ | CLI: `youtube_transcript_api VIDEO_ID --format json` | CLI: `getyoutubetranscript VIDEO --format json` |
224
+ | Proxies for cloud servers | Not needed |
225
+
226
+ ```python
227
+ # before
228
+ from youtube_transcript_api import YouTubeTranscriptApi
229
+ segments = YouTubeTranscriptApi().fetch("jNQXAC9IVRw").to_raw_data()
230
+
231
+ # after
232
+ from getyoutubetranscript import Client
233
+ segments = Client(api_key="sk_live_...").get_transcript("jNQXAC9IVRw", timestamps=True)["segments"]
234
+ ```
235
+
236
+ Not covered here: a list of preferred fallback languages, listing every available caption track, YouTube's machine translation of captions, and `preserve_formatting`. Request one language at a time with `language=`.
237
+
238
+ The response also includes the video title, channel name, channel URL, thumbnail and word count, which `youtube-transcript-api` does not return.
239
+
160
240
  ## Development
161
241
 
162
242
  ```bash
@@ -6,6 +6,8 @@
6
6
 
7
7
  The official Python SDK (`getyoutubetranscript`) for the [GetYouTubeTranscript](https://getyoutubetranscript.com) YouTube Transcript API. Get YouTube video transcripts, captions and subtitles (optionally with per-line timestamps) in Python without a Google API key, yt-dlp, or a headless browser. Get YouTube transcripts, search videos and channels, resolve channel handles, browse a channel's full upload history, search inside a channel, pull playlist contents, and check your credit balance, all with one typed client.
8
8
 
9
+ Export transcripts as plain text, timed text, JSON, SRT or WebVTT, from Python or the `getyoutubetranscript` command line. Getting `RequestBlocked` or `IpBlocked` from `youtube-transcript-api` on a cloud server? See [below](#getting-requestblocked-or-ipblocked).
10
+
9
11
  [![PyPI](https://img.shields.io/pypi/v/getyoutubetranscript)](https://pypi.org/project/getyoutubetranscript/)
10
12
 
11
13
  ## Install
@@ -65,13 +67,32 @@ print(result["segments"][0])
65
67
 
66
68
  Each segment is `{"start", "duration", "text"}` with `start` and `duration` in seconds. The `Segment` and `TranscriptData` typed dicts are importable from `getyoutubetranscript`.
67
69
 
70
+ ### Formats: text, timed text, JSON, SRT, WebVTT
71
+
72
+ Turn a transcript into a file format with the formatters. Timed text, SRT and WebVTT need per-line timing, so fetch with `timestamps=True` (same 1 credit).
73
+
74
+ ```python
75
+ from getyoutubetranscript import Client, to_srt, to_vtt, to_timed_text, to_json, to_text
76
+
77
+ result = client.get_transcript("5e37ZT3SQbk", timestamps=True)
78
+
79
+ open("video.srt", "w", encoding="utf-8").write(to_srt(result)) # SubRip subtitles
80
+ open("video.vtt", "w", encoding="utf-8").write(to_vtt(result)) # WebVTT subtitles
81
+ print(to_timed_text(result)) # "[0:03] So, Reed, education, ..." one line per caption
82
+ print(to_json(result)) # metadata, transcript and segments
83
+ print(to_text(result)) # one block of plain text
84
+ ```
85
+
86
+ `format_transcript(result, "srt")` does the same with the format as a string (`"text"`, `"timed"`, `"json"`, `"srt"`, `"vtt"`).
87
+
68
88
  ### Search
69
89
 
70
90
  ```python
71
- client.search("lofi beats", type="video", limit=10)
91
+ first_page = client.search("lofi beats", type="video", limit=10)
72
92
 
73
- # Pagination
74
- page2 = client.search(page_token=first_page["pagination"]["next_page_token"])
93
+ # Pagination: pass continuation_token back as page_token
94
+ if first_page.get("continuation_token"):
95
+ page2 = client.search(page_token=first_page["continuation_token"])
75
96
  ```
76
97
 
77
98
  ### Channels
@@ -102,6 +123,22 @@ while page["has_more"]:
102
123
  client.get_credits() # free - plan_credits_left, topup_credits_left, plan, rate_limit_per_minute
103
124
  ```
104
125
 
126
+ ## Command line
127
+
128
+ Installing the package also installs a `getyoutubetranscript` command.
129
+
130
+ ```bash
131
+ export GETYOUTUBETRANSCRIPT_API_KEY=sk_live_...
132
+
133
+ getyoutubetranscript https://youtu.be/5e37ZT3SQbk # plain text
134
+ getyoutubetranscript 5e37ZT3SQbk --format srt > video.srt # SRT subtitles
135
+ getyoutubetranscript 5e37ZT3SQbk --format timed --language en # [m:ss] lines
136
+ getyoutubetranscript VIDEO_1 VIDEO_2 --format json # one JSON list
137
+ getyoutubetranscript VIDEO_1 VIDEO_2 --format vtt --output-dir subs # subs/<video_id>.vtt
138
+ ```
139
+
140
+ Formats: `text` (default), `timed`, `json`, `srt`, `vtt`. Videos can be URLs or IDs. If one video fails, the rest still run and the command exits with status 1. `python -m getyoutubetranscript` works too.
141
+
105
142
  ## Error handling
106
143
 
107
144
  Every non-2xx or `{"success": false}` response raises `GetYouTubeTranscriptError` with the API's parsed error shape:
@@ -120,6 +157,49 @@ except GetYouTubeTranscriptError as e:
120
157
  print(e.response_body) # full parsed error body, e.g. {"creditsLeft": 0} on PAYMENT_REQUIRED
121
158
  ```
122
159
 
160
+ ## Getting RequestBlocked or IpBlocked?
161
+
162
+ If you use the open source `youtube-transcript-api` library, you have probably seen `RequestBlocked` or `IpBlocked` once your code runs on a server. YouTube blocks most IP addresses that belong to cloud providers (AWS, Google Cloud, Azure and others), and can also block a home IP that makes many requests. That library's own docs recommend rotating residential proxies as the workaround.
163
+
164
+ This SDK calls the GetYouTubeTranscript API instead of YouTube, so YouTube never sees your server's IP. There are no proxies to buy, rotate or debug, and the same code works on your laptop, a VPS, a serverless function or a CI job:
165
+
166
+ ```python
167
+ from getyoutubetranscript import Client
168
+
169
+ client = Client(api_key="sk_live_...")
170
+ result = client.get_transcript("https://www.youtube.com/watch?v=jNQXAC9IVRw", timestamps=True)
171
+ ```
172
+
173
+ The trade-off: it is a paid API with a free tier (each request uses credits), while `youtube-transcript-api` is free to run if you handle the blocking yourself.
174
+
175
+ ## Coming from youtube-transcript-api
176
+
177
+ The segment shape is the same (`text`, `start`, `duration`, in seconds), so most code ports directly.
178
+
179
+ | youtube-transcript-api | getyoutubetranscript |
180
+ | --- | --- |
181
+ | `YouTubeTranscriptApi().fetch(video_id)` | `client.get_transcript(video, timestamps=True)` |
182
+ | `fetched.to_raw_data()` | `result["segments"]` (already a list of dicts) |
183
+ | `fetch(video_id, languages=["de"])` | `get_transcript(video, language="de")` |
184
+ | Video ID only | Video ID or any YouTube URL (watch, youtu.be, Shorts, live) |
185
+ | `SRTFormatter()`, `WebVTTFormatter()`, `TextFormatter()`, `JSONFormatter()` | `to_srt`, `to_vtt`, `to_text`, `to_json` |
186
+ | CLI: `youtube_transcript_api VIDEO_ID --format json` | CLI: `getyoutubetranscript VIDEO --format json` |
187
+ | Proxies for cloud servers | Not needed |
188
+
189
+ ```python
190
+ # before
191
+ from youtube_transcript_api import YouTubeTranscriptApi
192
+ segments = YouTubeTranscriptApi().fetch("jNQXAC9IVRw").to_raw_data()
193
+
194
+ # after
195
+ from getyoutubetranscript import Client
196
+ segments = Client(api_key="sk_live_...").get_transcript("jNQXAC9IVRw", timestamps=True)["segments"]
197
+ ```
198
+
199
+ Not covered here: a list of preferred fallback languages, listing every available caption track, YouTube's machine translation of captions, and `preserve_formatting`. Request one language at a time with `language=`.
200
+
201
+ The response also includes the video title, channel name, channel URL, thumbnail and word count, which `youtube-transcript-api` does not return.
202
+
123
203
  ## Development
124
204
 
125
205
  ```bash
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "getyoutubetranscript"
7
- version = "0.2.2"
7
+ version = "0.3.1"
8
8
  description = "YouTube transcript API for Python: get YouTube video transcripts, captions and subtitles with timestamps, search YouTube, and list channel and playlist videos. No proxies or headless browser."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.9"
@@ -13,6 +13,7 @@ authors = [{ name = "tubeagentkit" }]
13
13
  keywords = [
14
14
  "youtube", "youtube-transcript", "youtube-transcripts", "youtube-transcript-api", "transcript", "transcripts",
15
15
  "captions", "subtitles", "youtube-captions", "youtube-subtitles", "timestamps", "youtube-api", "youtube-search",
16
+ "srt", "webvtt", "vtt", "cli",
16
17
  "llm", "rag", "ai", "api-client", "sdk",
17
18
  ]
18
19
  classifiers = [
@@ -37,6 +38,9 @@ dependencies = [
37
38
  "requests>=2.28",
38
39
  ]
39
40
 
41
+ [project.scripts]
42
+ getyoutubetranscript = "getyoutubetranscript.cli:main"
43
+
40
44
  [project.optional-dependencies]
41
45
  dev = [
42
46
  "pytest>=7.0",
@@ -10,16 +10,23 @@ See https://getyoutubetranscript.com/docs for the full API reference.
10
10
 
11
11
  from .client import Client, signup, verify_signup
12
12
  from .exceptions import GetYouTubeTranscriptError
13
+ from .formatters import format_transcript, to_json, to_srt, to_text, to_timed_text, to_vtt
13
14
  from .types import Segment, TranscriptData
14
15
 
15
- __version__ = "0.2.2"
16
+ __version__ = "0.3.1"
16
17
 
17
18
  __all__ = [
18
19
  "Client",
19
20
  "GetYouTubeTranscriptError",
20
21
  "Segment",
21
22
  "TranscriptData",
23
+ "format_transcript",
22
24
  "signup",
25
+ "to_json",
26
+ "to_srt",
27
+ "to_text",
28
+ "to_timed_text",
29
+ "to_vtt",
23
30
  "verify_signup",
24
31
  "__version__",
25
32
  ]
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
@@ -0,0 +1,100 @@
1
+ """Command line: ``getyoutubetranscript VIDEO [VIDEO ...] [--format srt]``.
2
+
3
+ Reads the API key from ``--api-key`` or the ``GETYOUTUBETRANSCRIPT_API_KEY``
4
+ environment variable.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import argparse
10
+ import os
11
+ import sys
12
+ from pathlib import Path
13
+ from typing import List, Optional
14
+
15
+ from . import __version__
16
+ from .client import Client
17
+ from .exceptions import GetYouTubeTranscriptError
18
+ from .formatters import FORMATS, TIMED_FORMATS, format_transcript, to_json
19
+
20
+ API_KEY_ENV = "GETYOUTUBETRANSCRIPT_API_KEY"
21
+
22
+
23
+ def _parser() -> argparse.ArgumentParser:
24
+ parser = argparse.ArgumentParser(
25
+ prog="getyoutubetranscript",
26
+ description="Get YouTube video transcripts as text, timed text, JSON, SRT or WebVTT.",
27
+ )
28
+ parser.add_argument("videos", nargs="+", metavar="VIDEO", help="YouTube video URL or 11-character video ID")
29
+ parser.add_argument("-f", "--format", choices=FORMATS, default="text", help="output format (default: text)")
30
+ parser.add_argument("-l", "--language", help="caption language code, e.g. en or es")
31
+ parser.add_argument(
32
+ "-o",
33
+ "--output-dir",
34
+ type=Path,
35
+ help="write one <video_id>.<format> file per video instead of printing (needed for several videos in a timed, SRT or VTT format)",
36
+ )
37
+ parser.add_argument("--api-key", help=f"API key (default: ${API_KEY_ENV})")
38
+ parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
39
+ return parser
40
+
41
+
42
+ def _extension(fmt: str) -> str:
43
+ return "txt" if fmt in ("text", "timed") else fmt
44
+
45
+
46
+ def main(argv: Optional[List[str]] = None, client: Optional[Client] = None) -> int:
47
+ try:
48
+ return _run(argv, client)
49
+ except BrokenPipeError:
50
+ # Output piped into a command that stopped reading (e.g. `| head`): exit quietly.
51
+ devnull = os.open(os.devnull, os.O_WRONLY)
52
+ os.dup2(devnull, sys.stdout.fileno())
53
+ return 0
54
+
55
+
56
+ def _run(argv: Optional[List[str]], client: Optional[Client]) -> int:
57
+ parser = _parser()
58
+ args = parser.parse_args(argv)
59
+
60
+ if args.format in TIMED_FORMATS and len(args.videos) > 1 and not args.output_dir:
61
+ parser.error(f"--output-dir is required to save {args.format} for more than one video")
62
+
63
+ if client is None:
64
+ api_key = args.api_key or os.environ.get(API_KEY_ENV)
65
+ if not api_key:
66
+ parser.error(f"no API key: pass --api-key or set {API_KEY_ENV} (get one at https://getyoutubetranscript.com)")
67
+ client = Client(api_key=api_key)
68
+
69
+ if args.output_dir:
70
+ args.output_dir.mkdir(parents=True, exist_ok=True)
71
+
72
+ failures = 0
73
+ json_results = []
74
+ for video in args.videos:
75
+ try:
76
+ data = client.get_transcript(
77
+ video, language=args.language, timestamps=args.format in TIMED_FORMATS or args.format == "json"
78
+ )
79
+ except GetYouTubeTranscriptError as e:
80
+ failures += 1
81
+ print(f"{video}: {e.message} ({e.code})", file=sys.stderr)
82
+ continue
83
+
84
+ if args.output_dir:
85
+ path = args.output_dir / f"{data['video_id']}.{_extension(args.format)}"
86
+ path.write_text(format_transcript(data, args.format), encoding="utf-8")
87
+ print(path, file=sys.stderr)
88
+ elif args.format == "json":
89
+ json_results.append(data)
90
+ else:
91
+ print(format_transcript(data, args.format))
92
+
93
+ if json_results:
94
+ print(to_json(json_results[0] if len(args.videos) == 1 else json_results))
95
+
96
+ return 1 if failures else 0
97
+
98
+
99
+ if __name__ == "__main__": # pragma: no cover
100
+ sys.exit(main())
@@ -184,9 +184,9 @@ class Client:
184
184
  Args:
185
185
  query: Search query. Required for a first page unless
186
186
  ``page_token`` is given.
187
- page_token: Continuation token from a previous response's
188
- ``pagination.next_page_token``, to fetch the next page. Treat
189
- as opaque - don't construct it yourself.
187
+ page_token: The ``continuation_token`` from a previous response,
188
+ to fetch the next page. Treat as opaque - don't construct it
189
+ yourself. Expires after 24 hours.
190
190
  type: Restrict results to ``"video"`` (default) or ``"channel"``.
191
191
  Never mixes both kinds in one response.
192
192
  country: Two-letter region code, e.g. ``"us"``.
@@ -196,7 +196,8 @@ class Client:
196
196
  Returns:
197
197
  dict with ``query``, and either ``video_results`` or
198
198
  ``channel_results`` depending on ``type``, plus
199
- ``pagination.next_page_token`` when more results are available.
199
+ ``continuation_token`` for the next page (missing or ``None`` when
200
+ there are no more pages, so read it with ``.get()``).
200
201
 
201
202
  Raises:
202
203
  ValueError: if neither ``query`` nor ``page_token`` is given.
@@ -0,0 +1,110 @@
1
+ """Turn a transcript from :meth:`Client.get_transcript` into text, timed text, JSON, SRT or WebVTT.
2
+
3
+ from getyoutubetranscript import Client
4
+ from getyoutubetranscript.formatters import to_srt
5
+
6
+ result = client.get_transcript("jNQXAC9IVRw", timestamps=True)
7
+ open("video.srt", "w", encoding="utf-8").write(to_srt(result))
8
+
9
+ Timed text, SRT and WebVTT need per-line timing, so fetch with ``timestamps=True``.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ from typing import Any, Callable, Dict, List, Mapping
16
+
17
+ from .types import Segment
18
+
19
+ FORMATS = ("text", "timed", "json", "srt", "vtt")
20
+ TIMED_FORMATS = ("timed", "srt", "vtt")
21
+
22
+
23
+ def _segments(data: Mapping[str, Any]) -> List[Segment]:
24
+ segments = data.get("segments")
25
+ if not segments:
26
+ raise ValueError(
27
+ "This format needs per-line timing. Fetch the transcript with timestamps=True."
28
+ )
29
+ return segments
30
+
31
+
32
+ def _clock(seconds: float, decimal_mark: str) -> str:
33
+ millis = max(0, int(round(seconds * 1000)))
34
+ hours, millis = divmod(millis, 3_600_000)
35
+ minutes, millis = divmod(millis, 60_000)
36
+ secs, millis = divmod(millis, 1000)
37
+ return f"{hours:02d}:{minutes:02d}:{secs:02d}{decimal_mark}{millis:03d}"
38
+
39
+
40
+ def _cues(data: Mapping[str, Any], decimal_mark: str) -> List[str]:
41
+ cues = []
42
+ for segment in _segments(data):
43
+ start = float(segment["start"])
44
+ end = start + float(segment["duration"])
45
+ timing = f"{_clock(start, decimal_mark)} --> {_clock(end, decimal_mark)}"
46
+ cues.append(f"{timing}\n{segment['text'].strip()}")
47
+ return cues
48
+
49
+
50
+ def to_text(data: Mapping[str, Any]) -> str:
51
+ """The transcript as one block of plain text."""
52
+ return data.get("transcript", "")
53
+
54
+
55
+ def _player_time(seconds: float) -> str:
56
+ total = max(0, int(seconds))
57
+ hours, rest = divmod(total, 3600)
58
+ minutes, secs = divmod(rest, 60)
59
+ return f"{hours}:{minutes:02d}:{secs:02d}" if hours else f"{minutes}:{secs:02d}"
60
+
61
+
62
+ def to_timed_text(data: Mapping[str, Any]) -> str:
63
+ """One ``[m:ss] text`` line per caption (``[h:mm:ss]`` past an hour). Needs ``segments``.
64
+
65
+ Easy for people and AI models to read and cite.
66
+ """
67
+ return "\n".join(f"[{_player_time(float(s['start']))}] {s['text'].strip()}" for s in _segments(data))
68
+
69
+
70
+ def to_json(data: Mapping[str, Any], **json_kwargs: Any) -> str:
71
+ """The full result (metadata, transcript and any segments) as a JSON string."""
72
+ json_kwargs.setdefault("ensure_ascii", False)
73
+ json_kwargs.setdefault("indent", 2)
74
+ return json.dumps(data, **json_kwargs)
75
+
76
+
77
+ def to_srt(data: Mapping[str, Any]) -> str:
78
+ """SubRip (.srt) subtitles. Needs ``segments`` (``timestamps=True``)."""
79
+ cues = _cues(data, ",")
80
+ return "\n\n".join(f"{index}\n{cue}" for index, cue in enumerate(cues, start=1)) + "\n"
81
+
82
+
83
+ def to_vtt(data: Mapping[str, Any]) -> str:
84
+ """WebVTT (.vtt) subtitles. Needs ``segments`` (``timestamps=True``)."""
85
+ cues = [_escape_vtt_text(cue) for cue in _cues(data, ".")]
86
+ return "WEBVTT\n\n" + "\n\n".join(cues) + "\n"
87
+
88
+
89
+ def _escape_vtt_text(cue: str) -> str:
90
+ # "-->" inside cue text would be read as a timing line, so escape it.
91
+ timing, _, text = cue.partition("\n")
92
+ return f"{timing}\n{text.replace('-->', '--&gt;')}"
93
+
94
+
95
+ _FORMATTERS: Dict[str, Callable[[Mapping[str, Any]], str]] = {
96
+ "text": to_text,
97
+ "timed": to_timed_text,
98
+ "json": to_json,
99
+ "srt": to_srt,
100
+ "vtt": to_vtt,
101
+ }
102
+
103
+
104
+ def format_transcript(data: Mapping[str, Any], fmt: str) -> str:
105
+ """Format a transcript as ``"text"``, ``"timed"``, ``"json"``, ``"srt"`` or ``"vtt"``."""
106
+ try:
107
+ formatter = _FORMATTERS[fmt]
108
+ except KeyError:
109
+ raise ValueError(f"Unknown format {fmt!r}. Choose one of: {', '.join(FORMATS)}.") from None
110
+ return formatter(data)
@@ -71,6 +71,18 @@ def test_search_live(client: Client):
71
71
  assert len(result["video_results"]) > 0
72
72
 
73
73
 
74
+ def test_search_pagination_live(client: Client):
75
+ """The README pagination example: continuation_token goes back in as page_token."""
76
+ first_page = client.search("lofi beats", type="video", limit=10)
77
+ assert "pagination" not in first_page
78
+ token = first_page.get("continuation_token")
79
+ assert token, "first page should offer a next page"
80
+ page2 = client.search(page_token=token)
81
+ assert page2["video_results"]
82
+ first_titles = {v.get("title") for v in first_page["video_results"]}
83
+ assert not first_titles & {v.get("title") for v in page2["video_results"]}
84
+
85
+
74
86
  def test_get_credits_live(client: Client):
75
87
  """Free endpoint - safe to run often."""
76
88
  result = client.get_credits()
@@ -0,0 +1,93 @@
1
+ """Unit tests for the command line (client is faked, no network)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+
7
+ import pytest
8
+
9
+ from getyoutubetranscript import GetYouTubeTranscriptError
10
+ from getyoutubetranscript.cli import API_KEY_ENV, main
11
+
12
+ SEGMENTS = [{"start": 0.0, "duration": 1.0, "text": "hi"}]
13
+
14
+
15
+ class FakeClient:
16
+ def __init__(self, fail=()):
17
+ self.calls = []
18
+ self.fail = set(fail)
19
+
20
+ def get_transcript(self, video, *, language=None, timestamps=False):
21
+ self.calls.append({"video": video, "language": language, "timestamps": timestamps})
22
+ if video in self.fail:
23
+ raise GetYouTubeTranscriptError("NOT_FOUND", "No transcript for this video.", 404)
24
+ data = {"video_id": video, "language_code": language or "en", "transcript": f"text of {video}"}
25
+ if timestamps:
26
+ data["segments"] = SEGMENTS
27
+ return data
28
+
29
+
30
+ def test_text_to_stdout_without_timestamps(capsys):
31
+ client = FakeClient()
32
+ assert main(["vid1"], client=client) == 0
33
+ assert capsys.readouterr().out == "text of vid1\n"
34
+ assert client.calls == [{"video": "vid1", "language": None, "timestamps": False}]
35
+
36
+
37
+ def test_srt_requests_timestamps_and_passes_language(capsys):
38
+ client = FakeClient()
39
+ assert main(["vid1", "-f", "srt", "-l", "es"], client=client) == 0
40
+ assert capsys.readouterr().out == "1\n00:00:00,000 --> 00:00:01,000\nhi\n\n"
41
+ assert client.calls[0] == {"video": "vid1", "language": "es", "timestamps": True}
42
+
43
+
44
+ def test_json_for_several_videos_is_one_list(capsys):
45
+ assert main(["a", "b", "-f", "json"], client=FakeClient()) == 0
46
+ out = json.loads(capsys.readouterr().out)
47
+ assert [item["video_id"] for item in out] == ["a", "b"]
48
+ assert out[0]["segments"] == SEGMENTS
49
+
50
+
51
+ def test_output_dir_writes_one_file_per_video(tmp_path):
52
+ assert main(["a", "b", "-f", "vtt", "-o", str(tmp_path)], client=FakeClient()) == 0
53
+ assert sorted(p.name for p in tmp_path.iterdir()) == ["a.vtt", "b.vtt"]
54
+ assert (tmp_path / "a.vtt").read_text(encoding="utf-8").startswith("WEBVTT\n\n00:00:00.000")
55
+
56
+
57
+ def test_timed_files_use_txt_extension(tmp_path):
58
+ assert main(["a", "-f", "timed", "-o", str(tmp_path)], client=FakeClient()) == 0
59
+ assert (tmp_path / "a.txt").read_text(encoding="utf-8") == "[0:00] hi"
60
+
61
+
62
+ def test_several_srt_videos_need_output_dir(capsys):
63
+ with pytest.raises(SystemExit) as exit_info:
64
+ main(["a", "b", "-f", "srt"], client=FakeClient())
65
+ assert exit_info.value.code == 2
66
+ assert "--output-dir" in capsys.readouterr().err
67
+
68
+
69
+ def test_keeps_going_after_a_failure_and_exits_1(capsys):
70
+ assert main(["bad", "good"], client=FakeClient(fail={"bad"})) == 1
71
+ captured = capsys.readouterr()
72
+ assert captured.out == "text of good\n"
73
+ assert "bad: No transcript for this video. (NOT_FOUND)" in captured.err
74
+
75
+
76
+ def test_missing_api_key_is_a_usage_error(monkeypatch, capsys):
77
+ monkeypatch.delenv(API_KEY_ENV, raising=False)
78
+ with pytest.raises(SystemExit) as exit_info:
79
+ main(["vid1"])
80
+ assert exit_info.value.code == 2
81
+ assert API_KEY_ENV in capsys.readouterr().err
82
+
83
+
84
+ def test_pipe_closed_early_exits_quietly(monkeypatch, tmp_path):
85
+ import getyoutubetranscript.cli as cli
86
+
87
+ def broken_pipe(*args, **kwargs):
88
+ raise BrokenPipeError
89
+
90
+ monkeypatch.setattr(cli, "_run", broken_pipe)
91
+ out = (tmp_path / "out").open("w")
92
+ monkeypatch.setattr("sys.stdout", out)
93
+ assert cli.main(["vid1"]) == 0
@@ -0,0 +1,73 @@
1
+ """Unit tests for the transcript formatters (no network)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+
7
+ import pytest
8
+
9
+ from getyoutubetranscript import format_transcript, to_json, to_srt, to_text, to_timed_text, to_vtt
10
+
11
+ DATA = {
12
+ "video_id": "abc12345678",
13
+ "language_code": "en",
14
+ "title": "A video",
15
+ "author_name": "A channel",
16
+ "transcript": "hello world. café --> next",
17
+ "word_count": 5,
18
+ "segments": [
19
+ {"start": 0.0, "duration": 2.5, "text": "hello world."},
20
+ {"start": 1.9996, "duration": 1.0, "text": "café --> next"},
21
+ {"start": 3725.25, "duration": 4.0, "text": " an hour in "},
22
+ ],
23
+ }
24
+ NO_SEGMENTS = {k: v for k, v in DATA.items() if k != "segments"}
25
+
26
+
27
+ def test_text_is_the_plain_transcript():
28
+ assert to_text(DATA) == "hello world. café --> next"
29
+ assert to_text(NO_SEGMENTS) == "hello world. café --> next"
30
+
31
+
32
+ def test_timed_text_uses_player_clock():
33
+ assert to_timed_text(DATA).splitlines() == [
34
+ "[0:00] hello world.",
35
+ "[0:01] café --> next",
36
+ "[1:02:05] an hour in",
37
+ ]
38
+
39
+
40
+ def test_srt_cues_numbering_and_rounding():
41
+ assert to_srt(DATA) == (
42
+ "1\n00:00:00,000 --> 00:00:02,500\nhello world.\n\n"
43
+ # 1.9996s rounds to 2.000s, never "00:00:01,1000"
44
+ "2\n00:00:02,000 --> 00:00:03,000\ncafé --> next\n\n"
45
+ "3\n01:02:05,250 --> 01:02:09,250\nan hour in\n"
46
+ )
47
+
48
+
49
+ def test_vtt_header_dot_millis_and_escaped_arrow():
50
+ assert to_vtt(DATA) == (
51
+ "WEBVTT\n\n"
52
+ "00:00:00.000 --> 00:00:02.500\nhello world.\n\n"
53
+ "00:00:02.000 --> 00:00:03.000\ncafé --&gt; next\n\n"
54
+ "01:02:05.250 --> 01:02:09.250\nan hour in\n"
55
+ )
56
+
57
+
58
+ def test_json_round_trips_and_keeps_unicode():
59
+ out = to_json(DATA)
60
+ assert json.loads(out) == DATA
61
+ assert "café" in out
62
+
63
+
64
+ @pytest.mark.parametrize("fn", [to_srt, to_vtt, to_timed_text])
65
+ def test_timed_formats_explain_missing_segments(fn):
66
+ with pytest.raises(ValueError, match="timestamps=True"):
67
+ fn(NO_SEGMENTS)
68
+
69
+
70
+ def test_format_transcript_dispatches_and_rejects_unknown():
71
+ assert format_transcript(DATA, "srt") == to_srt(DATA)
72
+ with pytest.raises(ValueError, match="Unknown format"):
73
+ format_transcript(DATA, "docx")