langchain-getyoutubetranscript 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langchain_getyoutubetranscript-0.1.0/.gitignore +7 -0
- langchain_getyoutubetranscript-0.1.0/LICENSE +21 -0
- langchain_getyoutubetranscript-0.1.0/PKG-INFO +123 -0
- langchain_getyoutubetranscript-0.1.0/README.md +92 -0
- langchain_getyoutubetranscript-0.1.0/langchain_getyoutubetranscript/__init__.py +6 -0
- langchain_getyoutubetranscript-0.1.0/langchain_getyoutubetranscript/_client.py +27 -0
- langchain_getyoutubetranscript-0.1.0/langchain_getyoutubetranscript/document_loaders.py +58 -0
- langchain_getyoutubetranscript-0.1.0/langchain_getyoutubetranscript/tools.py +97 -0
- langchain_getyoutubetranscript-0.1.0/pyproject.toml +48 -0
- langchain_getyoutubetranscript-0.1.0/tests/__init__.py +0 -0
- langchain_getyoutubetranscript-0.1.0/tests/integration_tests/__init__.py +0 -0
- langchain_getyoutubetranscript-0.1.0/tests/integration_tests/test_live.py +40 -0
- langchain_getyoutubetranscript-0.1.0/tests/unit_tests/__init__.py +0 -0
- langchain_getyoutubetranscript-0.1.0/tests/unit_tests/test_document_loaders.py +54 -0
- langchain_getyoutubetranscript-0.1.0/tests/unit_tests/test_tools.py +98 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 tubeagentkit
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: langchain-getyoutubetranscript
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: YouTube transcript tools and document loader for LangChain: get YouTube video transcripts with timestamps and search YouTube from agents and RAG pipelines, via the GetYouTubeTranscript API.
|
|
5
|
+
Project-URL: Homepage, https://getyoutubetranscript.com
|
|
6
|
+
Project-URL: Documentation, https://github.com/tubeagentkit/langchain-getyoutubetranscript#readme
|
|
7
|
+
Project-URL: Repository, https://github.com/tubeagentkit/langchain-getyoutubetranscript
|
|
8
|
+
Project-URL: API reference, https://getyoutubetranscript.com/docs
|
|
9
|
+
Project-URL: Get an API key, https://getyoutubetranscript.com/dashboard
|
|
10
|
+
Author: tubeagentkit
|
|
11
|
+
License: MIT
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Keywords: agents,captions,document-loader,langchain,llm,rag,subtitles,tools,transcript,youtube,youtube-transcript
|
|
14
|
+
Classifier: Development Status :: 4 - Beta
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Topic :: Multimedia :: Video
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Requires-Python: >=3.10
|
|
25
|
+
Requires-Dist: getyoutubetranscript<1.0.0,>=0.3.0
|
|
26
|
+
Requires-Dist: langchain-core<2.0.0,>=0.3.0
|
|
27
|
+
Provides-Extra: test
|
|
28
|
+
Requires-Dist: langchain-tests>=0.3.0; extra == 'test'
|
|
29
|
+
Requires-Dist: pytest>=7.0; extra == 'test'
|
|
30
|
+
Description-Content-Type: text/markdown
|
|
31
|
+
|
|
32
|
+
# langchain-getyoutubetranscript
|
|
33
|
+
|
|
34
|
+
YouTube transcript tools and a document loader for [LangChain](https://www.langchain.com), powered by the [GetYouTubeTranscript](https://getyoutubetranscript.com) API. Give agents the transcript of any YouTube video (optionally with `[m:ss]` timestamps) and YouTube search, or load transcripts into RAG pipelines.
|
|
35
|
+
|
|
36
|
+
The API fetches transcripts on its own servers, so it works from cloud servers and serverless functions without proxies, and without `RequestBlocked` / `IpBlocked` errors.
|
|
37
|
+
|
|
38
|
+
## Install
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
pip install langchain-getyoutubetranscript
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Get an API key at [getyoutubetranscript.com/dashboard](https://getyoutubetranscript.com/dashboard) (free tier included) and set it:
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
export GETYOUTUBETRANSCRIPT_API_KEY=sk_live_...
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Or pass `api_key="..."` to any class.
|
|
51
|
+
|
|
52
|
+
## Tools
|
|
53
|
+
|
|
54
|
+
| Class | Tool name | What it does |
|
|
55
|
+
| --- | --- | --- |
|
|
56
|
+
| `GetYouTubeTranscriptTool` | `youtube_transcript` | Transcript of a video (URL or ID) with title and channel. `timestamps=True` prefixes each line with `[m:ss]`. |
|
|
57
|
+
| `GetYouTubeTranscriptSearchTool` | `youtube_search` | Search YouTube for videos or channels. Returns JSON with a `continuation_token` for the next page. |
|
|
58
|
+
|
|
59
|
+
```python
|
|
60
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptTool
|
|
61
|
+
|
|
62
|
+
tool = GetYouTubeTranscriptTool()
|
|
63
|
+
print(tool.invoke({"video": "https://youtu.be/jNQXAC9IVRw", "timestamps": True}))
|
|
64
|
+
# Title: Me at the zoo
|
|
65
|
+
# Channel: jawed
|
|
66
|
+
# Language: en
|
|
67
|
+
#
|
|
68
|
+
# [0:01] All right, so here we are, in front of the elephants
|
|
69
|
+
# ...
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
API errors (no captions, invalid video, out of credits) come back to the agent as a short message instead of raising.
|
|
73
|
+
|
|
74
|
+
### With an agent
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
from langchain.agents import create_agent
|
|
78
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptSearchTool, GetYouTubeTranscriptTool
|
|
79
|
+
|
|
80
|
+
agent = create_agent(
|
|
81
|
+
model="anthropic:claude-sonnet-4-5",
|
|
82
|
+
tools=[GetYouTubeTranscriptSearchTool(), GetYouTubeTranscriptTool()],
|
|
83
|
+
)
|
|
84
|
+
agent.invoke(
|
|
85
|
+
{"messages": [{"role": "user", "content": "Find a short talk on transformers and summarize it with timestamps."}]}
|
|
86
|
+
)
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
## Document loader
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptLoader
|
|
93
|
+
|
|
94
|
+
docs = GetYouTubeTranscriptLoader(
|
|
95
|
+
["https://youtu.be/jNQXAC9IVRw", "5e37ZT3SQbk"],
|
|
96
|
+
language="en",
|
|
97
|
+
timestamps=False,
|
|
98
|
+
).load()
|
|
99
|
+
|
|
100
|
+
print(docs[0].metadata)
|
|
101
|
+
# {'source': 'https://www.youtube.com/watch?v=jNQXAC9IVRw', 'video_id': 'jNQXAC9IVRw',
|
|
102
|
+
# 'title': 'Me at the zoo', 'author_name': 'jawed', 'language_code': 'en', 'word_count': 39}
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
One `Document` per video. With `timestamps=True`, `page_content` is `[m:ss]` lines and `metadata["segments"]` holds `{start, duration, text}` per caption line (seconds).
|
|
106
|
+
|
|
107
|
+
## Pricing
|
|
108
|
+
|
|
109
|
+
Each transcript or search request uses one credit from your GetYouTubeTranscript account. Failed requests are not charged.
|
|
110
|
+
|
|
111
|
+
## Development
|
|
112
|
+
|
|
113
|
+
```bash
|
|
114
|
+
pip install -e ".[test]"
|
|
115
|
+
pytest # unit + LangChain standard tests, no network
|
|
116
|
+
GETYOUTUBETRANSCRIPT_API_KEY=... pytest tests/integration_tests # live, spends credits
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
## Links
|
|
120
|
+
|
|
121
|
+
- [GetYouTubeTranscript API docs](https://getyoutubetranscript.com/docs)
|
|
122
|
+
- [Python SDK](https://pypi.org/project/getyoutubetranscript/) (this package is built on it)
|
|
123
|
+
- [License: MIT](LICENSE)
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
# langchain-getyoutubetranscript
|
|
2
|
+
|
|
3
|
+
YouTube transcript tools and a document loader for [LangChain](https://www.langchain.com), powered by the [GetYouTubeTranscript](https://getyoutubetranscript.com) API. Give agents the transcript of any YouTube video (optionally with `[m:ss]` timestamps) and YouTube search, or load transcripts into RAG pipelines.
|
|
4
|
+
|
|
5
|
+
The API fetches transcripts on its own servers, so it works from cloud servers and serverless functions without proxies, and without `RequestBlocked` / `IpBlocked` errors.
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install langchain-getyoutubetranscript
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
Get an API key at [getyoutubetranscript.com/dashboard](https://getyoutubetranscript.com/dashboard) (free tier included) and set it:
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
export GETYOUTUBETRANSCRIPT_API_KEY=sk_live_...
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
Or pass `api_key="..."` to any class.
|
|
20
|
+
|
|
21
|
+
## Tools
|
|
22
|
+
|
|
23
|
+
| Class | Tool name | What it does |
|
|
24
|
+
| --- | --- | --- |
|
|
25
|
+
| `GetYouTubeTranscriptTool` | `youtube_transcript` | Transcript of a video (URL or ID) with title and channel. `timestamps=True` prefixes each line with `[m:ss]`. |
|
|
26
|
+
| `GetYouTubeTranscriptSearchTool` | `youtube_search` | Search YouTube for videos or channels. Returns JSON with a `continuation_token` for the next page. |
|
|
27
|
+
|
|
28
|
+
```python
|
|
29
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptTool
|
|
30
|
+
|
|
31
|
+
tool = GetYouTubeTranscriptTool()
|
|
32
|
+
print(tool.invoke({"video": "https://youtu.be/jNQXAC9IVRw", "timestamps": True}))
|
|
33
|
+
# Title: Me at the zoo
|
|
34
|
+
# Channel: jawed
|
|
35
|
+
# Language: en
|
|
36
|
+
#
|
|
37
|
+
# [0:01] All right, so here we are, in front of the elephants
|
|
38
|
+
# ...
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
API errors (no captions, invalid video, out of credits) come back to the agent as a short message instead of raising.
|
|
42
|
+
|
|
43
|
+
### With an agent
|
|
44
|
+
|
|
45
|
+
```python
|
|
46
|
+
from langchain.agents import create_agent
|
|
47
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptSearchTool, GetYouTubeTranscriptTool
|
|
48
|
+
|
|
49
|
+
agent = create_agent(
|
|
50
|
+
model="anthropic:claude-sonnet-4-5",
|
|
51
|
+
tools=[GetYouTubeTranscriptSearchTool(), GetYouTubeTranscriptTool()],
|
|
52
|
+
)
|
|
53
|
+
agent.invoke(
|
|
54
|
+
{"messages": [{"role": "user", "content": "Find a short talk on transformers and summarize it with timestamps."}]}
|
|
55
|
+
)
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
## Document loader
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptLoader
|
|
62
|
+
|
|
63
|
+
docs = GetYouTubeTranscriptLoader(
|
|
64
|
+
["https://youtu.be/jNQXAC9IVRw", "5e37ZT3SQbk"],
|
|
65
|
+
language="en",
|
|
66
|
+
timestamps=False,
|
|
67
|
+
).load()
|
|
68
|
+
|
|
69
|
+
print(docs[0].metadata)
|
|
70
|
+
# {'source': 'https://www.youtube.com/watch?v=jNQXAC9IVRw', 'video_id': 'jNQXAC9IVRw',
|
|
71
|
+
# 'title': 'Me at the zoo', 'author_name': 'jawed', 'language_code': 'en', 'word_count': 39}
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
One `Document` per video. With `timestamps=True`, `page_content` is `[m:ss]` lines and `metadata["segments"]` holds `{start, duration, text}` per caption line (seconds).
|
|
75
|
+
|
|
76
|
+
## Pricing
|
|
77
|
+
|
|
78
|
+
Each transcript or search request uses one credit from your GetYouTubeTranscript account. Failed requests are not charged.
|
|
79
|
+
|
|
80
|
+
## Development
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
pip install -e ".[test]"
|
|
84
|
+
pytest # unit + LangChain standard tests, no network
|
|
85
|
+
GETYOUTUBETRANSCRIPT_API_KEY=... pytest tests/integration_tests # live, spends credits
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## Links
|
|
89
|
+
|
|
90
|
+
- [GetYouTubeTranscript API docs](https://getyoutubetranscript.com/docs)
|
|
91
|
+
- [Python SDK](https://pypi.org/project/getyoutubetranscript/) (this package is built on it)
|
|
92
|
+
- [License: MIT](LICENSE)
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""LangChain integration for the GetYouTubeTranscript API: YouTube transcripts and search."""
|
|
2
|
+
|
|
3
|
+
from langchain_getyoutubetranscript.document_loaders import GetYouTubeTranscriptLoader
|
|
4
|
+
from langchain_getyoutubetranscript.tools import GetYouTubeTranscriptSearchTool, GetYouTubeTranscriptTool
|
|
5
|
+
|
|
6
|
+
__all__ = ["GetYouTubeTranscriptLoader", "GetYouTubeTranscriptSearchTool", "GetYouTubeTranscriptTool"]
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Shared API key handling for the tools and the loader."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from getyoutubetranscript import Client, to_timed_text
|
|
9
|
+
from pydantic import SecretStr
|
|
10
|
+
|
|
11
|
+
API_KEY_ENV = "GETYOUTUBETRANSCRIPT_API_KEY"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def make_client(api_key: SecretStr | None) -> Client:
|
|
15
|
+
if api_key is None or not api_key.get_secret_value():
|
|
16
|
+
raise ValueError(
|
|
17
|
+
f"A GetYouTubeTranscript API key is required. Pass api_key=... or set {API_KEY_ENV}. "
|
|
18
|
+
"Get one at https://getyoutubetranscript.com/dashboard"
|
|
19
|
+
)
|
|
20
|
+
return Client(api_key=api_key.get_secret_value())
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def format_transcript(result: Mapping[str, Any], timestamps: bool) -> str:
|
|
24
|
+
"""Title and channel header, then the transcript (``[m:ss]`` lines when timestamps)."""
|
|
25
|
+
body = to_timed_text(result) if timestamps and result.get("segments") else result.get("transcript", "")
|
|
26
|
+
header = f"Title: {result.get('title', '')}\nChannel: {result.get('author_name', '')}\nLanguage: {result.get('language_code', '')}"
|
|
27
|
+
return f"{header}\n\n{body}"
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Load YouTube transcripts as LangChain Documents."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterator, Sequence
|
|
6
|
+
|
|
7
|
+
from getyoutubetranscript import to_timed_text
|
|
8
|
+
from langchain_core.document_loaders import BaseLoader
|
|
9
|
+
from langchain_core.documents import Document
|
|
10
|
+
from langchain_core.utils import secret_from_env
|
|
11
|
+
from pydantic import SecretStr
|
|
12
|
+
|
|
13
|
+
from ._client import API_KEY_ENV, make_client
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class GetYouTubeTranscriptLoader(BaseLoader):
|
|
17
|
+
"""Load the transcripts of one or more YouTube videos, one Document per video.
|
|
18
|
+
|
|
19
|
+
.. code-block:: python
|
|
20
|
+
|
|
21
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptLoader
|
|
22
|
+
|
|
23
|
+
docs = GetYouTubeTranscriptLoader(["https://youtu.be/jNQXAC9IVRw"]).load()
|
|
24
|
+
|
|
25
|
+
``page_content`` is the transcript text (``[m:ss]`` lines when ``timestamps=True``).
|
|
26
|
+
``metadata`` has ``source``, ``video_id``, ``title``, ``author_name``,
|
|
27
|
+
``language_code`` and ``word_count``, plus ``segments`` when ``timestamps=True``.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(
|
|
31
|
+
self,
|
|
32
|
+
videos: str | Sequence[str],
|
|
33
|
+
*,
|
|
34
|
+
language: str | None = None,
|
|
35
|
+
timestamps: bool = False,
|
|
36
|
+
api_key: str | None = None,
|
|
37
|
+
) -> None:
|
|
38
|
+
self.videos = [videos] if isinstance(videos, str) else list(videos)
|
|
39
|
+
self.language = language
|
|
40
|
+
self.timestamps = timestamps
|
|
41
|
+
key = SecretStr(api_key) if api_key else secret_from_env(API_KEY_ENV, default=None)()
|
|
42
|
+
self._client = make_client(key)
|
|
43
|
+
|
|
44
|
+
def lazy_load(self) -> Iterator[Document]:
|
|
45
|
+
for video in self.videos:
|
|
46
|
+
result = self._client.get_transcript(video, language=self.language, timestamps=self.timestamps)
|
|
47
|
+
text = to_timed_text(result) if self.timestamps and result.get("segments") else result.get("transcript", "")
|
|
48
|
+
metadata = {
|
|
49
|
+
"source": f"https://www.youtube.com/watch?v={result['video_id']}",
|
|
50
|
+
"video_id": result["video_id"],
|
|
51
|
+
"title": result.get("title", ""),
|
|
52
|
+
"author_name": result.get("author_name", ""),
|
|
53
|
+
"language_code": result.get("language_code", ""),
|
|
54
|
+
"word_count": result.get("word_count", 0),
|
|
55
|
+
}
|
|
56
|
+
if self.timestamps and result.get("segments"):
|
|
57
|
+
metadata["segments"] = result["segments"]
|
|
58
|
+
yield Document(page_content=text, metadata=metadata)
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""LangChain tools for the GetYouTubeTranscript API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from typing import Literal
|
|
7
|
+
|
|
8
|
+
from getyoutubetranscript import Client, GetYouTubeTranscriptError
|
|
9
|
+
from langchain_core.tools import BaseTool, ToolException
|
|
10
|
+
from langchain_core.utils import secret_from_env
|
|
11
|
+
from pydantic import BaseModel, Field, PrivateAttr, SecretStr
|
|
12
|
+
|
|
13
|
+
from ._client import API_KEY_ENV, format_transcript, make_client
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class _ApiKeyMixin(BaseModel):
|
|
17
|
+
api_key: SecretStr | None = Field(default_factory=secret_from_env(API_KEY_ENV, default=None))
|
|
18
|
+
"""GetYouTubeTranscript API key. Defaults to the ``GETYOUTUBETRANSCRIPT_API_KEY`` env var."""
|
|
19
|
+
|
|
20
|
+
_client: Client | None = PrivateAttr(default=None)
|
|
21
|
+
|
|
22
|
+
@property
|
|
23
|
+
def client(self) -> Client:
|
|
24
|
+
if self._client is None:
|
|
25
|
+
self._client = make_client(self.api_key)
|
|
26
|
+
return self._client
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class TranscriptInput(BaseModel):
|
|
30
|
+
video: str = Field(description="YouTube video URL (watch, youtu.be, Shorts or live) or the 11-character video ID.")
|
|
31
|
+
language: str | None = Field(default=None, description="Optional caption language code, e.g. 'en' or 'es'.")
|
|
32
|
+
timestamps: bool = Field(
|
|
33
|
+
default=False,
|
|
34
|
+
description="Set to true to prefix each caption line with its start time, e.g. [1:05], for quoting or linking moments.",
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class GetYouTubeTranscriptTool(_ApiKeyMixin, BaseTool): # type: ignore[override]
|
|
39
|
+
"""Get the transcript of a YouTube video.
|
|
40
|
+
|
|
41
|
+
Setup:
|
|
42
|
+
``pip install langchain-getyoutubetranscript`` and set ``GETYOUTUBETRANSCRIPT_API_KEY``.
|
|
43
|
+
|
|
44
|
+
Invoke:
|
|
45
|
+
.. code-block:: python
|
|
46
|
+
|
|
47
|
+
tool = GetYouTubeTranscriptTool()
|
|
48
|
+
tool.invoke({"video": "https://youtu.be/jNQXAC9IVRw", "timestamps": True})
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
name: str = "youtube_transcript"
|
|
52
|
+
description: str = (
|
|
53
|
+
"Get the transcript (spoken text) of a YouTube video from its URL or video ID, with the title and channel. "
|
|
54
|
+
"Use it to summarize, quote or answer questions about a video."
|
|
55
|
+
)
|
|
56
|
+
args_schema: type[BaseModel] = TranscriptInput
|
|
57
|
+
handle_tool_error: bool = True
|
|
58
|
+
|
|
59
|
+
def _run(self, video: str, language: str | None = None, timestamps: bool = False, **kwargs: object) -> str:
|
|
60
|
+
try:
|
|
61
|
+
result = self.client.get_transcript(video, language=language, timestamps=timestamps)
|
|
62
|
+
except GetYouTubeTranscriptError as e:
|
|
63
|
+
raise ToolException(e.message) from e
|
|
64
|
+
return format_transcript(result, timestamps)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class SearchInput(BaseModel):
|
|
68
|
+
query: str = Field(description="What to search YouTube for.")
|
|
69
|
+
type: Literal["video", "channel"] = Field(default="video", description="Search for videos or channels.")
|
|
70
|
+
page_token: str | None = Field(
|
|
71
|
+
default=None, description="continuation_token from a previous search result, to get the next page."
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class GetYouTubeTranscriptSearchTool(_ApiKeyMixin, BaseTool): # type: ignore[override]
|
|
76
|
+
"""Search YouTube for videos or channels.
|
|
77
|
+
|
|
78
|
+
Invoke:
|
|
79
|
+
.. code-block:: python
|
|
80
|
+
|
|
81
|
+
GetYouTubeTranscriptSearchTool().invoke({"query": "linear algebra lecture"})
|
|
82
|
+
"""
|
|
83
|
+
|
|
84
|
+
name: str = "youtube_search"
|
|
85
|
+
description: str = (
|
|
86
|
+
"Search YouTube for videos or channels. Returns JSON with titles, video or channel IDs, links, channel, "
|
|
87
|
+
"views, length and upload date, plus a continuation_token for the next page."
|
|
88
|
+
)
|
|
89
|
+
args_schema: type[BaseModel] = SearchInput
|
|
90
|
+
handle_tool_error: bool = True
|
|
91
|
+
|
|
92
|
+
def _run(self, query: str, type: str = "video", page_token: str | None = None, **kwargs: object) -> str:
|
|
93
|
+
try:
|
|
94
|
+
result = self.client.search(query, type=type, page_token=page_token)
|
|
95
|
+
except GetYouTubeTranscriptError as e:
|
|
96
|
+
raise ToolException(e.message) from e
|
|
97
|
+
return json.dumps(result, ensure_ascii=False)
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "langchain-getyoutubetranscript"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "YouTube transcript tools and document loader for LangChain: get YouTube video transcripts with timestamps and search YouTube from agents and RAG pipelines, via the GetYouTubeTranscript API."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "tubeagentkit" }]
|
|
13
|
+
keywords = ["langchain", "youtube", "youtube-transcript", "transcript", "captions", "subtitles", "agents", "rag", "llm", "tools", "document-loader"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3.10",
|
|
20
|
+
"Programming Language :: Python :: 3.11",
|
|
21
|
+
"Programming Language :: Python :: 3.12",
|
|
22
|
+
"Programming Language :: Python :: 3.13",
|
|
23
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
24
|
+
"Topic :: Multimedia :: Video",
|
|
25
|
+
]
|
|
26
|
+
dependencies = [
|
|
27
|
+
"langchain-core>=0.3.0,<2.0.0",
|
|
28
|
+
"getyoutubetranscript>=0.3.0,<1.0.0",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[project.optional-dependencies]
|
|
32
|
+
test = ["pytest>=7.0", "langchain-tests>=0.3.0"]
|
|
33
|
+
|
|
34
|
+
[project.urls]
|
|
35
|
+
Homepage = "https://getyoutubetranscript.com"
|
|
36
|
+
Documentation = "https://github.com/tubeagentkit/langchain-getyoutubetranscript#readme"
|
|
37
|
+
Repository = "https://github.com/tubeagentkit/langchain-getyoutubetranscript"
|
|
38
|
+
"API reference" = "https://getyoutubetranscript.com/docs"
|
|
39
|
+
"Get an API key" = "https://getyoutubetranscript.com/dashboard"
|
|
40
|
+
|
|
41
|
+
[tool.hatch.build.targets.wheel]
|
|
42
|
+
packages = ["langchain_getyoutubetranscript"]
|
|
43
|
+
|
|
44
|
+
[tool.ruff]
|
|
45
|
+
line-length = 120
|
|
46
|
+
|
|
47
|
+
[tool.pytest.ini_options]
|
|
48
|
+
testpaths = ["tests/unit_tests"]
|
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Live tests against the real API. Skipped unless GETYOUTUBETRANSCRIPT_API_KEY is set; each run spends credits."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
|
|
8
|
+
import pytest
|
|
9
|
+
|
|
10
|
+
from langchain_getyoutubetranscript import (
|
|
11
|
+
GetYouTubeTranscriptLoader,
|
|
12
|
+
GetYouTubeTranscriptSearchTool,
|
|
13
|
+
GetYouTubeTranscriptTool,
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
pytestmark = pytest.mark.skipif(not os.environ.get("GETYOUTUBETRANSCRIPT_API_KEY"), reason="no API key")
|
|
17
|
+
VIDEO = "https://youtu.be/5e37ZT3SQbk"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def test_tool_timestamps():
|
|
21
|
+
out = GetYouTubeTranscriptTool().invoke({"video": VIDEO, "timestamps": True})
|
|
22
|
+
lines = [line for line in out.splitlines() if line.startswith("[")]
|
|
23
|
+
assert out.startswith("Title: ")
|
|
24
|
+
assert len(lines) > 400
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def test_tool_bad_video_returns_message():
|
|
28
|
+
out = GetYouTubeTranscriptTool().invoke({"video": "notavalidvideo"})
|
|
29
|
+
assert "YouTube video" in out
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def test_search_tool():
|
|
33
|
+
data = json.loads(GetYouTubeTranscriptSearchTool().invoke({"query": "linear algebra lecture"}))
|
|
34
|
+
assert data["video_results"]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_loader():
|
|
38
|
+
doc = GetYouTubeTranscriptLoader(VIDEO).load()[0]
|
|
39
|
+
assert doc.metadata["video_id"] == "5e37ZT3SQbk"
|
|
40
|
+
assert len(doc.page_content) > 10_000
|
|
File without changes
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from unittest.mock import MagicMock
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptLoader
|
|
8
|
+
|
|
9
|
+
RESULT = {
|
|
10
|
+
"video_id": "jNQXAC9IVRw",
|
|
11
|
+
"language_code": "en",
|
|
12
|
+
"title": "Me at the zoo",
|
|
13
|
+
"author_name": "jawed",
|
|
14
|
+
"transcript": "All right, so here we are",
|
|
15
|
+
"word_count": 6,
|
|
16
|
+
"segments": [{"start": 61.0, "duration": 2.0, "text": "All right, so here we are"}],
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _loader(**kwargs) -> tuple[GetYouTubeTranscriptLoader, MagicMock]:
|
|
21
|
+
loader = GetYouTubeTranscriptLoader(api_key="test-key", **kwargs)
|
|
22
|
+
client = MagicMock()
|
|
23
|
+
client.get_transcript.return_value = RESULT
|
|
24
|
+
loader._client = client
|
|
25
|
+
return loader, client
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def test_one_document_per_video_with_metadata():
|
|
29
|
+
loader, client = _loader(videos=["a", "b"], language="en")
|
|
30
|
+
docs = loader.load()
|
|
31
|
+
assert len(docs) == 2
|
|
32
|
+
assert client.get_transcript.call_args.kwargs == {"language": "en", "timestamps": False}
|
|
33
|
+
assert docs[0].page_content == "All right, so here we are"
|
|
34
|
+
assert docs[0].metadata == {
|
|
35
|
+
"source": "https://www.youtube.com/watch?v=jNQXAC9IVRw",
|
|
36
|
+
"video_id": "jNQXAC9IVRw",
|
|
37
|
+
"title": "Me at the zoo",
|
|
38
|
+
"author_name": "jawed",
|
|
39
|
+
"language_code": "en",
|
|
40
|
+
"word_count": 6,
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_timestamps_give_timed_text_and_segments():
|
|
45
|
+
loader, _ = _loader(videos="a", timestamps=True)
|
|
46
|
+
doc = loader.load()[0]
|
|
47
|
+
assert doc.page_content == "[1:01] All right, so here we are"
|
|
48
|
+
assert doc.metadata["segments"] == RESULT["segments"]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def test_missing_key(monkeypatch):
|
|
52
|
+
monkeypatch.delenv("GETYOUTUBETRANSCRIPT_API_KEY", raising=False)
|
|
53
|
+
with pytest.raises(ValueError, match="API key is required"):
|
|
54
|
+
GetYouTubeTranscriptLoader("a")
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from unittest.mock import MagicMock
|
|
5
|
+
|
|
6
|
+
import pytest
|
|
7
|
+
from getyoutubetranscript import GetYouTubeTranscriptError
|
|
8
|
+
from langchain_tests.unit_tests import ToolsUnitTests
|
|
9
|
+
|
|
10
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptSearchTool, GetYouTubeTranscriptTool
|
|
11
|
+
|
|
12
|
+
RESULT = {
|
|
13
|
+
"video_id": "jNQXAC9IVRw",
|
|
14
|
+
"language_code": "en",
|
|
15
|
+
"title": "Me at the zoo",
|
|
16
|
+
"author_name": "jawed",
|
|
17
|
+
"transcript": "All right, so here we are",
|
|
18
|
+
"word_count": 6,
|
|
19
|
+
"segments": [{"start": 1.2, "duration": 2.16, "text": "All right, so here we are"}],
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class TestTranscriptToolStandard(ToolsUnitTests):
|
|
24
|
+
@property
|
|
25
|
+
def tool_constructor(self) -> type[GetYouTubeTranscriptTool]:
|
|
26
|
+
return GetYouTubeTranscriptTool
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
def tool_constructor_params(self) -> dict:
|
|
30
|
+
return {"api_key": "test-key"}
|
|
31
|
+
|
|
32
|
+
@property
|
|
33
|
+
def tool_invoke_params_example(self) -> dict:
|
|
34
|
+
return {"video": "jNQXAC9IVRw", "language": "en", "timestamps": True}
|
|
35
|
+
|
|
36
|
+
@property
|
|
37
|
+
def init_from_env_params(self) -> tuple[dict, dict, dict]:
|
|
38
|
+
return ({"GETYOUTUBETRANSCRIPT_API_KEY": "env-key"}, {}, {"api_key": "env-key"})
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class TestSearchToolStandard(ToolsUnitTests):
|
|
42
|
+
@property
|
|
43
|
+
def tool_constructor(self) -> type[GetYouTubeTranscriptSearchTool]:
|
|
44
|
+
return GetYouTubeTranscriptSearchTool
|
|
45
|
+
|
|
46
|
+
@property
|
|
47
|
+
def tool_constructor_params(self) -> dict:
|
|
48
|
+
return {"api_key": "test-key"}
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def tool_invoke_params_example(self) -> dict:
|
|
52
|
+
return {"query": "linear algebra", "type": "video"}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _tool_with(cls, client: MagicMock):
|
|
56
|
+
tool = cls(api_key="test-key")
|
|
57
|
+
tool._client = client
|
|
58
|
+
return tool
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def test_transcript_plain_text():
|
|
62
|
+
client = MagicMock()
|
|
63
|
+
client.get_transcript.return_value = {k: v for k, v in RESULT.items() if k != "segments"}
|
|
64
|
+
out = _tool_with(GetYouTubeTranscriptTool, client).invoke({"video": "https://youtu.be/jNQXAC9IVRw"})
|
|
65
|
+
client.get_transcript.assert_called_once_with("https://youtu.be/jNQXAC9IVRw", language=None, timestamps=False)
|
|
66
|
+
assert out == "Title: Me at the zoo\nChannel: jawed\nLanguage: en\n\nAll right, so here we are"
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_transcript_timestamps():
|
|
70
|
+
client = MagicMock()
|
|
71
|
+
client.get_transcript.return_value = RESULT
|
|
72
|
+
out = _tool_with(GetYouTubeTranscriptTool, client).invoke({"video": "jNQXAC9IVRw", "timestamps": True})
|
|
73
|
+
assert out.endswith("\n\n[0:01] All right, so here we are")
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def test_api_error_is_returned_to_the_agent_not_raised():
|
|
77
|
+
client = MagicMock()
|
|
78
|
+
client.get_transcript.side_effect = GetYouTubeTranscriptError(
|
|
79
|
+
"INVALID_URL", "Not a recognizable YouTube video URL or id.", 400
|
|
80
|
+
)
|
|
81
|
+
out = _tool_with(GetYouTubeTranscriptTool, client).invoke({"video": "nope"})
|
|
82
|
+
assert out == "Not a recognizable YouTube video URL or id."
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def test_search_returns_json():
|
|
86
|
+
client = MagicMock()
|
|
87
|
+
client.search.return_value = {"video_results": [{"title": "x"}], "continuation_token": "t"}
|
|
88
|
+
out = _tool_with(GetYouTubeTranscriptSearchTool, client).invoke(
|
|
89
|
+
{"query": "q", "type": "channel", "page_token": "p"}
|
|
90
|
+
)
|
|
91
|
+
client.search.assert_called_once_with("q", type="channel", page_token="p")
|
|
92
|
+
assert json.loads(out)["continuation_token"] == "t"
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def test_missing_key_explains_how_to_get_one(monkeypatch):
|
|
96
|
+
monkeypatch.delenv("GETYOUTUBETRANSCRIPT_API_KEY", raising=False)
|
|
97
|
+
with pytest.raises(ValueError, match="GETYOUTUBETRANSCRIPT_API_KEY"):
|
|
98
|
+
_ = GetYouTubeTranscriptTool().client
|