langchain-getyoutubetranscript 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langchain_getyoutubetranscript/__init__.py +6 -0
- langchain_getyoutubetranscript/_client.py +27 -0
- langchain_getyoutubetranscript/document_loaders.py +58 -0
- langchain_getyoutubetranscript/tools.py +97 -0
- langchain_getyoutubetranscript-0.1.0.dist-info/METADATA +123 -0
- langchain_getyoutubetranscript-0.1.0.dist-info/RECORD +8 -0
- langchain_getyoutubetranscript-0.1.0.dist-info/WHEEL +4 -0
- langchain_getyoutubetranscript-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""LangChain integration for the GetYouTubeTranscript API: YouTube transcripts and search."""
|
|
2
|
+
|
|
3
|
+
from langchain_getyoutubetranscript.document_loaders import GetYouTubeTranscriptLoader
|
|
4
|
+
from langchain_getyoutubetranscript.tools import GetYouTubeTranscriptSearchTool, GetYouTubeTranscriptTool
|
|
5
|
+
|
|
6
|
+
__all__ = ["GetYouTubeTranscriptLoader", "GetYouTubeTranscriptSearchTool", "GetYouTubeTranscriptTool"]
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Shared API key handling for the tools and the loader."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from getyoutubetranscript import Client, to_timed_text
|
|
9
|
+
from pydantic import SecretStr
|
|
10
|
+
|
|
11
|
+
API_KEY_ENV = "GETYOUTUBETRANSCRIPT_API_KEY"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def make_client(api_key: SecretStr | None) -> Client:
|
|
15
|
+
if api_key is None or not api_key.get_secret_value():
|
|
16
|
+
raise ValueError(
|
|
17
|
+
f"A GetYouTubeTranscript API key is required. Pass api_key=... or set {API_KEY_ENV}. "
|
|
18
|
+
"Get one at https://getyoutubetranscript.com/dashboard"
|
|
19
|
+
)
|
|
20
|
+
return Client(api_key=api_key.get_secret_value())
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def format_transcript(result: Mapping[str, Any], timestamps: bool) -> str:
|
|
24
|
+
"""Title and channel header, then the transcript (``[m:ss]`` lines when timestamps)."""
|
|
25
|
+
body = to_timed_text(result) if timestamps and result.get("segments") else result.get("transcript", "")
|
|
26
|
+
header = f"Title: {result.get('title', '')}\nChannel: {result.get('author_name', '')}\nLanguage: {result.get('language_code', '')}"
|
|
27
|
+
return f"{header}\n\n{body}"
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Load YouTube transcripts as LangChain Documents."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterator, Sequence
|
|
6
|
+
|
|
7
|
+
from getyoutubetranscript import to_timed_text
|
|
8
|
+
from langchain_core.document_loaders import BaseLoader
|
|
9
|
+
from langchain_core.documents import Document
|
|
10
|
+
from langchain_core.utils import secret_from_env
|
|
11
|
+
from pydantic import SecretStr
|
|
12
|
+
|
|
13
|
+
from ._client import API_KEY_ENV, make_client
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class GetYouTubeTranscriptLoader(BaseLoader):
|
|
17
|
+
"""Load the transcripts of one or more YouTube videos, one Document per video.
|
|
18
|
+
|
|
19
|
+
.. code-block:: python
|
|
20
|
+
|
|
21
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptLoader
|
|
22
|
+
|
|
23
|
+
docs = GetYouTubeTranscriptLoader(["https://youtu.be/jNQXAC9IVRw"]).load()
|
|
24
|
+
|
|
25
|
+
``page_content`` is the transcript text (``[m:ss]`` lines when ``timestamps=True``).
|
|
26
|
+
``metadata`` has ``source``, ``video_id``, ``title``, ``author_name``,
|
|
27
|
+
``language_code`` and ``word_count``, plus ``segments`` when ``timestamps=True``.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(
|
|
31
|
+
self,
|
|
32
|
+
videos: str | Sequence[str],
|
|
33
|
+
*,
|
|
34
|
+
language: str | None = None,
|
|
35
|
+
timestamps: bool = False,
|
|
36
|
+
api_key: str | None = None,
|
|
37
|
+
) -> None:
|
|
38
|
+
self.videos = [videos] if isinstance(videos, str) else list(videos)
|
|
39
|
+
self.language = language
|
|
40
|
+
self.timestamps = timestamps
|
|
41
|
+
key = SecretStr(api_key) if api_key else secret_from_env(API_KEY_ENV, default=None)()
|
|
42
|
+
self._client = make_client(key)
|
|
43
|
+
|
|
44
|
+
def lazy_load(self) -> Iterator[Document]:
|
|
45
|
+
for video in self.videos:
|
|
46
|
+
result = self._client.get_transcript(video, language=self.language, timestamps=self.timestamps)
|
|
47
|
+
text = to_timed_text(result) if self.timestamps and result.get("segments") else result.get("transcript", "")
|
|
48
|
+
metadata = {
|
|
49
|
+
"source": f"https://www.youtube.com/watch?v={result['video_id']}",
|
|
50
|
+
"video_id": result["video_id"],
|
|
51
|
+
"title": result.get("title", ""),
|
|
52
|
+
"author_name": result.get("author_name", ""),
|
|
53
|
+
"language_code": result.get("language_code", ""),
|
|
54
|
+
"word_count": result.get("word_count", 0),
|
|
55
|
+
}
|
|
56
|
+
if self.timestamps and result.get("segments"):
|
|
57
|
+
metadata["segments"] = result["segments"]
|
|
58
|
+
yield Document(page_content=text, metadata=metadata)
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""LangChain tools for the GetYouTubeTranscript API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from typing import Literal
|
|
7
|
+
|
|
8
|
+
from getyoutubetranscript import Client, GetYouTubeTranscriptError
|
|
9
|
+
from langchain_core.tools import BaseTool, ToolException
|
|
10
|
+
from langchain_core.utils import secret_from_env
|
|
11
|
+
from pydantic import BaseModel, Field, PrivateAttr, SecretStr
|
|
12
|
+
|
|
13
|
+
from ._client import API_KEY_ENV, format_transcript, make_client
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class _ApiKeyMixin(BaseModel):
|
|
17
|
+
api_key: SecretStr | None = Field(default_factory=secret_from_env(API_KEY_ENV, default=None))
|
|
18
|
+
"""GetYouTubeTranscript API key. Defaults to the ``GETYOUTUBETRANSCRIPT_API_KEY`` env var."""
|
|
19
|
+
|
|
20
|
+
_client: Client | None = PrivateAttr(default=None)
|
|
21
|
+
|
|
22
|
+
@property
|
|
23
|
+
def client(self) -> Client:
|
|
24
|
+
if self._client is None:
|
|
25
|
+
self._client = make_client(self.api_key)
|
|
26
|
+
return self._client
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class TranscriptInput(BaseModel):
|
|
30
|
+
video: str = Field(description="YouTube video URL (watch, youtu.be, Shorts or live) or the 11-character video ID.")
|
|
31
|
+
language: str | None = Field(default=None, description="Optional caption language code, e.g. 'en' or 'es'.")
|
|
32
|
+
timestamps: bool = Field(
|
|
33
|
+
default=False,
|
|
34
|
+
description="Set to true to prefix each caption line with its start time, e.g. [1:05], for quoting or linking moments.",
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class GetYouTubeTranscriptTool(_ApiKeyMixin, BaseTool): # type: ignore[override]
|
|
39
|
+
"""Get the transcript of a YouTube video.
|
|
40
|
+
|
|
41
|
+
Setup:
|
|
42
|
+
``pip install langchain-getyoutubetranscript`` and set ``GETYOUTUBETRANSCRIPT_API_KEY``.
|
|
43
|
+
|
|
44
|
+
Invoke:
|
|
45
|
+
.. code-block:: python
|
|
46
|
+
|
|
47
|
+
tool = GetYouTubeTranscriptTool()
|
|
48
|
+
tool.invoke({"video": "https://youtu.be/jNQXAC9IVRw", "timestamps": True})
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
name: str = "youtube_transcript"
|
|
52
|
+
description: str = (
|
|
53
|
+
"Get the transcript (spoken text) of a YouTube video from its URL or video ID, with the title and channel. "
|
|
54
|
+
"Use it to summarize, quote or answer questions about a video."
|
|
55
|
+
)
|
|
56
|
+
args_schema: type[BaseModel] = TranscriptInput
|
|
57
|
+
handle_tool_error: bool = True
|
|
58
|
+
|
|
59
|
+
def _run(self, video: str, language: str | None = None, timestamps: bool = False, **kwargs: object) -> str:
|
|
60
|
+
try:
|
|
61
|
+
result = self.client.get_transcript(video, language=language, timestamps=timestamps)
|
|
62
|
+
except GetYouTubeTranscriptError as e:
|
|
63
|
+
raise ToolException(e.message) from e
|
|
64
|
+
return format_transcript(result, timestamps)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class SearchInput(BaseModel):
|
|
68
|
+
query: str = Field(description="What to search YouTube for.")
|
|
69
|
+
type: Literal["video", "channel"] = Field(default="video", description="Search for videos or channels.")
|
|
70
|
+
page_token: str | None = Field(
|
|
71
|
+
default=None, description="continuation_token from a previous search result, to get the next page."
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class GetYouTubeTranscriptSearchTool(_ApiKeyMixin, BaseTool): # type: ignore[override]
|
|
76
|
+
"""Search YouTube for videos or channels.
|
|
77
|
+
|
|
78
|
+
Invoke:
|
|
79
|
+
.. code-block:: python
|
|
80
|
+
|
|
81
|
+
GetYouTubeTranscriptSearchTool().invoke({"query": "linear algebra lecture"})
|
|
82
|
+
"""
|
|
83
|
+
|
|
84
|
+
name: str = "youtube_search"
|
|
85
|
+
description: str = (
|
|
86
|
+
"Search YouTube for videos or channels. Returns JSON with titles, video or channel IDs, links, channel, "
|
|
87
|
+
"views, length and upload date, plus a continuation_token for the next page."
|
|
88
|
+
)
|
|
89
|
+
args_schema: type[BaseModel] = SearchInput
|
|
90
|
+
handle_tool_error: bool = True
|
|
91
|
+
|
|
92
|
+
def _run(self, query: str, type: str = "video", page_token: str | None = None, **kwargs: object) -> str:
|
|
93
|
+
try:
|
|
94
|
+
result = self.client.search(query, type=type, page_token=page_token)
|
|
95
|
+
except GetYouTubeTranscriptError as e:
|
|
96
|
+
raise ToolException(e.message) from e
|
|
97
|
+
return json.dumps(result, ensure_ascii=False)
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: langchain-getyoutubetranscript
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: YouTube transcript tools and document loader for LangChain: get YouTube video transcripts with timestamps and search YouTube from agents and RAG pipelines, via the GetYouTubeTranscript API.
|
|
5
|
+
Project-URL: Homepage, https://getyoutubetranscript.com
|
|
6
|
+
Project-URL: Documentation, https://github.com/tubeagentkit/langchain-getyoutubetranscript#readme
|
|
7
|
+
Project-URL: Repository, https://github.com/tubeagentkit/langchain-getyoutubetranscript
|
|
8
|
+
Project-URL: API reference, https://getyoutubetranscript.com/docs
|
|
9
|
+
Project-URL: Get an API key, https://getyoutubetranscript.com/dashboard
|
|
10
|
+
Author: tubeagentkit
|
|
11
|
+
License: MIT
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Keywords: agents,captions,document-loader,langchain,llm,rag,subtitles,tools,transcript,youtube,youtube-transcript
|
|
14
|
+
Classifier: Development Status :: 4 - Beta
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Topic :: Multimedia :: Video
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Requires-Python: >=3.10
|
|
25
|
+
Requires-Dist: getyoutubetranscript<1.0.0,>=0.3.0
|
|
26
|
+
Requires-Dist: langchain-core<2.0.0,>=0.3.0
|
|
27
|
+
Provides-Extra: test
|
|
28
|
+
Requires-Dist: langchain-tests>=0.3.0; extra == 'test'
|
|
29
|
+
Requires-Dist: pytest>=7.0; extra == 'test'
|
|
30
|
+
Description-Content-Type: text/markdown
|
|
31
|
+
|
|
32
|
+
# langchain-getyoutubetranscript
|
|
33
|
+
|
|
34
|
+
YouTube transcript tools and a document loader for [LangChain](https://www.langchain.com), powered by the [GetYouTubeTranscript](https://getyoutubetranscript.com) API. Give agents the transcript of any YouTube video (optionally with `[m:ss]` timestamps) and YouTube search, or load transcripts into RAG pipelines.
|
|
35
|
+
|
|
36
|
+
The API fetches transcripts on its own servers, so it works from cloud servers and serverless functions without proxies, and without `RequestBlocked` / `IpBlocked` errors.
|
|
37
|
+
|
|
38
|
+
## Install
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
pip install langchain-getyoutubetranscript
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Get an API key at [getyoutubetranscript.com/dashboard](https://getyoutubetranscript.com/dashboard) (free tier included) and set it:
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
export GETYOUTUBETRANSCRIPT_API_KEY=sk_live_...
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Or pass `api_key="..."` to any class.
|
|
51
|
+
|
|
52
|
+
## Tools
|
|
53
|
+
|
|
54
|
+
| Class | Tool name | What it does |
|
|
55
|
+
| --- | --- | --- |
|
|
56
|
+
| `GetYouTubeTranscriptTool` | `youtube_transcript` | Transcript of a video (URL or ID) with title and channel. `timestamps=True` prefixes each line with `[m:ss]`. |
|
|
57
|
+
| `GetYouTubeTranscriptSearchTool` | `youtube_search` | Search YouTube for videos or channels. Returns JSON with a `continuation_token` for the next page. |
|
|
58
|
+
|
|
59
|
+
```python
|
|
60
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptTool
|
|
61
|
+
|
|
62
|
+
tool = GetYouTubeTranscriptTool()
|
|
63
|
+
print(tool.invoke({"video": "https://youtu.be/jNQXAC9IVRw", "timestamps": True}))
|
|
64
|
+
# Title: Me at the zoo
|
|
65
|
+
# Channel: jawed
|
|
66
|
+
# Language: en
|
|
67
|
+
#
|
|
68
|
+
# [0:01] All right, so here we are, in front of the elephants
|
|
69
|
+
# ...
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
API errors (no captions, invalid video, out of credits) come back to the agent as a short message instead of raising.
|
|
73
|
+
|
|
74
|
+
### With an agent
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
from langchain.agents import create_agent
|
|
78
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptSearchTool, GetYouTubeTranscriptTool
|
|
79
|
+
|
|
80
|
+
agent = create_agent(
|
|
81
|
+
model="anthropic:claude-sonnet-4-5",
|
|
82
|
+
tools=[GetYouTubeTranscriptSearchTool(), GetYouTubeTranscriptTool()],
|
|
83
|
+
)
|
|
84
|
+
agent.invoke(
|
|
85
|
+
{"messages": [{"role": "user", "content": "Find a short talk on transformers and summarize it with timestamps."}]}
|
|
86
|
+
)
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
## Document loader
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
from langchain_getyoutubetranscript import GetYouTubeTranscriptLoader
|
|
93
|
+
|
|
94
|
+
docs = GetYouTubeTranscriptLoader(
|
|
95
|
+
["https://youtu.be/jNQXAC9IVRw", "5e37ZT3SQbk"],
|
|
96
|
+
language="en",
|
|
97
|
+
timestamps=False,
|
|
98
|
+
).load()
|
|
99
|
+
|
|
100
|
+
print(docs[0].metadata)
|
|
101
|
+
# {'source': 'https://www.youtube.com/watch?v=jNQXAC9IVRw', 'video_id': 'jNQXAC9IVRw',
|
|
102
|
+
# 'title': 'Me at the zoo', 'author_name': 'jawed', 'language_code': 'en', 'word_count': 39}
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
One `Document` per video. With `timestamps=True`, `page_content` is `[m:ss]` lines and `metadata["segments"]` holds `{start, duration, text}` per caption line (seconds).
|
|
106
|
+
|
|
107
|
+
## Pricing
|
|
108
|
+
|
|
109
|
+
Each transcript or search request uses one credit from your GetYouTubeTranscript account. Failed requests are not charged.
|
|
110
|
+
|
|
111
|
+
## Development
|
|
112
|
+
|
|
113
|
+
```bash
|
|
114
|
+
pip install -e ".[test]"
|
|
115
|
+
pytest # unit + LangChain standard tests, no network
|
|
116
|
+
GETYOUTUBETRANSCRIPT_API_KEY=... pytest tests/integration_tests # live, spends credits
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
## Links
|
|
120
|
+
|
|
121
|
+
- [GetYouTubeTranscript API docs](https://getyoutubetranscript.com/docs)
|
|
122
|
+
- [Python SDK](https://pypi.org/project/getyoutubetranscript/) (this package is built on it)
|
|
123
|
+
- [License: MIT](LICENSE)
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
langchain_getyoutubetranscript/__init__.py,sha256=p1klBbJkwmkW8vk2WQKW0oJ0mCWLtasCnzh3zMpO7a4,392
|
|
2
|
+
langchain_getyoutubetranscript/_client.py,sha256=Hkmu8rFdduFGexcrjrPTPKTg9PT6EFs8_NDzA9Ue4nk,1106
|
|
3
|
+
langchain_getyoutubetranscript/document_loaders.py,sha256=q6WNNSDXMVUCXxqK4xCi-ch-jbafUZnihWjiW0ZFtbI,2379
|
|
4
|
+
langchain_getyoutubetranscript/tools.py,sha256=jFDpfwPtfTYalsqHXTKt2JCsLRUu2GjpjWTdXyiCO40,3816
|
|
5
|
+
langchain_getyoutubetranscript-0.1.0.dist-info/METADATA,sha256=16nCMZ12WJPwhQpC744m7WpfK0_3e6UHsNDXOjn9FQM,4907
|
|
6
|
+
langchain_getyoutubetranscript-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
7
|
+
langchain_getyoutubetranscript-0.1.0.dist-info/licenses/LICENSE,sha256=9BV44fzXqxv9egaqFNr-TvzhlWpNBTVX8TqxF6bIWek,1069
|
|
8
|
+
langchain_getyoutubetranscript-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 tubeagentkit
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|