langchain-getyoutubetranscript 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,7 @@
1
+ dist/
2
+ __pycache__/
3
+ *.egg-info/
4
+ .venv/
5
+ .pytest_cache/
6
+ .ruff_cache/
7
+ .mypy_cache/
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 tubeagentkit
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,123 @@
1
+ Metadata-Version: 2.5
2
+ Name: langchain-getyoutubetranscript
3
+ Version: 0.1.0
4
+ Summary: YouTube transcript tools and document loader for LangChain: get YouTube video transcripts with timestamps and search YouTube from agents and RAG pipelines, via the GetYouTubeTranscript API.
5
+ Project-URL: Homepage, https://getyoutubetranscript.com
6
+ Project-URL: Documentation, https://github.com/tubeagentkit/langchain-getyoutubetranscript#readme
7
+ Project-URL: Repository, https://github.com/tubeagentkit/langchain-getyoutubetranscript
8
+ Project-URL: API reference, https://getyoutubetranscript.com/docs
9
+ Project-URL: Get an API key, https://getyoutubetranscript.com/dashboard
10
+ Author: tubeagentkit
11
+ License: MIT
12
+ License-File: LICENSE
13
+ Keywords: agents,captions,document-loader,langchain,llm,rag,subtitles,tools,transcript,youtube,youtube-transcript
14
+ Classifier: Development Status :: 4 - Beta
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: License :: OSI Approved :: MIT License
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Programming Language :: Python :: 3.13
22
+ Classifier: Topic :: Multimedia :: Video
23
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
24
+ Requires-Python: >=3.10
25
+ Requires-Dist: getyoutubetranscript<1.0.0,>=0.3.0
26
+ Requires-Dist: langchain-core<2.0.0,>=0.3.0
27
+ Provides-Extra: test
28
+ Requires-Dist: langchain-tests>=0.3.0; extra == 'test'
29
+ Requires-Dist: pytest>=7.0; extra == 'test'
30
+ Description-Content-Type: text/markdown
31
+
32
+ # langchain-getyoutubetranscript
33
+
34
+ YouTube transcript tools and a document loader for [LangChain](https://www.langchain.com), powered by the [GetYouTubeTranscript](https://getyoutubetranscript.com) API. Give agents the transcript of any YouTube video (optionally with `[m:ss]` timestamps) and YouTube search, or load transcripts into RAG pipelines.
35
+
36
+ The API fetches transcripts on its own servers, so it works from cloud servers and serverless functions without proxies, and without `RequestBlocked` / `IpBlocked` errors.
37
+
38
+ ## Install
39
+
40
+ ```bash
41
+ pip install langchain-getyoutubetranscript
42
+ ```
43
+
44
+ Get an API key at [getyoutubetranscript.com/dashboard](https://getyoutubetranscript.com/dashboard) (free tier included) and set it:
45
+
46
+ ```bash
47
+ export GETYOUTUBETRANSCRIPT_API_KEY=sk_live_...
48
+ ```
49
+
50
+ Or pass `api_key="..."` to any class.
51
+
52
+ ## Tools
53
+
54
+ | Class | Tool name | What it does |
55
+ | --- | --- | --- |
56
+ | `GetYouTubeTranscriptTool` | `youtube_transcript` | Transcript of a video (URL or ID) with title and channel. `timestamps=True` prefixes each line with `[m:ss]`. |
57
+ | `GetYouTubeTranscriptSearchTool` | `youtube_search` | Search YouTube for videos or channels. Returns JSON with a `continuation_token` for the next page. |
58
+
59
+ ```python
60
+ from langchain_getyoutubetranscript import GetYouTubeTranscriptTool
61
+
62
+ tool = GetYouTubeTranscriptTool()
63
+ print(tool.invoke({"video": "https://youtu.be/jNQXAC9IVRw", "timestamps": True}))
64
+ # Title: Me at the zoo
65
+ # Channel: jawed
66
+ # Language: en
67
+ #
68
+ # [0:01] All right, so here we are, in front of the elephants
69
+ # ...
70
+ ```
71
+
72
+ API errors (no captions, invalid video, out of credits) come back to the agent as a short message instead of raising.
73
+
74
+ ### With an agent
75
+
76
+ ```python
77
+ from langchain.agents import create_agent
78
+ from langchain_getyoutubetranscript import GetYouTubeTranscriptSearchTool, GetYouTubeTranscriptTool
79
+
80
+ agent = create_agent(
81
+ model="anthropic:claude-sonnet-4-5",
82
+ tools=[GetYouTubeTranscriptSearchTool(), GetYouTubeTranscriptTool()],
83
+ )
84
+ agent.invoke(
85
+ {"messages": [{"role": "user", "content": "Find a short talk on transformers and summarize it with timestamps."}]}
86
+ )
87
+ ```
88
+
89
+ ## Document loader
90
+
91
+ ```python
92
+ from langchain_getyoutubetranscript import GetYouTubeTranscriptLoader
93
+
94
+ docs = GetYouTubeTranscriptLoader(
95
+ ["https://youtu.be/jNQXAC9IVRw", "5e37ZT3SQbk"],
96
+ language="en",
97
+ timestamps=False,
98
+ ).load()
99
+
100
+ print(docs[0].metadata)
101
+ # {'source': 'https://www.youtube.com/watch?v=jNQXAC9IVRw', 'video_id': 'jNQXAC9IVRw',
102
+ # 'title': 'Me at the zoo', 'author_name': 'jawed', 'language_code': 'en', 'word_count': 39}
103
+ ```
104
+
105
+ One `Document` per video. With `timestamps=True`, `page_content` is `[m:ss]` lines and `metadata["segments"]` holds `{start, duration, text}` per caption line (seconds).
106
+
107
+ ## Pricing
108
+
109
+ Each transcript or search request uses one credit from your GetYouTubeTranscript account. Failed requests are not charged.
110
+
111
+ ## Development
112
+
113
+ ```bash
114
+ pip install -e ".[test]"
115
+ pytest # unit + LangChain standard tests, no network
116
+ GETYOUTUBETRANSCRIPT_API_KEY=... pytest tests/integration_tests # live, spends credits
117
+ ```
118
+
119
+ ## Links
120
+
121
+ - [GetYouTubeTranscript API docs](https://getyoutubetranscript.com/docs)
122
+ - [Python SDK](https://pypi.org/project/getyoutubetranscript/) (this package is built on it)
123
+ - [License: MIT](LICENSE)
@@ -0,0 +1,92 @@
1
+ # langchain-getyoutubetranscript
2
+
3
+ YouTube transcript tools and a document loader for [LangChain](https://www.langchain.com), powered by the [GetYouTubeTranscript](https://getyoutubetranscript.com) API. Give agents the transcript of any YouTube video (optionally with `[m:ss]` timestamps) and YouTube search, or load transcripts into RAG pipelines.
4
+
5
+ The API fetches transcripts on its own servers, so it works from cloud servers and serverless functions without proxies, and without `RequestBlocked` / `IpBlocked` errors.
6
+
7
+ ## Install
8
+
9
+ ```bash
10
+ pip install langchain-getyoutubetranscript
11
+ ```
12
+
13
+ Get an API key at [getyoutubetranscript.com/dashboard](https://getyoutubetranscript.com/dashboard) (free tier included) and set it:
14
+
15
+ ```bash
16
+ export GETYOUTUBETRANSCRIPT_API_KEY=sk_live_...
17
+ ```
18
+
19
+ Or pass `api_key="..."` to any class.
20
+
21
+ ## Tools
22
+
23
+ | Class | Tool name | What it does |
24
+ | --- | --- | --- |
25
+ | `GetYouTubeTranscriptTool` | `youtube_transcript` | Transcript of a video (URL or ID) with title and channel. `timestamps=True` prefixes each line with `[m:ss]`. |
26
+ | `GetYouTubeTranscriptSearchTool` | `youtube_search` | Search YouTube for videos or channels. Returns JSON with a `continuation_token` for the next page. |
27
+
28
+ ```python
29
+ from langchain_getyoutubetranscript import GetYouTubeTranscriptTool
30
+
31
+ tool = GetYouTubeTranscriptTool()
32
+ print(tool.invoke({"video": "https://youtu.be/jNQXAC9IVRw", "timestamps": True}))
33
+ # Title: Me at the zoo
34
+ # Channel: jawed
35
+ # Language: en
36
+ #
37
+ # [0:01] All right, so here we are, in front of the elephants
38
+ # ...
39
+ ```
40
+
41
+ API errors (no captions, invalid video, out of credits) come back to the agent as a short message instead of raising.
42
+
43
+ ### With an agent
44
+
45
+ ```python
46
+ from langchain.agents import create_agent
47
+ from langchain_getyoutubetranscript import GetYouTubeTranscriptSearchTool, GetYouTubeTranscriptTool
48
+
49
+ agent = create_agent(
50
+ model="anthropic:claude-sonnet-4-5",
51
+ tools=[GetYouTubeTranscriptSearchTool(), GetYouTubeTranscriptTool()],
52
+ )
53
+ agent.invoke(
54
+ {"messages": [{"role": "user", "content": "Find a short talk on transformers and summarize it with timestamps."}]}
55
+ )
56
+ ```
57
+
58
+ ## Document loader
59
+
60
+ ```python
61
+ from langchain_getyoutubetranscript import GetYouTubeTranscriptLoader
62
+
63
+ docs = GetYouTubeTranscriptLoader(
64
+ ["https://youtu.be/jNQXAC9IVRw", "5e37ZT3SQbk"],
65
+ language="en",
66
+ timestamps=False,
67
+ ).load()
68
+
69
+ print(docs[0].metadata)
70
+ # {'source': 'https://www.youtube.com/watch?v=jNQXAC9IVRw', 'video_id': 'jNQXAC9IVRw',
71
+ # 'title': 'Me at the zoo', 'author_name': 'jawed', 'language_code': 'en', 'word_count': 39}
72
+ ```
73
+
74
+ One `Document` per video. With `timestamps=True`, `page_content` is `[m:ss]` lines and `metadata["segments"]` holds `{start, duration, text}` per caption line (seconds).
75
+
76
+ ## Pricing
77
+
78
+ Each transcript or search request uses one credit from your GetYouTubeTranscript account. Failed requests are not charged.
79
+
80
+ ## Development
81
+
82
+ ```bash
83
+ pip install -e ".[test]"
84
+ pytest # unit + LangChain standard tests, no network
85
+ GETYOUTUBETRANSCRIPT_API_KEY=... pytest tests/integration_tests # live, spends credits
86
+ ```
87
+
88
+ ## Links
89
+
90
+ - [GetYouTubeTranscript API docs](https://getyoutubetranscript.com/docs)
91
+ - [Python SDK](https://pypi.org/project/getyoutubetranscript/) (this package is built on it)
92
+ - [License: MIT](LICENSE)
@@ -0,0 +1,6 @@
1
+ """LangChain integration for the GetYouTubeTranscript API: YouTube transcripts and search."""
2
+
3
+ from langchain_getyoutubetranscript.document_loaders import GetYouTubeTranscriptLoader
4
+ from langchain_getyoutubetranscript.tools import GetYouTubeTranscriptSearchTool, GetYouTubeTranscriptTool
5
+
6
+ __all__ = ["GetYouTubeTranscriptLoader", "GetYouTubeTranscriptSearchTool", "GetYouTubeTranscriptTool"]
@@ -0,0 +1,27 @@
1
+ """Shared API key handling for the tools and the loader."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping
6
+ from typing import Any
7
+
8
+ from getyoutubetranscript import Client, to_timed_text
9
+ from pydantic import SecretStr
10
+
11
+ API_KEY_ENV = "GETYOUTUBETRANSCRIPT_API_KEY"
12
+
13
+
14
+ def make_client(api_key: SecretStr | None) -> Client:
15
+ if api_key is None or not api_key.get_secret_value():
16
+ raise ValueError(
17
+ f"A GetYouTubeTranscript API key is required. Pass api_key=... or set {API_KEY_ENV}. "
18
+ "Get one at https://getyoutubetranscript.com/dashboard"
19
+ )
20
+ return Client(api_key=api_key.get_secret_value())
21
+
22
+
23
+ def format_transcript(result: Mapping[str, Any], timestamps: bool) -> str:
24
+ """Title and channel header, then the transcript (``[m:ss]`` lines when timestamps)."""
25
+ body = to_timed_text(result) if timestamps and result.get("segments") else result.get("transcript", "")
26
+ header = f"Title: {result.get('title', '')}\nChannel: {result.get('author_name', '')}\nLanguage: {result.get('language_code', '')}"
27
+ return f"{header}\n\n{body}"
@@ -0,0 +1,58 @@
1
+ """Load YouTube transcripts as LangChain Documents."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Iterator, Sequence
6
+
7
+ from getyoutubetranscript import to_timed_text
8
+ from langchain_core.document_loaders import BaseLoader
9
+ from langchain_core.documents import Document
10
+ from langchain_core.utils import secret_from_env
11
+ from pydantic import SecretStr
12
+
13
+ from ._client import API_KEY_ENV, make_client
14
+
15
+
16
+ class GetYouTubeTranscriptLoader(BaseLoader):
17
+ """Load the transcripts of one or more YouTube videos, one Document per video.
18
+
19
+ .. code-block:: python
20
+
21
+ from langchain_getyoutubetranscript import GetYouTubeTranscriptLoader
22
+
23
+ docs = GetYouTubeTranscriptLoader(["https://youtu.be/jNQXAC9IVRw"]).load()
24
+
25
+ ``page_content`` is the transcript text (``[m:ss]`` lines when ``timestamps=True``).
26
+ ``metadata`` has ``source``, ``video_id``, ``title``, ``author_name``,
27
+ ``language_code`` and ``word_count``, plus ``segments`` when ``timestamps=True``.
28
+ """
29
+
30
+ def __init__(
31
+ self,
32
+ videos: str | Sequence[str],
33
+ *,
34
+ language: str | None = None,
35
+ timestamps: bool = False,
36
+ api_key: str | None = None,
37
+ ) -> None:
38
+ self.videos = [videos] if isinstance(videos, str) else list(videos)
39
+ self.language = language
40
+ self.timestamps = timestamps
41
+ key = SecretStr(api_key) if api_key else secret_from_env(API_KEY_ENV, default=None)()
42
+ self._client = make_client(key)
43
+
44
+ def lazy_load(self) -> Iterator[Document]:
45
+ for video in self.videos:
46
+ result = self._client.get_transcript(video, language=self.language, timestamps=self.timestamps)
47
+ text = to_timed_text(result) if self.timestamps and result.get("segments") else result.get("transcript", "")
48
+ metadata = {
49
+ "source": f"https://www.youtube.com/watch?v={result['video_id']}",
50
+ "video_id": result["video_id"],
51
+ "title": result.get("title", ""),
52
+ "author_name": result.get("author_name", ""),
53
+ "language_code": result.get("language_code", ""),
54
+ "word_count": result.get("word_count", 0),
55
+ }
56
+ if self.timestamps and result.get("segments"):
57
+ metadata["segments"] = result["segments"]
58
+ yield Document(page_content=text, metadata=metadata)
@@ -0,0 +1,97 @@
1
+ """LangChain tools for the GetYouTubeTranscript API."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from typing import Literal
7
+
8
+ from getyoutubetranscript import Client, GetYouTubeTranscriptError
9
+ from langchain_core.tools import BaseTool, ToolException
10
+ from langchain_core.utils import secret_from_env
11
+ from pydantic import BaseModel, Field, PrivateAttr, SecretStr
12
+
13
+ from ._client import API_KEY_ENV, format_transcript, make_client
14
+
15
+
16
+ class _ApiKeyMixin(BaseModel):
17
+ api_key: SecretStr | None = Field(default_factory=secret_from_env(API_KEY_ENV, default=None))
18
+ """GetYouTubeTranscript API key. Defaults to the ``GETYOUTUBETRANSCRIPT_API_KEY`` env var."""
19
+
20
+ _client: Client | None = PrivateAttr(default=None)
21
+
22
+ @property
23
+ def client(self) -> Client:
24
+ if self._client is None:
25
+ self._client = make_client(self.api_key)
26
+ return self._client
27
+
28
+
29
+ class TranscriptInput(BaseModel):
30
+ video: str = Field(description="YouTube video URL (watch, youtu.be, Shorts or live) or the 11-character video ID.")
31
+ language: str | None = Field(default=None, description="Optional caption language code, e.g. 'en' or 'es'.")
32
+ timestamps: bool = Field(
33
+ default=False,
34
+ description="Set to true to prefix each caption line with its start time, e.g. [1:05], for quoting or linking moments.",
35
+ )
36
+
37
+
38
+ class GetYouTubeTranscriptTool(_ApiKeyMixin, BaseTool): # type: ignore[override]
39
+ """Get the transcript of a YouTube video.
40
+
41
+ Setup:
42
+ ``pip install langchain-getyoutubetranscript`` and set ``GETYOUTUBETRANSCRIPT_API_KEY``.
43
+
44
+ Invoke:
45
+ .. code-block:: python
46
+
47
+ tool = GetYouTubeTranscriptTool()
48
+ tool.invoke({"video": "https://youtu.be/jNQXAC9IVRw", "timestamps": True})
49
+ """
50
+
51
+ name: str = "youtube_transcript"
52
+ description: str = (
53
+ "Get the transcript (spoken text) of a YouTube video from its URL or video ID, with the title and channel. "
54
+ "Use it to summarize, quote or answer questions about a video."
55
+ )
56
+ args_schema: type[BaseModel] = TranscriptInput
57
+ handle_tool_error: bool = True
58
+
59
+ def _run(self, video: str, language: str | None = None, timestamps: bool = False, **kwargs: object) -> str:
60
+ try:
61
+ result = self.client.get_transcript(video, language=language, timestamps=timestamps)
62
+ except GetYouTubeTranscriptError as e:
63
+ raise ToolException(e.message) from e
64
+ return format_transcript(result, timestamps)
65
+
66
+
67
+ class SearchInput(BaseModel):
68
+ query: str = Field(description="What to search YouTube for.")
69
+ type: Literal["video", "channel"] = Field(default="video", description="Search for videos or channels.")
70
+ page_token: str | None = Field(
71
+ default=None, description="continuation_token from a previous search result, to get the next page."
72
+ )
73
+
74
+
75
+ class GetYouTubeTranscriptSearchTool(_ApiKeyMixin, BaseTool): # type: ignore[override]
76
+ """Search YouTube for videos or channels.
77
+
78
+ Invoke:
79
+ .. code-block:: python
80
+
81
+ GetYouTubeTranscriptSearchTool().invoke({"query": "linear algebra lecture"})
82
+ """
83
+
84
+ name: str = "youtube_search"
85
+ description: str = (
86
+ "Search YouTube for videos or channels. Returns JSON with titles, video or channel IDs, links, channel, "
87
+ "views, length and upload date, plus a continuation_token for the next page."
88
+ )
89
+ args_schema: type[BaseModel] = SearchInput
90
+ handle_tool_error: bool = True
91
+
92
+ def _run(self, query: str, type: str = "video", page_token: str | None = None, **kwargs: object) -> str:
93
+ try:
94
+ result = self.client.search(query, type=type, page_token=page_token)
95
+ except GetYouTubeTranscriptError as e:
96
+ raise ToolException(e.message) from e
97
+ return json.dumps(result, ensure_ascii=False)
@@ -0,0 +1,48 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "langchain-getyoutubetranscript"
7
+ version = "0.1.0"
8
+ description = "YouTube transcript tools and document loader for LangChain: get YouTube video transcripts with timestamps and search YouTube from agents and RAG pipelines, via the GetYouTubeTranscript API."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = { text = "MIT" }
12
+ authors = [{ name = "tubeagentkit" }]
13
+ keywords = ["langchain", "youtube", "youtube-transcript", "transcript", "captions", "subtitles", "agents", "rag", "llm", "tools", "document-loader"]
14
+ classifiers = [
15
+ "Development Status :: 4 - Beta",
16
+ "Intended Audience :: Developers",
17
+ "License :: OSI Approved :: MIT License",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3.10",
20
+ "Programming Language :: Python :: 3.11",
21
+ "Programming Language :: Python :: 3.12",
22
+ "Programming Language :: Python :: 3.13",
23
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
24
+ "Topic :: Multimedia :: Video",
25
+ ]
26
+ dependencies = [
27
+ "langchain-core>=0.3.0,<2.0.0",
28
+ "getyoutubetranscript>=0.3.0,<1.0.0",
29
+ ]
30
+
31
+ [project.optional-dependencies]
32
+ test = ["pytest>=7.0", "langchain-tests>=0.3.0"]
33
+
34
+ [project.urls]
35
+ Homepage = "https://getyoutubetranscript.com"
36
+ Documentation = "https://github.com/tubeagentkit/langchain-getyoutubetranscript#readme"
37
+ Repository = "https://github.com/tubeagentkit/langchain-getyoutubetranscript"
38
+ "API reference" = "https://getyoutubetranscript.com/docs"
39
+ "Get an API key" = "https://getyoutubetranscript.com/dashboard"
40
+
41
+ [tool.hatch.build.targets.wheel]
42
+ packages = ["langchain_getyoutubetranscript"]
43
+
44
+ [tool.ruff]
45
+ line-length = 120
46
+
47
+ [tool.pytest.ini_options]
48
+ testpaths = ["tests/unit_tests"]
File without changes
@@ -0,0 +1,40 @@
1
+ """Live tests against the real API. Skipped unless GETYOUTUBETRANSCRIPT_API_KEY is set; each run spends credits."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import os
7
+
8
+ import pytest
9
+
10
+ from langchain_getyoutubetranscript import (
11
+ GetYouTubeTranscriptLoader,
12
+ GetYouTubeTranscriptSearchTool,
13
+ GetYouTubeTranscriptTool,
14
+ )
15
+
16
+ pytestmark = pytest.mark.skipif(not os.environ.get("GETYOUTUBETRANSCRIPT_API_KEY"), reason="no API key")
17
+ VIDEO = "https://youtu.be/5e37ZT3SQbk"
18
+
19
+
20
+ def test_tool_timestamps():
21
+ out = GetYouTubeTranscriptTool().invoke({"video": VIDEO, "timestamps": True})
22
+ lines = [line for line in out.splitlines() if line.startswith("[")]
23
+ assert out.startswith("Title: ")
24
+ assert len(lines) > 400
25
+
26
+
27
+ def test_tool_bad_video_returns_message():
28
+ out = GetYouTubeTranscriptTool().invoke({"video": "notavalidvideo"})
29
+ assert "YouTube video" in out
30
+
31
+
32
+ def test_search_tool():
33
+ data = json.loads(GetYouTubeTranscriptSearchTool().invoke({"query": "linear algebra lecture"}))
34
+ assert data["video_results"]
35
+
36
+
37
+ def test_loader():
38
+ doc = GetYouTubeTranscriptLoader(VIDEO).load()[0]
39
+ assert doc.metadata["video_id"] == "5e37ZT3SQbk"
40
+ assert len(doc.page_content) > 10_000
@@ -0,0 +1,54 @@
1
+ from __future__ import annotations
2
+
3
+ from unittest.mock import MagicMock
4
+
5
+ import pytest
6
+
7
+ from langchain_getyoutubetranscript import GetYouTubeTranscriptLoader
8
+
9
+ RESULT = {
10
+ "video_id": "jNQXAC9IVRw",
11
+ "language_code": "en",
12
+ "title": "Me at the zoo",
13
+ "author_name": "jawed",
14
+ "transcript": "All right, so here we are",
15
+ "word_count": 6,
16
+ "segments": [{"start": 61.0, "duration": 2.0, "text": "All right, so here we are"}],
17
+ }
18
+
19
+
20
+ def _loader(**kwargs) -> tuple[GetYouTubeTranscriptLoader, MagicMock]:
21
+ loader = GetYouTubeTranscriptLoader(api_key="test-key", **kwargs)
22
+ client = MagicMock()
23
+ client.get_transcript.return_value = RESULT
24
+ loader._client = client
25
+ return loader, client
26
+
27
+
28
+ def test_one_document_per_video_with_metadata():
29
+ loader, client = _loader(videos=["a", "b"], language="en")
30
+ docs = loader.load()
31
+ assert len(docs) == 2
32
+ assert client.get_transcript.call_args.kwargs == {"language": "en", "timestamps": False}
33
+ assert docs[0].page_content == "All right, so here we are"
34
+ assert docs[0].metadata == {
35
+ "source": "https://www.youtube.com/watch?v=jNQXAC9IVRw",
36
+ "video_id": "jNQXAC9IVRw",
37
+ "title": "Me at the zoo",
38
+ "author_name": "jawed",
39
+ "language_code": "en",
40
+ "word_count": 6,
41
+ }
42
+
43
+
44
+ def test_timestamps_give_timed_text_and_segments():
45
+ loader, _ = _loader(videos="a", timestamps=True)
46
+ doc = loader.load()[0]
47
+ assert doc.page_content == "[1:01] All right, so here we are"
48
+ assert doc.metadata["segments"] == RESULT["segments"]
49
+
50
+
51
+ def test_missing_key(monkeypatch):
52
+ monkeypatch.delenv("GETYOUTUBETRANSCRIPT_API_KEY", raising=False)
53
+ with pytest.raises(ValueError, match="API key is required"):
54
+ GetYouTubeTranscriptLoader("a")
@@ -0,0 +1,98 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from unittest.mock import MagicMock
5
+
6
+ import pytest
7
+ from getyoutubetranscript import GetYouTubeTranscriptError
8
+ from langchain_tests.unit_tests import ToolsUnitTests
9
+
10
+ from langchain_getyoutubetranscript import GetYouTubeTranscriptSearchTool, GetYouTubeTranscriptTool
11
+
12
+ RESULT = {
13
+ "video_id": "jNQXAC9IVRw",
14
+ "language_code": "en",
15
+ "title": "Me at the zoo",
16
+ "author_name": "jawed",
17
+ "transcript": "All right, so here we are",
18
+ "word_count": 6,
19
+ "segments": [{"start": 1.2, "duration": 2.16, "text": "All right, so here we are"}],
20
+ }
21
+
22
+
23
+ class TestTranscriptToolStandard(ToolsUnitTests):
24
+ @property
25
+ def tool_constructor(self) -> type[GetYouTubeTranscriptTool]:
26
+ return GetYouTubeTranscriptTool
27
+
28
+ @property
29
+ def tool_constructor_params(self) -> dict:
30
+ return {"api_key": "test-key"}
31
+
32
+ @property
33
+ def tool_invoke_params_example(self) -> dict:
34
+ return {"video": "jNQXAC9IVRw", "language": "en", "timestamps": True}
35
+
36
+ @property
37
+ def init_from_env_params(self) -> tuple[dict, dict, dict]:
38
+ return ({"GETYOUTUBETRANSCRIPT_API_KEY": "env-key"}, {}, {"api_key": "env-key"})
39
+
40
+
41
+ class TestSearchToolStandard(ToolsUnitTests):
42
+ @property
43
+ def tool_constructor(self) -> type[GetYouTubeTranscriptSearchTool]:
44
+ return GetYouTubeTranscriptSearchTool
45
+
46
+ @property
47
+ def tool_constructor_params(self) -> dict:
48
+ return {"api_key": "test-key"}
49
+
50
+ @property
51
+ def tool_invoke_params_example(self) -> dict:
52
+ return {"query": "linear algebra", "type": "video"}
53
+
54
+
55
+ def _tool_with(cls, client: MagicMock):
56
+ tool = cls(api_key="test-key")
57
+ tool._client = client
58
+ return tool
59
+
60
+
61
+ def test_transcript_plain_text():
62
+ client = MagicMock()
63
+ client.get_transcript.return_value = {k: v for k, v in RESULT.items() if k != "segments"}
64
+ out = _tool_with(GetYouTubeTranscriptTool, client).invoke({"video": "https://youtu.be/jNQXAC9IVRw"})
65
+ client.get_transcript.assert_called_once_with("https://youtu.be/jNQXAC9IVRw", language=None, timestamps=False)
66
+ assert out == "Title: Me at the zoo\nChannel: jawed\nLanguage: en\n\nAll right, so here we are"
67
+
68
+
69
+ def test_transcript_timestamps():
70
+ client = MagicMock()
71
+ client.get_transcript.return_value = RESULT
72
+ out = _tool_with(GetYouTubeTranscriptTool, client).invoke({"video": "jNQXAC9IVRw", "timestamps": True})
73
+ assert out.endswith("\n\n[0:01] All right, so here we are")
74
+
75
+
76
+ def test_api_error_is_returned_to_the_agent_not_raised():
77
+ client = MagicMock()
78
+ client.get_transcript.side_effect = GetYouTubeTranscriptError(
79
+ "INVALID_URL", "Not a recognizable YouTube video URL or id.", 400
80
+ )
81
+ out = _tool_with(GetYouTubeTranscriptTool, client).invoke({"video": "nope"})
82
+ assert out == "Not a recognizable YouTube video URL or id."
83
+
84
+
85
+ def test_search_returns_json():
86
+ client = MagicMock()
87
+ client.search.return_value = {"video_results": [{"title": "x"}], "continuation_token": "t"}
88
+ out = _tool_with(GetYouTubeTranscriptSearchTool, client).invoke(
89
+ {"query": "q", "type": "channel", "page_token": "p"}
90
+ )
91
+ client.search.assert_called_once_with("q", type="channel", page_token="p")
92
+ assert json.loads(out)["continuation_token"] == "t"
93
+
94
+
95
+ def test_missing_key_explains_how_to_get_one(monkeypatch):
96
+ monkeypatch.delenv("GETYOUTUBETRANSCRIPT_API_KEY", raising=False)
97
+ with pytest.raises(ValueError, match="GETYOUTUBETRANSCRIPT_API_KEY"):
98
+ _ = GetYouTubeTranscriptTool().client