langchain-diffbot 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langchain_diffbot/__init__.py +39 -0
- langchain_diffbot/_base.py +110 -0
- langchain_diffbot/chat_models.py +122 -0
- langchain_diffbot/document_loaders.py +144 -0
- langchain_diffbot/py.typed +0 -0
- langchain_diffbot/retrievers.py +297 -0
- langchain_diffbot/tools.py +435 -0
- langchain_diffbot-0.1.0.dist-info/METADATA +255 -0
- langchain_diffbot-0.1.0.dist-info/RECORD +11 -0
- langchain_diffbot-0.1.0.dist-info/WHEEL +4 -0
- langchain_diffbot-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""LangChain integration for Diffbot.
|
|
2
|
+
|
|
3
|
+
Thin layer over the official `diffbot-python` SDK. Every public class accepts
|
|
4
|
+
either a `diffbot_api_token` (or `DIFFBOT_API_TOKEN` env var) or a pre-built
|
|
5
|
+
`diffbot.Diffbot` / `diffbot.DiffbotAsync` client via the `client` /
|
|
6
|
+
`async_client` fields — anything the SDK can do, you can do via these classes.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from langchain_diffbot.chat_models import ChatDiffbot
|
|
10
|
+
from langchain_diffbot.document_loaders import (
|
|
11
|
+
DiffbotCrawlLoader,
|
|
12
|
+
DiffbotExtractLoader,
|
|
13
|
+
)
|
|
14
|
+
from langchain_diffbot.retrievers import (
|
|
15
|
+
DiffbotKnowledgeGraphRetriever,
|
|
16
|
+
DiffbotWebSearchRetriever,
|
|
17
|
+
)
|
|
18
|
+
from langchain_diffbot.tools import (
|
|
19
|
+
DiffbotDQLProbeTool,
|
|
20
|
+
DiffbotEntitiesTool,
|
|
21
|
+
DiffbotExtractTool,
|
|
22
|
+
DiffbotKnowledgeGraphTool,
|
|
23
|
+
DiffbotOntologyTool,
|
|
24
|
+
DiffbotWebSearchTool,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"ChatDiffbot",
|
|
29
|
+
"DiffbotCrawlLoader",
|
|
30
|
+
"DiffbotDQLProbeTool",
|
|
31
|
+
"DiffbotEntitiesTool",
|
|
32
|
+
"DiffbotExtractLoader",
|
|
33
|
+
"DiffbotExtractTool",
|
|
34
|
+
"DiffbotKnowledgeGraphRetriever",
|
|
35
|
+
"DiffbotKnowledgeGraphTool",
|
|
36
|
+
"DiffbotOntologyTool",
|
|
37
|
+
"DiffbotWebSearchRetriever",
|
|
38
|
+
"DiffbotWebSearchTool",
|
|
39
|
+
]
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
"""Shared base for langchain-diffbot components.
|
|
2
|
+
|
|
3
|
+
Every public class in this package inherits from `_BaseDiffbotComponent`. The
|
|
4
|
+
mixin holds the token / timeout / optional pre-built SDK clients and exposes
|
|
5
|
+
two context managers (`_sync_db`, `_async_db`) that the components use to
|
|
6
|
+
acquire a `diffbot.Diffbot` / `diffbot.DiffbotAsync` for a single call.
|
|
7
|
+
|
|
8
|
+
Bring-your-own-client: if the user supplies `client=...` or
|
|
9
|
+
`async_client=...`, we use it as-is and **do not close it** — the user owns
|
|
10
|
+
the lifecycle. Otherwise we construct a fresh SDK client per call and close
|
|
11
|
+
it on exit (same per-call lifecycle as the previous hand-rolled httpx
|
|
12
|
+
wrapper).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import os
|
|
18
|
+
from collections.abc import AsyncIterator, Iterator
|
|
19
|
+
from contextlib import asynccontextmanager, contextmanager
|
|
20
|
+
|
|
21
|
+
from diffbot import Diffbot, DiffbotAsync
|
|
22
|
+
from pydantic import BaseModel, ConfigDict, Field, SecretStr, model_validator
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class _BaseDiffbotComponent(BaseModel):
|
|
26
|
+
"""Mixin holding token, timeout, and optional pre-built SDK clients.
|
|
27
|
+
|
|
28
|
+
Concrete classes inherit from this *and* a LangChain base
|
|
29
|
+
(`BaseRetriever`, `BaseTool`, `BaseDocumentLoader`, `BaseChatModel`).
|
|
30
|
+
Both are Pydantic models, so their fields merge.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
model_config = ConfigDict(arbitrary_types_allowed=True)
|
|
34
|
+
|
|
35
|
+
diffbot_api_token: SecretStr | None = Field(default=None)
|
|
36
|
+
"""Diffbot API token. Falls back to `DIFFBOT_API_TOKEN`.
|
|
37
|
+
|
|
38
|
+
Not required when both `client` and `async_client` are supplied.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
timeout: float = 30.0
|
|
42
|
+
"""HTTP timeout (seconds) for SDK clients we construct ourselves.
|
|
43
|
+
|
|
44
|
+
Ignored when `client` / `async_client` are supplied.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
client: Diffbot | None = Field(default=None, exclude=True, repr=False)
|
|
48
|
+
"""Optional pre-built sync SDK client.
|
|
49
|
+
|
|
50
|
+
If set, we use it as-is and do not close it.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
async_client: DiffbotAsync | None = Field(default=None, exclude=True, repr=False)
|
|
54
|
+
"""Optional pre-built async SDK client.
|
|
55
|
+
|
|
56
|
+
If set, we use it as-is and do not close it.
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
@model_validator(mode="after")
|
|
60
|
+
def _resolve_token(self) -> _BaseDiffbotComponent:
|
|
61
|
+
# If the user gave us a client (for either side), we can't be sure
|
|
62
|
+
# they'll use the other side — but token resolution shouldn't block
|
|
63
|
+
# construction in that case. Defer the missing-token error to call time.
|
|
64
|
+
if self.client is not None or self.async_client is not None:
|
|
65
|
+
return self
|
|
66
|
+
if (
|
|
67
|
+
self.diffbot_api_token is None
|
|
68
|
+
or not self.diffbot_api_token.get_secret_value()
|
|
69
|
+
):
|
|
70
|
+
env_token = os.environ.get("DIFFBOT_API_TOKEN", "")
|
|
71
|
+
if not env_token:
|
|
72
|
+
msg = (
|
|
73
|
+
"A Diffbot API token is required. Pass `diffbot_api_token=...`, "
|
|
74
|
+
"set the `DIFFBOT_API_TOKEN` environment variable, or supply a "
|
|
75
|
+
"pre-built `client` / `async_client`."
|
|
76
|
+
)
|
|
77
|
+
raise ValueError(msg)
|
|
78
|
+
self.diffbot_api_token = SecretStr(env_token)
|
|
79
|
+
return self
|
|
80
|
+
|
|
81
|
+
def _token(self) -> str:
|
|
82
|
+
if (
|
|
83
|
+
self.diffbot_api_token is None
|
|
84
|
+
or not self.diffbot_api_token.get_secret_value()
|
|
85
|
+
):
|
|
86
|
+
msg = (
|
|
87
|
+
"A Diffbot API token is required for this call. Pass "
|
|
88
|
+
"`diffbot_api_token=...`, set `DIFFBOT_API_TOKEN`, or supply a "
|
|
89
|
+
"pre-built client."
|
|
90
|
+
)
|
|
91
|
+
raise ValueError(msg)
|
|
92
|
+
return self.diffbot_api_token.get_secret_value()
|
|
93
|
+
|
|
94
|
+
@contextmanager
|
|
95
|
+
def _sync_db(self) -> Iterator[Diffbot]:
|
|
96
|
+
"""Yield a `Diffbot` for one call. Closes only clients we constructed."""
|
|
97
|
+
if self.client is not None:
|
|
98
|
+
yield self.client
|
|
99
|
+
return
|
|
100
|
+
with Diffbot(token=self._token(), timeout=self.timeout) as db:
|
|
101
|
+
yield db
|
|
102
|
+
|
|
103
|
+
@asynccontextmanager
|
|
104
|
+
async def _async_db(self) -> AsyncIterator[DiffbotAsync]:
|
|
105
|
+
"""Yield a `DiffbotAsync` for one call. Closes only clients we constructed."""
|
|
106
|
+
if self.async_client is not None:
|
|
107
|
+
yield self.async_client
|
|
108
|
+
return
|
|
109
|
+
async with DiffbotAsync(token=self._token(), timeout=self.timeout) as db:
|
|
110
|
+
yield db
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""ChatDiffbot — LangChain chat model wrapping Diffbot's LLM RAG `ask` endpoint."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import AsyncIterator, Iterator
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from langchain_core.callbacks import (
|
|
9
|
+
AsyncCallbackManagerForLLMRun,
|
|
10
|
+
CallbackManagerForLLMRun,
|
|
11
|
+
)
|
|
12
|
+
from langchain_core.language_models.chat_models import BaseChatModel
|
|
13
|
+
from langchain_core.messages import (
|
|
14
|
+
AIMessage,
|
|
15
|
+
AIMessageChunk,
|
|
16
|
+
BaseMessage,
|
|
17
|
+
HumanMessage,
|
|
18
|
+
SystemMessage,
|
|
19
|
+
)
|
|
20
|
+
from langchain_core.outputs import ChatGeneration, ChatGenerationChunk, ChatResult
|
|
21
|
+
|
|
22
|
+
from langchain_diffbot._base import _BaseDiffbotComponent
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _to_diffbot_messages(messages: list[BaseMessage]) -> list[dict[str, str]]:
|
|
26
|
+
out: list[dict[str, str]] = []
|
|
27
|
+
for m in messages:
|
|
28
|
+
if isinstance(m, HumanMessage):
|
|
29
|
+
role = "user"
|
|
30
|
+
elif isinstance(m, AIMessage):
|
|
31
|
+
role = "assistant"
|
|
32
|
+
elif isinstance(m, SystemMessage):
|
|
33
|
+
role = "system"
|
|
34
|
+
else:
|
|
35
|
+
# Fall back on the message's `type` attribute for anything exotic.
|
|
36
|
+
role = m.type if isinstance(m.type, str) else "user"
|
|
37
|
+
content = m.content if isinstance(m.content, str) else str(m.content)
|
|
38
|
+
out.append({"role": role, "content": content})
|
|
39
|
+
return out
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class ChatDiffbot(_BaseDiffbotComponent, BaseChatModel):
|
|
43
|
+
"""Chat model backed by Diffbot's LLM RAG (`ask`) endpoint.
|
|
44
|
+
|
|
45
|
+
The SDK streams tokens natively, so this class implements `_stream` and
|
|
46
|
+
`_astream`. `_generate` / `_agenerate` aggregate the stream into a single
|
|
47
|
+
`ChatGeneration`.
|
|
48
|
+
|
|
49
|
+
Example:
|
|
50
|
+
```python
|
|
51
|
+
from langchain_diffbot import ChatDiffbot
|
|
52
|
+
|
|
53
|
+
llm = ChatDiffbot()
|
|
54
|
+
llm.invoke("What's the capital of France?")
|
|
55
|
+
```
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def _llm_type(self) -> str:
|
|
60
|
+
return "diffbot"
|
|
61
|
+
|
|
62
|
+
def _stream(
|
|
63
|
+
self,
|
|
64
|
+
messages: list[BaseMessage],
|
|
65
|
+
stop: list[str] | None = None,
|
|
66
|
+
run_manager: CallbackManagerForLLMRun | None = None,
|
|
67
|
+
**kwargs: Any,
|
|
68
|
+
) -> Iterator[ChatGenerationChunk]:
|
|
69
|
+
payload = _to_diffbot_messages(messages)
|
|
70
|
+
with self._sync_db() as db:
|
|
71
|
+
for chunk in db.ask(payload):
|
|
72
|
+
if run_manager is not None:
|
|
73
|
+
run_manager.on_llm_new_token(chunk)
|
|
74
|
+
yield ChatGenerationChunk(message=AIMessageChunk(content=chunk))
|
|
75
|
+
|
|
76
|
+
async def _astream(
|
|
77
|
+
self,
|
|
78
|
+
messages: list[BaseMessage],
|
|
79
|
+
stop: list[str] | None = None,
|
|
80
|
+
run_manager: AsyncCallbackManagerForLLMRun | None = None,
|
|
81
|
+
**kwargs: Any,
|
|
82
|
+
) -> AsyncIterator[ChatGenerationChunk]:
|
|
83
|
+
payload = _to_diffbot_messages(messages)
|
|
84
|
+
async with self._async_db() as db:
|
|
85
|
+
async for chunk in db.ask(payload):
|
|
86
|
+
if run_manager is not None:
|
|
87
|
+
await run_manager.on_llm_new_token(chunk)
|
|
88
|
+
yield ChatGenerationChunk(message=AIMessageChunk(content=chunk))
|
|
89
|
+
|
|
90
|
+
def _generate(
|
|
91
|
+
self,
|
|
92
|
+
messages: list[BaseMessage],
|
|
93
|
+
stop: list[str] | None = None,
|
|
94
|
+
run_manager: CallbackManagerForLLMRun | None = None,
|
|
95
|
+
**kwargs: Any,
|
|
96
|
+
) -> ChatResult:
|
|
97
|
+
parts: list[str] = []
|
|
98
|
+
for chunk in self._stream(
|
|
99
|
+
messages, stop=stop, run_manager=run_manager, **kwargs
|
|
100
|
+
):
|
|
101
|
+
content = chunk.message.content
|
|
102
|
+
parts.append(content if isinstance(content, str) else str(content))
|
|
103
|
+
return ChatResult(
|
|
104
|
+
generations=[ChatGeneration(message=AIMessage(content="".join(parts)))]
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
async def _agenerate(
|
|
108
|
+
self,
|
|
109
|
+
messages: list[BaseMessage],
|
|
110
|
+
stop: list[str] | None = None,
|
|
111
|
+
run_manager: AsyncCallbackManagerForLLMRun | None = None,
|
|
112
|
+
**kwargs: Any,
|
|
113
|
+
) -> ChatResult:
|
|
114
|
+
parts: list[str] = []
|
|
115
|
+
async for chunk in self._astream(
|
|
116
|
+
messages, stop=stop, run_manager=run_manager, **kwargs
|
|
117
|
+
):
|
|
118
|
+
content = chunk.message.content
|
|
119
|
+
parts.append(content if isinstance(content, str) else str(content))
|
|
120
|
+
return ChatResult(
|
|
121
|
+
generations=[ChatGeneration(message=AIMessage(content="".join(parts)))]
|
|
122
|
+
)
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""Diffbot document loaders — batch ingestion via Extract and Crawl."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import AsyncIterator, Callable, Iterator
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from diffbot.crawl import CrawlEvent, CrawlEventType
|
|
9
|
+
from langchain_core.document_loaders import BaseLoader
|
|
10
|
+
from langchain_core.documents import Document
|
|
11
|
+
from pydantic import Field
|
|
12
|
+
|
|
13
|
+
from langchain_diffbot._base import _BaseDiffbotComponent
|
|
14
|
+
|
|
15
|
+
ExtractDocumentMapper = Callable[[str, dict[str, Any]], Document]
|
|
16
|
+
"""`(url, raw_response) -> Document`."""
|
|
17
|
+
|
|
18
|
+
CrawlEventMapper = Callable[[CrawlEvent], Document | None]
|
|
19
|
+
"""`(event) -> Document | None`. Return None to skip the event."""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _default_extract_mapper(url: str, raw: dict[str, Any]) -> Document:
|
|
23
|
+
objects = raw.get("objects") or []
|
|
24
|
+
first = objects[0] if objects else {}
|
|
25
|
+
page_content = first.get("text") or raw.get("markdown") or ""
|
|
26
|
+
metadata = {
|
|
27
|
+
"url": url,
|
|
28
|
+
"title": first.get("title") or raw.get("title"),
|
|
29
|
+
"pageUrl": first.get("pageUrl") or raw.get("url"),
|
|
30
|
+
"resolvedPageUrl": first.get("resolvedPageUrl"),
|
|
31
|
+
"type": first.get("type") or raw.get("type"),
|
|
32
|
+
}
|
|
33
|
+
return Document(page_content=page_content, metadata=metadata)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _default_crawl_mapper(event: CrawlEvent) -> Document | None:
|
|
37
|
+
# Non-URL events (e.g. JOB_CREATED) are skipped by default. The Document's
|
|
38
|
+
# page_content is the URL itself — the crawl SDK only surfaces URLs, not
|
|
39
|
+
# page contents. Chain with `DiffbotExtractLoader` to fetch content.
|
|
40
|
+
if event.event_type != CrawlEventType.URL_PROCESSED:
|
|
41
|
+
return None
|
|
42
|
+
url = event.details.get("url", "")
|
|
43
|
+
return Document(
|
|
44
|
+
page_content=url,
|
|
45
|
+
metadata={
|
|
46
|
+
"url": url,
|
|
47
|
+
"status": event.details.get("status"),
|
|
48
|
+
"crawl_timestamp": event.timestamp,
|
|
49
|
+
},
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class DiffbotExtractLoader(_BaseDiffbotComponent, BaseLoader):
|
|
54
|
+
"""Loader that calls `extract` on each of `urls` and yields a `Document`.
|
|
55
|
+
|
|
56
|
+
Example:
|
|
57
|
+
```python
|
|
58
|
+
from langchain_diffbot import DiffbotExtractLoader
|
|
59
|
+
|
|
60
|
+
docs = DiffbotExtractLoader(
|
|
61
|
+
urls=["https://example.com", "https://diffbot.com"],
|
|
62
|
+
).load()
|
|
63
|
+
```
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
urls: list[str]
|
|
67
|
+
"""URLs to extract."""
|
|
68
|
+
|
|
69
|
+
api: str = "analyze"
|
|
70
|
+
"""Diffbot extract API. Defaults to `analyze`."""
|
|
71
|
+
|
|
72
|
+
fmt: str = "markdown"
|
|
73
|
+
"""Output format. `markdown` uses Diffbot's LLM-optimized mode."""
|
|
74
|
+
|
|
75
|
+
document_mapper: ExtractDocumentMapper | None = None
|
|
76
|
+
"""Optional `(url, raw_response) -> Document` override."""
|
|
77
|
+
|
|
78
|
+
def _to_doc(self, url: str, raw: dict[str, Any]) -> Document:
|
|
79
|
+
mapper = self.document_mapper or _default_extract_mapper
|
|
80
|
+
return mapper(url, raw)
|
|
81
|
+
|
|
82
|
+
def lazy_load(self) -> Iterator[Document]:
|
|
83
|
+
"""Yield one `Document` per URL, calling `extract` synchronously."""
|
|
84
|
+
with self._sync_db() as db:
|
|
85
|
+
for url in self.urls:
|
|
86
|
+
raw = db.extract(url, api=self.api, fmt=self.fmt)
|
|
87
|
+
yield self._to_doc(url, raw)
|
|
88
|
+
|
|
89
|
+
async def alazy_load(self) -> AsyncIterator[Document]:
|
|
90
|
+
"""Yield one `Document` per URL, calling `extract` asynchronously."""
|
|
91
|
+
async with self._async_db() as db:
|
|
92
|
+
for url in self.urls:
|
|
93
|
+
raw = await db.extract(url, api=self.api, fmt=self.fmt)
|
|
94
|
+
yield self._to_doc(url, raw)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class DiffbotCrawlLoader(_BaseDiffbotComponent, BaseLoader):
|
|
98
|
+
"""Loader that drives a Diffbot crawl and yields a `Document` per URL.
|
|
99
|
+
|
|
100
|
+
By default `page_content` is the crawled URL itself (the crawl SDK only
|
|
101
|
+
yields URL events, not page contents). To fetch content per URL, chain
|
|
102
|
+
this with `DiffbotExtractLoader`.
|
|
103
|
+
|
|
104
|
+
Defaults `watch=True` so URL events are actually emitted — with
|
|
105
|
+
`watch=False` the SDK only yields a single JOB_CREATED event, which the
|
|
106
|
+
default mapper skips.
|
|
107
|
+
"""
|
|
108
|
+
|
|
109
|
+
site: str
|
|
110
|
+
"""Seed URL for the crawl."""
|
|
111
|
+
|
|
112
|
+
crawl_kwargs: dict[str, Any] = Field(default_factory=dict)
|
|
113
|
+
"""Extra kwargs passed to `Diffbot.crawl(site, **crawl_kwargs)`.
|
|
114
|
+
|
|
115
|
+
Defaults to `{"watch": True}` if not overridden.
|
|
116
|
+
"""
|
|
117
|
+
|
|
118
|
+
event_mapper: CrawlEventMapper | None = None
|
|
119
|
+
"""Optional `(event) -> Document | None` override. Return None to skip."""
|
|
120
|
+
|
|
121
|
+
def _kwargs(self) -> dict[str, Any]:
|
|
122
|
+
kw = dict(self.crawl_kwargs)
|
|
123
|
+
kw.setdefault("watch", True)
|
|
124
|
+
return kw
|
|
125
|
+
|
|
126
|
+
def _map_event(self, event: CrawlEvent) -> Document | None:
|
|
127
|
+
mapper = self.event_mapper or _default_crawl_mapper
|
|
128
|
+
return mapper(event)
|
|
129
|
+
|
|
130
|
+
def lazy_load(self) -> Iterator[Document]:
|
|
131
|
+
"""Drive a crawl synchronously and yield a `Document` per mapped event."""
|
|
132
|
+
with self._sync_db() as db:
|
|
133
|
+
for event in db.crawl(self.site, **self._kwargs()):
|
|
134
|
+
doc = self._map_event(event)
|
|
135
|
+
if doc is not None:
|
|
136
|
+
yield doc
|
|
137
|
+
|
|
138
|
+
async def alazy_load(self) -> AsyncIterator[Document]:
|
|
139
|
+
"""Drive a crawl asynchronously and yield a `Document` per mapped event."""
|
|
140
|
+
async with self._async_db() as db:
|
|
141
|
+
async for event in db.crawl(self.site, **self._kwargs()):
|
|
142
|
+
doc = self._map_event(event)
|
|
143
|
+
if doc is not None:
|
|
144
|
+
yield doc
|
|
File without changes
|