langchain-diffbot 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,39 @@
1
+ """LangChain integration for Diffbot.
2
+
3
+ Thin layer over the official `diffbot-python` SDK. Every public class accepts
4
+ either a `diffbot_api_token` (or `DIFFBOT_API_TOKEN` env var) or a pre-built
5
+ `diffbot.Diffbot` / `diffbot.DiffbotAsync` client via the `client` /
6
+ `async_client` fields — anything the SDK can do, you can do via these classes.
7
+ """
8
+
9
+ from langchain_diffbot.chat_models import ChatDiffbot
10
+ from langchain_diffbot.document_loaders import (
11
+ DiffbotCrawlLoader,
12
+ DiffbotExtractLoader,
13
+ )
14
+ from langchain_diffbot.retrievers import (
15
+ DiffbotKnowledgeGraphRetriever,
16
+ DiffbotWebSearchRetriever,
17
+ )
18
+ from langchain_diffbot.tools import (
19
+ DiffbotDQLProbeTool,
20
+ DiffbotEntitiesTool,
21
+ DiffbotExtractTool,
22
+ DiffbotKnowledgeGraphTool,
23
+ DiffbotOntologyTool,
24
+ DiffbotWebSearchTool,
25
+ )
26
+
27
+ __all__ = [
28
+ "ChatDiffbot",
29
+ "DiffbotCrawlLoader",
30
+ "DiffbotDQLProbeTool",
31
+ "DiffbotEntitiesTool",
32
+ "DiffbotExtractLoader",
33
+ "DiffbotExtractTool",
34
+ "DiffbotKnowledgeGraphRetriever",
35
+ "DiffbotKnowledgeGraphTool",
36
+ "DiffbotOntologyTool",
37
+ "DiffbotWebSearchRetriever",
38
+ "DiffbotWebSearchTool",
39
+ ]
@@ -0,0 +1,110 @@
1
+ """Shared base for langchain-diffbot components.
2
+
3
+ Every public class in this package inherits from `_BaseDiffbotComponent`. The
4
+ mixin holds the token / timeout / optional pre-built SDK clients and exposes
5
+ two context managers (`_sync_db`, `_async_db`) that the components use to
6
+ acquire a `diffbot.Diffbot` / `diffbot.DiffbotAsync` for a single call.
7
+
8
+ Bring-your-own-client: if the user supplies `client=...` or
9
+ `async_client=...`, we use it as-is and **do not close it** — the user owns
10
+ the lifecycle. Otherwise we construct a fresh SDK client per call and close
11
+ it on exit (same per-call lifecycle as the previous hand-rolled httpx
12
+ wrapper).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import os
18
+ from collections.abc import AsyncIterator, Iterator
19
+ from contextlib import asynccontextmanager, contextmanager
20
+
21
+ from diffbot import Diffbot, DiffbotAsync
22
+ from pydantic import BaseModel, ConfigDict, Field, SecretStr, model_validator
23
+
24
+
25
+ class _BaseDiffbotComponent(BaseModel):
26
+ """Mixin holding token, timeout, and optional pre-built SDK clients.
27
+
28
+ Concrete classes inherit from this *and* a LangChain base
29
+ (`BaseRetriever`, `BaseTool`, `BaseDocumentLoader`, `BaseChatModel`).
30
+ Both are Pydantic models, so their fields merge.
31
+ """
32
+
33
+ model_config = ConfigDict(arbitrary_types_allowed=True)
34
+
35
+ diffbot_api_token: SecretStr | None = Field(default=None)
36
+ """Diffbot API token. Falls back to `DIFFBOT_API_TOKEN`.
37
+
38
+ Not required when both `client` and `async_client` are supplied.
39
+ """
40
+
41
+ timeout: float = 30.0
42
+ """HTTP timeout (seconds) for SDK clients we construct ourselves.
43
+
44
+ Ignored when `client` / `async_client` are supplied.
45
+ """
46
+
47
+ client: Diffbot | None = Field(default=None, exclude=True, repr=False)
48
+ """Optional pre-built sync SDK client.
49
+
50
+ If set, we use it as-is and do not close it.
51
+ """
52
+
53
+ async_client: DiffbotAsync | None = Field(default=None, exclude=True, repr=False)
54
+ """Optional pre-built async SDK client.
55
+
56
+ If set, we use it as-is and do not close it.
57
+ """
58
+
59
+ @model_validator(mode="after")
60
+ def _resolve_token(self) -> _BaseDiffbotComponent:
61
+ # If the user gave us a client (for either side), we can't be sure
62
+ # they'll use the other side — but token resolution shouldn't block
63
+ # construction in that case. Defer the missing-token error to call time.
64
+ if self.client is not None or self.async_client is not None:
65
+ return self
66
+ if (
67
+ self.diffbot_api_token is None
68
+ or not self.diffbot_api_token.get_secret_value()
69
+ ):
70
+ env_token = os.environ.get("DIFFBOT_API_TOKEN", "")
71
+ if not env_token:
72
+ msg = (
73
+ "A Diffbot API token is required. Pass `diffbot_api_token=...`, "
74
+ "set the `DIFFBOT_API_TOKEN` environment variable, or supply a "
75
+ "pre-built `client` / `async_client`."
76
+ )
77
+ raise ValueError(msg)
78
+ self.diffbot_api_token = SecretStr(env_token)
79
+ return self
80
+
81
+ def _token(self) -> str:
82
+ if (
83
+ self.diffbot_api_token is None
84
+ or not self.diffbot_api_token.get_secret_value()
85
+ ):
86
+ msg = (
87
+ "A Diffbot API token is required for this call. Pass "
88
+ "`diffbot_api_token=...`, set `DIFFBOT_API_TOKEN`, or supply a "
89
+ "pre-built client."
90
+ )
91
+ raise ValueError(msg)
92
+ return self.diffbot_api_token.get_secret_value()
93
+
94
+ @contextmanager
95
+ def _sync_db(self) -> Iterator[Diffbot]:
96
+ """Yield a `Diffbot` for one call. Closes only clients we constructed."""
97
+ if self.client is not None:
98
+ yield self.client
99
+ return
100
+ with Diffbot(token=self._token(), timeout=self.timeout) as db:
101
+ yield db
102
+
103
+ @asynccontextmanager
104
+ async def _async_db(self) -> AsyncIterator[DiffbotAsync]:
105
+ """Yield a `DiffbotAsync` for one call. Closes only clients we constructed."""
106
+ if self.async_client is not None:
107
+ yield self.async_client
108
+ return
109
+ async with DiffbotAsync(token=self._token(), timeout=self.timeout) as db:
110
+ yield db
@@ -0,0 +1,122 @@
1
+ """ChatDiffbot — LangChain chat model wrapping Diffbot's LLM RAG `ask` endpoint."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import AsyncIterator, Iterator
6
+ from typing import Any
7
+
8
+ from langchain_core.callbacks import (
9
+ AsyncCallbackManagerForLLMRun,
10
+ CallbackManagerForLLMRun,
11
+ )
12
+ from langchain_core.language_models.chat_models import BaseChatModel
13
+ from langchain_core.messages import (
14
+ AIMessage,
15
+ AIMessageChunk,
16
+ BaseMessage,
17
+ HumanMessage,
18
+ SystemMessage,
19
+ )
20
+ from langchain_core.outputs import ChatGeneration, ChatGenerationChunk, ChatResult
21
+
22
+ from langchain_diffbot._base import _BaseDiffbotComponent
23
+
24
+
25
+ def _to_diffbot_messages(messages: list[BaseMessage]) -> list[dict[str, str]]:
26
+ out: list[dict[str, str]] = []
27
+ for m in messages:
28
+ if isinstance(m, HumanMessage):
29
+ role = "user"
30
+ elif isinstance(m, AIMessage):
31
+ role = "assistant"
32
+ elif isinstance(m, SystemMessage):
33
+ role = "system"
34
+ else:
35
+ # Fall back on the message's `type` attribute for anything exotic.
36
+ role = m.type if isinstance(m.type, str) else "user"
37
+ content = m.content if isinstance(m.content, str) else str(m.content)
38
+ out.append({"role": role, "content": content})
39
+ return out
40
+
41
+
42
+ class ChatDiffbot(_BaseDiffbotComponent, BaseChatModel):
43
+ """Chat model backed by Diffbot's LLM RAG (`ask`) endpoint.
44
+
45
+ The SDK streams tokens natively, so this class implements `_stream` and
46
+ `_astream`. `_generate` / `_agenerate` aggregate the stream into a single
47
+ `ChatGeneration`.
48
+
49
+ Example:
50
+ ```python
51
+ from langchain_diffbot import ChatDiffbot
52
+
53
+ llm = ChatDiffbot()
54
+ llm.invoke("What's the capital of France?")
55
+ ```
56
+ """
57
+
58
+ @property
59
+ def _llm_type(self) -> str:
60
+ return "diffbot"
61
+
62
+ def _stream(
63
+ self,
64
+ messages: list[BaseMessage],
65
+ stop: list[str] | None = None,
66
+ run_manager: CallbackManagerForLLMRun | None = None,
67
+ **kwargs: Any,
68
+ ) -> Iterator[ChatGenerationChunk]:
69
+ payload = _to_diffbot_messages(messages)
70
+ with self._sync_db() as db:
71
+ for chunk in db.ask(payload):
72
+ if run_manager is not None:
73
+ run_manager.on_llm_new_token(chunk)
74
+ yield ChatGenerationChunk(message=AIMessageChunk(content=chunk))
75
+
76
+ async def _astream(
77
+ self,
78
+ messages: list[BaseMessage],
79
+ stop: list[str] | None = None,
80
+ run_manager: AsyncCallbackManagerForLLMRun | None = None,
81
+ **kwargs: Any,
82
+ ) -> AsyncIterator[ChatGenerationChunk]:
83
+ payload = _to_diffbot_messages(messages)
84
+ async with self._async_db() as db:
85
+ async for chunk in db.ask(payload):
86
+ if run_manager is not None:
87
+ await run_manager.on_llm_new_token(chunk)
88
+ yield ChatGenerationChunk(message=AIMessageChunk(content=chunk))
89
+
90
+ def _generate(
91
+ self,
92
+ messages: list[BaseMessage],
93
+ stop: list[str] | None = None,
94
+ run_manager: CallbackManagerForLLMRun | None = None,
95
+ **kwargs: Any,
96
+ ) -> ChatResult:
97
+ parts: list[str] = []
98
+ for chunk in self._stream(
99
+ messages, stop=stop, run_manager=run_manager, **kwargs
100
+ ):
101
+ content = chunk.message.content
102
+ parts.append(content if isinstance(content, str) else str(content))
103
+ return ChatResult(
104
+ generations=[ChatGeneration(message=AIMessage(content="".join(parts)))]
105
+ )
106
+
107
+ async def _agenerate(
108
+ self,
109
+ messages: list[BaseMessage],
110
+ stop: list[str] | None = None,
111
+ run_manager: AsyncCallbackManagerForLLMRun | None = None,
112
+ **kwargs: Any,
113
+ ) -> ChatResult:
114
+ parts: list[str] = []
115
+ async for chunk in self._astream(
116
+ messages, stop=stop, run_manager=run_manager, **kwargs
117
+ ):
118
+ content = chunk.message.content
119
+ parts.append(content if isinstance(content, str) else str(content))
120
+ return ChatResult(
121
+ generations=[ChatGeneration(message=AIMessage(content="".join(parts)))]
122
+ )
@@ -0,0 +1,144 @@
1
+ """Diffbot document loaders — batch ingestion via Extract and Crawl."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import AsyncIterator, Callable, Iterator
6
+ from typing import Any
7
+
8
+ from diffbot.crawl import CrawlEvent, CrawlEventType
9
+ from langchain_core.document_loaders import BaseLoader
10
+ from langchain_core.documents import Document
11
+ from pydantic import Field
12
+
13
+ from langchain_diffbot._base import _BaseDiffbotComponent
14
+
15
+ ExtractDocumentMapper = Callable[[str, dict[str, Any]], Document]
16
+ """`(url, raw_response) -> Document`."""
17
+
18
+ CrawlEventMapper = Callable[[CrawlEvent], Document | None]
19
+ """`(event) -> Document | None`. Return None to skip the event."""
20
+
21
+
22
+ def _default_extract_mapper(url: str, raw: dict[str, Any]) -> Document:
23
+ objects = raw.get("objects") or []
24
+ first = objects[0] if objects else {}
25
+ page_content = first.get("text") or raw.get("markdown") or ""
26
+ metadata = {
27
+ "url": url,
28
+ "title": first.get("title") or raw.get("title"),
29
+ "pageUrl": first.get("pageUrl") or raw.get("url"),
30
+ "resolvedPageUrl": first.get("resolvedPageUrl"),
31
+ "type": first.get("type") or raw.get("type"),
32
+ }
33
+ return Document(page_content=page_content, metadata=metadata)
34
+
35
+
36
+ def _default_crawl_mapper(event: CrawlEvent) -> Document | None:
37
+ # Non-URL events (e.g. JOB_CREATED) are skipped by default. The Document's
38
+ # page_content is the URL itself — the crawl SDK only surfaces URLs, not
39
+ # page contents. Chain with `DiffbotExtractLoader` to fetch content.
40
+ if event.event_type != CrawlEventType.URL_PROCESSED:
41
+ return None
42
+ url = event.details.get("url", "")
43
+ return Document(
44
+ page_content=url,
45
+ metadata={
46
+ "url": url,
47
+ "status": event.details.get("status"),
48
+ "crawl_timestamp": event.timestamp,
49
+ },
50
+ )
51
+
52
+
53
+ class DiffbotExtractLoader(_BaseDiffbotComponent, BaseLoader):
54
+ """Loader that calls `extract` on each of `urls` and yields a `Document`.
55
+
56
+ Example:
57
+ ```python
58
+ from langchain_diffbot import DiffbotExtractLoader
59
+
60
+ docs = DiffbotExtractLoader(
61
+ urls=["https://example.com", "https://diffbot.com"],
62
+ ).load()
63
+ ```
64
+ """
65
+
66
+ urls: list[str]
67
+ """URLs to extract."""
68
+
69
+ api: str = "analyze"
70
+ """Diffbot extract API. Defaults to `analyze`."""
71
+
72
+ fmt: str = "markdown"
73
+ """Output format. `markdown` uses Diffbot's LLM-optimized mode."""
74
+
75
+ document_mapper: ExtractDocumentMapper | None = None
76
+ """Optional `(url, raw_response) -> Document` override."""
77
+
78
+ def _to_doc(self, url: str, raw: dict[str, Any]) -> Document:
79
+ mapper = self.document_mapper or _default_extract_mapper
80
+ return mapper(url, raw)
81
+
82
+ def lazy_load(self) -> Iterator[Document]:
83
+ """Yield one `Document` per URL, calling `extract` synchronously."""
84
+ with self._sync_db() as db:
85
+ for url in self.urls:
86
+ raw = db.extract(url, api=self.api, fmt=self.fmt)
87
+ yield self._to_doc(url, raw)
88
+
89
+ async def alazy_load(self) -> AsyncIterator[Document]:
90
+ """Yield one `Document` per URL, calling `extract` asynchronously."""
91
+ async with self._async_db() as db:
92
+ for url in self.urls:
93
+ raw = await db.extract(url, api=self.api, fmt=self.fmt)
94
+ yield self._to_doc(url, raw)
95
+
96
+
97
+ class DiffbotCrawlLoader(_BaseDiffbotComponent, BaseLoader):
98
+ """Loader that drives a Diffbot crawl and yields a `Document` per URL.
99
+
100
+ By default `page_content` is the crawled URL itself (the crawl SDK only
101
+ yields URL events, not page contents). To fetch content per URL, chain
102
+ this with `DiffbotExtractLoader`.
103
+
104
+ Defaults `watch=True` so URL events are actually emitted — with
105
+ `watch=False` the SDK only yields a single JOB_CREATED event, which the
106
+ default mapper skips.
107
+ """
108
+
109
+ site: str
110
+ """Seed URL for the crawl."""
111
+
112
+ crawl_kwargs: dict[str, Any] = Field(default_factory=dict)
113
+ """Extra kwargs passed to `Diffbot.crawl(site, **crawl_kwargs)`.
114
+
115
+ Defaults to `{"watch": True}` if not overridden.
116
+ """
117
+
118
+ event_mapper: CrawlEventMapper | None = None
119
+ """Optional `(event) -> Document | None` override. Return None to skip."""
120
+
121
+ def _kwargs(self) -> dict[str, Any]:
122
+ kw = dict(self.crawl_kwargs)
123
+ kw.setdefault("watch", True)
124
+ return kw
125
+
126
+ def _map_event(self, event: CrawlEvent) -> Document | None:
127
+ mapper = self.event_mapper or _default_crawl_mapper
128
+ return mapper(event)
129
+
130
+ def lazy_load(self) -> Iterator[Document]:
131
+ """Drive a crawl synchronously and yield a `Document` per mapped event."""
132
+ with self._sync_db() as db:
133
+ for event in db.crawl(self.site, **self._kwargs()):
134
+ doc = self._map_event(event)
135
+ if doc is not None:
136
+ yield doc
137
+
138
+ async def alazy_load(self) -> AsyncIterator[Document]:
139
+ """Drive a crawl asynchronously and yield a `Document` per mapped event."""
140
+ async with self._async_db() as db:
141
+ async for event in db.crawl(self.site, **self._kwargs()):
142
+ doc = self._map_event(event)
143
+ if doc is not None:
144
+ yield doc
File without changes