langchain-diffbot 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langchain_diffbot/__init__.py +39 -0
- langchain_diffbot/_base.py +110 -0
- langchain_diffbot/chat_models.py +122 -0
- langchain_diffbot/document_loaders.py +144 -0
- langchain_diffbot/py.typed +0 -0
- langchain_diffbot/retrievers.py +297 -0
- langchain_diffbot/tools.py +435 -0
- langchain_diffbot-0.1.0.dist-info/METADATA +255 -0
- langchain_diffbot-0.1.0.dist-info/RECORD +11 -0
- langchain_diffbot-0.1.0.dist-info/WHEEL +4 -0
- langchain_diffbot-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,435 @@
|
|
|
1
|
+
"""Diffbot tools — agent-callable wrappers around individual SDK methods.
|
|
2
|
+
|
|
3
|
+
Each tool is a thin BaseTool around one `diffbot` method. Args schemas mirror
|
|
4
|
+
the SDK signatures one-for-one so agents calling these tools see the same
|
|
5
|
+
shape as a direct SDK call.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any, Literal
|
|
11
|
+
|
|
12
|
+
from diffbot import Ontology
|
|
13
|
+
from diffbot.errors import ExtractionError
|
|
14
|
+
from langchain_core.callbacks import (
|
|
15
|
+
AsyncCallbackManagerForToolRun,
|
|
16
|
+
CallbackManagerForToolRun,
|
|
17
|
+
)
|
|
18
|
+
from langchain_core.tools import BaseTool
|
|
19
|
+
from pydantic import BaseModel, Field, PrivateAttr
|
|
20
|
+
|
|
21
|
+
from langchain_diffbot._base import _BaseDiffbotComponent
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class _DiffbotExtractInput(BaseModel):
|
|
25
|
+
url: str = Field(description="URL to extract structured content from.")
|
|
26
|
+
api: str = Field(
|
|
27
|
+
default="analyze",
|
|
28
|
+
description=(
|
|
29
|
+
"Diffbot extract API to call. Defaults to `analyze` "
|
|
30
|
+
"(auto-detects content type)."
|
|
31
|
+
),
|
|
32
|
+
)
|
|
33
|
+
fmt: str = Field(
|
|
34
|
+
default="markdown",
|
|
35
|
+
description="Output format. `markdown` uses Diffbot's LLM-optimized mode.",
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class DiffbotExtractTool(_BaseDiffbotComponent, BaseTool):
|
|
40
|
+
"""Tool that extracts structured content from a URL via Diffbot's analyze API.
|
|
41
|
+
|
|
42
|
+
Returns a small dict so the agent doesn't have to wade through the full
|
|
43
|
+
raw response. On extraction failure (a 200 response with an `errorCode`
|
|
44
|
+
body) returns a structured error dict instead of raising, so the agent
|
|
45
|
+
can react. Auth / rate-limit errors propagate as exceptions — those are
|
|
46
|
+
infra problems, not per-call signals.
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
name: str = "diffbot_extract"
|
|
50
|
+
description: str = (
|
|
51
|
+
"Extract structured content (title, text, type, resolved URL) from a "
|
|
52
|
+
"single web page. Use for reading the contents of a known URL."
|
|
53
|
+
)
|
|
54
|
+
args_schema: type[BaseModel] = _DiffbotExtractInput
|
|
55
|
+
|
|
56
|
+
@staticmethod
|
|
57
|
+
def _shape_response(raw: dict[str, Any]) -> dict[str, Any]:
|
|
58
|
+
objects = raw.get("objects") or []
|
|
59
|
+
first = objects[0] if objects else {}
|
|
60
|
+
return {
|
|
61
|
+
"content": first.get("text") or raw.get("markdown") or "",
|
|
62
|
+
"title": first.get("title") or raw.get("title"),
|
|
63
|
+
"pageUrl": first.get("pageUrl") or raw.get("url"),
|
|
64
|
+
"resolvedPageUrl": first.get("resolvedPageUrl"),
|
|
65
|
+
"type": first.get("type") or raw.get("type"),
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
def _run(
|
|
69
|
+
self,
|
|
70
|
+
url: str,
|
|
71
|
+
api: str = "analyze",
|
|
72
|
+
fmt: str = "markdown",
|
|
73
|
+
run_manager: CallbackManagerForToolRun | None = None,
|
|
74
|
+
) -> dict[str, Any]:
|
|
75
|
+
try:
|
|
76
|
+
with self._sync_db() as db:
|
|
77
|
+
raw = db.extract(url, api=api, fmt=fmt)
|
|
78
|
+
except ExtractionError as e:
|
|
79
|
+
return {"error": str(e), "errorCode": e.error_code}
|
|
80
|
+
return self._shape_response(raw)
|
|
81
|
+
|
|
82
|
+
async def _arun(
|
|
83
|
+
self,
|
|
84
|
+
url: str,
|
|
85
|
+
api: str = "analyze",
|
|
86
|
+
fmt: str = "markdown",
|
|
87
|
+
run_manager: AsyncCallbackManagerForToolRun | None = None,
|
|
88
|
+
) -> dict[str, Any]:
|
|
89
|
+
try:
|
|
90
|
+
async with self._async_db() as db:
|
|
91
|
+
raw = await db.extract(url, api=api, fmt=fmt)
|
|
92
|
+
except ExtractionError as e:
|
|
93
|
+
return {"error": str(e), "errorCode": e.error_code}
|
|
94
|
+
return self._shape_response(raw)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class _DiffbotWebSearchInput(BaseModel):
|
|
98
|
+
text: str = Field(description="Natural-language search query.")
|
|
99
|
+
num_results: int | None = Field(
|
|
100
|
+
default=None, description="Max results to return. Server default if unset."
|
|
101
|
+
)
|
|
102
|
+
max_tokens: int | None = Field(
|
|
103
|
+
default=None, description="Optional total content-token cap."
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
class DiffbotWebSearchTool(_BaseDiffbotComponent, BaseTool):
|
|
108
|
+
"""Tool that performs a Diffbot web search and returns the raw result list.
|
|
109
|
+
|
|
110
|
+
Use this when the agent needs the search results as-is (with score, title,
|
|
111
|
+
pageUrl, content). For LangChain `Document` output use
|
|
112
|
+
`DiffbotWebSearchRetriever` instead.
|
|
113
|
+
"""
|
|
114
|
+
|
|
115
|
+
name: str = "diffbot_web_search"
|
|
116
|
+
description: str = (
|
|
117
|
+
"Search the web via Diffbot. Returns a list of results, each with "
|
|
118
|
+
"title, pageUrl, score, and content."
|
|
119
|
+
)
|
|
120
|
+
args_schema: type[BaseModel] = _DiffbotWebSearchInput
|
|
121
|
+
|
|
122
|
+
def _run(
|
|
123
|
+
self,
|
|
124
|
+
text: str,
|
|
125
|
+
num_results: int | None = None,
|
|
126
|
+
max_tokens: int | None = None,
|
|
127
|
+
run_manager: CallbackManagerForToolRun | None = None,
|
|
128
|
+
) -> list[dict[str, Any]]:
|
|
129
|
+
with self._sync_db() as db:
|
|
130
|
+
body = db.web_search(text, num_results=num_results, max_tokens=max_tokens)
|
|
131
|
+
return body.get("search_results", [])
|
|
132
|
+
|
|
133
|
+
async def _arun(
|
|
134
|
+
self,
|
|
135
|
+
text: str,
|
|
136
|
+
num_results: int | None = None,
|
|
137
|
+
max_tokens: int | None = None,
|
|
138
|
+
run_manager: AsyncCallbackManagerForToolRun | None = None,
|
|
139
|
+
) -> list[dict[str, Any]]:
|
|
140
|
+
async with self._async_db() as db:
|
|
141
|
+
body = await db.web_search(
|
|
142
|
+
text, num_results=num_results, max_tokens=max_tokens
|
|
143
|
+
)
|
|
144
|
+
return body.get("search_results", [])
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
class _DiffbotEntitiesInput(BaseModel):
|
|
148
|
+
text: str = Field(description="Text to extract entities and sentiment from.")
|
|
149
|
+
lang: str = Field(default="auto", description="Language hint (`auto` to detect).")
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
class DiffbotEntitiesTool(_BaseDiffbotComponent, BaseTool):
|
|
153
|
+
"""Tool that identifies entities and sentiment in text via Diffbot NLP.
|
|
154
|
+
|
|
155
|
+
Returns the SDK response dict as-is — it's small (entity list + sentiment).
|
|
156
|
+
Entity IDs in the response can be looked up in the KG via
|
|
157
|
+
`DiffbotKnowledgeGraphTool` or `DiffbotKnowledgeGraphRetriever` using
|
|
158
|
+
`id:or("id1","id2",...)`.
|
|
159
|
+
"""
|
|
160
|
+
|
|
161
|
+
name: str = "diffbot_entities"
|
|
162
|
+
description: str = (
|
|
163
|
+
"Identify entities (people, organizations, places, ...) and sentiment "
|
|
164
|
+
"in a piece of text. Returns entity IDs that can be looked up in the "
|
|
165
|
+
"Diffbot Knowledge Graph."
|
|
166
|
+
)
|
|
167
|
+
args_schema: type[BaseModel] = _DiffbotEntitiesInput
|
|
168
|
+
|
|
169
|
+
def _run(
|
|
170
|
+
self,
|
|
171
|
+
text: str,
|
|
172
|
+
lang: str = "auto",
|
|
173
|
+
run_manager: CallbackManagerForToolRun | None = None,
|
|
174
|
+
) -> dict[str, Any]:
|
|
175
|
+
with self._sync_db() as db:
|
|
176
|
+
return db.entities(text, lang=lang)
|
|
177
|
+
|
|
178
|
+
async def _arun(
|
|
179
|
+
self,
|
|
180
|
+
text: str,
|
|
181
|
+
lang: str = "auto",
|
|
182
|
+
run_manager: AsyncCallbackManagerForToolRun | None = None,
|
|
183
|
+
) -> dict[str, Any]:
|
|
184
|
+
async with self._async_db() as db:
|
|
185
|
+
return await db.entities(text, lang=lang)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
class _DiffbotDQLInput(BaseModel):
|
|
189
|
+
query: str = Field(
|
|
190
|
+
description='DQL query, e.g. `type:Organization name:"Diffbot"`.'
|
|
191
|
+
)
|
|
192
|
+
size: int = Field(default=10, description="Max results.")
|
|
193
|
+
from_: int = Field(default=0, description="Result offset.")
|
|
194
|
+
filter: str | None = Field(
|
|
195
|
+
default=None, description="Optional DQL filter expression."
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
class DiffbotKnowledgeGraphTool(_BaseDiffbotComponent, BaseTool):
|
|
200
|
+
"""Tool that runs a DQL query against the Diffbot Knowledge Graph.
|
|
201
|
+
|
|
202
|
+
Returns the raw response dict (with `data`, `hits`, etc.). For LangChain
|
|
203
|
+
`Document` output use `DiffbotKnowledgeGraphRetriever` instead.
|
|
204
|
+
|
|
205
|
+
Best for agents that have been instructed in DQL syntax.
|
|
206
|
+
"""
|
|
207
|
+
|
|
208
|
+
name: str = "diffbot_knowledge_graph"
|
|
209
|
+
description: str = (
|
|
210
|
+
"Query the Diffbot Knowledge Graph with a DQL expression "
|
|
211
|
+
'(e.g. `type:Organization location.city.name:"Boston"`). '
|
|
212
|
+
"Returns the raw response — use only if you know DQL syntax."
|
|
213
|
+
)
|
|
214
|
+
args_schema: type[BaseModel] = _DiffbotDQLInput
|
|
215
|
+
|
|
216
|
+
def _run(
|
|
217
|
+
self,
|
|
218
|
+
query: str,
|
|
219
|
+
size: int = 10,
|
|
220
|
+
from_: int = 0,
|
|
221
|
+
filter: str | None = None,
|
|
222
|
+
run_manager: CallbackManagerForToolRun | None = None,
|
|
223
|
+
) -> dict[str, Any]:
|
|
224
|
+
with self._sync_db() as db:
|
|
225
|
+
body = db.dql(query, size=size, from_=from_, filter=filter)
|
|
226
|
+
if not isinstance(body, dict):
|
|
227
|
+
msg = "Unexpected non-JSON DQL response."
|
|
228
|
+
raise TypeError(msg)
|
|
229
|
+
return body
|
|
230
|
+
|
|
231
|
+
async def _arun(
|
|
232
|
+
self,
|
|
233
|
+
query: str,
|
|
234
|
+
size: int = 10,
|
|
235
|
+
from_: int = 0,
|
|
236
|
+
filter: str | None = None,
|
|
237
|
+
run_manager: AsyncCallbackManagerForToolRun | None = None,
|
|
238
|
+
) -> dict[str, Any]:
|
|
239
|
+
async with self._async_db() as db:
|
|
240
|
+
body = await db.dql(query, size=size, from_=from_, filter=filter)
|
|
241
|
+
if not isinstance(body, dict):
|
|
242
|
+
msg = "Unexpected non-JSON DQL response."
|
|
243
|
+
raise TypeError(msg)
|
|
244
|
+
return body
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
_OntologyOp = Literal[
|
|
248
|
+
"types", "composites", "enums", "taxonomies", "fields", "taxonomy", "enum", "search"
|
|
249
|
+
]
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _ontology_lookup(
|
|
253
|
+
ont: Ontology, op: str, name: str | None, search: str | None
|
|
254
|
+
) -> list[str] | dict[str, str]:
|
|
255
|
+
"""Run one ontology navigation op, returning a list or a recoverable error dict.
|
|
256
|
+
|
|
257
|
+
`KeyError` (unknown type/taxonomy/enum) and missing-argument `ValueError`s are
|
|
258
|
+
returned as `{"error": ...}` so the agent can correct itself and retry rather
|
|
259
|
+
than the tool call raising.
|
|
260
|
+
"""
|
|
261
|
+
try:
|
|
262
|
+
if op == "types":
|
|
263
|
+
return ont.types()
|
|
264
|
+
if op == "composites":
|
|
265
|
+
return ont.composites()
|
|
266
|
+
if op == "enums":
|
|
267
|
+
return ont.enums()
|
|
268
|
+
if op == "taxonomies":
|
|
269
|
+
return ont.taxonomies()
|
|
270
|
+
if op == "fields":
|
|
271
|
+
if not name:
|
|
272
|
+
msg = "op='fields' requires `name` (the entity type or composite)."
|
|
273
|
+
raise ValueError(msg)
|
|
274
|
+
fields = ont.fields_for(name)
|
|
275
|
+
return [
|
|
276
|
+
Ontology.format_field(n, m)
|
|
277
|
+
for n, m in Ontology.filter_fields(fields, search)
|
|
278
|
+
]
|
|
279
|
+
if op == "taxonomy":
|
|
280
|
+
if not name:
|
|
281
|
+
msg = "op='taxonomy' requires `name` (the taxonomy)."
|
|
282
|
+
raise ValueError(msg)
|
|
283
|
+
return ont.taxonomy_values(name, search)
|
|
284
|
+
if op == "enum":
|
|
285
|
+
if not name:
|
|
286
|
+
msg = "op='enum' requires `name` (the enum)."
|
|
287
|
+
raise ValueError(msg)
|
|
288
|
+
return ont.enum_values(name)
|
|
289
|
+
if op == "search":
|
|
290
|
+
if not name:
|
|
291
|
+
msg = "op='search' requires `name` (the regex to match)."
|
|
292
|
+
raise ValueError(msg)
|
|
293
|
+
return ont.find_named(name)
|
|
294
|
+
msg = f"Unknown op {op!r}."
|
|
295
|
+
raise ValueError(msg)
|
|
296
|
+
except (KeyError, ValueError) as exc:
|
|
297
|
+
return {
|
|
298
|
+
"error": str(exc).strip('"'),
|
|
299
|
+
"hint": (
|
|
300
|
+
"Valid ops: types, composites, enums, taxonomies, fields, "
|
|
301
|
+
"taxonomy, enum, search. List names first (e.g. op='types') "
|
|
302
|
+
"before drilling into fields/taxonomy/enum."
|
|
303
|
+
),
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
class _DiffbotOntologyInput(BaseModel):
|
|
308
|
+
op: _OntologyOp = Field(
|
|
309
|
+
description=(
|
|
310
|
+
"Which part of the ontology to inspect: `types`/`composites`/`enums`/"
|
|
311
|
+
"`taxonomies` list names; `fields` lists the fields of a type or "
|
|
312
|
+
"composite; `taxonomy`/`enum` list a named taxonomy's/enum's values; "
|
|
313
|
+
"`search` matches any name anywhere in the ontology by regex."
|
|
314
|
+
)
|
|
315
|
+
)
|
|
316
|
+
name: str | None = Field(
|
|
317
|
+
default=None,
|
|
318
|
+
description=(
|
|
319
|
+
"Target name for `fields` (a type/composite), `taxonomy`, or `enum`; "
|
|
320
|
+
"the regex pattern for `search`. Unused by the list ops."
|
|
321
|
+
),
|
|
322
|
+
)
|
|
323
|
+
search: str | None = Field(
|
|
324
|
+
default=None,
|
|
325
|
+
description=(
|
|
326
|
+
"Optional case-insensitive regex to filter `fields` or `taxonomy` results."
|
|
327
|
+
),
|
|
328
|
+
)
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
class DiffbotOntologyTool(_BaseDiffbotComponent, BaseTool):
|
|
332
|
+
"""Tool that navigates the Diffbot Knowledge Graph ontology.
|
|
333
|
+
|
|
334
|
+
Lets an agent discover real entity types, field paths, taxonomy values, and
|
|
335
|
+
enum values *before* writing a DQL query — so it constructs valid queries
|
|
336
|
+
instead of guessing field names. The ontology is fetched once over HTTP via
|
|
337
|
+
`Diffbot.dql_fetch_ontology()` and cached in memory on the tool instance for
|
|
338
|
+
the rest of its lifetime (pass `refresh=True` in a call to re-fetch).
|
|
339
|
+
"""
|
|
340
|
+
|
|
341
|
+
name: str = "diffbot_ontology"
|
|
342
|
+
description: str = (
|
|
343
|
+
"Inspect the Diffbot Knowledge Graph schema to build valid DQL. Ops: "
|
|
344
|
+
"`types`/`composites`/`enums`/`taxonomies` (list names), `fields` "
|
|
345
|
+
"(fields of a type/composite — pass `name`), `taxonomy`/`enum` (values "
|
|
346
|
+
"of a named taxonomy/enum — pass `name`), `search` (regex over all "
|
|
347
|
+
"names — pass the pattern as `name`). Look fields up here before "
|
|
348
|
+
"writing a DQL query."
|
|
349
|
+
)
|
|
350
|
+
args_schema: type[BaseModel] = _DiffbotOntologyInput
|
|
351
|
+
|
|
352
|
+
_ontology: Ontology | None = PrivateAttr(default=None)
|
|
353
|
+
|
|
354
|
+
def _run(
|
|
355
|
+
self,
|
|
356
|
+
op: str,
|
|
357
|
+
name: str | None = None,
|
|
358
|
+
search: str | None = None,
|
|
359
|
+
run_manager: CallbackManagerForToolRun | None = None,
|
|
360
|
+
) -> list[str] | dict[str, str]:
|
|
361
|
+
if self._ontology is None:
|
|
362
|
+
with self._sync_db() as db:
|
|
363
|
+
self._ontology = db.dql_fetch_ontology()
|
|
364
|
+
return _ontology_lookup(self._ontology, op, name, search)
|
|
365
|
+
|
|
366
|
+
async def _arun(
|
|
367
|
+
self,
|
|
368
|
+
op: str,
|
|
369
|
+
name: str | None = None,
|
|
370
|
+
search: str | None = None,
|
|
371
|
+
run_manager: AsyncCallbackManagerForToolRun | None = None,
|
|
372
|
+
) -> list[str] | dict[str, str]:
|
|
373
|
+
if self._ontology is None:
|
|
374
|
+
async with self._async_db() as db:
|
|
375
|
+
self._ontology = await db.dql_fetch_ontology()
|
|
376
|
+
return _ontology_lookup(self._ontology, op, name, search)
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
class _DiffbotDQLProbeInput(BaseModel):
|
|
380
|
+
queries: list[str] = Field(
|
|
381
|
+
description=(
|
|
382
|
+
"DQL query variants to probe. Each is run with size=0 (hit count only)."
|
|
383
|
+
)
|
|
384
|
+
)
|
|
385
|
+
workers: int = Field(default=8, description="Max concurrent requests.")
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
class DiffbotDQLProbeTool(_BaseDiffbotComponent, BaseTool):
|
|
389
|
+
"""Tool that probes DQL query variants in parallel, returning hit counts only.
|
|
390
|
+
|
|
391
|
+
Each query runs with `size=0`, so the response carries the match count but no
|
|
392
|
+
entity data — cheap and fast. Use it to check a query's selectivity (too
|
|
393
|
+
broad? too narrow?) and compare variants before committing to a full
|
|
394
|
+
`diffbot_knowledge_graph` query. Backed by `Diffbot.dql_parallel()`.
|
|
395
|
+
"""
|
|
396
|
+
|
|
397
|
+
name: str = "diffbot_dql_probe"
|
|
398
|
+
description: str = (
|
|
399
|
+
"Probe one or more DQL query variants in parallel and get the hit count "
|
|
400
|
+
"for each (size=0, no entity data). Use to validate that a query is "
|
|
401
|
+
"well-shaped — not matching zero results or millions — before running it."
|
|
402
|
+
)
|
|
403
|
+
args_schema: type[BaseModel] = _DiffbotDQLProbeInput
|
|
404
|
+
|
|
405
|
+
@staticmethod
|
|
406
|
+
def _shape(queries: list[str], results: list[Any]) -> list[dict[str, Any]]:
|
|
407
|
+
return [
|
|
408
|
+
{
|
|
409
|
+
"query": q,
|
|
410
|
+
"hits": r.get("hits") if isinstance(r, dict) else None,
|
|
411
|
+
}
|
|
412
|
+
for q, r in zip(queries, results, strict=False)
|
|
413
|
+
]
|
|
414
|
+
|
|
415
|
+
def _run(
|
|
416
|
+
self,
|
|
417
|
+
queries: list[str],
|
|
418
|
+
workers: int = 8,
|
|
419
|
+
run_manager: CallbackManagerForToolRun | None = None,
|
|
420
|
+
) -> list[dict[str, Any]]:
|
|
421
|
+
reqs = [{"query": q, "size": 0} for q in queries]
|
|
422
|
+
with self._sync_db() as db:
|
|
423
|
+
results = db.dql_parallel(reqs, workers=workers)
|
|
424
|
+
return self._shape(queries, results)
|
|
425
|
+
|
|
426
|
+
async def _arun(
|
|
427
|
+
self,
|
|
428
|
+
queries: list[str],
|
|
429
|
+
workers: int = 8,
|
|
430
|
+
run_manager: AsyncCallbackManagerForToolRun | None = None,
|
|
431
|
+
) -> list[dict[str, Any]]:
|
|
432
|
+
reqs = [{"query": q, "size": 0} for q in queries]
|
|
433
|
+
async with self._async_db() as db:
|
|
434
|
+
results = await db.dql_parallel(reqs, workers=workers)
|
|
435
|
+
return self._shape(queries, results)
|
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: langchain-diffbot
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: LangChain integration for the Diffbot Knowledge Graph and Extract APIs
|
|
5
|
+
Project-URL: Homepage, https://www.diffbot.com/
|
|
6
|
+
Project-URL: Repository, https://github.com/diffbot/langchain-diffbot
|
|
7
|
+
Project-URL: Issues, https://github.com/diffbot/langchain-diffbot/issues
|
|
8
|
+
Author: Diffbot
|
|
9
|
+
License: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Requires-Python: <4.0,>=3.10
|
|
21
|
+
Requires-Dist: diffbot-python>=0.2.1
|
|
22
|
+
Requires-Dist: httpx<1.0,>=0.27
|
|
23
|
+
Requires-Dist: langchain-core<2.0,>=1.0
|
|
24
|
+
Provides-Extra: examples
|
|
25
|
+
Requires-Dist: fastapi<1.0,>=0.115; extra == 'examples'
|
|
26
|
+
Requires-Dist: langchain-anthropic<2.0,>=1.4; extra == 'examples'
|
|
27
|
+
Requires-Dist: langchain<2.0,>=1.3; extra == 'examples'
|
|
28
|
+
Requires-Dist: langsmith<1.0,>=0.1; extra == 'examples'
|
|
29
|
+
Requires-Dist: python-dotenv<2.0,>=1.0; extra == 'examples'
|
|
30
|
+
Requires-Dist: uvicorn[standard]<1.0,>=0.30; extra == 'examples'
|
|
31
|
+
Description-Content-Type: text/markdown
|
|
32
|
+
|
|
33
|
+
# langchain-diffbot
|
|
34
|
+
|
|
35
|
+
A thin LangChain integration over the official [`diffbot-python`](https://github.com/diffbot/diffbot-python) SDK. Every Diffbot API gets the closest LangChain primitive:
|
|
36
|
+
|
|
37
|
+
| Diffbot API | LangChain class(es) |
|
|
38
|
+
| --- | --- |
|
|
39
|
+
| Knowledge Graph (DQL) | `DiffbotKnowledgeGraphRetriever`, `DiffbotKnowledgeGraphTool` |
|
|
40
|
+
| Web Search | `DiffbotWebSearchRetriever`, `DiffbotWebSearchTool` |
|
|
41
|
+
| Extract (Analyze) | `DiffbotExtractTool`, `DiffbotExtractLoader` |
|
|
42
|
+
| NLP entities | `DiffbotEntitiesTool` |
|
|
43
|
+
| Crawl | `DiffbotCrawlLoader` |
|
|
44
|
+
| LLM RAG (`ask`) | `ChatDiffbot` (with native streaming) |
|
|
45
|
+
|
|
46
|
+
## Installation
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install langchain-diffbot
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Authentication
|
|
53
|
+
|
|
54
|
+
Get an API token at https://app.diffbot.com/get-started/ and export it:
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
export DIFFBOT_API_TOKEN="..."
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
Every class also accepts `diffbot_api_token=...` directly, or a pre-built `diffbot.Diffbot` client via `client=...` (see [Bring-your-own-client](#bring-your-own-client) below).
|
|
61
|
+
|
|
62
|
+
## Quickstart — Knowledge Graph retriever
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
from langchain_diffbot import DiffbotKnowledgeGraphRetriever
|
|
66
|
+
|
|
67
|
+
retriever = DiffbotKnowledgeGraphRetriever(k=5)
|
|
68
|
+
docs = retriever.invoke("type:Organization industries:\"Artificial Intelligence\" location.city.name:\"Boston\"")
|
|
69
|
+
for d in docs:
|
|
70
|
+
print(d.metadata["name"], "—", d.page_content[:120])
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
The query string is a [DQL (Diffbot Query Language)](https://docs.diffbot.com/reference/dql-quickstart) expression.
|
|
74
|
+
|
|
75
|
+
## Shaping the output
|
|
76
|
+
|
|
77
|
+
Diffbot KG entities and web-search results are large. Dumping them straight into an LLM prompt can blow past per-minute input-token limits in a single call. Both retrievers expose three shaping knobs:
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
from langchain_core.documents import Document
|
|
81
|
+
from langchain_diffbot import DiffbotKnowledgeGraphRetriever
|
|
82
|
+
|
|
83
|
+
# 1. Project only the top-level fields you care about. Drops everything else
|
|
84
|
+
# from `metadata`. Recommended for agent / tool-use scenarios.
|
|
85
|
+
retriever = DiffbotKnowledgeGraphRetriever(
|
|
86
|
+
k=5,
|
|
87
|
+
fields=["id", "type", "name", "homepageUri", "nbEmployees"],
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
# 2. Choose which field becomes `page_content`. First non-empty value wins.
|
|
91
|
+
retriever = DiffbotKnowledgeGraphRetriever(
|
|
92
|
+
content_fields=["summary", "description", "name"],
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
# 3. For total control, pass a `document_mapper` that turns a raw entity
|
|
96
|
+
# dict into whatever Document shape you want.
|
|
97
|
+
def mapper(entity: dict) -> Document:
|
|
98
|
+
return Document(
|
|
99
|
+
page_content=entity.get("summary", ""),
|
|
100
|
+
metadata={"id": entity["id"], "name": entity["name"]},
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
retriever = DiffbotKnowledgeGraphRetriever(document_mapper=mapper)
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
`fields` and `content_fields` are ignored when `document_mapper` is set. The same knobs work on `DiffbotWebSearchRetriever`.
|
|
107
|
+
|
|
108
|
+
## Web search
|
|
109
|
+
|
|
110
|
+
```python
|
|
111
|
+
from langchain_diffbot import DiffbotWebSearchRetriever
|
|
112
|
+
|
|
113
|
+
web = DiffbotWebSearchRetriever(k=5, fields=["title", "pageUrl", "score"])
|
|
114
|
+
docs = web.invoke("diffbot knowledge graph llm grounding")
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
## Extract a URL
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
from langchain_diffbot import DiffbotExtractTool, DiffbotExtractLoader
|
|
121
|
+
|
|
122
|
+
# Single URL
|
|
123
|
+
tool = DiffbotExtractTool()
|
|
124
|
+
page = tool.invoke({"url": "https://www.diffbot.com/products/extract/"})
|
|
125
|
+
|
|
126
|
+
# Batch — yields one Document per URL, sync or async
|
|
127
|
+
loader = DiffbotExtractLoader(urls=["https://example.com", "https://diffbot.com"])
|
|
128
|
+
for doc in loader.lazy_load():
|
|
129
|
+
print(doc.metadata["title"], doc.page_content[:200])
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
`DiffbotExtractTool` returns a structured `{"error": ..., "errorCode": ...}` dict when Diffbot reports an extraction failure (200 with `errorCode`), so agents can react and try another URL instead of catching an exception. Auth / rate-limit errors propagate as `diffbot.errors.AuthError` / `RateLimitError`.
|
|
133
|
+
|
|
134
|
+
## ChatDiffbot
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
from langchain_core.messages import HumanMessage
|
|
138
|
+
from langchain_diffbot import ChatDiffbot
|
|
139
|
+
|
|
140
|
+
llm = ChatDiffbot()
|
|
141
|
+
|
|
142
|
+
for chunk in llm.stream([HumanMessage(content="What is the Diffbot Knowledge Graph?")]):
|
|
143
|
+
print(chunk.content, end="", flush=True)
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
`_stream` / `_astream` are native — no thread-pool fallback. `.invoke()` aggregates the stream into a single message.
|
|
147
|
+
|
|
148
|
+
## Using a retriever in a chain
|
|
149
|
+
|
|
150
|
+
The retrievers are standard `BaseRetriever`s, so they slot into LCEL like any other:
|
|
151
|
+
|
|
152
|
+
```python
|
|
153
|
+
from langchain_anthropic import ChatAnthropic
|
|
154
|
+
from langchain_core.output_parsers import StrOutputParser
|
|
155
|
+
from langchain_core.prompts import ChatPromptTemplate
|
|
156
|
+
from langchain_core.runnables import RunnablePassthrough
|
|
157
|
+
from langchain_diffbot import DiffbotKnowledgeGraphRetriever
|
|
158
|
+
|
|
159
|
+
retriever = DiffbotKnowledgeGraphRetriever(
|
|
160
|
+
k=5,
|
|
161
|
+
fields=["id", "name", "homepageUri", "nbEmployees", "industries"],
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
prompt = ChatPromptTemplate.from_template(
|
|
165
|
+
"Answer using only this Diffbot KG context:\n\n{context}\n\nQuestion: {question}"
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _format(docs):
|
|
170
|
+
return "\n---\n".join(
|
|
171
|
+
f"{d.metadata.get('name')} (id={d.metadata.get('id')}): {d.page_content}"
|
|
172
|
+
for d in docs
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
chain = (
|
|
177
|
+
{"context": retriever | _format, "question": RunnablePassthrough()}
|
|
178
|
+
| prompt
|
|
179
|
+
| ChatAnthropic(model="claude-sonnet-4-6")
|
|
180
|
+
| StrOutputParser()
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
chain.invoke('type:Organization location.city.name:"Boston" industries:"Biotech"')
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
## Bring-your-own-client
|
|
187
|
+
|
|
188
|
+
Every class accepts a pre-built `diffbot.Diffbot` (or `diffbot.DiffbotAsync`) via `client` / `async_client`. The package uses it as-is and **does not close it** — you own the lifecycle. This is the escape hatch for anything the SDK supports that's not re-exposed as a field (custom URLs, `transport=`, shared connection pools, custom headers).
|
|
189
|
+
|
|
190
|
+
```python
|
|
191
|
+
from diffbot import Diffbot
|
|
192
|
+
from langchain_diffbot import DiffbotKnowledgeGraphRetriever
|
|
193
|
+
|
|
194
|
+
# One client shared across many retriever calls (no per-call httpx pool churn)
|
|
195
|
+
shared = Diffbot(token="...", timeout=60.0)
|
|
196
|
+
retriever = DiffbotKnowledgeGraphRetriever(client=shared, k=5)
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
## Examples
|
|
200
|
+
|
|
201
|
+
The [`examples/`](./examples) folder has runnable demos:
|
|
202
|
+
|
|
203
|
+
- [`examples/quickstart/`](./examples/quickstart) — full tour: every public class, output shaping, async, and a multi-tool research agent.
|
|
204
|
+
- [`examples/company_research/`](./examples/company_research) — the same multi-tool agent as a one-shot CLI: `cd examples && python -m company_research "your question"`. The agent combines KG search + web search + URL extract.
|
|
205
|
+
|
|
206
|
+
Both need `langchain` + `langchain-anthropic` on top of the base package — install the extra:
|
|
207
|
+
|
|
208
|
+
```bash
|
|
209
|
+
pip install "langchain-diffbot[examples]"
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
## Development
|
|
213
|
+
|
|
214
|
+
```bash
|
|
215
|
+
uv sync --all-groups
|
|
216
|
+
uv run pytest tests/unit_tests
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
## Releasing
|
|
220
|
+
|
|
221
|
+
Tokens are stored in macOS Keychain so the Makefile can pull them automatically — no plaintext on disk, no shell-history leaks. First-time setup:
|
|
222
|
+
|
|
223
|
+
```bash
|
|
224
|
+
make set-token-testpypi # prompts; input is hidden as you paste
|
|
225
|
+
make set-token-pypi # same, for real PyPI
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
Both targets read with `bash read -rsp` (hidden input), overwrite any existing entry, and never put the token in `make` output or shell history. Re-run either one any time you rotate a token.
|
|
229
|
+
|
|
230
|
+
Then the release flow per version:
|
|
231
|
+
|
|
232
|
+
```bash
|
|
233
|
+
# 1. Bump the version (edits pyproject.toml in place via `uv version --bump`)
|
|
234
|
+
make bump-patch # 0.1.0 → 0.1.1
|
|
235
|
+
# or: make bump-minor # 0.1.0 → 0.2.0
|
|
236
|
+
# or: make bump-major # 0.1.0 → 1.0.0
|
|
237
|
+
|
|
238
|
+
# 2. Publish to TestPyPI and verify installable
|
|
239
|
+
make release-test
|
|
240
|
+
make verify-release-test
|
|
241
|
+
|
|
242
|
+
# 3. Publish to real PyPI (prompts for the version to confirm)
|
|
243
|
+
make release
|
|
244
|
+
make verify-release
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
`make release-test` and `make release` will refuse to publish if the current `pyproject.toml` version is already on the target index — so the workflow is "bump → release-test → verify → release → verify".
|
|
248
|
+
|
|
249
|
+
To rotate: revoke the old token at https://pypi.org/manage/account/token/ (or the TestPyPI equivalent), then run `make set-token-pypi` / `make set-token-testpypi` again — it overwrites the existing Keychain entry without prompting.
|
|
250
|
+
|
|
251
|
+
Integration tests hit the live Diffbot API and require `DIFFBOT_API_TOKEN`:
|
|
252
|
+
|
|
253
|
+
```bash
|
|
254
|
+
uv run pytest tests/integration_tests
|
|
255
|
+
```
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
langchain_diffbot/__init__.py,sha256=buggsYKg8Jeb0svJV4PgUKSuvK4CuTXVP1j9HEhSHX0,1139
|
|
2
|
+
langchain_diffbot/_base.py,sha256=eKJ6tZXSkbijTIL0Sal01w0IGSmBtcgo7r0O3CYCZ28,4233
|
|
3
|
+
langchain_diffbot/chat_models.py,sha256=hxwSoOHT27NgAbVQdLh6z93DS_PzCvqmPfGt3H3w3wc,4193
|
|
4
|
+
langchain_diffbot/document_loaders.py,sha256=L7bE69gwnIPn7cqC3oeccuxMe6f3Lz6n7ovdq5H_BBA,5252
|
|
5
|
+
langchain_diffbot/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
6
|
+
langchain_diffbot/retrievers.py,sha256=flW6CVfJUANGOgNbEdEc0zEsiRBwLICthlXBnbzRLPg,9903
|
|
7
|
+
langchain_diffbot/tools.py,sha256=9jrie3sHjDMYh3f_hYg448w2M_tn7tvpaqm_wHXrRT0,15782
|
|
8
|
+
langchain_diffbot-0.1.0.dist-info/METADATA,sha256=298LPyP0yGBLFaooTZgphjEWZNuzK7T7hwt-8KLA2xU,9440
|
|
9
|
+
langchain_diffbot-0.1.0.dist-info/WHEEL,sha256=qtCwoSJWgHk21S1Kb4ihdzI2rlJ1ZKaIurTj_ngOhyQ,87
|
|
10
|
+
langchain_diffbot-0.1.0.dist-info/licenses/LICENSE,sha256=ctCt1arYbBpFK3id9IIqaBMAaU3gwnWTmnbzw4bVDSo,1064
|
|
11
|
+
langchain_diffbot-0.1.0.dist-info/RECORD,,
|