diffbot 3.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffbot/__init__.py +63 -0
- diffbot/_auth.py +41 -0
- diffbot/ask.py +195 -0
- diffbot/cli/__init__.py +430 -0
- diffbot/cli/__main__.py +4 -0
- diffbot/cli/_common.py +21 -0
- diffbot/cli/dql.py +310 -0
- diffbot/cli/entities.py +155 -0
- diffbot/cli/ontology.py +74 -0
- diffbot/client.py +362 -0
- diffbot/crawl.py +270 -0
- diffbot/errors.py +51 -0
- diffbot/extract.py +45 -0
- diffbot/kg.py +128 -0
- diffbot/nlp.py +37 -0
- diffbot/ontology.py +160 -0
- diffbot/web_search.py +44 -0
- diffbot-3.0.0.dist-info/METADATA +410 -0
- diffbot-3.0.0.dist-info/RECORD +22 -0
- diffbot-3.0.0.dist-info/WHEEL +4 -0
- diffbot-3.0.0.dist-info/entry_points.txt +2 -0
- diffbot-3.0.0.dist-info/licenses/LICENSE +21 -0
diffbot/extract.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Diffbot Analyze API: extract structured content from a URL."""
|
|
2
|
+
|
|
3
|
+
from typing import TYPE_CHECKING, Any, Dict
|
|
4
|
+
|
|
5
|
+
if TYPE_CHECKING:
|
|
6
|
+
from .client import Diffbot, DiffbotAsync
|
|
7
|
+
|
|
8
|
+
from .errors import ExtractionError
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _build_params(client: Any, url: str, fmt: str) -> Dict[str, Any]:
|
|
12
|
+
params = {"token": client.token, "url": url, "timeout": 30000}
|
|
13
|
+
if fmt == "markdown":
|
|
14
|
+
params["mode"] = "llm"
|
|
15
|
+
return params
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _parse_response(client: Any, data: Dict[str, Any]) -> Dict[str, Any]:
|
|
19
|
+
if "errorCode" in data:
|
|
20
|
+
raise ExtractionError(data["errorCode"], data.get("error", ""))
|
|
21
|
+
return data
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _normalize_url(url: str) -> str:
|
|
25
|
+
return url if url.startswith("http") else f"https://{url}"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def extract(client: "Diffbot", url: str, api: str = "analyze", fmt: str = "markdown") -> Dict[str, Any]:
|
|
29
|
+
url = _normalize_url(url)
|
|
30
|
+
response = client._http.get(
|
|
31
|
+
f"{client.analyze_url}/{api}",
|
|
32
|
+
params=_build_params(client, url, fmt),
|
|
33
|
+
)
|
|
34
|
+
client._raise_for_status(response)
|
|
35
|
+
return _parse_response(client, response.json())
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
async def extract_async(client: "DiffbotAsync", url: str, api: str = "analyze", fmt: str = "markdown") -> Dict[str, Any]:
|
|
39
|
+
url = _normalize_url(url)
|
|
40
|
+
response = await client._http.get(
|
|
41
|
+
f"{client.analyze_url}/{api}",
|
|
42
|
+
params=_build_params(client, url, fmt),
|
|
43
|
+
)
|
|
44
|
+
client._raise_for_status(response)
|
|
45
|
+
return _parse_response(client, response.json())
|
diffbot/kg.py
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Diffbot Knowledge Graph APIs: DQL search and entity enhancement."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import pathlib
|
|
5
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
6
|
+
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Sequence, Union
|
|
7
|
+
|
|
8
|
+
from .ontology import Ontology
|
|
9
|
+
|
|
10
|
+
if TYPE_CHECKING:
|
|
11
|
+
from .client import Diffbot, DiffbotAsync
|
|
12
|
+
|
|
13
|
+
KG_DQL_ENDPOINT = "https://kg.diffbot.com/kg/v3/dql"
|
|
14
|
+
KG_ONTOLOGY_ENDPOINT = "https://kg.diffbot.com/kg/ontology"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _build_dql_params(
|
|
18
|
+
client: Any,
|
|
19
|
+
query: str,
|
|
20
|
+
size: int,
|
|
21
|
+
from_: int,
|
|
22
|
+
format: str,
|
|
23
|
+
filter: Optional[str],
|
|
24
|
+
exportspec: Optional[str],
|
|
25
|
+
extra: Optional[Dict[str, str]],
|
|
26
|
+
) -> Dict[str, Any]:
|
|
27
|
+
params: Dict[str, Any] = {"token": client.token, "query": query, "size": size}
|
|
28
|
+
if from_:
|
|
29
|
+
params["from"] = from_
|
|
30
|
+
if format != "json":
|
|
31
|
+
params["format"] = format
|
|
32
|
+
if filter is not None:
|
|
33
|
+
params["filter"] = filter
|
|
34
|
+
if exportspec is not None:
|
|
35
|
+
params["exportspec"] = exportspec
|
|
36
|
+
if extra:
|
|
37
|
+
params.update(extra)
|
|
38
|
+
return params
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def dql(
|
|
42
|
+
client: "Diffbot",
|
|
43
|
+
query: str,
|
|
44
|
+
*,
|
|
45
|
+
size: int = 10,
|
|
46
|
+
from_: int = 0,
|
|
47
|
+
format: str = "json",
|
|
48
|
+
filter: Optional[str] = None,
|
|
49
|
+
exportspec: Optional[str] = None,
|
|
50
|
+
extra: Optional[Dict[str, str]] = None,
|
|
51
|
+
raw: bool = False,
|
|
52
|
+
) -> Union[Dict[str, Any], bytes]:
|
|
53
|
+
params = _build_dql_params(client, query, size, from_, format, filter, exportspec, extra)
|
|
54
|
+
response = client._http.get(client.dql_url, params=params)
|
|
55
|
+
client._raise_for_status(response)
|
|
56
|
+
return response.content if raw else response.json()
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
async def dql_async(
|
|
60
|
+
client: "DiffbotAsync",
|
|
61
|
+
query: str,
|
|
62
|
+
*,
|
|
63
|
+
size: int = 10,
|
|
64
|
+
from_: int = 0,
|
|
65
|
+
format: str = "json",
|
|
66
|
+
filter: Optional[str] = None,
|
|
67
|
+
exportspec: Optional[str] = None,
|
|
68
|
+
extra: Optional[Dict[str, str]] = None,
|
|
69
|
+
raw: bool = False,
|
|
70
|
+
) -> Union[Dict[str, Any], bytes]:
|
|
71
|
+
params = _build_dql_params(client, query, size, from_, format, filter, exportspec, extra)
|
|
72
|
+
response = await client._http.get(client.dql_url, params=params)
|
|
73
|
+
client._raise_for_status(response)
|
|
74
|
+
return response.content if raw else response.json()
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def dql_parallel(
|
|
78
|
+
client: "Diffbot",
|
|
79
|
+
queries: Sequence[Dict[str, Any]],
|
|
80
|
+
*,
|
|
81
|
+
workers: int = 8,
|
|
82
|
+
) -> List[Union[Dict[str, Any], bytes]]:
|
|
83
|
+
if not queries:
|
|
84
|
+
return []
|
|
85
|
+
with ThreadPoolExecutor(max_workers=min(workers, len(queries))) as ex:
|
|
86
|
+
return list(ex.map(lambda q: dql(client, **q), queries))
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
async def dql_parallel_async(
|
|
90
|
+
client: "DiffbotAsync",
|
|
91
|
+
queries: Sequence[Dict[str, Any]],
|
|
92
|
+
*,
|
|
93
|
+
workers: int = 8,
|
|
94
|
+
) -> List[Union[Dict[str, Any], bytes]]:
|
|
95
|
+
if not queries:
|
|
96
|
+
return []
|
|
97
|
+
sem = asyncio.Semaphore(workers)
|
|
98
|
+
|
|
99
|
+
async def _one(q: Dict[str, Any]) -> Union[Dict[str, Any], bytes]:
|
|
100
|
+
async with sem:
|
|
101
|
+
return await dql_async(client, **q)
|
|
102
|
+
|
|
103
|
+
return await asyncio.gather(*(_one(q) for q in queries))
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def dql_refresh_ontology(client: "Diffbot", dest: pathlib.Path) -> None:
|
|
107
|
+
response = client._http.get(client.ontology_url)
|
|
108
|
+
client._raise_for_status(response)
|
|
109
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
110
|
+
dest.write_bytes(response.content)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def dql_fetch_ontology(client: "Diffbot") -> Ontology:
|
|
114
|
+
"""Download the ontology and return it as a queryable :class:`Ontology`.
|
|
115
|
+
|
|
116
|
+
Performs no caching — the caller decides whether and where to hold onto the
|
|
117
|
+
result. Use :func:`dql_refresh_ontology` instead to persist raw bytes to disk.
|
|
118
|
+
"""
|
|
119
|
+
response = client._http.get(client.ontology_url)
|
|
120
|
+
client._raise_for_status(response)
|
|
121
|
+
return Ontology.from_json(response.content)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
async def dql_fetch_ontology_async(client: "DiffbotAsync") -> Ontology:
|
|
125
|
+
"""Async variant of :func:`dql_fetch_ontology`."""
|
|
126
|
+
response = await client._http.get(client.ontology_url)
|
|
127
|
+
client._raise_for_status(response)
|
|
128
|
+
return Ontology.from_json(response.content)
|
diffbot/nlp.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Diffbot NLP API: entity identification, resolution, and sentiment."""
|
|
2
|
+
|
|
3
|
+
from typing import TYPE_CHECKING, Any, Dict
|
|
4
|
+
|
|
5
|
+
if TYPE_CHECKING:
|
|
6
|
+
from .client import Diffbot, DiffbotAsync
|
|
7
|
+
|
|
8
|
+
NLP_BASE = "https://nl.diffbot.com/v1/"
|
|
9
|
+
NLP_FIELDS = "entities,sentiment"
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def entities(
|
|
13
|
+
client: "Diffbot",
|
|
14
|
+
text: str,
|
|
15
|
+
*,
|
|
16
|
+
lang: str = "auto",
|
|
17
|
+
) -> Dict[str, Any]:
|
|
18
|
+
params = {"token": client.token, "fields": NLP_FIELDS}
|
|
19
|
+
payload = [{"lang": lang, "format": "plain text", "content": text}]
|
|
20
|
+
response = client._http.post(client.nlp_url, params=params, json=payload)
|
|
21
|
+
client._raise_for_status(response)
|
|
22
|
+
data = response.json()
|
|
23
|
+
return data[0] if isinstance(data, list) else data
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
async def entities_async(
|
|
27
|
+
client: "DiffbotAsync",
|
|
28
|
+
text: str,
|
|
29
|
+
*,
|
|
30
|
+
lang: str = "auto",
|
|
31
|
+
) -> Dict[str, Any]:
|
|
32
|
+
params = {"token": client.token, "fields": NLP_FIELDS}
|
|
33
|
+
payload = [{"lang": lang, "format": "plain text", "content": text}]
|
|
34
|
+
response = await client._http.post(client.nlp_url, params=params, json=payload)
|
|
35
|
+
client._raise_for_status(response)
|
|
36
|
+
data = response.json()
|
|
37
|
+
return data[0] if isinstance(data, list) else data
|
diffbot/ontology.py
ADDED
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""In-memory navigation of the Diffbot Knowledge Graph ontology.
|
|
2
|
+
|
|
3
|
+
The ontology is a JSON document describing the Knowledge Graph's entity types,
|
|
4
|
+
composite types, enums, and taxonomies. An agent constructing DQL needs it to
|
|
5
|
+
look up real field paths and taxonomy values instead of guessing them.
|
|
6
|
+
|
|
7
|
+
This module is pure and storage-agnostic: build an :class:`Ontology` from
|
|
8
|
+
already-parsed data (or from raw JSON / a file path) and query it. How the
|
|
9
|
+
ontology document is fetched, and whether or where it is cached, is left
|
|
10
|
+
entirely to the caller — the `db` CLI caches it on disk at
|
|
11
|
+
``~/.diffbot/ontology.json``; an in-process consumer (e.g. langchain) can cache
|
|
12
|
+
the :class:`Ontology` in memory. Fetch a fresh one over HTTP with
|
|
13
|
+
:meth:`diffbot.Diffbot.dql_fetch_ontology`.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import json
|
|
17
|
+
import pathlib
|
|
18
|
+
import re
|
|
19
|
+
from typing import Any, Dict, List, Optional, Tuple, Union
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class Ontology:
|
|
23
|
+
"""Queryable view over a parsed Diffbot ontology document.
|
|
24
|
+
|
|
25
|
+
The instance holds the parsed document on :attr:`data` and exposes pure
|
|
26
|
+
lookup methods over it. Nothing here performs I/O — construct with already
|
|
27
|
+
parsed data, or use :meth:`from_json` / :meth:`from_path` for convenience.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(self, data: Dict[str, Any]):
|
|
31
|
+
self.data = data
|
|
32
|
+
|
|
33
|
+
@classmethod
|
|
34
|
+
def from_json(cls, raw: Union[str, bytes]) -> "Ontology":
|
|
35
|
+
"""Build from a raw JSON string or bytes (e.g. an HTTP response body)."""
|
|
36
|
+
return cls(json.loads(raw))
|
|
37
|
+
|
|
38
|
+
@classmethod
|
|
39
|
+
def from_path(cls, path: Union[str, pathlib.Path]) -> "Ontology":
|
|
40
|
+
"""Build from a JSON file on disk."""
|
|
41
|
+
return cls(json.loads(pathlib.Path(path).read_text()))
|
|
42
|
+
|
|
43
|
+
def types(self) -> List[str]:
|
|
44
|
+
"""All entity type names (e.g. ``Organization``, ``Person``)."""
|
|
45
|
+
return sorted(self.data.get("types", {}).keys())
|
|
46
|
+
|
|
47
|
+
def composites(self) -> List[str]:
|
|
48
|
+
"""All composite type names (e.g. ``Location``, ``Employment``)."""
|
|
49
|
+
return sorted(self.data.get("composites", {}).keys())
|
|
50
|
+
|
|
51
|
+
def enums(self) -> List[str]:
|
|
52
|
+
"""All enum type names (e.g. ``Language``, ``Gender``)."""
|
|
53
|
+
return sorted(self.data.get("enums", {}).keys())
|
|
54
|
+
|
|
55
|
+
def taxonomies(self) -> List[str]:
|
|
56
|
+
"""All taxonomy names (e.g. ``OrganizationCategory``)."""
|
|
57
|
+
return sorted(self.data.get("taxonomies", {}).keys())
|
|
58
|
+
|
|
59
|
+
@staticmethod
|
|
60
|
+
def _fields_of(container: Dict[str, Any], type_name: str) -> Dict[str, Any]:
|
|
61
|
+
entry = container.get(type_name)
|
|
62
|
+
if entry is None:
|
|
63
|
+
raise KeyError(f"Unknown name: {type_name}")
|
|
64
|
+
return entry.get("fields", {})
|
|
65
|
+
|
|
66
|
+
def fields_for(self, type_name: str) -> Dict[str, Any]:
|
|
67
|
+
"""Return the field map of an entity type or composite.
|
|
68
|
+
|
|
69
|
+
Auto-routes: ``type_name`` may be an entity type (``Organization``) or a
|
|
70
|
+
composite (``Location``). Raises ``KeyError`` if it is neither.
|
|
71
|
+
"""
|
|
72
|
+
types = self.data.get("types", {})
|
|
73
|
+
composites = self.data.get("composites", {})
|
|
74
|
+
if type_name in types:
|
|
75
|
+
return self._fields_of(types, type_name)
|
|
76
|
+
if type_name in composites:
|
|
77
|
+
return self._fields_of(composites, type_name)
|
|
78
|
+
raise KeyError(f"{type_name} is not a known entity type or composite")
|
|
79
|
+
|
|
80
|
+
@staticmethod
|
|
81
|
+
def filter_fields(
|
|
82
|
+
fields: Dict[str, Any],
|
|
83
|
+
search: Optional[str],
|
|
84
|
+
include_deprecated: bool = False,
|
|
85
|
+
) -> List[Tuple[str, Dict[str, Any]]]:
|
|
86
|
+
"""Filter a field map by a name regex, dropping deprecated by default."""
|
|
87
|
+
pattern = re.compile(search, re.IGNORECASE) if search else None
|
|
88
|
+
out = []
|
|
89
|
+
for name, meta in fields.items():
|
|
90
|
+
if not include_deprecated and meta.get("isDeprecated"):
|
|
91
|
+
continue
|
|
92
|
+
if pattern and not pattern.search(name):
|
|
93
|
+
continue
|
|
94
|
+
out.append((name, meta))
|
|
95
|
+
return out
|
|
96
|
+
|
|
97
|
+
def taxonomy_values(self, name: str, search: Optional[str] = None) -> List[str]:
|
|
98
|
+
"""Flatten a taxonomy's values (recursing into children), optionally filtered."""
|
|
99
|
+
tax = self.data.get("taxonomies", {}).get(name)
|
|
100
|
+
if tax is None:
|
|
101
|
+
raise KeyError(f"Unknown taxonomy: {name}")
|
|
102
|
+
pattern = re.compile(search, re.IGNORECASE) if search else None
|
|
103
|
+
out: List[str] = []
|
|
104
|
+
|
|
105
|
+
def walk(node: Dict[str, Any]) -> None:
|
|
106
|
+
n = node.get("name")
|
|
107
|
+
if n and (pattern is None or pattern.search(n)):
|
|
108
|
+
out.append(n)
|
|
109
|
+
for child in node.get("children", []) or []:
|
|
110
|
+
walk(child)
|
|
111
|
+
|
|
112
|
+
for cat in tax.get("categories", []) or []:
|
|
113
|
+
walk(cat)
|
|
114
|
+
return out
|
|
115
|
+
|
|
116
|
+
def enum_values(self, name: str) -> List[str]:
|
|
117
|
+
"""Return the allowed values of an enum."""
|
|
118
|
+
enum = self.data.get("enums", {}).get(name)
|
|
119
|
+
if enum is None:
|
|
120
|
+
raise KeyError(f"Unknown enum: {name}")
|
|
121
|
+
return list(enum.get("values", []))
|
|
122
|
+
|
|
123
|
+
def find_named(self, search: str) -> List[str]:
|
|
124
|
+
"""Fallback search: every ``name`` anywhere in the document matching a regex."""
|
|
125
|
+
pattern = re.compile(search, re.IGNORECASE)
|
|
126
|
+
found = set()
|
|
127
|
+
|
|
128
|
+
def walk(node: Any) -> None:
|
|
129
|
+
if isinstance(node, dict):
|
|
130
|
+
n = node.get("name")
|
|
131
|
+
if isinstance(n, str) and pattern.search(n):
|
|
132
|
+
found.add(n)
|
|
133
|
+
for v in node.values():
|
|
134
|
+
walk(v)
|
|
135
|
+
elif isinstance(node, list):
|
|
136
|
+
for v in node:
|
|
137
|
+
walk(v)
|
|
138
|
+
|
|
139
|
+
walk(self.data)
|
|
140
|
+
return sorted(found)
|
|
141
|
+
|
|
142
|
+
@staticmethod
|
|
143
|
+
def format_field(name: str, meta: Dict[str, Any]) -> str:
|
|
144
|
+
"""Render one field as ``<name>: [<type>] [flags...]`` for display."""
|
|
145
|
+
t = meta.get("type", "?")
|
|
146
|
+
if t == "LinkedEntity":
|
|
147
|
+
le = meta.get("leType") or []
|
|
148
|
+
if le:
|
|
149
|
+
t = f"LinkedEntity ({le[0]})"
|
|
150
|
+
flags = []
|
|
151
|
+
if meta.get("isList"):
|
|
152
|
+
flags.append("isList")
|
|
153
|
+
if meta.get("isComposite"):
|
|
154
|
+
flags.append("isComposite")
|
|
155
|
+
if meta.get("isEnum"):
|
|
156
|
+
flags.append("isEnum")
|
|
157
|
+
if meta.get("isDeprecated"):
|
|
158
|
+
flags.append("DEPRECATED")
|
|
159
|
+
suffix = "".join(f" [{f}]" for f in flags)
|
|
160
|
+
return f"{name}: [{t}]{suffix}"
|
diffbot/web_search.py
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""Diffbot web search API."""
|
|
2
|
+
|
|
3
|
+
from typing import TYPE_CHECKING, Any, Dict, List, Optional
|
|
4
|
+
|
|
5
|
+
if TYPE_CHECKING:
|
|
6
|
+
from .client import Diffbot, DiffbotAsync
|
|
7
|
+
|
|
8
|
+
WEB_SEARCH_BASE = "https://llm.diffbot.com/api/v1/web_search"
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def web_search(
|
|
12
|
+
client: "Diffbot",
|
|
13
|
+
text: str,
|
|
14
|
+
*,
|
|
15
|
+
num_results: Optional[int] = None,
|
|
16
|
+
max_tokens: Optional[int] = None,
|
|
17
|
+
) -> Dict[str, Any]:
|
|
18
|
+
headers = {"Authorization": f"Bearer {client.token}"}
|
|
19
|
+
params: Dict[str, Any] = {"text": text}
|
|
20
|
+
if num_results is not None:
|
|
21
|
+
params["size"] = num_results
|
|
22
|
+
if max_tokens is not None:
|
|
23
|
+
params["maxTokens"] = max_tokens
|
|
24
|
+
response = client._http.get(client.web_search_url, headers=headers, params=params)
|
|
25
|
+
client._raise_for_status(response)
|
|
26
|
+
return response.json()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
async def web_search_async(
|
|
30
|
+
client: "DiffbotAsync",
|
|
31
|
+
text: str,
|
|
32
|
+
*,
|
|
33
|
+
num_results: Optional[int] = None,
|
|
34
|
+
max_tokens: Optional[int] = None,
|
|
35
|
+
) -> Dict[str, Any]:
|
|
36
|
+
headers = {"Authorization": f"Bearer {client.token}"}
|
|
37
|
+
params: Dict[str, Any] = {"text": text}
|
|
38
|
+
if num_results is not None:
|
|
39
|
+
params["size"] = num_results
|
|
40
|
+
if max_tokens is not None:
|
|
41
|
+
params["maxTokens"] = max_tokens
|
|
42
|
+
response = await client._http.get(client.web_search_url, headers=headers, params=params)
|
|
43
|
+
client._raise_for_status(response)
|
|
44
|
+
return response.json()
|