aiblocklist 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aiblocklist/__init__.py +42 -0
- aiblocklist/__main__.py +36 -0
- aiblocklist/client.py +330 -0
- aiblocklist-1.0.0.dist-info/LICENSE +21 -0
- aiblocklist-1.0.0.dist-info/METADATA +401 -0
- aiblocklist-1.0.0.dist-info/RECORD +8 -0
- aiblocklist-1.0.0.dist-info/WHEEL +5 -0
- aiblocklist-1.0.0.dist-info/top_level.txt +1 -0
aiblocklist/__init__.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""
|
|
2
|
+
aiblocklist: Python client for the AI Tools Blocklist API at
|
|
3
|
+
https://www.aitoolsblocklist.com.
|
|
4
|
+
|
|
5
|
+
Classify any domain as an AI tool (20,000+ domains, 18 categories, daily refresh,
|
|
6
|
+
vendor training-on-your-data verdicts) and download the feed database for DNS
|
|
7
|
+
sinkholes, proxies, firewalls and SOAR playbooks.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from .client import (
|
|
11
|
+
AIBlocklistClient,
|
|
12
|
+
Feeds,
|
|
13
|
+
Lookup,
|
|
14
|
+
AIBlocklistError,
|
|
15
|
+
AuthenticationError,
|
|
16
|
+
QuotaError,
|
|
17
|
+
PlanError,
|
|
18
|
+
BadRequestError,
|
|
19
|
+
NotFoundError,
|
|
20
|
+
ServiceUnavailableError,
|
|
21
|
+
clean_domain,
|
|
22
|
+
DATA_USE_FIELDS,
|
|
23
|
+
CLAUSE_FIELDS,
|
|
24
|
+
__version__,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"AIBlocklistClient",
|
|
29
|
+
"Feeds",
|
|
30
|
+
"Lookup",
|
|
31
|
+
"AIBlocklistError",
|
|
32
|
+
"AuthenticationError",
|
|
33
|
+
"QuotaError",
|
|
34
|
+
"PlanError",
|
|
35
|
+
"BadRequestError",
|
|
36
|
+
"NotFoundError",
|
|
37
|
+
"ServiceUnavailableError",
|
|
38
|
+
"clean_domain",
|
|
39
|
+
"DATA_USE_FIELDS",
|
|
40
|
+
"CLAUSE_FIELDS",
|
|
41
|
+
"__version__",
|
|
42
|
+
]
|
aiblocklist/__main__.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Command line entry: print the classification of one or more domains as JSON.
|
|
3
|
+
|
|
4
|
+
python -m aiblocklist chatgpt.com midjourney.com
|
|
5
|
+
ATB_API_KEY=... python -m aiblocklist --stats
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import os
|
|
10
|
+
import sys
|
|
11
|
+
|
|
12
|
+
from .client import AIBlocklistClient, AIBlocklistError
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def main(argv=None) -> int:
|
|
16
|
+
args = list(sys.argv[1:] if argv is None else argv)
|
|
17
|
+
if not args or args == ["-h"] or args == ["--help"]:
|
|
18
|
+
print(__doc__.strip())
|
|
19
|
+
return 2
|
|
20
|
+
client = AIBlocklistClient(os.environ.get("ATB_API_KEY"))
|
|
21
|
+
try:
|
|
22
|
+
if args == ["--stats"]:
|
|
23
|
+
print(json.dumps(client.stats(), indent=2))
|
|
24
|
+
return 0
|
|
25
|
+
results = client.check_many(args)
|
|
26
|
+
for r in results:
|
|
27
|
+
print(json.dumps(dict(r), ensure_ascii=False))
|
|
28
|
+
return 0 if all(r.blocked is False or r.blocked is True for r in results) else 1
|
|
29
|
+
except AIBlocklistError as exc:
|
|
30
|
+
print(json.dumps({"error": exc.__class__.__name__, "status": exc.status,
|
|
31
|
+
"message": str(exc)}), file=sys.stderr)
|
|
32
|
+
return 1
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
if __name__ == "__main__":
|
|
36
|
+
sys.exit(main())
|
aiblocklist/client.py
ADDED
|
@@ -0,0 +1,330 @@
|
|
|
1
|
+
"""
|
|
2
|
+
aiblocklist: Python client for the AI Tools Blocklist API
|
|
3
|
+
(https://www.aitoolsblocklist.com).
|
|
4
|
+
|
|
5
|
+
Lookup API (one lookup charged per call):
|
|
6
|
+
GET https://www.aitoolsblocklist.com/api/check?domain=<domain>
|
|
7
|
+
header X-API-Key: <key>
|
|
8
|
+
|
|
9
|
+
Database API (feed and database plans):
|
|
10
|
+
GET https://www.aitoolsblocklist.com/api/database/?action=status
|
|
11
|
+
GET https://www.aitoolsblocklist.com/api/database/?action=database_info
|
|
12
|
+
GET https://www.aitoolsblocklist.com/api/database/?action=download_database
|
|
13
|
+
GET https://www.aitoolsblocklist.com/api/database/?action=download_categories
|
|
14
|
+
|
|
15
|
+
Public endpoints (no key):
|
|
16
|
+
GET https://www.aitoolsblocklist.com/api/stats.php
|
|
17
|
+
GET https://www.aitoolsblocklist.com/api/data-use-clause.php?d=<domain>&f=<field>
|
|
18
|
+
|
|
19
|
+
Only ``requests`` is required.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
import time
|
|
23
|
+
from typing import Dict, Iterable, List, Optional
|
|
24
|
+
|
|
25
|
+
import requests
|
|
26
|
+
|
|
27
|
+
__version__ = "1.0.0"
|
|
28
|
+
|
|
29
|
+
DEFAULT_BASE_URL = "https://www.aitoolsblocklist.com"
|
|
30
|
+
DEFAULT_TIMEOUT = 30
|
|
31
|
+
DOWNLOAD_TIMEOUT = 300
|
|
32
|
+
USER_AGENT = "aiblocklist-python/%s (+https://www.aitoolsblocklist.com)" % __version__
|
|
33
|
+
|
|
34
|
+
DATA_USE_FIELDS = ("trains_on_data", "opt_out_available", "enterprise_no_training",
|
|
35
|
+
"api_no_training", "terms_checked")
|
|
36
|
+
CLAUSE_FIELDS = ("trains_consumer_default", "optout_available", "enterprise_no_training",
|
|
37
|
+
"api_no_training")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class AIBlocklistError(Exception):
|
|
41
|
+
"""Base exception. ``status`` and ``body`` carry the HTTP status and parsed response."""
|
|
42
|
+
|
|
43
|
+
def __init__(self, message: str, status: int = 0, body: Optional[dict] = None):
|
|
44
|
+
super().__init__(message)
|
|
45
|
+
self.status = status
|
|
46
|
+
self.body = body or {}
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class AuthenticationError(AIBlocklistError):
|
|
50
|
+
"""401: no key, or a key that matches no account."""
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class QuotaError(AIBlocklistError):
|
|
54
|
+
"""403: inactive account or monthly lookup quota exhausted."""
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class PlanError(AIBlocklistError):
|
|
58
|
+
"""403 on the database endpoints: the plan is lookup-only. ``body['plan']`` names it."""
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class BadRequestError(AIBlocklistError):
|
|
62
|
+
"""400: domain missing, or an unknown clause field."""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class NotFoundError(AIBlocklistError):
|
|
66
|
+
"""404: database file not available for the account."""
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class ServiceUnavailableError(AIBlocklistError):
|
|
70
|
+
"""503: lookup service busy. Retried before it is raised."""
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def clean_domain(domain: str) -> str:
|
|
74
|
+
"""Reduce a URL or host string to a bare lowercase domain."""
|
|
75
|
+
d = str(domain).strip().lower()
|
|
76
|
+
if "://" in d:
|
|
77
|
+
d = d.split("://", 1)[1]
|
|
78
|
+
d = d.split("/")[0].split("?")[0].split("#")[0]
|
|
79
|
+
if d.startswith("www."):
|
|
80
|
+
d = d[4:]
|
|
81
|
+
d = d.split("@")[-1].split(":")[0]
|
|
82
|
+
return d
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class Lookup(dict):
|
|
86
|
+
"""One lookup response. A dict with the raw fields plus convenience properties."""
|
|
87
|
+
|
|
88
|
+
@property
|
|
89
|
+
def domain(self) -> str:
|
|
90
|
+
return self.get("domain", "")
|
|
91
|
+
|
|
92
|
+
@property
|
|
93
|
+
def blocked(self) -> bool:
|
|
94
|
+
return self.get("blocked") is True
|
|
95
|
+
|
|
96
|
+
@property
|
|
97
|
+
def is_ai_tool(self) -> bool:
|
|
98
|
+
return self.blocked
|
|
99
|
+
|
|
100
|
+
@property
|
|
101
|
+
def primary_category(self) -> Optional[str]:
|
|
102
|
+
return self.get("primary_category")
|
|
103
|
+
|
|
104
|
+
@property
|
|
105
|
+
def ai_type(self) -> Optional[str]:
|
|
106
|
+
return self.get("ai_type")
|
|
107
|
+
|
|
108
|
+
@property
|
|
109
|
+
def category_names(self) -> List[str]:
|
|
110
|
+
return [c.get("category") for c in self.get("categories", []) if c.get("category")]
|
|
111
|
+
|
|
112
|
+
@property
|
|
113
|
+
def subcategory_names(self) -> List[str]:
|
|
114
|
+
return [c.get("subcategory") for c in self.get("categories", []) if c.get("subcategory")]
|
|
115
|
+
|
|
116
|
+
@property
|
|
117
|
+
def trains_on_data(self) -> Optional[str]:
|
|
118
|
+
return self.get("trains_on_data")
|
|
119
|
+
|
|
120
|
+
@property
|
|
121
|
+
def data_use(self) -> Dict[str, Optional[str]]:
|
|
122
|
+
return {f: self.get(f) for f in DATA_USE_FIELDS}
|
|
123
|
+
|
|
124
|
+
@property
|
|
125
|
+
def quota_remaining(self) -> Optional[int]:
|
|
126
|
+
q = self.get("quota_remaining")
|
|
127
|
+
return q if isinstance(q, int) else None
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
class Feeds:
|
|
131
|
+
"""Database and feed downloads for feed and database plans."""
|
|
132
|
+
|
|
133
|
+
def __init__(self, client: "AIBlocklistClient"):
|
|
134
|
+
self._client = client
|
|
135
|
+
|
|
136
|
+
def _url(self, action: str) -> str:
|
|
137
|
+
return "%s/api/database/?action=%s" % (self._client.base_url, action)
|
|
138
|
+
|
|
139
|
+
def status(self) -> Dict:
|
|
140
|
+
"""Plan, key state, subscribed file, last update."""
|
|
141
|
+
return self._client._json(self._url("status"), with_key=True)
|
|
142
|
+
|
|
143
|
+
def database_info(self) -> Dict:
|
|
144
|
+
"""File name, last update, size in bytes for the database this key is entitled to."""
|
|
145
|
+
return self._client._json(self._url("database_info"), with_key=True)
|
|
146
|
+
|
|
147
|
+
def download_database(self, dest_path: str) -> Dict:
|
|
148
|
+
"""Stream the full classified CSV to ``dest_path``. Returns {"file", "bytes"}."""
|
|
149
|
+
return self._client._download(self._url("download_database"), dest_path)
|
|
150
|
+
|
|
151
|
+
def download_categories(self, dest_path: str) -> Dict:
|
|
152
|
+
"""Stream the categories CSV (category and subcategory tree) to ``dest_path``."""
|
|
153
|
+
return self._client._download(self._url("download_categories"), dest_path)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
class AIBlocklistClient:
|
|
157
|
+
"""
|
|
158
|
+
Client for the AI Tools Blocklist API.
|
|
159
|
+
|
|
160
|
+
Args:
|
|
161
|
+
api_key: key from the account area, sent as ``X-API-Key``. Optional for the
|
|
162
|
+
public methods ``stats()``, ``categories()`` and ``clause()``.
|
|
163
|
+
base_url: defaults to https://www.aitoolsblocklist.com
|
|
164
|
+
timeout: per-request timeout in seconds (downloads use at least 300)
|
|
165
|
+
max_retries: retries on 503 and network errors
|
|
166
|
+
|
|
167
|
+
Example:
|
|
168
|
+
>>> from aiblocklist import AIBlocklistClient
|
|
169
|
+
>>> client = AIBlocklistClient("YOUR_API_KEY")
|
|
170
|
+
>>> r = client.check("chatgpt.com")
|
|
171
|
+
>>> r.blocked, r.primary_category, r.trains_on_data
|
|
172
|
+
(True, 'Text & Language', 'opt_out_default')
|
|
173
|
+
"""
|
|
174
|
+
|
|
175
|
+
def __init__(self, api_key: Optional[str] = None, base_url: str = DEFAULT_BASE_URL,
|
|
176
|
+
timeout: int = DEFAULT_TIMEOUT, max_retries: int = 2):
|
|
177
|
+
self.api_key = api_key or None
|
|
178
|
+
self.base_url = base_url.rstrip("/")
|
|
179
|
+
self.timeout = timeout
|
|
180
|
+
self.max_retries = max_retries
|
|
181
|
+
self._session = requests.Session()
|
|
182
|
+
self._session.headers.update({"Accept": "application/json", "User-Agent": USER_AGENT})
|
|
183
|
+
self.feeds = Feeds(self)
|
|
184
|
+
|
|
185
|
+
# ---- internals ----
|
|
186
|
+
|
|
187
|
+
def _headers(self, with_key: bool) -> Dict[str, str]:
|
|
188
|
+
if not with_key:
|
|
189
|
+
return {}
|
|
190
|
+
if not self.api_key:
|
|
191
|
+
raise AuthenticationError("api_key is required for this call", 401)
|
|
192
|
+
return {"X-API-Key": self.api_key}
|
|
193
|
+
|
|
194
|
+
@staticmethod
|
|
195
|
+
def _body(resp: requests.Response) -> Dict:
|
|
196
|
+
try:
|
|
197
|
+
data = resp.json()
|
|
198
|
+
return data if isinstance(data, dict) else {"data": data}
|
|
199
|
+
except ValueError:
|
|
200
|
+
return {"message": (resp.text or "")[:200]}
|
|
201
|
+
|
|
202
|
+
@classmethod
|
|
203
|
+
def _raise(cls, resp: requests.Response) -> None:
|
|
204
|
+
data = cls._body(resp)
|
|
205
|
+
msg = data.get("message") or data.get("error") or ("HTTP %s" % resp.status_code)
|
|
206
|
+
if resp.status_code == 400:
|
|
207
|
+
raise BadRequestError(msg, 400, data)
|
|
208
|
+
if resp.status_code == 401:
|
|
209
|
+
raise AuthenticationError(msg, 401, data)
|
|
210
|
+
if resp.status_code == 403:
|
|
211
|
+
if data.get("plan"):
|
|
212
|
+
raise PlanError(msg, 403, data)
|
|
213
|
+
raise QuotaError(msg, 403, data)
|
|
214
|
+
if resp.status_code == 404:
|
|
215
|
+
raise NotFoundError(msg, 404, data)
|
|
216
|
+
if resp.status_code == 503:
|
|
217
|
+
raise ServiceUnavailableError(msg, 503, data)
|
|
218
|
+
raise AIBlocklistError(msg, resp.status_code, data)
|
|
219
|
+
|
|
220
|
+
def _json(self, url: str, with_key: bool) -> Dict:
|
|
221
|
+
headers = self._headers(with_key)
|
|
222
|
+
last_exc = None
|
|
223
|
+
for attempt in range(self.max_retries + 1):
|
|
224
|
+
try:
|
|
225
|
+
resp = self._session.get(url, headers=headers, timeout=self.timeout)
|
|
226
|
+
except requests.RequestException as exc:
|
|
227
|
+
last_exc = exc
|
|
228
|
+
if attempt < self.max_retries:
|
|
229
|
+
time.sleep(1.0 * (attempt + 1))
|
|
230
|
+
continue
|
|
231
|
+
raise AIBlocklistError("request failed: %s" % exc) from exc
|
|
232
|
+
if resp.status_code == 200:
|
|
233
|
+
return self._body(resp)
|
|
234
|
+
if resp.status_code == 503 and attempt < self.max_retries:
|
|
235
|
+
time.sleep(1.5 * (attempt + 1))
|
|
236
|
+
continue
|
|
237
|
+
self._raise(resp)
|
|
238
|
+
raise AIBlocklistError("request failed after retries: %s" % last_exc)
|
|
239
|
+
|
|
240
|
+
def _download(self, url: str, dest_path: str) -> Dict:
|
|
241
|
+
if not dest_path:
|
|
242
|
+
raise ValueError("dest_path is required")
|
|
243
|
+
resp = self._session.get(url, headers=self._headers(True),
|
|
244
|
+
timeout=max(self.timeout, DOWNLOAD_TIMEOUT), stream=True)
|
|
245
|
+
ctype = resp.headers.get("Content-Type", "")
|
|
246
|
+
if resp.status_code == 200 and "application/json" not in ctype:
|
|
247
|
+
total = 0
|
|
248
|
+
with open(dest_path, "wb") as fh:
|
|
249
|
+
for chunk in resp.iter_content(chunk_size=1 << 16):
|
|
250
|
+
if chunk:
|
|
251
|
+
fh.write(chunk)
|
|
252
|
+
total += len(chunk)
|
|
253
|
+
return {"file": dest_path, "bytes": total}
|
|
254
|
+
self._raise(resp)
|
|
255
|
+
return {}
|
|
256
|
+
|
|
257
|
+
# ---- lookup API (key required) ----
|
|
258
|
+
|
|
259
|
+
def check(self, domain: str) -> Lookup:
|
|
260
|
+
"""Classify one domain. One lookup is charged."""
|
|
261
|
+
d = clean_domain(domain)
|
|
262
|
+
if not d:
|
|
263
|
+
raise BadRequestError("domain is empty", 400)
|
|
264
|
+
url = "%s/api/check?domain=%s" % (self.base_url, requests.utils.quote(d, safe=""))
|
|
265
|
+
return Lookup(self._json(url, with_key=True))
|
|
266
|
+
|
|
267
|
+
def is_blocked(self, domain: str) -> bool:
|
|
268
|
+
"""True when the domain is a known AI tool."""
|
|
269
|
+
return self.check(domain).blocked
|
|
270
|
+
|
|
271
|
+
def check_many(self, domains: Iterable[str], pause: float = 0.0) -> List[Lookup]:
|
|
272
|
+
"""Sequential lookups, one charged per unique domain. Duplicates are checked once."""
|
|
273
|
+
seen: Dict[str, Lookup] = {}
|
|
274
|
+
out: List[Lookup] = []
|
|
275
|
+
for raw in domains:
|
|
276
|
+
d = clean_domain(raw)
|
|
277
|
+
if not d:
|
|
278
|
+
continue
|
|
279
|
+
if d not in seen:
|
|
280
|
+
seen[d] = self.check(d)
|
|
281
|
+
if pause:
|
|
282
|
+
time.sleep(pause)
|
|
283
|
+
out.append(seen[d])
|
|
284
|
+
return out
|
|
285
|
+
|
|
286
|
+
def data_use(self, domain: str) -> Dict[str, Optional[str]]:
|
|
287
|
+
"""The five vendor data-use fields for one domain."""
|
|
288
|
+
return self.check(domain).data_use
|
|
289
|
+
|
|
290
|
+
# ---- public endpoints (no key) ----
|
|
291
|
+
|
|
292
|
+
def stats(self) -> Dict:
|
|
293
|
+
"""Total tools, the 18 categories with counts, subcategory counts."""
|
|
294
|
+
return self._json("%s/api/stats.php" % self.base_url, with_key=False)
|
|
295
|
+
|
|
296
|
+
def categories(self) -> List[Dict]:
|
|
297
|
+
"""The category list from the public stats: [{name, total, subcategories}, ...]."""
|
|
298
|
+
return self.stats().get("categories", [])
|
|
299
|
+
|
|
300
|
+
def clause(self, domain: str, field: str) -> Optional[Dict]:
|
|
301
|
+
"""
|
|
302
|
+
The verbatim vendor clause behind a public data-use field, as {"clause", "url"},
|
|
303
|
+
or None when nothing is published for that domain and field.
|
|
304
|
+
|
|
305
|
+
field: trains_consumer_default | optout_available | enterprise_no_training | api_no_training
|
|
306
|
+
"""
|
|
307
|
+
if field not in CLAUSE_FIELDS:
|
|
308
|
+
raise BadRequestError("field must be one of %s" % ", ".join(CLAUSE_FIELDS), 400)
|
|
309
|
+
d = clean_domain(domain)
|
|
310
|
+
url = "%s/api/data-use-clause.php?d=%s&f=%s" % (
|
|
311
|
+
self.base_url, requests.utils.quote(d, safe=""), field)
|
|
312
|
+
resp = self._session.get(url, timeout=self.timeout)
|
|
313
|
+
if resp.status_code == 200:
|
|
314
|
+
data = self._body(resp)
|
|
315
|
+
return data if data.get("clause") else None
|
|
316
|
+
if resp.status_code in (400, 404):
|
|
317
|
+
return None
|
|
318
|
+
self._raise(resp)
|
|
319
|
+
return None
|
|
320
|
+
|
|
321
|
+
# ---- lifecycle ----
|
|
322
|
+
|
|
323
|
+
def close(self) -> None:
|
|
324
|
+
self._session.close()
|
|
325
|
+
|
|
326
|
+
def __enter__(self):
|
|
327
|
+
return self
|
|
328
|
+
|
|
329
|
+
def __exit__(self, *exc):
|
|
330
|
+
self.close()
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Alpha Quantum
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,401 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: aiblocklist
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: AI blocklist client: classify any domain as an AI tool (20,000+ domains, 18 categories, daily refresh, vendor training-on-your-data verdicts) and download the feed database for DNS sinkholes, proxies, firewalls and SOAR playbooks.
|
|
5
|
+
Home-page: https://www.aitoolsblocklist.com
|
|
6
|
+
Author: Alpha Quantum
|
|
7
|
+
Author-email: info@alpha-quantum.com
|
|
8
|
+
License: MIT
|
|
9
|
+
Project-URL: Homepage, https://www.aitoolsblocklist.com
|
|
10
|
+
Project-URL: Documentation, https://www.aitoolsblocklist.com/ai-blocklist-api.php
|
|
11
|
+
Project-URL: Source, https://github.com/explainableaixai/aiblocklist
|
|
12
|
+
Project-URL: Tracker, https://www.aitoolsblocklist.com/contact.php
|
|
13
|
+
Project-URL: AI Tools Blocklist, https://www.aitoolsblocklist.com
|
|
14
|
+
Project-URL: AI Agent Allow List, https://www.aiagentallowlist.com
|
|
15
|
+
Project-URL: Shadow AI Tools, https://www.shadowaitools.com
|
|
16
|
+
Keywords: ai blocklist,ai tools blocklist,block ai tools,shadow ai,dns sinkhole,external dynamic list,web filtering,secure web gateway,dlp,acceptable use policy,generative ai,ai governance,domain classification,training on your data
|
|
17
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
18
|
+
Classifier: Intended Audience :: Developers
|
|
19
|
+
Classifier: Intended Audience :: Information Technology
|
|
20
|
+
Classifier: Intended Audience :: System Administrators
|
|
21
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
22
|
+
Classifier: Programming Language :: Python :: 3
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.7
|
|
24
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
25
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
26
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
27
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
28
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
29
|
+
Classifier: Topic :: Internet :: Proxy Servers
|
|
30
|
+
Classifier: Topic :: Security
|
|
31
|
+
Classifier: Topic :: System :: Networking :: Firewalls
|
|
32
|
+
Requires-Python: >=3.7
|
|
33
|
+
Description-Content-Type: text/markdown
|
|
34
|
+
License-File: LICENSE
|
|
35
|
+
Requires-Dist: requests >=2.20.0
|
|
36
|
+
|
|
37
|
+
# aiblocklist
|
|
38
|
+
|
|
39
|
+
Python client for the AI Tools Blocklist API, a database of [classified AI tool domains](https://www.aitoolsblocklist.com) built for the teams that run proxies, DNS resolvers, DLP pipelines and security automation. One call classifies a domain; one download gives you the whole list for local matching.
|
|
40
|
+
|
|
41
|
+
What each record carries:
|
|
42
|
+
|
|
43
|
+
- 20,000+ AI-tool domains in 18 functional categories with subcategories, multi-label, rebuilt daily
|
|
44
|
+
- an AI type: `ai_native` for tools that are the AI, `ai_enabled` for products with an AI feature inside
|
|
45
|
+
- the vendor's data-use position: trains on input, opt-out available, enterprise tier exempt, API tier exempt, and the date the terms were checked
|
|
46
|
+
- for feed and database plans, the full CSV plus hosted EDL, PAC, hosts and DNS feeds
|
|
47
|
+
|
|
48
|
+
Only `requests` is required. Python 3.7 and newer.
|
|
49
|
+
|
|
50
|
+
---
|
|
51
|
+
|
|
52
|
+
## Installation
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
pip install aiblocklist
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
## Quick start
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
from aiblocklist import AIBlocklistClient
|
|
62
|
+
|
|
63
|
+
client = AIBlocklistClient("YOUR_API_KEY")
|
|
64
|
+
|
|
65
|
+
r = client.check("chatgpt.com")
|
|
66
|
+
r.blocked # True
|
|
67
|
+
r.primary_category # "Text & Language"
|
|
68
|
+
r.ai_type # "ai_native"
|
|
69
|
+
r.category_names # ["Text & Language"]
|
|
70
|
+
r.subcategory_names # ["General assistants & chatbots"]
|
|
71
|
+
r.trains_on_data # "opt_out_default"
|
|
72
|
+
r.terms_checked # "2026-09-17"
|
|
73
|
+
r.quota_remaining # lookups left in the current 30-day cycle
|
|
74
|
+
|
|
75
|
+
client.check("example.com").blocked # False
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
`check()` returns a `Lookup`, a `dict` subclass with the raw JSON and convenience properties, so `r["blocked"]` and `r.blocked` are the same value. The key comes from the account area at aitoolsblocklist.com and is sent as `X-API-Key`. Public methods (`stats()`, `categories()`, `clause()`) work without a key.
|
|
79
|
+
|
|
80
|
+
From the command line:
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
export ATB_API_KEY=YOUR_API_KEY
|
|
84
|
+
python -m aiblocklist chatgpt.com midjourney.com example.com
|
|
85
|
+
python -m aiblocklist --stats
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## Methods
|
|
89
|
+
|
|
90
|
+
| Method | Endpoint | Key | Returns |
|
|
91
|
+
|---|---|---|---|
|
|
92
|
+
| `check(domain)` | `GET /api/check?domain=` | yes | `Lookup` |
|
|
93
|
+
| `is_blocked(domain)` | same | yes | `bool` |
|
|
94
|
+
| `check_many(domains, pause=0)` | same, sequential, deduplicated | yes | `list[Lookup]` |
|
|
95
|
+
| `data_use(domain)` | same | yes | dict of the five data-use fields |
|
|
96
|
+
| `feeds.status()` | `GET /api/database/?action=status` | feed or database plan | plan and file state |
|
|
97
|
+
| `feeds.database_info()` | `GET /api/database/?action=database_info` | feed or database plan | file name, timestamp, size |
|
|
98
|
+
| `feeds.download_database(path)` | `GET /api/database/?action=download_database` | feed or database plan | `{"file", "bytes"}` |
|
|
99
|
+
| `feeds.download_categories(path)` | `GET /api/database/?action=download_categories` | feed or database plan | `{"file", "bytes"}` |
|
|
100
|
+
| `stats()` | `GET /api/stats.php` | no | totals and 18 categories with counts |
|
|
101
|
+
| `categories()` | same | no | the category list |
|
|
102
|
+
| `clause(domain, field)` | `GET /api/data-use-clause.php` | no | `{"clause", "url"}` or `None` |
|
|
103
|
+
|
|
104
|
+
Pass a bare domain or any URL; `clean_domain()` strips scheme, path, port, credentials and a leading `www.`. Subdomains resolve to the registrable domain on the server.
|
|
105
|
+
|
|
106
|
+
## The response, field by field
|
|
107
|
+
|
|
108
|
+
```json
|
|
109
|
+
{
|
|
110
|
+
"domain": "chatgpt.com",
|
|
111
|
+
"blocked": true,
|
|
112
|
+
"primary_category": "Text & Language",
|
|
113
|
+
"ai_type": "ai_native",
|
|
114
|
+
"categories": [{"category": "Text & Language", "subcategory": "General assistants & chatbots"}],
|
|
115
|
+
"trains_on_data": "opt_out_default",
|
|
116
|
+
"opt_out_available": "yes",
|
|
117
|
+
"enterprise_no_training": "yes",
|
|
118
|
+
"api_no_training": "yes",
|
|
119
|
+
"terms_checked": "2026-09-17",
|
|
120
|
+
"quota_remaining": 9999986
|
|
121
|
+
}
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
| Field | Values |
|
|
125
|
+
|---|---|
|
|
126
|
+
| `blocked` | `true` when the domain is a known AI tool, otherwise `false` with an empty `categories` list |
|
|
127
|
+
| `primary_category` | one of the 18 categories |
|
|
128
|
+
| `ai_type` | `ai_native` or `ai_enabled` |
|
|
129
|
+
| `categories` | every category and subcategory the tool belongs to |
|
|
130
|
+
| `trains_on_data` | `yes`, `no`, `opt_out_default`, `unstated` |
|
|
131
|
+
| `opt_out_available` | `yes`, `no`, `unstated` |
|
|
132
|
+
| `enterprise_no_training` | `yes`, `no`, `unstated` |
|
|
133
|
+
| `api_no_training` | `yes`, `no`, `unstated` |
|
|
134
|
+
| `terms_checked` | ISO date the vendor terms were last read |
|
|
135
|
+
| `quota_remaining` | lookups left on the plan |
|
|
136
|
+
|
|
137
|
+
Both found and not-found are HTTP 200. Errors are raised as exceptions (see below).
|
|
138
|
+
|
|
139
|
+
## Worked examples
|
|
140
|
+
|
|
141
|
+
### 1. FastAPI middleware for an internal egress service
|
|
142
|
+
|
|
143
|
+
An internal HTTP egress service (or any FastAPI app that proxies outbound requests) checks the destination host once, keeps the verdict for a day, and enforces a category policy. Assistants that train on input by default are allowed only with a header that the DLP layer downstream can key on.
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
import time
|
|
147
|
+
from fastapi import FastAPI, Request
|
|
148
|
+
from fastapi.responses import JSONResponse
|
|
149
|
+
from aiblocklist import AIBlocklistClient, AIBlocklistError
|
|
150
|
+
|
|
151
|
+
app = FastAPI()
|
|
152
|
+
client = AIBlocklistClient("YOUR_API_KEY")
|
|
153
|
+
cache = {} # host -> (verdict, expires)
|
|
154
|
+
TTL = 86400
|
|
155
|
+
BLOCK = {"Image & Visual", "Audio & Voice", "Companions & Social"}
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def verdict_for(host: str) -> str:
|
|
159
|
+
now = time.time()
|
|
160
|
+
hit = cache.get(host)
|
|
161
|
+
if hit and hit[1] > now:
|
|
162
|
+
return hit[0]
|
|
163
|
+
try:
|
|
164
|
+
r = client.check(host)
|
|
165
|
+
except AIBlocklistError:
|
|
166
|
+
return "allow" # fail open on API trouble, log it
|
|
167
|
+
verdict = "allow"
|
|
168
|
+
if r.blocked:
|
|
169
|
+
if BLOCK & set(r.category_names):
|
|
170
|
+
verdict = "block"
|
|
171
|
+
elif r.trains_on_data in ("yes", "opt_out_default"):
|
|
172
|
+
verdict = "warn"
|
|
173
|
+
cache[host] = (verdict, now + TTL)
|
|
174
|
+
return verdict
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
@app.middleware("http")
|
|
178
|
+
async def ai_policy(request: Request, call_next):
|
|
179
|
+
host = request.headers.get("x-target-host", "")
|
|
180
|
+
v = verdict_for(host) if host else "allow"
|
|
181
|
+
if v == "block":
|
|
182
|
+
return JSONResponse({"error": "AI tool blocked by policy", "host": host}, status_code=403)
|
|
183
|
+
response = await call_next(request)
|
|
184
|
+
if v == "warn":
|
|
185
|
+
response.headers["X-AI-Policy"] = "trains-on-input"
|
|
186
|
+
return response
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
The same shape works as a Flask `before_request` hook. With a one-day cache, a gateway that sees a few thousand distinct hosts uses a few thousand lookups a month.
|
|
190
|
+
|
|
191
|
+
### 2. Classifying a proxy export with pandas
|
|
192
|
+
|
|
193
|
+
Take the unique hosts from a proxy or DNS export, classify them, and produce a table of AI tools with their categories and training terms. `check_many()` deduplicates, so the cost is one lookup per unique host.
|
|
194
|
+
|
|
195
|
+
```python
|
|
196
|
+
import pandas as pd
|
|
197
|
+
from aiblocklist import AIBlocklistClient
|
|
198
|
+
|
|
199
|
+
client = AIBlocklistClient("YOUR_API_KEY")
|
|
200
|
+
|
|
201
|
+
log = pd.read_csv("proxy_export.csv") # columns: timestamp, user, host, bytes
|
|
202
|
+
hosts = log["host"].dropna().unique().tolist()
|
|
203
|
+
|
|
204
|
+
results = client.check_many(hosts, pause=0.05)
|
|
205
|
+
found = pd.DataFrame([
|
|
206
|
+
{
|
|
207
|
+
"domain": r.domain,
|
|
208
|
+
"primary_category": r.primary_category,
|
|
209
|
+
"subcategories": "; ".join(r.subcategory_names),
|
|
210
|
+
"ai_type": r.ai_type,
|
|
211
|
+
"trains_on_data": r.trains_on_data,
|
|
212
|
+
"enterprise_no_training": r["enterprise_no_training"],
|
|
213
|
+
"terms_checked": r.terms_checked,
|
|
214
|
+
}
|
|
215
|
+
for r in results if r.blocked
|
|
216
|
+
])
|
|
217
|
+
|
|
218
|
+
# join back to users so the report says who reached what
|
|
219
|
+
usage = log.merge(found, left_on="host", right_on="domain")
|
|
220
|
+
summary = usage.groupby(["domain", "primary_category", "trains_on_data"])["user"].nunique()
|
|
221
|
+
print(summary.sort_values(ascending=False).head(25))
|
|
222
|
+
found.to_csv("ai_tools_found.csv", index=False)
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
For exports with hundreds of thousands of lines, the hosted [shadow AI inventory](https://www.shadowaitools.com) service does this with per-user breakdowns, sanctioned lists and a PDF evidence pack, without writing the join yourself.
|
|
226
|
+
|
|
227
|
+
### 3. A nightly feed refresh for a DNS resolver
|
|
228
|
+
|
|
229
|
+
Feed and database plans download the full CSV. This cron job fetches it, keeps only the categories the organisation blocks, and writes an Unbound local-zone file. Swap the output format for dnsmasq, a hosts file or a firewall EDL as needed.
|
|
230
|
+
|
|
231
|
+
```python
|
|
232
|
+
#!/usr/bin/env python3
|
|
233
|
+
import csv, os, subprocess
|
|
234
|
+
from aiblocklist import AIBlocklistClient, PlanError
|
|
235
|
+
|
|
236
|
+
client = AIBlocklistClient(os.environ["ATB_API_KEY"])
|
|
237
|
+
BLOCK = {"Image & Visual", "Audio & Voice", "Companions & Social", "Agents & Automation"}
|
|
238
|
+
CSV_PATH = "/var/lib/aiblocklist/ai_tools_full.csv"
|
|
239
|
+
ZONE_TMP = "/etc/unbound/unbound.conf.d/ai-block.conf.tmp"
|
|
240
|
+
ZONE = "/etc/unbound/unbound.conf.d/ai-block.conf"
|
|
241
|
+
|
|
242
|
+
try:
|
|
243
|
+
info = client.feeds.database_info()
|
|
244
|
+
except PlanError as exc:
|
|
245
|
+
raise SystemExit("this key is lookup-only: %s" % exc.body.get("plan"))
|
|
246
|
+
|
|
247
|
+
print("database %s updated %s (%s)" % (info["database_file"], info["last_updated"], info["file_size_human"]))
|
|
248
|
+
client.feeds.download_database(CSV_PATH)
|
|
249
|
+
|
|
250
|
+
n = 0
|
|
251
|
+
with open(CSV_PATH, newline="", encoding="utf-8") as src, open(ZONE_TMP, "w") as out:
|
|
252
|
+
out.write("server:\n")
|
|
253
|
+
for row in csv.DictReader(src):
|
|
254
|
+
if row.get("primary_category") in BLOCK:
|
|
255
|
+
out.write(' local-zone: "%s." always_nxdomain\n' % row["domain"])
|
|
256
|
+
n += 1
|
|
257
|
+
os.replace(ZONE_TMP, ZONE)
|
|
258
|
+
subprocess.run(["unbound-control", "reload"], check=False)
|
|
259
|
+
print("%d domains sinkholed" % n)
|
|
260
|
+
```
|
|
261
|
+
|
|
262
|
+
```
|
|
263
|
+
30 5 * * * /usr/bin/python3 /opt/aiblocklist/refresh_zone.py >> /var/log/aiblocklist.log 2>&1
|
|
264
|
+
```
|
|
265
|
+
|
|
266
|
+
### 4. Citing the vendor's own terms
|
|
267
|
+
|
|
268
|
+
`clause()` returns the verbatim sentence behind a public data-use field with the URL it came from, which is what an approval ticket or a policy exception needs.
|
|
269
|
+
|
|
270
|
+
```python
|
|
271
|
+
c = client.clause("chatgpt.com", "trains_consumer_default")
|
|
272
|
+
if c:
|
|
273
|
+
print(c["clause"])
|
|
274
|
+
print(c["url"])
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
Fields: `trains_consumer_default`, `optout_available`, `enterprise_no_training`, `api_no_training`.
|
|
278
|
+
|
|
279
|
+
## Error handling
|
|
280
|
+
|
|
281
|
+
```python
|
|
282
|
+
from aiblocklist import (
|
|
283
|
+
AIBlocklistError, AuthenticationError, QuotaError, PlanError,
|
|
284
|
+
BadRequestError, NotFoundError, ServiceUnavailableError,
|
|
285
|
+
)
|
|
286
|
+
|
|
287
|
+
try:
|
|
288
|
+
r = client.check("notion.so")
|
|
289
|
+
except AuthenticationError: # 401: no key or unknown key
|
|
290
|
+
...
|
|
291
|
+
except QuotaError as exc: # 403: inactive account or quota exhausted
|
|
292
|
+
print(exc.status, exc.body)
|
|
293
|
+
except ServiceUnavailableError: # 503 after retries
|
|
294
|
+
...
|
|
295
|
+
except AIBlocklistError: # anything else, including network failures
|
|
296
|
+
...
|
|
297
|
+
```
|
|
298
|
+
|
|
299
|
+
| Exception | HTTP | Meaning |
|
|
300
|
+
|---|---|---|
|
|
301
|
+
| `BadRequestError` | 400 | empty domain, or an unknown clause field |
|
|
302
|
+
| `AuthenticationError` | 401 | no key, or a key that matches no account |
|
|
303
|
+
| `QuotaError` | 403 | inactive account or monthly quota exhausted |
|
|
304
|
+
| `PlanError` | 403 | database endpoints on a lookup-only plan; `body["plan"]` names it |
|
|
305
|
+
| `NotFoundError` | 404 | database file not available for the account |
|
|
306
|
+
| `ServiceUnavailableError` | 503 | lookup service busy; retried `max_retries` times first |
|
|
307
|
+
|
|
308
|
+
The client is a context manager, so the HTTP session closes cleanly:
|
|
309
|
+
|
|
310
|
+
```python
|
|
311
|
+
with AIBlocklistClient("YOUR_API_KEY") as client:
|
|
312
|
+
print(client.stats()["total_tools"])
|
|
313
|
+
```
|
|
314
|
+
|
|
315
|
+
## Configuration
|
|
316
|
+
|
|
317
|
+
```python
|
|
318
|
+
client = AIBlocklistClient(
|
|
319
|
+
api_key="YOUR_API_KEY",
|
|
320
|
+
base_url="https://www.aitoolsblocklist.com", # default
|
|
321
|
+
timeout=30, # seconds; downloads use at least 300
|
|
322
|
+
max_retries=2, # on 503 and network errors
|
|
323
|
+
)
|
|
324
|
+
```
|
|
325
|
+
|
|
326
|
+
Keep the key out of source: read it from an environment variable or a secrets manager and pass it to the constructor. The command line entry reads `ATB_API_KEY` for that reason. Lookups are metered per call, so cache verdicts in your own process or datastore for a day; the list is rebuilt daily and a shorter cache buys nothing. Downloads are streamed to disk in 64 KB chunks, so a full database file never sits in memory.
|
|
327
|
+
|
|
328
|
+
## Why a classified AI tool list belongs in the security stack
|
|
329
|
+
|
|
330
|
+
The risk in AI tool usage is not the tool, it is the input. A contract pasted into a summariser, a customer list dropped into a spreadsheet assistant, a repository fed to a code explainer: each is an outbound request to a hostname, and the hostname is the only thing every control point on the network can see without an agent or TLS inspection. Classifying that hostname is the step that turns a policy into an enforceable rule.
|
|
331
|
+
|
|
332
|
+
### Data loss prevention needs the destination, not just the content
|
|
333
|
+
|
|
334
|
+
[Data loss prevention](https://en.wikipedia.org/wiki/Data_loss_prevention_software) systems inspect what leaves. They cannot decide whether leaving is acceptable without knowing where it is going and what the recipient does with it. The [AI tool classification database](https://www.aitoolsblocklist.com) supplies both: the functional category of the destination and the vendor's own position on training, opt-out and enterprise exemptions. A DLP rule that says "block source code to AI assistants that train on input" is expressible only with that data.
|
|
335
|
+
|
|
336
|
+
### Frameworks ask for an inventory first
|
|
337
|
+
|
|
338
|
+
The [NIST AI Risk Management Framework](https://www.nist.gov/itl/ai-risk-management-framework) starts its Map function with knowing which AI systems are in use and which third parties receive data. The [ENISA](https://www.enisa.europa.eu/) guidance on AI cybersecurity takes the same order: identify, then govern. Under the [EU AI Act](https://eur-lex.europa.eu/eli/reg/2024/1689/oj), deployers carry obligations that depend on knowing which systems staff use. None of this is possible from a survey; it is possible from network evidence joined to a classified list.
|
|
339
|
+
|
|
340
|
+
### Prompt-borne exposure is a recognised risk class
|
|
341
|
+
|
|
342
|
+
The [OWASP Top 10 for LLM Applications](https://owasp.org/www-project-top-10-for-large-language-model-applications/) lists sensitive information disclosure among its leading risks. For an organisation that does not build LLM applications but whose staff use hundreds of them, the mitigation is at the network edge: know which destinations are AI tools, permit the ones whose terms are acceptable, and block or warn on the rest. Categories and subcategories make that policy fine-grained; the daily rebuild keeps it current as tools launch.
|
|
343
|
+
|
|
344
|
+
## Neighbouring products
|
|
345
|
+
|
|
346
|
+
This package governs what people reach. What an organisation's own AI agents may open on the web is the reverse question, answered by the [AI agent allow list](https://www.aiagentallowlist.com): verified page-type URLs for 40 million+ domains, up to 28 types per domain, and a per-URL allow or deny verdict so a browsing agent can read documentation and pricing while staying off login, checkout and upload pages. Its client is `aiagentallowlist`. Together the two give a [per-URL policy for browsing agents](https://www.aiagentallowlist.com) and a per-domain policy for staff.
|
|
347
|
+
|
|
348
|
+
Before either policy is written, most teams want to [find the AI tools employees use](https://www.shadowaitools.com) today. That service takes a DNS, proxy or firewall export, matches it against this same list, and returns the tools found with categories, risk flags, training verdicts, a per-user breakdown and a dated CSV and PDF.
|
|
349
|
+
|
|
350
|
+
## Frequently asked questions
|
|
351
|
+
|
|
352
|
+
**What is the AI Tools Blocklist?**
|
|
353
|
+
A daily-refreshed database of 20,000+ AI-tool domains classified into 18 functional categories with subcategories, with the vendor's data-use terms attached to each record. It is delivered as a lookup API, as downloadable CSV and JSON, and as hosted feeds in EDL, PAC, hosts and DNS formats from [aitoolsblocklist.com](https://www.aitoolsblocklist.com).
|
|
354
|
+
|
|
355
|
+
**How do I check whether a domain is an AI tool in Python?**
|
|
356
|
+
`pip install aiblocklist`, then `AIBlocklistClient("YOUR_API_KEY").check("domain.com").blocked`. The same `Lookup` object carries the category, subcategories, AI type and the five data-use fields.
|
|
357
|
+
|
|
358
|
+
**How do I find out whether an AI vendor trains on my data?**
|
|
359
|
+
`client.data_use("domain.com")` returns `trains_on_data`, `opt_out_available`, `enterprise_no_training`, `api_no_training` and `terms_checked`. `client.clause("domain.com", "trains_consumer_default")` returns the verbatim clause with its URL.
|
|
360
|
+
|
|
361
|
+
**Can I download the whole list?**
|
|
362
|
+
Feed and database plans can: `client.feeds.download_database(path)` streams the CSV, `download_categories(path)` the category tree. The Lookup API plan is lookup-only and receives a `PlanError` on those calls.
|
|
363
|
+
|
|
364
|
+
**Is there a bulk endpoint?**
|
|
365
|
+
No. `check_many()` runs sequential lookups, deduplicated, with an optional pause. For large lists, download the database once and match locally.
|
|
366
|
+
|
|
367
|
+
**How current is the list?**
|
|
368
|
+
Rebuilt daily. `client.stats()` shows the live totals; `client.feeds.database_info()` shows the timestamp of your file.
|
|
369
|
+
|
|
370
|
+
**Does the list work with schools and CIPA filtering?**
|
|
371
|
+
Yes. Districts use the categories to allow approved tutors while blocking essay generators, deepfake tools and companion chatbots, alongside the general filtering database at [cipawebfiltering.com](https://www.cipawebfiltering.com).
|
|
372
|
+
|
|
373
|
+
**Who builds it?**
|
|
374
|
+
Alpha Quantum, also behind the [website categorization API](https://www.websitecategorizationapi.com), the [web filtering database](https://www.webfilteringdatabase.com), the [AI agent allow list](https://www.aiagentallowlist.com) and the [shadow AI inventory](https://www.shadowaitools.com) service.
|
|
375
|
+
|
|
376
|
+
## Related packages
|
|
377
|
+
|
|
378
|
+
- [`aiblocklist` on npm](https://www.npmjs.com/package/aiblocklist), the Node.js version of this client
|
|
379
|
+
- [`aitoolsblocklist`](https://pypi.org/project/aitoolsblocklist/) on PyPI and [on npm](https://www.npmjs.com/package/aitoolsblocklist), the original AI Tools Blocklist client
|
|
380
|
+
- [`shadowaitools`](https://pypi.org/project/shadowaitools/) on PyPI and [on npm](https://www.npmjs.com/package/shadowaitools), local log scanner for shadow AI tools
|
|
381
|
+
- [`aiagentallowlist`](https://pypi.org/project/aiagentallowlist/) on PyPI and [on npm](https://www.npmjs.com/package/aiagentallowlist), client for the AI agent allow list
|
|
382
|
+
- [`phishingdetectionapi`](https://pypi.org/project/phishingdetectionapi/) on PyPI and [on npm](https://www.npmjs.com/package/phishingdetectionapi), from [phishingdetectionapi.com](https://www.phishingdetectionapi.com)
|
|
383
|
+
- [`webfilteringdatabase`](https://www.npmjs.com/package/webfilteringdatabase) on npm, from [webfilteringdatabase.com](https://www.webfilteringdatabase.com)
|
|
384
|
+
- [`websiteclassificationapi`](https://pypi.org/project/websiteclassificationapi/) on PyPI and [`websitecategorization`](https://www.npmjs.com/package/websitecategorization) on npm, from [websitecategorizationapi.com](https://www.websitecategorizationapi.com)
|
|
385
|
+
- [`cipawebfiltering`](https://pypi.org/project/cipawebfiltering/) on PyPI and [on npm](https://www.npmjs.com/package/cipawebfiltering)
|
|
386
|
+
- [PII detection API](https://www.piidetectionapi.com) for prompts and DLP logs
|
|
387
|
+
|
|
388
|
+
Source: [github.com/explainableaixai/aiblocklist](https://github.com/explainableaixai/aiblocklist), mirror at [gitlab.com/url-classifications/aiblocklist](https://gitlab.com/url-classifications/aiblocklist).
|
|
389
|
+
|
|
390
|
+
## Links
|
|
391
|
+
|
|
392
|
+
- Product, plans and feeds: [https://www.aitoolsblocklist.com](https://www.aitoolsblocklist.com)
|
|
393
|
+
- NIST AI Risk Management Framework: [https://www.nist.gov/itl/ai-risk-management-framework](https://www.nist.gov/itl/ai-risk-management-framework)
|
|
394
|
+
- OWASP Top 10 for LLM Applications: [https://owasp.org/www-project-top-10-for-large-language-model-applications/](https://owasp.org/www-project-top-10-for-large-language-model-applications/)
|
|
395
|
+
- ENISA: [https://www.enisa.europa.eu/](https://www.enisa.europa.eu/)
|
|
396
|
+
- EU AI Act, Regulation (EU) 2024/1689: [https://eur-lex.europa.eu/eli/reg/2024/1689/oj](https://eur-lex.europa.eu/eli/reg/2024/1689/oj)
|
|
397
|
+
- Data loss prevention software: [https://en.wikipedia.org/wiki/Data_loss_prevention_software](https://en.wikipedia.org/wiki/Data_loss_prevention_software)
|
|
398
|
+
|
|
399
|
+
## License
|
|
400
|
+
|
|
401
|
+
MIT
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
aiblocklist/__init__.py,sha256=VRLOrS3LNlc20aJrc_Zc-YCQAnLrNrTGwNrlF9bwyJs,918
|
|
2
|
+
aiblocklist/__main__.py,sha256=OnzT2EYofGJfr6AXa9qzyFrwwZ7xiCRxr-ZhINuAPUE,1117
|
|
3
|
+
aiblocklist/client.py,sha256=TC1Wxj_rJhJLwE_3Dil6F3dXEXNUUD3ebRdrEoMw1bc,12017
|
|
4
|
+
aiblocklist-1.0.0.dist-info/LICENSE,sha256=-kjrwollysEMZkPQkJIq5zyefl9XyPw5egz8knSXiB4,1070
|
|
5
|
+
aiblocklist-1.0.0.dist-info/METADATA,sha256=5-SQD55x30M4PxHaXoyxOCnqpH8tywF7OFaYGJO40yU,21620
|
|
6
|
+
aiblocklist-1.0.0.dist-info/WHEEL,sha256=BNRMDyzLkkcmlv0J8ppDQkk2VED33SesJDynr9ED1gc,91
|
|
7
|
+
aiblocklist-1.0.0.dist-info/top_level.txt,sha256=GC2FB1-k_7Zi9rFUIdW57PeZH4DnkheskywOCEGoLBY,12
|
|
8
|
+
aiblocklist-1.0.0.dist-info/RECORD,,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
aiblocklist
|