selldatatoai 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- selldatatoai-1.0.0/LICENSE +21 -0
- selldatatoai-1.0.0/MANIFEST.in +1 -0
- selldatatoai-1.0.0/PKG-INFO +317 -0
- selldatatoai-1.0.0/README.md +289 -0
- selldatatoai-1.0.0/selldatatoai/__init__.py +5 -0
- selldatatoai-1.0.0/selldatatoai/__main__.py +64 -0
- selldatatoai-1.0.0/selldatatoai/client.py +88 -0
- selldatatoai-1.0.0/selldatatoai.egg-info/PKG-INFO +317 -0
- selldatatoai-1.0.0/selldatatoai.egg-info/SOURCES.txt +14 -0
- selldatatoai-1.0.0/selldatatoai.egg-info/dependency_links.txt +1 -0
- selldatatoai-1.0.0/selldatatoai.egg-info/entry_points.txt +2 -0
- selldatatoai-1.0.0/selldatatoai.egg-info/requires.txt +1 -0
- selldatatoai-1.0.0/selldatatoai.egg-info/top_level.txt +1 -0
- selldatatoai-1.0.0/setup.cfg +4 -0
- selldatatoai-1.0.0/setup.py +40 -0
- selldatatoai-1.0.0/tests/test_client.py +82 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Alpha Quantum
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
include README.md LICENSE
|
|
@@ -0,0 +1,317 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: selldatatoai
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Python client for the Data Asset Score API from selldatatoai.com: score company domains 0 to 100 for the data AI buyers want, one at a time or in batches of 100.
|
|
5
|
+
Home-page: https://www.selldatatoai.com
|
|
6
|
+
Author: Alpha Quantum
|
|
7
|
+
Author-email: info@alpha-quantum.com
|
|
8
|
+
License: MIT
|
|
9
|
+
Project-URL: Homepage, https://www.selldatatoai.com
|
|
10
|
+
Project-URL: Documentation, https://www.selldatatoai.com/api/
|
|
11
|
+
Project-URL: Source, https://github.com/explainableaixai/selldatatoai-python
|
|
12
|
+
Project-URL: Pricing, https://www.selldatatoai.com/pricing/
|
|
13
|
+
Project-URL: Demo, https://www.selldatatoai.com/data-asset-score/
|
|
14
|
+
Keywords: sell data to ai,data asset score,ai training data,ai data broker,company data api,lead scoring,firmographics,technographics,data licensing,b2b data,domain intelligence
|
|
15
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
16
|
+
Classifier: Intended Audience :: Developers
|
|
17
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
18
|
+
Classifier: Operating System :: OS Independent
|
|
19
|
+
Classifier: Programming Language :: Python :: 3
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.7
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
23
|
+
Classifier: Topic :: Office/Business
|
|
24
|
+
Requires-Python: >=3.7
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
License-File: LICENSE
|
|
27
|
+
Requires-Dist: requests>=2.20
|
|
28
|
+
|
|
29
|
+
# selldatatoai for Python
|
|
30
|
+
|
|
31
|
+
`selldatatoai` scores companies for AI data deals from Python. You pass a website; the API answers with a 0 to 100 [Data Asset Score](https://www.selldatatoai.com/data-asset-score/), a grade, the data the company likely holds and how long it has been online.
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
pip install selldatatoai
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Requires Python 3.7 or newer and `requests`. The package also installs a `selldatatoai` command for scoring a text file of domains into a CSV.
|
|
38
|
+
|
|
39
|
+
The service behind it is [selldatatoai.com](https://www.selldatatoai.com/), which keeps an index of 102 million domains, 99.99% of the active internet, together with each domain's history.
|
|
40
|
+
|
|
41
|
+
## Who this is for
|
|
42
|
+
|
|
43
|
+
- **Data brokers** who source data partners for AI labs and need to know which companies are worth a call.
|
|
44
|
+
- **Data companies** that resell records and want to rank prospects by the data they hold.
|
|
45
|
+
- **Referral partners** in data programs who screen companies before they introduce them.
|
|
46
|
+
- **Analysts** who study which sectors hold the operational records that AI training buys.
|
|
47
|
+
|
|
48
|
+
If you are new to the trade itself, the step-by-step guide on [how to sell data to AI companies](https://www.selldatatoai.com/how-to-sell-data-to-ai-companies/) covers the deal from first contact to delivery.
|
|
49
|
+
|
|
50
|
+
## Five lines to a score
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
import os
|
|
54
|
+
from selldatatoai import DataAssetScoreClient
|
|
55
|
+
|
|
56
|
+
client = DataAssetScoreClient(os.environ["SDA_API_KEY"])
|
|
57
|
+
s = client.score("example.com")
|
|
58
|
+
print(s["data_asset_score"], s["grade"], [a["label"] for a in s["likely_data_assets"]])
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
The result is a plain `dict` with the API field names. A trimmed real answer looks like this:
|
|
62
|
+
|
|
63
|
+
```json
|
|
64
|
+
{
|
|
65
|
+
"domain": "promega.com",
|
|
66
|
+
"data_asset_score": 88,
|
|
67
|
+
"grade": "A",
|
|
68
|
+
"status": "active",
|
|
69
|
+
"verified_active": true,
|
|
70
|
+
"iab_category": "Business and Finance > Industries > Pharmaceutical Industry",
|
|
71
|
+
"country": "United States",
|
|
72
|
+
"history": {"first_seen_year": 1993, "years_online": 33, "founded_year": null, "pre_ai_years": 30},
|
|
73
|
+
"pre_ai_archive_likely": true,
|
|
74
|
+
"data_systems": [
|
|
75
|
+
{"system": "Jira / Confluence", "data_type": "Work tickets and internal wiki", "type": "work_tickets"},
|
|
76
|
+
{"system": "Webex", "data_type": "Calls and meetings", "type": "calls"}
|
|
77
|
+
],
|
|
78
|
+
"likely_data_assets": [
|
|
79
|
+
{"type": "support_tickets", "label": "Support tickets and chat transcripts"},
|
|
80
|
+
{"type": "knowledge_base", "label": "Knowledge base, SOPs and documentation"}
|
|
81
|
+
],
|
|
82
|
+
"data_coverage": "full",
|
|
83
|
+
"cached": true
|
|
84
|
+
}
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## API surface
|
|
88
|
+
|
|
89
|
+
| Python call | Endpoint | Cost |
|
|
90
|
+
|---|---|---|
|
|
91
|
+
| `client.score(domain)` | `GET /api/v1/score` | 1 lookup |
|
|
92
|
+
| `client.submit_batch(domains)` | `POST /api/v1/score/batch` | 1 lookup per valid, unique domain |
|
|
93
|
+
| `client.get_batch(batch_id)` | `GET /api/v1/score/batch?id=` | free |
|
|
94
|
+
| `client.wait_for_batch(batch_id, interval=5, max_wait=300)` | polls `get_batch` | free |
|
|
95
|
+
| `client.score_many(domains, interval=5, max_wait=300, on_batch=None)` | batches of 100 | 1 lookup per valid, unique domain |
|
|
96
|
+
| `client.usage()` | `GET /api/v1/usage` | free |
|
|
97
|
+
|
|
98
|
+
Constructor:
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
DataAssetScoreClient(api_key, base_url="https://www.selldatatoai.com/api/v1", timeout=60, session=None)
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Pass your own `requests.Session` if you need retries, a proxy or connection pooling shared with other code. The key travels in the `X-API-Key` header only.
|
|
105
|
+
|
|
106
|
+
Every field is described in the [REST API reference](https://www.selldatatoai.com/api/).
|
|
107
|
+
|
|
108
|
+
## Handling failures
|
|
109
|
+
|
|
110
|
+
All HTTP errors raise `SellDataToAIError`. It carries `.status` (HTTP code), `.code` (the API error string) and `.body` (the decoded answer).
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
from selldatatoai import SellDataToAIError
|
|
114
|
+
|
|
115
|
+
def safe_score(client, domain):
|
|
116
|
+
try:
|
|
117
|
+
return client.score(domain)
|
|
118
|
+
except SellDataToAIError as e:
|
|
119
|
+
if e.code == "invalid_domain":
|
|
120
|
+
return None # bad input, skip it
|
|
121
|
+
if e.code == "monthly_limit_reached":
|
|
122
|
+
raise SystemExit("Out of lookups until the 1st (UTC)")
|
|
123
|
+
raise # 401, network trouble, anything else
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
Codes you can meet:
|
|
127
|
+
|
|
128
|
+
- `missing_api_key`, `invalid_api_key` (401): no key, or the plan is not active.
|
|
129
|
+
- `invalid_domain` (400): not a valid domain.
|
|
130
|
+
- `no_domains`, `too_many_domains` (400): a batch needs 1 to 100 entries.
|
|
131
|
+
- `monthly_limit_reached` (429): limits reset on the first day of each month, UTC.
|
|
132
|
+
- `batch_busy` (429): three of your batches are still open.
|
|
133
|
+
- `batch_not_found` (404): unknown id, or older than 7 days.
|
|
134
|
+
- `not_available` (410): company lists are not served by the API.
|
|
135
|
+
|
|
136
|
+
## Recipes
|
|
137
|
+
|
|
138
|
+
### Recipe 1: score a spreadsheet with pandas
|
|
139
|
+
|
|
140
|
+
```python
|
|
141
|
+
import os
|
|
142
|
+
import pandas as pd
|
|
143
|
+
from selldatatoai import DataAssetScoreClient
|
|
144
|
+
|
|
145
|
+
client = DataAssetScoreClient(os.environ["SDA_API_KEY"])
|
|
146
|
+
df = pd.read_csv("prospects.csv") # needs a 'website' column
|
|
147
|
+
|
|
148
|
+
items = client.score_many(df["website"].dropna().tolist())
|
|
149
|
+
rows = []
|
|
150
|
+
for it in items:
|
|
151
|
+
r = it.get("result") or {}
|
|
152
|
+
rows.append({
|
|
153
|
+
"website": it["input"],
|
|
154
|
+
"score": r.get("data_asset_score"),
|
|
155
|
+
"grade": r.get("grade"),
|
|
156
|
+
"status": r.get("status", it["status"]),
|
|
157
|
+
"pre_ai_years": (r.get("history") or {}).get("pre_ai_years"),
|
|
158
|
+
"systems": ", ".join(s["system"] for s in r.get("data_systems", [])),
|
|
159
|
+
})
|
|
160
|
+
|
|
161
|
+
scored = df.merge(pd.DataFrame(rows), on="website", how="left")
|
|
162
|
+
scored.sort_values("score", ascending=False).to_csv("prospects_scored.csv", index=False)
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
### Recipe 2: a FastAPI route for a partner form
|
|
166
|
+
|
|
167
|
+
```python
|
|
168
|
+
import os
|
|
169
|
+
from fastapi import FastAPI, HTTPException
|
|
170
|
+
from pydantic import BaseModel
|
|
171
|
+
from selldatatoai import DataAssetScoreClient, SellDataToAIError
|
|
172
|
+
|
|
173
|
+
app = FastAPI()
|
|
174
|
+
sda = DataAssetScoreClient(os.environ["SDA_API_KEY"])
|
|
175
|
+
|
|
176
|
+
class Signup(BaseModel):
|
|
177
|
+
company: str
|
|
178
|
+
website: str
|
|
179
|
+
|
|
180
|
+
@app.post("/signup")
|
|
181
|
+
def signup(body: Signup):
|
|
182
|
+
try:
|
|
183
|
+
s = sda.score(body.website)
|
|
184
|
+
except SellDataToAIError as e:
|
|
185
|
+
if e.status == 400:
|
|
186
|
+
raise HTTPException(422, "That website does not look valid")
|
|
187
|
+
raise HTTPException(503, "Scoring is unavailable, try again shortly")
|
|
188
|
+
tier = "priority" if s["grade"] in ("A", "B") and s["verified_active"] else "standard"
|
|
189
|
+
return {"company": body.company, "tier": tier, "score": s["data_asset_score"]}
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
The client is synchronous. Inside an async framework, FastAPI runs plain `def` routes in a thread pool, which is what you want here.
|
|
193
|
+
|
|
194
|
+
### Recipe 3: parallel single lookups with a thread pool
|
|
195
|
+
|
|
196
|
+
Batches are the better tool for big lists. For a few dozen domains where you want each answer as soon as it is ready, a small pool works well:
|
|
197
|
+
|
|
198
|
+
```python
|
|
199
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
200
|
+
|
|
201
|
+
domains = ["example-one.com", "example-two.com", "example-three.com"]
|
|
202
|
+
with ThreadPoolExecutor(max_workers=4) as pool:
|
|
203
|
+
futures = {pool.submit(client.score, d): d for d in domains}
|
|
204
|
+
for f in as_completed(futures):
|
|
205
|
+
d = futures[f]
|
|
206
|
+
try:
|
|
207
|
+
print(d, f.result()["data_asset_score"])
|
|
208
|
+
except SellDataToAIError as e:
|
|
209
|
+
print(d, "failed:", e.code)
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
Keep the pool small. Each worker is one open request, and every call still counts against your monthly lookups.
|
|
213
|
+
|
|
214
|
+
### Recipe 4: the command line
|
|
215
|
+
|
|
216
|
+
```bash
|
|
217
|
+
export SDA_API_KEY=xxxx
|
|
218
|
+
selldatatoai score example.com # JSON for one company
|
|
219
|
+
selldatatoai usage # plan and remaining lookups
|
|
220
|
+
selldatatoai file domains.txt out.csv # one domain per line, '#' lines skipped
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
`python -m selldatatoai` works the same way if the script directory is not on your `PATH`. The CSV has one row per input line, in input order, with score, grade, company status, years online, pre-AI years, systems and likely data assets.
|
|
224
|
+
|
|
225
|
+
## How batches behave
|
|
226
|
+
|
|
227
|
+
1. You send up to 100 domains.
|
|
228
|
+
2. Lookups are charged right away: one per valid, unique domain.
|
|
229
|
+
3. Domains scored in the last 30 days are `done` in the first answer.
|
|
230
|
+
4. The rest finish in the background, usually within a minute.
|
|
231
|
+
5. You poll with the batch id. Polling is free.
|
|
232
|
+
6. Results stay in the order you sent and are kept for 7 days.
|
|
233
|
+
|
|
234
|
+
You can hold three open batches per key. `score_many` sends them one after another, so it never trips that limit.
|
|
235
|
+
|
|
236
|
+
## Making sense of the numbers
|
|
237
|
+
|
|
238
|
+
The score gathers several factor groups into one number. The [how the Data Asset Score works](https://www.selldatatoai.com/how-the-data-asset-score-works/) page describes the data behind each group. The exact weights are not published.
|
|
239
|
+
|
|
240
|
+
A few reading tips:
|
|
241
|
+
|
|
242
|
+
- **Grade first, score second.** Two companies at 71 and 74 are in the same band. A and B grades are where most buyers start.
|
|
243
|
+
- **Check `status`.** A high score on a `winding_down_or_acquired` company means the records may still exist, but the seller has changed.
|
|
244
|
+
- **Look at `pre_ai_years`.** Records written before 2023 are free of AI-generated text, which some buyers value highly.
|
|
245
|
+
- **Read `data_coverage`.** `limited` means fewer signals were available, so the score is less certain.
|
|
246
|
+
|
|
247
|
+
Sector context helps too. Life sciences companies tend to hold lab, study and R&D records, which is why buyers of [AI training data companies](https://www.selldatatoai.com/ai-training-data-companies/) research often start with that sector. A ready file of the top 3,000 US firms in that space is described on the [list of biotech companies](https://www.selldatatoai.com/list-of-biotech-companies/) page.
|
|
248
|
+
|
|
249
|
+
## Grouping companies by the data they hold
|
|
250
|
+
|
|
251
|
+
Buyers rarely ask for "data". They ask for a type: support conversations, engineering tickets, recorded calls, contracts, design files. The `type` keys in `data_systems` and `likely_data_assets` let you build those groups in a few lines.
|
|
252
|
+
|
|
253
|
+
```python
|
|
254
|
+
from collections import defaultdict
|
|
255
|
+
|
|
256
|
+
by_type = defaultdict(list)
|
|
257
|
+
for it in items: # items from score_many()
|
|
258
|
+
r = it.get("result")
|
|
259
|
+
if not r or not r["verified_active"]:
|
|
260
|
+
continue
|
|
261
|
+
for a in r["likely_data_assets"]:
|
|
262
|
+
by_type[a["type"]].append((r["data_asset_score"], r["domain"]))
|
|
263
|
+
|
|
264
|
+
for t, rows in sorted(by_type.items()):
|
|
265
|
+
top = sorted(rows, reverse=True)[:10]
|
|
266
|
+
print(t, len(rows), "companies, top:", ", ".join(d for _, d in top))
|
|
267
|
+
```
|
|
268
|
+
|
|
269
|
+
Two keys deserve a note:
|
|
270
|
+
|
|
271
|
+
- `data_systems` lists systems the company is seen to run, each mapped to the data type it stores. It is the stronger signal, because a running system means the records exist and can be exported.
|
|
272
|
+
- `likely_data_assets` lists what the company probably holds based on its sector and footprint. Use it to widen a search, not to promise a buyer anything.
|
|
273
|
+
|
|
274
|
+
When a buyer asks for one data type across a whole market, the website also sells ready lists by data type next to the sector lists.
|
|
275
|
+
|
|
276
|
+
## Plans
|
|
277
|
+
|
|
278
|
+
Paid plans only, from $99 a month: Basic $99 for 5,000 lookups, Pro $299 for 25,000 with batch scoring, Scale $799 for 100,000 with batch scoring. The website demo allows 5 checks a day if you want to see the output before you subscribe. Your key appears in your dashboard as soon as the payment completes.
|
|
279
|
+
|
|
280
|
+
Company lists are not part of any API plan. They are one-time files of the top 3,000 US companies per sector, from $249 a list, delivered by email links that work for 30 days.
|
|
281
|
+
|
|
282
|
+
## FAQ
|
|
283
|
+
|
|
284
|
+
### What does the selldatatoai package do?
|
|
285
|
+
|
|
286
|
+
The selldatatoai package is the Python client for the Data Asset Score API at selldatatoai.com. It scores a company website from 0 to 100 for the data AI buyers want and returns the grade, status, data systems, likely data assets and web history.
|
|
287
|
+
|
|
288
|
+
### Does it support asyncio?
|
|
289
|
+
|
|
290
|
+
Not directly. Run calls in a thread pool (`asyncio.to_thread` on Python 3.9+) or use the batch endpoint, which does the parallel work on the server.
|
|
291
|
+
|
|
292
|
+
### What counts as a lookup?
|
|
293
|
+
|
|
294
|
+
Each scored domain, fresh or cached. A batch counts each valid, unique domain once. Usage and polling calls are free.
|
|
295
|
+
|
|
296
|
+
### Can I pass full URLs?
|
|
297
|
+
|
|
298
|
+
Yes. `https://www.example.com/contact` and `example.com` score the same company.
|
|
299
|
+
|
|
300
|
+
### Are results stored?
|
|
301
|
+
|
|
302
|
+
A result is reused for 30 days, then the domain is scored again. Batches are deleted after 7 days.
|
|
303
|
+
|
|
304
|
+
### Is there a sandbox key?
|
|
305
|
+
|
|
306
|
+
No. Plans are paid, and the demo on the website is the only free access.
|
|
307
|
+
|
|
308
|
+
## Links
|
|
309
|
+
|
|
310
|
+
- Homepage: https://www.selldatatoai.com/
|
|
311
|
+
- Source code: https://github.com/explainableaixai/selldatatoai-python
|
|
312
|
+
- Python packaging guide: [packaging.python.org](https://packaging.python.org/en/latest/tutorials/installing-packages/)
|
|
313
|
+
- Thread pools in the standard library: [docs.python.org](https://docs.python.org/3/library/concurrent.futures.html)
|
|
314
|
+
|
|
315
|
+
## License
|
|
316
|
+
|
|
317
|
+
MIT, Copyright (c) 2026 Alpha Quantum. Questions: info@alpha-quantum.com
|
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
# selldatatoai for Python
|
|
2
|
+
|
|
3
|
+
`selldatatoai` scores companies for AI data deals from Python. You pass a website; the API answers with a 0 to 100 [Data Asset Score](https://www.selldatatoai.com/data-asset-score/), a grade, the data the company likely holds and how long it has been online.
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
pip install selldatatoai
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
Requires Python 3.7 or newer and `requests`. The package also installs a `selldatatoai` command for scoring a text file of domains into a CSV.
|
|
10
|
+
|
|
11
|
+
The service behind it is [selldatatoai.com](https://www.selldatatoai.com/), which keeps an index of 102 million domains, 99.99% of the active internet, together with each domain's history.
|
|
12
|
+
|
|
13
|
+
## Who this is for
|
|
14
|
+
|
|
15
|
+
- **Data brokers** who source data partners for AI labs and need to know which companies are worth a call.
|
|
16
|
+
- **Data companies** that resell records and want to rank prospects by the data they hold.
|
|
17
|
+
- **Referral partners** in data programs who screen companies before they introduce them.
|
|
18
|
+
- **Analysts** who study which sectors hold the operational records that AI training buys.
|
|
19
|
+
|
|
20
|
+
If you are new to the trade itself, the step-by-step guide on [how to sell data to AI companies](https://www.selldatatoai.com/how-to-sell-data-to-ai-companies/) covers the deal from first contact to delivery.
|
|
21
|
+
|
|
22
|
+
## Five lines to a score
|
|
23
|
+
|
|
24
|
+
```python
|
|
25
|
+
import os
|
|
26
|
+
from selldatatoai import DataAssetScoreClient
|
|
27
|
+
|
|
28
|
+
client = DataAssetScoreClient(os.environ["SDA_API_KEY"])
|
|
29
|
+
s = client.score("example.com")
|
|
30
|
+
print(s["data_asset_score"], s["grade"], [a["label"] for a in s["likely_data_assets"]])
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
The result is a plain `dict` with the API field names. A trimmed real answer looks like this:
|
|
34
|
+
|
|
35
|
+
```json
|
|
36
|
+
{
|
|
37
|
+
"domain": "promega.com",
|
|
38
|
+
"data_asset_score": 88,
|
|
39
|
+
"grade": "A",
|
|
40
|
+
"status": "active",
|
|
41
|
+
"verified_active": true,
|
|
42
|
+
"iab_category": "Business and Finance > Industries > Pharmaceutical Industry",
|
|
43
|
+
"country": "United States",
|
|
44
|
+
"history": {"first_seen_year": 1993, "years_online": 33, "founded_year": null, "pre_ai_years": 30},
|
|
45
|
+
"pre_ai_archive_likely": true,
|
|
46
|
+
"data_systems": [
|
|
47
|
+
{"system": "Jira / Confluence", "data_type": "Work tickets and internal wiki", "type": "work_tickets"},
|
|
48
|
+
{"system": "Webex", "data_type": "Calls and meetings", "type": "calls"}
|
|
49
|
+
],
|
|
50
|
+
"likely_data_assets": [
|
|
51
|
+
{"type": "support_tickets", "label": "Support tickets and chat transcripts"},
|
|
52
|
+
{"type": "knowledge_base", "label": "Knowledge base, SOPs and documentation"}
|
|
53
|
+
],
|
|
54
|
+
"data_coverage": "full",
|
|
55
|
+
"cached": true
|
|
56
|
+
}
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## API surface
|
|
60
|
+
|
|
61
|
+
| Python call | Endpoint | Cost |
|
|
62
|
+
|---|---|---|
|
|
63
|
+
| `client.score(domain)` | `GET /api/v1/score` | 1 lookup |
|
|
64
|
+
| `client.submit_batch(domains)` | `POST /api/v1/score/batch` | 1 lookup per valid, unique domain |
|
|
65
|
+
| `client.get_batch(batch_id)` | `GET /api/v1/score/batch?id=` | free |
|
|
66
|
+
| `client.wait_for_batch(batch_id, interval=5, max_wait=300)` | polls `get_batch` | free |
|
|
67
|
+
| `client.score_many(domains, interval=5, max_wait=300, on_batch=None)` | batches of 100 | 1 lookup per valid, unique domain |
|
|
68
|
+
| `client.usage()` | `GET /api/v1/usage` | free |
|
|
69
|
+
|
|
70
|
+
Constructor:
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
DataAssetScoreClient(api_key, base_url="https://www.selldatatoai.com/api/v1", timeout=60, session=None)
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Pass your own `requests.Session` if you need retries, a proxy or connection pooling shared with other code. The key travels in the `X-API-Key` header only.
|
|
77
|
+
|
|
78
|
+
Every field is described in the [REST API reference](https://www.selldatatoai.com/api/).
|
|
79
|
+
|
|
80
|
+
## Handling failures
|
|
81
|
+
|
|
82
|
+
All HTTP errors raise `SellDataToAIError`. It carries `.status` (HTTP code), `.code` (the API error string) and `.body` (the decoded answer).
|
|
83
|
+
|
|
84
|
+
```python
|
|
85
|
+
from selldatatoai import SellDataToAIError
|
|
86
|
+
|
|
87
|
+
def safe_score(client, domain):
|
|
88
|
+
try:
|
|
89
|
+
return client.score(domain)
|
|
90
|
+
except SellDataToAIError as e:
|
|
91
|
+
if e.code == "invalid_domain":
|
|
92
|
+
return None # bad input, skip it
|
|
93
|
+
if e.code == "monthly_limit_reached":
|
|
94
|
+
raise SystemExit("Out of lookups until the 1st (UTC)")
|
|
95
|
+
raise # 401, network trouble, anything else
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
Codes you can meet:
|
|
99
|
+
|
|
100
|
+
- `missing_api_key`, `invalid_api_key` (401): no key, or the plan is not active.
|
|
101
|
+
- `invalid_domain` (400): not a valid domain.
|
|
102
|
+
- `no_domains`, `too_many_domains` (400): a batch needs 1 to 100 entries.
|
|
103
|
+
- `monthly_limit_reached` (429): limits reset on the first day of each month, UTC.
|
|
104
|
+
- `batch_busy` (429): three of your batches are still open.
|
|
105
|
+
- `batch_not_found` (404): unknown id, or older than 7 days.
|
|
106
|
+
- `not_available` (410): company lists are not served by the API.
|
|
107
|
+
|
|
108
|
+
## Recipes
|
|
109
|
+
|
|
110
|
+
### Recipe 1: score a spreadsheet with pandas
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
import os
|
|
114
|
+
import pandas as pd
|
|
115
|
+
from selldatatoai import DataAssetScoreClient
|
|
116
|
+
|
|
117
|
+
client = DataAssetScoreClient(os.environ["SDA_API_KEY"])
|
|
118
|
+
df = pd.read_csv("prospects.csv") # needs a 'website' column
|
|
119
|
+
|
|
120
|
+
items = client.score_many(df["website"].dropna().tolist())
|
|
121
|
+
rows = []
|
|
122
|
+
for it in items:
|
|
123
|
+
r = it.get("result") or {}
|
|
124
|
+
rows.append({
|
|
125
|
+
"website": it["input"],
|
|
126
|
+
"score": r.get("data_asset_score"),
|
|
127
|
+
"grade": r.get("grade"),
|
|
128
|
+
"status": r.get("status", it["status"]),
|
|
129
|
+
"pre_ai_years": (r.get("history") or {}).get("pre_ai_years"),
|
|
130
|
+
"systems": ", ".join(s["system"] for s in r.get("data_systems", [])),
|
|
131
|
+
})
|
|
132
|
+
|
|
133
|
+
scored = df.merge(pd.DataFrame(rows), on="website", how="left")
|
|
134
|
+
scored.sort_values("score", ascending=False).to_csv("prospects_scored.csv", index=False)
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
### Recipe 2: a FastAPI route for a partner form
|
|
138
|
+
|
|
139
|
+
```python
|
|
140
|
+
import os
|
|
141
|
+
from fastapi import FastAPI, HTTPException
|
|
142
|
+
from pydantic import BaseModel
|
|
143
|
+
from selldatatoai import DataAssetScoreClient, SellDataToAIError
|
|
144
|
+
|
|
145
|
+
app = FastAPI()
|
|
146
|
+
sda = DataAssetScoreClient(os.environ["SDA_API_KEY"])
|
|
147
|
+
|
|
148
|
+
class Signup(BaseModel):
|
|
149
|
+
company: str
|
|
150
|
+
website: str
|
|
151
|
+
|
|
152
|
+
@app.post("/signup")
|
|
153
|
+
def signup(body: Signup):
|
|
154
|
+
try:
|
|
155
|
+
s = sda.score(body.website)
|
|
156
|
+
except SellDataToAIError as e:
|
|
157
|
+
if e.status == 400:
|
|
158
|
+
raise HTTPException(422, "That website does not look valid")
|
|
159
|
+
raise HTTPException(503, "Scoring is unavailable, try again shortly")
|
|
160
|
+
tier = "priority" if s["grade"] in ("A", "B") and s["verified_active"] else "standard"
|
|
161
|
+
return {"company": body.company, "tier": tier, "score": s["data_asset_score"]}
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
The client is synchronous. Inside an async framework, FastAPI runs plain `def` routes in a thread pool, which is what you want here.
|
|
165
|
+
|
|
166
|
+
### Recipe 3: parallel single lookups with a thread pool
|
|
167
|
+
|
|
168
|
+
Batches are the better tool for big lists. For a few dozen domains where you want each answer as soon as it is ready, a small pool works well:
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
172
|
+
|
|
173
|
+
domains = ["example-one.com", "example-two.com", "example-three.com"]
|
|
174
|
+
with ThreadPoolExecutor(max_workers=4) as pool:
|
|
175
|
+
futures = {pool.submit(client.score, d): d for d in domains}
|
|
176
|
+
for f in as_completed(futures):
|
|
177
|
+
d = futures[f]
|
|
178
|
+
try:
|
|
179
|
+
print(d, f.result()["data_asset_score"])
|
|
180
|
+
except SellDataToAIError as e:
|
|
181
|
+
print(d, "failed:", e.code)
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
Keep the pool small. Each worker is one open request, and every call still counts against your monthly lookups.
|
|
185
|
+
|
|
186
|
+
### Recipe 4: the command line
|
|
187
|
+
|
|
188
|
+
```bash
|
|
189
|
+
export SDA_API_KEY=xxxx
|
|
190
|
+
selldatatoai score example.com # JSON for one company
|
|
191
|
+
selldatatoai usage # plan and remaining lookups
|
|
192
|
+
selldatatoai file domains.txt out.csv # one domain per line, '#' lines skipped
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
`python -m selldatatoai` works the same way if the script directory is not on your `PATH`. The CSV has one row per input line, in input order, with score, grade, company status, years online, pre-AI years, systems and likely data assets.
|
|
196
|
+
|
|
197
|
+
## How batches behave
|
|
198
|
+
|
|
199
|
+
1. You send up to 100 domains.
|
|
200
|
+
2. Lookups are charged right away: one per valid, unique domain.
|
|
201
|
+
3. Domains scored in the last 30 days are `done` in the first answer.
|
|
202
|
+
4. The rest finish in the background, usually within a minute.
|
|
203
|
+
5. You poll with the batch id. Polling is free.
|
|
204
|
+
6. Results stay in the order you sent and are kept for 7 days.
|
|
205
|
+
|
|
206
|
+
You can hold three open batches per key. `score_many` sends them one after another, so it never trips that limit.
|
|
207
|
+
|
|
208
|
+
## Making sense of the numbers
|
|
209
|
+
|
|
210
|
+
The score gathers several factor groups into one number. The [how the Data Asset Score works](https://www.selldatatoai.com/how-the-data-asset-score-works/) page describes the data behind each group. The exact weights are not published.
|
|
211
|
+
|
|
212
|
+
A few reading tips:
|
|
213
|
+
|
|
214
|
+
- **Grade first, score second.** Two companies at 71 and 74 are in the same band. A and B grades are where most buyers start.
|
|
215
|
+
- **Check `status`.** A high score on a `winding_down_or_acquired` company means the records may still exist, but the seller has changed.
|
|
216
|
+
- **Look at `pre_ai_years`.** Records written before 2023 are free of AI-generated text, which some buyers value highly.
|
|
217
|
+
- **Read `data_coverage`.** `limited` means fewer signals were available, so the score is less certain.
|
|
218
|
+
|
|
219
|
+
Sector context helps too. Life sciences companies tend to hold lab, study and R&D records, which is why buyers of [AI training data companies](https://www.selldatatoai.com/ai-training-data-companies/) research often start with that sector. A ready file of the top 3,000 US firms in that space is described on the [list of biotech companies](https://www.selldatatoai.com/list-of-biotech-companies/) page.
|
|
220
|
+
|
|
221
|
+
## Grouping companies by the data they hold
|
|
222
|
+
|
|
223
|
+
Buyers rarely ask for "data". They ask for a type: support conversations, engineering tickets, recorded calls, contracts, design files. The `type` keys in `data_systems` and `likely_data_assets` let you build those groups in a few lines.
|
|
224
|
+
|
|
225
|
+
```python
|
|
226
|
+
from collections import defaultdict
|
|
227
|
+
|
|
228
|
+
by_type = defaultdict(list)
|
|
229
|
+
for it in items: # items from score_many()
|
|
230
|
+
r = it.get("result")
|
|
231
|
+
if not r or not r["verified_active"]:
|
|
232
|
+
continue
|
|
233
|
+
for a in r["likely_data_assets"]:
|
|
234
|
+
by_type[a["type"]].append((r["data_asset_score"], r["domain"]))
|
|
235
|
+
|
|
236
|
+
for t, rows in sorted(by_type.items()):
|
|
237
|
+
top = sorted(rows, reverse=True)[:10]
|
|
238
|
+
print(t, len(rows), "companies, top:", ", ".join(d for _, d in top))
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
Two keys deserve a note:
|
|
242
|
+
|
|
243
|
+
- `data_systems` lists systems the company is seen to run, each mapped to the data type it stores. It is the stronger signal, because a running system means the records exist and can be exported.
|
|
244
|
+
- `likely_data_assets` lists what the company probably holds based on its sector and footprint. Use it to widen a search, not to promise a buyer anything.
|
|
245
|
+
|
|
246
|
+
When a buyer asks for one data type across a whole market, the website also sells ready lists by data type next to the sector lists.
|
|
247
|
+
|
|
248
|
+
## Plans
|
|
249
|
+
|
|
250
|
+
Paid plans only, from $99 a month: Basic $99 for 5,000 lookups, Pro $299 for 25,000 with batch scoring, Scale $799 for 100,000 with batch scoring. The website demo allows 5 checks a day if you want to see the output before you subscribe. Your key appears in your dashboard as soon as the payment completes.
|
|
251
|
+
|
|
252
|
+
Company lists are not part of any API plan. They are one-time files of the top 3,000 US companies per sector, from $249 a list, delivered by email links that work for 30 days.
|
|
253
|
+
|
|
254
|
+
## FAQ
|
|
255
|
+
|
|
256
|
+
### What does the selldatatoai package do?
|
|
257
|
+
|
|
258
|
+
The selldatatoai package is the Python client for the Data Asset Score API at selldatatoai.com. It scores a company website from 0 to 100 for the data AI buyers want and returns the grade, status, data systems, likely data assets and web history.
|
|
259
|
+
|
|
260
|
+
### Does it support asyncio?
|
|
261
|
+
|
|
262
|
+
Not directly. Run calls in a thread pool (`asyncio.to_thread` on Python 3.9+) or use the batch endpoint, which does the parallel work on the server.
|
|
263
|
+
|
|
264
|
+
### What counts as a lookup?
|
|
265
|
+
|
|
266
|
+
Each scored domain, fresh or cached. A batch counts each valid, unique domain once. Usage and polling calls are free.
|
|
267
|
+
|
|
268
|
+
### Can I pass full URLs?
|
|
269
|
+
|
|
270
|
+
Yes. `https://www.example.com/contact` and `example.com` score the same company.
|
|
271
|
+
|
|
272
|
+
### Are results stored?
|
|
273
|
+
|
|
274
|
+
A result is reused for 30 days, then the domain is scored again. Batches are deleted after 7 days.
|
|
275
|
+
|
|
276
|
+
### Is there a sandbox key?
|
|
277
|
+
|
|
278
|
+
No. Plans are paid, and the demo on the website is the only free access.
|
|
279
|
+
|
|
280
|
+
## Links
|
|
281
|
+
|
|
282
|
+
- Homepage: https://www.selldatatoai.com/
|
|
283
|
+
- Source code: https://github.com/explainableaixai/selldatatoai-python
|
|
284
|
+
- Python packaging guide: [packaging.python.org](https://packaging.python.org/en/latest/tutorials/installing-packages/)
|
|
285
|
+
- Thread pools in the standard library: [docs.python.org](https://docs.python.org/3/library/concurrent.futures.html)
|
|
286
|
+
|
|
287
|
+
## License
|
|
288
|
+
|
|
289
|
+
MIT, Copyright (c) 2026 Alpha Quantum. Questions: info@alpha-quantum.com
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
"""selldatatoai: Python client for the Data Asset Score API (https://www.selldatatoai.com/api/)."""
|
|
2
|
+
from .client import BATCH_MAX, VERSION, DataAssetScoreClient, SellDataToAIError
|
|
3
|
+
|
|
4
|
+
__version__ = VERSION
|
|
5
|
+
__all__ = ["DataAssetScoreClient", "SellDataToAIError", "BATCH_MAX", "__version__"]
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""python -m selldatatoai score example.com | python -m selldatatoai file domains.txt out.csv | python -m selldatatoai usage
|
|
2
|
+
|
|
3
|
+
The key is read from the SDA_API_KEY environment variable.
|
|
4
|
+
"""
|
|
5
|
+
import csv
|
|
6
|
+
import json
|
|
7
|
+
import os
|
|
8
|
+
import sys
|
|
9
|
+
|
|
10
|
+
from .client import DataAssetScoreClient, SellDataToAIError
|
|
11
|
+
|
|
12
|
+
FIELDS = ["input", "domain", "status", "data_asset_score", "grade", "company_status", "verified_active",
|
|
13
|
+
"iab_category", "country", "years_online", "pre_ai_years", "data_systems", "likely_data_assets"]
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _row(item):
|
|
17
|
+
r = item.get("result") or {}
|
|
18
|
+
h = r.get("history") or {}
|
|
19
|
+
return {
|
|
20
|
+
"input": item.get("input"), "domain": item.get("domain"), "status": item.get("status"),
|
|
21
|
+
"data_asset_score": r.get("data_asset_score"), "grade": r.get("grade"), "company_status": r.get("status"),
|
|
22
|
+
"verified_active": r.get("verified_active"), "iab_category": r.get("iab_category"), "country": r.get("country"),
|
|
23
|
+
"years_online": h.get("years_online"), "pre_ai_years": h.get("pre_ai_years"),
|
|
24
|
+
"data_systems": "; ".join(s["system"] for s in r.get("data_systems") or []),
|
|
25
|
+
"likely_data_assets": "; ".join(a["label"] for a in r.get("likely_data_assets") or []),
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def main(argv):
|
|
30
|
+
key = os.environ.get("SDA_API_KEY")
|
|
31
|
+
if not key or len(argv) < 1:
|
|
32
|
+
print(__doc__)
|
|
33
|
+
return 2
|
|
34
|
+
c = DataAssetScoreClient(key)
|
|
35
|
+
try:
|
|
36
|
+
if argv[0] == "score" and len(argv) > 1:
|
|
37
|
+
print(json.dumps(c.score(argv[1]), indent=2))
|
|
38
|
+
elif argv[0] == "usage":
|
|
39
|
+
print(json.dumps(c.usage(), indent=2))
|
|
40
|
+
elif argv[0] == "file" and len(argv) > 2:
|
|
41
|
+
with open(argv[1], encoding="utf-8") as f:
|
|
42
|
+
domains = [l.strip() for l in f if l.strip() and not l.startswith("#")]
|
|
43
|
+
items = c.score_many(domains, on_batch=lambda b: print("batch %s done" % b["id"], file=sys.stderr))
|
|
44
|
+
with open(argv[2], "w", newline="", encoding="utf-8") as f:
|
|
45
|
+
w = csv.DictWriter(f, fieldnames=FIELDS)
|
|
46
|
+
w.writeheader()
|
|
47
|
+
for it in items:
|
|
48
|
+
w.writerow(_row(it))
|
|
49
|
+
print("wrote %d rows to %s" % (len(items), argv[2]))
|
|
50
|
+
else:
|
|
51
|
+
print(__doc__)
|
|
52
|
+
return 2
|
|
53
|
+
except SellDataToAIError as e:
|
|
54
|
+
print("error %s %s: %s" % (e.status, e.code, e), file=sys.stderr)
|
|
55
|
+
return 1
|
|
56
|
+
return 0
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
if __name__ == "__main__":
|
|
60
|
+
sys.exit(main(sys.argv[1:]))
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _entry():
|
|
64
|
+
sys.exit(main(sys.argv[1:]))
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Client for the Data Asset Score API at https://www.selldatatoai.com/api/"""
|
|
2
|
+
import time
|
|
3
|
+
|
|
4
|
+
import requests
|
|
5
|
+
|
|
6
|
+
__all__ = ["DataAssetScoreClient", "SellDataToAIError", "BATCH_MAX"]
|
|
7
|
+
|
|
8
|
+
VERSION = "1.0.0"
|
|
9
|
+
DEFAULT_BASE = "https://www.selldatatoai.com/api/v1"
|
|
10
|
+
USER_AGENT = "selldatatoai-python/%s (+https://www.selldatatoai.com)" % VERSION
|
|
11
|
+
BATCH_MAX = 100
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class SellDataToAIError(Exception):
|
|
15
|
+
"""Raised for any non-2xx answer. `status` is the HTTP code, `code` the API error string."""
|
|
16
|
+
|
|
17
|
+
def __init__(self, status, code=None, message=None, body=None):
|
|
18
|
+
super().__init__(message or code or "HTTP %s" % status)
|
|
19
|
+
self.status = status
|
|
20
|
+
self.code = code
|
|
21
|
+
self.body = body
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class DataAssetScoreClient:
|
|
25
|
+
def __init__(self, api_key, base_url=DEFAULT_BASE, timeout=60, session=None):
|
|
26
|
+
if not api_key:
|
|
27
|
+
raise ValueError("An API key is required. Plans: https://www.selldatatoai.com/pricing/")
|
|
28
|
+
self.base_url = base_url.rstrip("/")
|
|
29
|
+
self.timeout = timeout
|
|
30
|
+
self.session = session or requests.Session()
|
|
31
|
+
self.session.headers.update({"X-API-Key": api_key, "User-Agent": USER_AGENT, "Accept": "application/json"})
|
|
32
|
+
|
|
33
|
+
def _request(self, method, path, params=None, json=None):
|
|
34
|
+
r = self.session.request(method, self.base_url + path, params=params, json=json, timeout=self.timeout)
|
|
35
|
+
try:
|
|
36
|
+
data = r.json()
|
|
37
|
+
except ValueError:
|
|
38
|
+
data = None
|
|
39
|
+
if 200 <= r.status_code < 300 and data is not None:
|
|
40
|
+
return data
|
|
41
|
+
code = data.get("error") if isinstance(data, dict) else None
|
|
42
|
+
msg = data.get("message") if isinstance(data, dict) else None
|
|
43
|
+
raise SellDataToAIError(r.status_code, code, msg or "HTTP %s" % r.status_code, data if data is not None else r.text)
|
|
44
|
+
|
|
45
|
+
def score(self, domain):
|
|
46
|
+
"""Score one company domain. Full URLs are accepted."""
|
|
47
|
+
if not domain:
|
|
48
|
+
raise ValueError("domain is required")
|
|
49
|
+
return self._request("GET", "/score", params={"domain": str(domain).strip()})
|
|
50
|
+
|
|
51
|
+
def submit_batch(self, domains):
|
|
52
|
+
"""Submit up to 100 domains; returns the batch dict (id, status, results...)."""
|
|
53
|
+
domains = list(domains)
|
|
54
|
+
if not domains:
|
|
55
|
+
raise ValueError("domains must not be empty")
|
|
56
|
+
if len(domains) > BATCH_MAX:
|
|
57
|
+
raise ValueError("a batch takes at most %d domains" % BATCH_MAX)
|
|
58
|
+
return self._request("POST", "/score/batch", json={"domains": domains})
|
|
59
|
+
|
|
60
|
+
def get_batch(self, batch_id):
|
|
61
|
+
"""Poll a batch. Polling does not use lookups."""
|
|
62
|
+
return self._request("GET", "/score/batch", params={"id": batch_id})
|
|
63
|
+
|
|
64
|
+
def wait_for_batch(self, batch_id, interval=5, max_wait=300):
|
|
65
|
+
start = time.time()
|
|
66
|
+
while True:
|
|
67
|
+
b = self.get_batch(batch_id)
|
|
68
|
+
if b.get("status") == "done":
|
|
69
|
+
return b
|
|
70
|
+
if time.time() - start > max_wait:
|
|
71
|
+
raise SellDataToAIError(0, "batch_timeout", "batch %s still processing after %ss" % (batch_id, max_wait), b)
|
|
72
|
+
time.sleep(interval)
|
|
73
|
+
|
|
74
|
+
def score_many(self, domains, interval=5, max_wait=300, on_batch=None):
|
|
75
|
+
"""Score any number of domains in batches of 100; returns result items in input order."""
|
|
76
|
+
domains = list(domains)
|
|
77
|
+
out = []
|
|
78
|
+
for i in range(0, len(domains), BATCH_MAX):
|
|
79
|
+
sent = self.submit_batch(domains[i:i + BATCH_MAX])
|
|
80
|
+
done = sent if sent.get("status") == "done" else self.wait_for_batch(sent["id"], interval, max_wait)
|
|
81
|
+
out.extend(done["results"])
|
|
82
|
+
if on_batch:
|
|
83
|
+
on_batch(done)
|
|
84
|
+
return out
|
|
85
|
+
|
|
86
|
+
def usage(self):
|
|
87
|
+
"""Plan, lookups used this month and lookups remaining."""
|
|
88
|
+
return self._request("GET", "/usage")
|
|
@@ -0,0 +1,317 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: selldatatoai
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Python client for the Data Asset Score API from selldatatoai.com: score company domains 0 to 100 for the data AI buyers want, one at a time or in batches of 100.
|
|
5
|
+
Home-page: https://www.selldatatoai.com
|
|
6
|
+
Author: Alpha Quantum
|
|
7
|
+
Author-email: info@alpha-quantum.com
|
|
8
|
+
License: MIT
|
|
9
|
+
Project-URL: Homepage, https://www.selldatatoai.com
|
|
10
|
+
Project-URL: Documentation, https://www.selldatatoai.com/api/
|
|
11
|
+
Project-URL: Source, https://github.com/explainableaixai/selldatatoai-python
|
|
12
|
+
Project-URL: Pricing, https://www.selldatatoai.com/pricing/
|
|
13
|
+
Project-URL: Demo, https://www.selldatatoai.com/data-asset-score/
|
|
14
|
+
Keywords: sell data to ai,data asset score,ai training data,ai data broker,company data api,lead scoring,firmographics,technographics,data licensing,b2b data,domain intelligence
|
|
15
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
16
|
+
Classifier: Intended Audience :: Developers
|
|
17
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
18
|
+
Classifier: Operating System :: OS Independent
|
|
19
|
+
Classifier: Programming Language :: Python :: 3
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.7
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
23
|
+
Classifier: Topic :: Office/Business
|
|
24
|
+
Requires-Python: >=3.7
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
License-File: LICENSE
|
|
27
|
+
Requires-Dist: requests>=2.20
|
|
28
|
+
|
|
29
|
+
# selldatatoai for Python
|
|
30
|
+
|
|
31
|
+
`selldatatoai` scores companies for AI data deals from Python. You pass a website; the API answers with a 0 to 100 [Data Asset Score](https://www.selldatatoai.com/data-asset-score/), a grade, the data the company likely holds and how long it has been online.
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
pip install selldatatoai
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Requires Python 3.7 or newer and `requests`. The package also installs a `selldatatoai` command for scoring a text file of domains into a CSV.
|
|
38
|
+
|
|
39
|
+
The service behind it is [selldatatoai.com](https://www.selldatatoai.com/), which keeps an index of 102 million domains, 99.99% of the active internet, together with each domain's history.
|
|
40
|
+
|
|
41
|
+
## Who this is for
|
|
42
|
+
|
|
43
|
+
- **Data brokers** who source data partners for AI labs and need to know which companies are worth a call.
|
|
44
|
+
- **Data companies** that resell records and want to rank prospects by the data they hold.
|
|
45
|
+
- **Referral partners** in data programs who screen companies before they introduce them.
|
|
46
|
+
- **Analysts** who study which sectors hold the operational records that AI training buys.
|
|
47
|
+
|
|
48
|
+
If you are new to the trade itself, the step-by-step guide on [how to sell data to AI companies](https://www.selldatatoai.com/how-to-sell-data-to-ai-companies/) covers the deal from first contact to delivery.
|
|
49
|
+
|
|
50
|
+
## Five lines to a score
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
import os
|
|
54
|
+
from selldatatoai import DataAssetScoreClient
|
|
55
|
+
|
|
56
|
+
client = DataAssetScoreClient(os.environ["SDA_API_KEY"])
|
|
57
|
+
s = client.score("example.com")
|
|
58
|
+
print(s["data_asset_score"], s["grade"], [a["label"] for a in s["likely_data_assets"]])
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
The result is a plain `dict` with the API field names. A trimmed real answer looks like this:
|
|
62
|
+
|
|
63
|
+
```json
|
|
64
|
+
{
|
|
65
|
+
"domain": "promega.com",
|
|
66
|
+
"data_asset_score": 88,
|
|
67
|
+
"grade": "A",
|
|
68
|
+
"status": "active",
|
|
69
|
+
"verified_active": true,
|
|
70
|
+
"iab_category": "Business and Finance > Industries > Pharmaceutical Industry",
|
|
71
|
+
"country": "United States",
|
|
72
|
+
"history": {"first_seen_year": 1993, "years_online": 33, "founded_year": null, "pre_ai_years": 30},
|
|
73
|
+
"pre_ai_archive_likely": true,
|
|
74
|
+
"data_systems": [
|
|
75
|
+
{"system": "Jira / Confluence", "data_type": "Work tickets and internal wiki", "type": "work_tickets"},
|
|
76
|
+
{"system": "Webex", "data_type": "Calls and meetings", "type": "calls"}
|
|
77
|
+
],
|
|
78
|
+
"likely_data_assets": [
|
|
79
|
+
{"type": "support_tickets", "label": "Support tickets and chat transcripts"},
|
|
80
|
+
{"type": "knowledge_base", "label": "Knowledge base, SOPs and documentation"}
|
|
81
|
+
],
|
|
82
|
+
"data_coverage": "full",
|
|
83
|
+
"cached": true
|
|
84
|
+
}
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## API surface
|
|
88
|
+
|
|
89
|
+
| Python call | Endpoint | Cost |
|
|
90
|
+
|---|---|---|
|
|
91
|
+
| `client.score(domain)` | `GET /api/v1/score` | 1 lookup |
|
|
92
|
+
| `client.submit_batch(domains)` | `POST /api/v1/score/batch` | 1 lookup per valid, unique domain |
|
|
93
|
+
| `client.get_batch(batch_id)` | `GET /api/v1/score/batch?id=` | free |
|
|
94
|
+
| `client.wait_for_batch(batch_id, interval=5, max_wait=300)` | polls `get_batch` | free |
|
|
95
|
+
| `client.score_many(domains, interval=5, max_wait=300, on_batch=None)` | batches of 100 | 1 lookup per valid, unique domain |
|
|
96
|
+
| `client.usage()` | `GET /api/v1/usage` | free |
|
|
97
|
+
|
|
98
|
+
Constructor:
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
DataAssetScoreClient(api_key, base_url="https://www.selldatatoai.com/api/v1", timeout=60, session=None)
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Pass your own `requests.Session` if you need retries, a proxy or connection pooling shared with other code. The key travels in the `X-API-Key` header only.
|
|
105
|
+
|
|
106
|
+
Every field is described in the [REST API reference](https://www.selldatatoai.com/api/).
|
|
107
|
+
|
|
108
|
+
## Handling failures
|
|
109
|
+
|
|
110
|
+
All HTTP errors raise `SellDataToAIError`. It carries `.status` (HTTP code), `.code` (the API error string) and `.body` (the decoded answer).
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
from selldatatoai import SellDataToAIError
|
|
114
|
+
|
|
115
|
+
def safe_score(client, domain):
|
|
116
|
+
try:
|
|
117
|
+
return client.score(domain)
|
|
118
|
+
except SellDataToAIError as e:
|
|
119
|
+
if e.code == "invalid_domain":
|
|
120
|
+
return None # bad input, skip it
|
|
121
|
+
if e.code == "monthly_limit_reached":
|
|
122
|
+
raise SystemExit("Out of lookups until the 1st (UTC)")
|
|
123
|
+
raise # 401, network trouble, anything else
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
Codes you can meet:
|
|
127
|
+
|
|
128
|
+
- `missing_api_key`, `invalid_api_key` (401): no key, or the plan is not active.
|
|
129
|
+
- `invalid_domain` (400): not a valid domain.
|
|
130
|
+
- `no_domains`, `too_many_domains` (400): a batch needs 1 to 100 entries.
|
|
131
|
+
- `monthly_limit_reached` (429): limits reset on the first day of each month, UTC.
|
|
132
|
+
- `batch_busy` (429): three of your batches are still open.
|
|
133
|
+
- `batch_not_found` (404): unknown id, or older than 7 days.
|
|
134
|
+
- `not_available` (410): company lists are not served by the API.
|
|
135
|
+
|
|
136
|
+
## Recipes
|
|
137
|
+
|
|
138
|
+
### Recipe 1: score a spreadsheet with pandas
|
|
139
|
+
|
|
140
|
+
```python
|
|
141
|
+
import os
|
|
142
|
+
import pandas as pd
|
|
143
|
+
from selldatatoai import DataAssetScoreClient
|
|
144
|
+
|
|
145
|
+
client = DataAssetScoreClient(os.environ["SDA_API_KEY"])
|
|
146
|
+
df = pd.read_csv("prospects.csv") # needs a 'website' column
|
|
147
|
+
|
|
148
|
+
items = client.score_many(df["website"].dropna().tolist())
|
|
149
|
+
rows = []
|
|
150
|
+
for it in items:
|
|
151
|
+
r = it.get("result") or {}
|
|
152
|
+
rows.append({
|
|
153
|
+
"website": it["input"],
|
|
154
|
+
"score": r.get("data_asset_score"),
|
|
155
|
+
"grade": r.get("grade"),
|
|
156
|
+
"status": r.get("status", it["status"]),
|
|
157
|
+
"pre_ai_years": (r.get("history") or {}).get("pre_ai_years"),
|
|
158
|
+
"systems": ", ".join(s["system"] for s in r.get("data_systems", [])),
|
|
159
|
+
})
|
|
160
|
+
|
|
161
|
+
scored = df.merge(pd.DataFrame(rows), on="website", how="left")
|
|
162
|
+
scored.sort_values("score", ascending=False).to_csv("prospects_scored.csv", index=False)
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
### Recipe 2: a FastAPI route for a partner form
|
|
166
|
+
|
|
167
|
+
```python
|
|
168
|
+
import os
|
|
169
|
+
from fastapi import FastAPI, HTTPException
|
|
170
|
+
from pydantic import BaseModel
|
|
171
|
+
from selldatatoai import DataAssetScoreClient, SellDataToAIError
|
|
172
|
+
|
|
173
|
+
app = FastAPI()
|
|
174
|
+
sda = DataAssetScoreClient(os.environ["SDA_API_KEY"])
|
|
175
|
+
|
|
176
|
+
class Signup(BaseModel):
|
|
177
|
+
company: str
|
|
178
|
+
website: str
|
|
179
|
+
|
|
180
|
+
@app.post("/signup")
|
|
181
|
+
def signup(body: Signup):
|
|
182
|
+
try:
|
|
183
|
+
s = sda.score(body.website)
|
|
184
|
+
except SellDataToAIError as e:
|
|
185
|
+
if e.status == 400:
|
|
186
|
+
raise HTTPException(422, "That website does not look valid")
|
|
187
|
+
raise HTTPException(503, "Scoring is unavailable, try again shortly")
|
|
188
|
+
tier = "priority" if s["grade"] in ("A", "B") and s["verified_active"] else "standard"
|
|
189
|
+
return {"company": body.company, "tier": tier, "score": s["data_asset_score"]}
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
The client is synchronous. Inside an async framework, FastAPI runs plain `def` routes in a thread pool, which is what you want here.
|
|
193
|
+
|
|
194
|
+
### Recipe 3: parallel single lookups with a thread pool
|
|
195
|
+
|
|
196
|
+
Batches are the better tool for big lists. For a few dozen domains where you want each answer as soon as it is ready, a small pool works well:
|
|
197
|
+
|
|
198
|
+
```python
|
|
199
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
200
|
+
|
|
201
|
+
domains = ["example-one.com", "example-two.com", "example-three.com"]
|
|
202
|
+
with ThreadPoolExecutor(max_workers=4) as pool:
|
|
203
|
+
futures = {pool.submit(client.score, d): d for d in domains}
|
|
204
|
+
for f in as_completed(futures):
|
|
205
|
+
d = futures[f]
|
|
206
|
+
try:
|
|
207
|
+
print(d, f.result()["data_asset_score"])
|
|
208
|
+
except SellDataToAIError as e:
|
|
209
|
+
print(d, "failed:", e.code)
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
Keep the pool small. Each worker is one open request, and every call still counts against your monthly lookups.
|
|
213
|
+
|
|
214
|
+
### Recipe 4: the command line
|
|
215
|
+
|
|
216
|
+
```bash
|
|
217
|
+
export SDA_API_KEY=xxxx
|
|
218
|
+
selldatatoai score example.com # JSON for one company
|
|
219
|
+
selldatatoai usage # plan and remaining lookups
|
|
220
|
+
selldatatoai file domains.txt out.csv # one domain per line, '#' lines skipped
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
`python -m selldatatoai` works the same way if the script directory is not on your `PATH`. The CSV has one row per input line, in input order, with score, grade, company status, years online, pre-AI years, systems and likely data assets.
|
|
224
|
+
|
|
225
|
+
## How batches behave
|
|
226
|
+
|
|
227
|
+
1. You send up to 100 domains.
|
|
228
|
+
2. Lookups are charged right away: one per valid, unique domain.
|
|
229
|
+
3. Domains scored in the last 30 days are `done` in the first answer.
|
|
230
|
+
4. The rest finish in the background, usually within a minute.
|
|
231
|
+
5. You poll with the batch id. Polling is free.
|
|
232
|
+
6. Results stay in the order you sent and are kept for 7 days.
|
|
233
|
+
|
|
234
|
+
You can hold three open batches per key. `score_many` sends them one after another, so it never trips that limit.
|
|
235
|
+
|
|
236
|
+
## Making sense of the numbers
|
|
237
|
+
|
|
238
|
+
The score gathers several factor groups into one number. The [how the Data Asset Score works](https://www.selldatatoai.com/how-the-data-asset-score-works/) page describes the data behind each group. The exact weights are not published.
|
|
239
|
+
|
|
240
|
+
A few reading tips:
|
|
241
|
+
|
|
242
|
+
- **Grade first, score second.** Two companies at 71 and 74 are in the same band. A and B grades are where most buyers start.
|
|
243
|
+
- **Check `status`.** A high score on a `winding_down_or_acquired` company means the records may still exist, but the seller has changed.
|
|
244
|
+
- **Look at `pre_ai_years`.** Records written before 2023 are free of AI-generated text, which some buyers value highly.
|
|
245
|
+
- **Read `data_coverage`.** `limited` means fewer signals were available, so the score is less certain.
|
|
246
|
+
|
|
247
|
+
Sector context helps too. Life sciences companies tend to hold lab, study and R&D records, which is why buyers of [AI training data companies](https://www.selldatatoai.com/ai-training-data-companies/) research often start with that sector. A ready file of the top 3,000 US firms in that space is described on the [list of biotech companies](https://www.selldatatoai.com/list-of-biotech-companies/) page.
|
|
248
|
+
|
|
249
|
+
## Grouping companies by the data they hold
|
|
250
|
+
|
|
251
|
+
Buyers rarely ask for "data". They ask for a type: support conversations, engineering tickets, recorded calls, contracts, design files. The `type` keys in `data_systems` and `likely_data_assets` let you build those groups in a few lines.
|
|
252
|
+
|
|
253
|
+
```python
|
|
254
|
+
from collections import defaultdict
|
|
255
|
+
|
|
256
|
+
by_type = defaultdict(list)
|
|
257
|
+
for it in items: # items from score_many()
|
|
258
|
+
r = it.get("result")
|
|
259
|
+
if not r or not r["verified_active"]:
|
|
260
|
+
continue
|
|
261
|
+
for a in r["likely_data_assets"]:
|
|
262
|
+
by_type[a["type"]].append((r["data_asset_score"], r["domain"]))
|
|
263
|
+
|
|
264
|
+
for t, rows in sorted(by_type.items()):
|
|
265
|
+
top = sorted(rows, reverse=True)[:10]
|
|
266
|
+
print(t, len(rows), "companies, top:", ", ".join(d for _, d in top))
|
|
267
|
+
```
|
|
268
|
+
|
|
269
|
+
Two keys deserve a note:
|
|
270
|
+
|
|
271
|
+
- `data_systems` lists systems the company is seen to run, each mapped to the data type it stores. It is the stronger signal, because a running system means the records exist and can be exported.
|
|
272
|
+
- `likely_data_assets` lists what the company probably holds based on its sector and footprint. Use it to widen a search, not to promise a buyer anything.
|
|
273
|
+
|
|
274
|
+
When a buyer asks for one data type across a whole market, the website also sells ready lists by data type next to the sector lists.
|
|
275
|
+
|
|
276
|
+
## Plans
|
|
277
|
+
|
|
278
|
+
Paid plans only, from $99 a month: Basic $99 for 5,000 lookups, Pro $299 for 25,000 with batch scoring, Scale $799 for 100,000 with batch scoring. The website demo allows 5 checks a day if you want to see the output before you subscribe. Your key appears in your dashboard as soon as the payment completes.
|
|
279
|
+
|
|
280
|
+
Company lists are not part of any API plan. They are one-time files of the top 3,000 US companies per sector, from $249 a list, delivered by email links that work for 30 days.
|
|
281
|
+
|
|
282
|
+
## FAQ
|
|
283
|
+
|
|
284
|
+
### What does the selldatatoai package do?
|
|
285
|
+
|
|
286
|
+
The selldatatoai package is the Python client for the Data Asset Score API at selldatatoai.com. It scores a company website from 0 to 100 for the data AI buyers want and returns the grade, status, data systems, likely data assets and web history.
|
|
287
|
+
|
|
288
|
+
### Does it support asyncio?
|
|
289
|
+
|
|
290
|
+
Not directly. Run calls in a thread pool (`asyncio.to_thread` on Python 3.9+) or use the batch endpoint, which does the parallel work on the server.
|
|
291
|
+
|
|
292
|
+
### What counts as a lookup?
|
|
293
|
+
|
|
294
|
+
Each scored domain, fresh or cached. A batch counts each valid, unique domain once. Usage and polling calls are free.
|
|
295
|
+
|
|
296
|
+
### Can I pass full URLs?
|
|
297
|
+
|
|
298
|
+
Yes. `https://www.example.com/contact` and `example.com` score the same company.
|
|
299
|
+
|
|
300
|
+
### Are results stored?
|
|
301
|
+
|
|
302
|
+
A result is reused for 30 days, then the domain is scored again. Batches are deleted after 7 days.
|
|
303
|
+
|
|
304
|
+
### Is there a sandbox key?
|
|
305
|
+
|
|
306
|
+
No. Plans are paid, and the demo on the website is the only free access.
|
|
307
|
+
|
|
308
|
+
## Links
|
|
309
|
+
|
|
310
|
+
- Homepage: https://www.selldatatoai.com/
|
|
311
|
+
- Source code: https://github.com/explainableaixai/selldatatoai-python
|
|
312
|
+
- Python packaging guide: [packaging.python.org](https://packaging.python.org/en/latest/tutorials/installing-packages/)
|
|
313
|
+
- Thread pools in the standard library: [docs.python.org](https://docs.python.org/3/library/concurrent.futures.html)
|
|
314
|
+
|
|
315
|
+
## License
|
|
316
|
+
|
|
317
|
+
MIT, Copyright (c) 2026 Alpha Quantum. Questions: info@alpha-quantum.com
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
MANIFEST.in
|
|
3
|
+
README.md
|
|
4
|
+
setup.py
|
|
5
|
+
selldatatoai/__init__.py
|
|
6
|
+
selldatatoai/__main__.py
|
|
7
|
+
selldatatoai/client.py
|
|
8
|
+
selldatatoai.egg-info/PKG-INFO
|
|
9
|
+
selldatatoai.egg-info/SOURCES.txt
|
|
10
|
+
selldatatoai.egg-info/dependency_links.txt
|
|
11
|
+
selldatatoai.egg-info/entry_points.txt
|
|
12
|
+
selldatatoai.egg-info/requires.txt
|
|
13
|
+
selldatatoai.egg-info/top_level.txt
|
|
14
|
+
tests/test_client.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
requests>=2.20
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
selldatatoai
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import pathlib
|
|
2
|
+
from setuptools import setup, find_packages
|
|
3
|
+
|
|
4
|
+
README = (pathlib.Path(__file__).parent / "README.md").read_text(encoding="utf-8")
|
|
5
|
+
|
|
6
|
+
setup(
|
|
7
|
+
name="selldatatoai",
|
|
8
|
+
version="1.0.0",
|
|
9
|
+
description="Python client for the Data Asset Score API from selldatatoai.com: score company domains 0 to 100 for the data AI buyers want, one at a time or in batches of 100.",
|
|
10
|
+
long_description=README,
|
|
11
|
+
long_description_content_type="text/markdown",
|
|
12
|
+
author="Alpha Quantum",
|
|
13
|
+
author_email="info@alpha-quantum.com",
|
|
14
|
+
url="https://www.selldatatoai.com",
|
|
15
|
+
project_urls={
|
|
16
|
+
"Homepage": "https://www.selldatatoai.com",
|
|
17
|
+
"Documentation": "https://www.selldatatoai.com/api/",
|
|
18
|
+
"Source": "https://github.com/explainableaixai/selldatatoai-python",
|
|
19
|
+
"Pricing": "https://www.selldatatoai.com/pricing/",
|
|
20
|
+
"Demo": "https://www.selldatatoai.com/data-asset-score/",
|
|
21
|
+
},
|
|
22
|
+
license="MIT",
|
|
23
|
+
packages=find_packages(exclude=("tests", "test")),
|
|
24
|
+
install_requires=["requests>=2.20"],
|
|
25
|
+
python_requires=">=3.7",
|
|
26
|
+
entry_points={"console_scripts": ["selldatatoai=selldatatoai.__main__:_entry"]},
|
|
27
|
+
keywords=["sell data to ai", "data asset score", "ai training data", "ai data broker", "company data api",
|
|
28
|
+
"lead scoring", "firmographics", "technographics", "data licensing", "b2b data", "domain intelligence"],
|
|
29
|
+
classifiers=[
|
|
30
|
+
"Development Status :: 5 - Production/Stable",
|
|
31
|
+
"Intended Audience :: Developers",
|
|
32
|
+
"License :: OSI Approved :: MIT License",
|
|
33
|
+
"Operating System :: OS Independent",
|
|
34
|
+
"Programming Language :: Python :: 3",
|
|
35
|
+
"Programming Language :: Python :: 3.7",
|
|
36
|
+
"Programming Language :: Python :: 3.12",
|
|
37
|
+
"Topic :: Internet :: WWW/HTTP",
|
|
38
|
+
"Topic :: Office/Business",
|
|
39
|
+
],
|
|
40
|
+
)
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import threading
|
|
3
|
+
import unittest
|
|
4
|
+
from http.server import BaseHTTPRequestHandler, HTTPServer
|
|
5
|
+
from urllib.parse import parse_qs, urlparse
|
|
6
|
+
|
|
7
|
+
from selldatatoai import DataAssetScoreClient, SellDataToAIError
|
|
8
|
+
|
|
9
|
+
RESULT = {"domain": "example-labs.com", "data_asset_score": 82, "grade": "A", "status": "active"}
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Stub(BaseHTTPRequestHandler):
|
|
13
|
+
polls = 0
|
|
14
|
+
|
|
15
|
+
def log_message(self, *a):
|
|
16
|
+
pass
|
|
17
|
+
|
|
18
|
+
def _send(self, code, obj):
|
|
19
|
+
b = json.dumps(obj).encode()
|
|
20
|
+
self.send_response(code)
|
|
21
|
+
self.send_header("Content-Type", "application/json")
|
|
22
|
+
self.send_header("Content-Length", str(len(b)))
|
|
23
|
+
self.end_headers()
|
|
24
|
+
self.wfile.write(b)
|
|
25
|
+
|
|
26
|
+
def do_GET(self):
|
|
27
|
+
u = urlparse(self.path)
|
|
28
|
+
q = parse_qs(u.query)
|
|
29
|
+
if self.headers.get("X-API-Key") != "k":
|
|
30
|
+
return self._send(401, {"error": "invalid_api_key"})
|
|
31
|
+
if u.path == "/v1/score":
|
|
32
|
+
if q.get("domain") == ["bad"]:
|
|
33
|
+
return self._send(400, {"error": "invalid_domain", "message": "Pass a domain"})
|
|
34
|
+
return self._send(200, RESULT)
|
|
35
|
+
if u.path == "/v1/usage":
|
|
36
|
+
return self._send(200, {"plan": "pro", "remaining": 7})
|
|
37
|
+
if u.path == "/v1/score/batch":
|
|
38
|
+
Stub.polls += 1
|
|
39
|
+
if Stub.polls < 2:
|
|
40
|
+
return self._send(200, {"id": "b_1", "status": "processing", "results": []})
|
|
41
|
+
return self._send(200, {"id": "b_1", "status": "done", "results": [{"input": "a.com", "status": "done", "result": RESULT}]})
|
|
42
|
+
self._send(404, {"error": "unknown_endpoint"})
|
|
43
|
+
|
|
44
|
+
def do_POST(self):
|
|
45
|
+
n = int(self.headers.get("Content-Length", 0))
|
|
46
|
+
body = json.loads(self.rfile.read(n))
|
|
47
|
+
self._send(202, {"id": "b_1", "status": "processing", "total": len(body["domains"]), "results": []})
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class ClientTest(unittest.TestCase):
|
|
51
|
+
@classmethod
|
|
52
|
+
def setUpClass(cls):
|
|
53
|
+
cls.srv = HTTPServer(("127.0.0.1", 0), Stub)
|
|
54
|
+
threading.Thread(target=cls.srv.serve_forever, daemon=True).start()
|
|
55
|
+
cls.base = "http://127.0.0.1:%d/v1" % cls.srv.server_port
|
|
56
|
+
|
|
57
|
+
@classmethod
|
|
58
|
+
def tearDownClass(cls):
|
|
59
|
+
cls.srv.shutdown()
|
|
60
|
+
|
|
61
|
+
def test_score_usage(self):
|
|
62
|
+
c = DataAssetScoreClient("k", base_url=self.base)
|
|
63
|
+
self.assertEqual(c.score("example-labs.com")["grade"], "A")
|
|
64
|
+
self.assertEqual(c.usage()["remaining"], 7)
|
|
65
|
+
|
|
66
|
+
def test_errors(self):
|
|
67
|
+
c = DataAssetScoreClient("k", base_url=self.base)
|
|
68
|
+
with self.assertRaises(SellDataToAIError) as e:
|
|
69
|
+
c.score("bad")
|
|
70
|
+
self.assertEqual((e.exception.status, e.exception.code), (400, "invalid_domain"))
|
|
71
|
+
with self.assertRaises(SellDataToAIError) as e:
|
|
72
|
+
DataAssetScoreClient("x", base_url=self.base).usage()
|
|
73
|
+
self.assertEqual(e.exception.status, 401)
|
|
74
|
+
|
|
75
|
+
def test_score_many(self):
|
|
76
|
+
c = DataAssetScoreClient("k", base_url=self.base)
|
|
77
|
+
items = c.score_many(["a.com"], interval=0.01)
|
|
78
|
+
self.assertEqual(items[0]["result"]["data_asset_score"], 82)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
if __name__ == "__main__":
|
|
82
|
+
unittest.main()
|