cookielessaudiences 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cookielessaudiences-1.0.0/LICENSE +21 -0
- cookielessaudiences-1.0.0/MANIFEST.in +2 -0
- cookielessaudiences-1.0.0/PKG-INFO +155 -0
- cookielessaudiences-1.0.0/README.md +128 -0
- cookielessaudiences-1.0.0/cookielessaudiences/__init__.py +4 -0
- cookielessaudiences-1.0.0/cookielessaudiences/client.py +103 -0
- cookielessaudiences-1.0.0/cookielessaudiences.egg-info/PKG-INFO +155 -0
- cookielessaudiences-1.0.0/cookielessaudiences.egg-info/SOURCES.txt +11 -0
- cookielessaudiences-1.0.0/cookielessaudiences.egg-info/dependency_links.txt +1 -0
- cookielessaudiences-1.0.0/cookielessaudiences.egg-info/requires.txt +1 -0
- cookielessaudiences-1.0.0/cookielessaudiences.egg-info/top_level.txt +1 -0
- cookielessaudiences-1.0.0/setup.cfg +4 -0
- cookielessaudiences-1.0.0/setup.py +40 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Alpha Quantum
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: cookielessaudiences
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Python client for the Cookieless Audiences API: page-level audience segmentation (demographics, interests, purchase intent, B2B firmographics, personas) and IAB content categorization for any URL, with no cookies and no PII.
|
|
5
|
+
Home-page: https://www.cookielessaudiences.com
|
|
6
|
+
Author: Alpha Quantum
|
|
7
|
+
Author-email: info@alpha-quantum.com
|
|
8
|
+
License: MIT
|
|
9
|
+
Project-URL: Homepage, https://www.cookielessaudiences.com
|
|
10
|
+
Project-URL: Documentation, https://www.cookielessaudiences.com/api.php
|
|
11
|
+
Project-URL: Source, https://github.com/explainableaixai/cookielessaudiences
|
|
12
|
+
Project-URL: Mirror, https://gitlab.com/url-classifications/cookielessaudiences
|
|
13
|
+
Project-URL: Tracker, https://www.cookielessaudiences.com/contact.php
|
|
14
|
+
Project-URL: Pricing, https://www.cookielessaudiences.com/pricing.php
|
|
15
|
+
Description: # cookielessaudiences (Python)
|
|
16
|
+
|
|
17
|
+
Cookieless audience data for a URL, as a Python call. You pass a page and receive demographics, interests, purchase intent, B2B firmographics and personas, with no cookies and no personal data involved.
|
|
18
|
+
|
|
19
|
+
Built on `requests`, works on Python 3.7 and later.
|
|
20
|
+
|
|
21
|
+
## Setup
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install cookielessaudiences
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
An API key comes with any plan. If you are still comparing options, read how [cookieless audience segmentation](https://www.cookielessaudiences.com/features/cookieless-audience-segmentation.php) works before you buy.
|
|
28
|
+
|
|
29
|
+
## Quick start
|
|
30
|
+
|
|
31
|
+
```python
|
|
32
|
+
from cookielessaudiences import CookielessAudiences, labels_for
|
|
33
|
+
|
|
34
|
+
client = CookielessAudiences("YOUR_API_KEY")
|
|
35
|
+
result = client.segment("https://example.com/blog")
|
|
36
|
+
|
|
37
|
+
print(result["audience_type"])
|
|
38
|
+
print(result["demographics"]["income_level"])
|
|
39
|
+
print(labels_for(result))
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Reading the response
|
|
43
|
+
|
|
44
|
+
Think of the structured result as five questions about the reader of a page:
|
|
45
|
+
|
|
46
|
+
1. Who are they? `demographics` and `b2b`
|
|
47
|
+
2. What do they like? `interests`, two tiers of `INT.*` codes
|
|
48
|
+
3. What might they buy? `purchase_intent`, `PI.*` codes
|
|
49
|
+
4. Which named audiences fit? `personas`
|
|
50
|
+
5. What kind of page is it? `content_context`
|
|
51
|
+
|
|
52
|
+
`labels_for(result)` turns the codes in questions 2 and 3 into names you can show to a person.
|
|
53
|
+
|
|
54
|
+
## API surface
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
client.segment(url, structured=True)
|
|
58
|
+
client.categorize(url, confidence=True, root_fallback=False)
|
|
59
|
+
client.categorize_text(text, confidence=True)
|
|
60
|
+
client.segment_many(urls, workers=8)
|
|
61
|
+
CookielessAudiences.vocabularies()
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
`segment_many` returns a dict of url to result. A URL that failed holds its exception instead, so one bad page never stops a batch.
|
|
65
|
+
|
|
66
|
+
## Errors
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
from cookielessaudiences import CookielessAudiencesError
|
|
70
|
+
|
|
71
|
+
try:
|
|
72
|
+
client.segment(url)
|
|
73
|
+
except CookielessAudiencesError as exc:
|
|
74
|
+
print(exc.status, exc.body)
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
The HTTP layer is plain; the signal lives in `status` inside the JSON. 401 is a bad key, 403 means the key is inactive or credits are gone, 410 and 411 mean the page could not be read.
|
|
78
|
+
|
|
79
|
+
## Example: audience report for a media plan
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
import csv
|
|
83
|
+
from cookielessaudiences import CookielessAudiences, labels_for
|
|
84
|
+
|
|
85
|
+
client = CookielessAudiences("YOUR_API_KEY")
|
|
86
|
+
sites = [line.strip() for line in open("sites.txt") if line.strip()]
|
|
87
|
+
results = client.segment_many(sites, workers=10)
|
|
88
|
+
|
|
89
|
+
with open("plan.csv", "w", newline="") as fh:
|
|
90
|
+
out = csv.writer(fh)
|
|
91
|
+
out.writerow(["site", "type", "age", "income", "top_interest", "top_intent"])
|
|
92
|
+
for site, res in results.items():
|
|
93
|
+
if isinstance(res, Exception):
|
|
94
|
+
out.writerow([site, "error", "", "", "", ""])
|
|
95
|
+
continue
|
|
96
|
+
names = labels_for(res)
|
|
97
|
+
out.writerow([
|
|
98
|
+
site,
|
|
99
|
+
res.get("audience_type"),
|
|
100
|
+
"|".join(res["demographics"].get("age_bracket", [])),
|
|
101
|
+
res["demographics"].get("income_level"),
|
|
102
|
+
(names["interests"] or [""])[0],
|
|
103
|
+
(names["purchase_intent"] or [""])[0],
|
|
104
|
+
])
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
This is the shape most [website audience demographics](https://www.cookielessaudiences.com/features/website-audience-demographics.php) work takes: many sites in, one row per site out.
|
|
108
|
+
|
|
109
|
+
## Example: pandas enrichment
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
import pandas as pd
|
|
113
|
+
|
|
114
|
+
df = pd.read_csv("accounts.csv") # column: url
|
|
115
|
+
found = client.segment_many(df["url"].tolist())
|
|
116
|
+
df["audience_type"] = [r.get("audience_type") if isinstance(r, dict) else None for r in found.values()]
|
|
117
|
+
df["seniority"] = [
|
|
118
|
+
",".join(r.get("b2b", {}).get("seniority", [])) if isinstance(r, dict) else None
|
|
119
|
+
for r in found.values()
|
|
120
|
+
]
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
## Example: IAB categories for a text snippet
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
snippet = "Best budget laptops for students heading to university"
|
|
127
|
+
cats = client.categorize_text(snippet)
|
|
128
|
+
for name, score in cats["iab_classification"]:
|
|
129
|
+
print(name, score)
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
## Frequently asked
|
|
133
|
+
|
|
134
|
+
**Can I get the full word list for a field?** Yes. `CookielessAudiences.vocabularies()` returns every enumerated value and needs no key.
|
|
135
|
+
|
|
136
|
+
**What about scale?** The service accepts up to 50 parallel threads by default, about 270 URLs a minute.
|
|
137
|
+
|
|
138
|
+
**Which fields are free text?** None in the structured shape. Small fields are closed lists; open-ended signals are mapped to canonical codes or dropped.
|
|
139
|
+
|
|
140
|
+
## Package notes
|
|
141
|
+
|
|
142
|
+
MIT licensed. Questions go to info@alpha-quantum.com. Version 1.0.0.
|
|
143
|
+
|
|
144
|
+
Keywords: cookieless,audience segmentation,iab audience taxonomy,iab categorization,contextual targeting,seller defined audiences,purchase intent,adtech,media planning,inventory curation,audience data,privacy-safe advertising,domain audience data
|
|
145
|
+
Platform: UNKNOWN
|
|
146
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
147
|
+
Classifier: Intended Audience :: Developers
|
|
148
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
149
|
+
Classifier: Programming Language :: Python :: 3
|
|
150
|
+
Classifier: Programming Language :: Python :: 3.7
|
|
151
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
152
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
153
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
154
|
+
Requires-Python: >=3.7
|
|
155
|
+
Description-Content-Type: text/markdown
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
# cookielessaudiences (Python)
|
|
2
|
+
|
|
3
|
+
Cookieless audience data for a URL, as a Python call. You pass a page and receive demographics, interests, purchase intent, B2B firmographics and personas, with no cookies and no personal data involved.
|
|
4
|
+
|
|
5
|
+
Built on `requests`, works on Python 3.7 and later.
|
|
6
|
+
|
|
7
|
+
## Setup
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install cookielessaudiences
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
An API key comes with any plan. If you are still comparing options, read how [cookieless audience segmentation](https://www.cookielessaudiences.com/features/cookieless-audience-segmentation.php) works before you buy.
|
|
14
|
+
|
|
15
|
+
## Quick start
|
|
16
|
+
|
|
17
|
+
```python
|
|
18
|
+
from cookielessaudiences import CookielessAudiences, labels_for
|
|
19
|
+
|
|
20
|
+
client = CookielessAudiences("YOUR_API_KEY")
|
|
21
|
+
result = client.segment("https://example.com/blog")
|
|
22
|
+
|
|
23
|
+
print(result["audience_type"])
|
|
24
|
+
print(result["demographics"]["income_level"])
|
|
25
|
+
print(labels_for(result))
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Reading the response
|
|
29
|
+
|
|
30
|
+
Think of the structured result as five questions about the reader of a page:
|
|
31
|
+
|
|
32
|
+
1. Who are they? `demographics` and `b2b`
|
|
33
|
+
2. What do they like? `interests`, two tiers of `INT.*` codes
|
|
34
|
+
3. What might they buy? `purchase_intent`, `PI.*` codes
|
|
35
|
+
4. Which named audiences fit? `personas`
|
|
36
|
+
5. What kind of page is it? `content_context`
|
|
37
|
+
|
|
38
|
+
`labels_for(result)` turns the codes in questions 2 and 3 into names you can show to a person.
|
|
39
|
+
|
|
40
|
+
## API surface
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
client.segment(url, structured=True)
|
|
44
|
+
client.categorize(url, confidence=True, root_fallback=False)
|
|
45
|
+
client.categorize_text(text, confidence=True)
|
|
46
|
+
client.segment_many(urls, workers=8)
|
|
47
|
+
CookielessAudiences.vocabularies()
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
`segment_many` returns a dict of url to result. A URL that failed holds its exception instead, so one bad page never stops a batch.
|
|
51
|
+
|
|
52
|
+
## Errors
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
from cookielessaudiences import CookielessAudiencesError
|
|
56
|
+
|
|
57
|
+
try:
|
|
58
|
+
client.segment(url)
|
|
59
|
+
except CookielessAudiencesError as exc:
|
|
60
|
+
print(exc.status, exc.body)
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
The HTTP layer is plain; the signal lives in `status` inside the JSON. 401 is a bad key, 403 means the key is inactive or credits are gone, 410 and 411 mean the page could not be read.
|
|
64
|
+
|
|
65
|
+
## Example: audience report for a media plan
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
import csv
|
|
69
|
+
from cookielessaudiences import CookielessAudiences, labels_for
|
|
70
|
+
|
|
71
|
+
client = CookielessAudiences("YOUR_API_KEY")
|
|
72
|
+
sites = [line.strip() for line in open("sites.txt") if line.strip()]
|
|
73
|
+
results = client.segment_many(sites, workers=10)
|
|
74
|
+
|
|
75
|
+
with open("plan.csv", "w", newline="") as fh:
|
|
76
|
+
out = csv.writer(fh)
|
|
77
|
+
out.writerow(["site", "type", "age", "income", "top_interest", "top_intent"])
|
|
78
|
+
for site, res in results.items():
|
|
79
|
+
if isinstance(res, Exception):
|
|
80
|
+
out.writerow([site, "error", "", "", "", ""])
|
|
81
|
+
continue
|
|
82
|
+
names = labels_for(res)
|
|
83
|
+
out.writerow([
|
|
84
|
+
site,
|
|
85
|
+
res.get("audience_type"),
|
|
86
|
+
"|".join(res["demographics"].get("age_bracket", [])),
|
|
87
|
+
res["demographics"].get("income_level"),
|
|
88
|
+
(names["interests"] or [""])[0],
|
|
89
|
+
(names["purchase_intent"] or [""])[0],
|
|
90
|
+
])
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
This is the shape most [website audience demographics](https://www.cookielessaudiences.com/features/website-audience-demographics.php) work takes: many sites in, one row per site out.
|
|
94
|
+
|
|
95
|
+
## Example: pandas enrichment
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
import pandas as pd
|
|
99
|
+
|
|
100
|
+
df = pd.read_csv("accounts.csv") # column: url
|
|
101
|
+
found = client.segment_many(df["url"].tolist())
|
|
102
|
+
df["audience_type"] = [r.get("audience_type") if isinstance(r, dict) else None for r in found.values()]
|
|
103
|
+
df["seniority"] = [
|
|
104
|
+
",".join(r.get("b2b", {}).get("seniority", [])) if isinstance(r, dict) else None
|
|
105
|
+
for r in found.values()
|
|
106
|
+
]
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## Example: IAB categories for a text snippet
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
snippet = "Best budget laptops for students heading to university"
|
|
113
|
+
cats = client.categorize_text(snippet)
|
|
114
|
+
for name, score in cats["iab_classification"]:
|
|
115
|
+
print(name, score)
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
## Frequently asked
|
|
119
|
+
|
|
120
|
+
**Can I get the full word list for a field?** Yes. `CookielessAudiences.vocabularies()` returns every enumerated value and needs no key.
|
|
121
|
+
|
|
122
|
+
**What about scale?** The service accepts up to 50 parallel threads by default, about 270 URLs a minute.
|
|
123
|
+
|
|
124
|
+
**Which fields are free text?** None in the structured shape. Small fields are closed lists; open-ended signals are mapped to canonical codes or dropped.
|
|
125
|
+
|
|
126
|
+
## Package notes
|
|
127
|
+
|
|
128
|
+
MIT licensed. Questions go to info@alpha-quantum.com. Version 1.0.0.
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""Client for the Cookieless Audiences API (https://www.cookielessaudiences.com)."""
|
|
2
|
+
from typing import Any, Dict, Iterable, List, Optional
|
|
3
|
+
|
|
4
|
+
import requests
|
|
5
|
+
|
|
6
|
+
__version__ = "1.0.0"
|
|
7
|
+
BASE = "https://www.cookielessaudiences.com"
|
|
8
|
+
|
|
9
|
+
STATUS_TEXT = {
|
|
10
|
+
400: "Bad request, check the parameters",
|
|
11
|
+
401: "Invalid API key",
|
|
12
|
+
403: "Key not active or monthly credits used up",
|
|
13
|
+
407: "Missing data_type, must be 'url' or 'text'",
|
|
14
|
+
410: "Not enough content in the page or text",
|
|
15
|
+
411: "The URL content could not be fetched",
|
|
16
|
+
500: "General error, check the request or contact support",
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class CookielessAudiencesError(Exception):
|
|
21
|
+
"""Raised when the JSON body carries a status other than 200."""
|
|
22
|
+
|
|
23
|
+
def __init__(self, status: int, message: str, body: Optional[dict] = None):
|
|
24
|
+
super().__init__("[%s] %s" % (status, message))
|
|
25
|
+
self.status = status
|
|
26
|
+
self.body = body or {}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class CookielessAudiences:
|
|
30
|
+
def __init__(self, api_key: str, timeout: float = 120.0, session: Optional[requests.Session] = None):
|
|
31
|
+
if not api_key:
|
|
32
|
+
raise ValueError("An API key is required")
|
|
33
|
+
self.api_key = api_key
|
|
34
|
+
self.timeout = timeout
|
|
35
|
+
self.session = session or requests.Session()
|
|
36
|
+
self.session.headers["User-Agent"] = "cookielessaudiences-python/%s (+https://www.cookielessaudiences.com)" % __version__
|
|
37
|
+
|
|
38
|
+
def _post(self, path: str, form: Dict[str, str]) -> Dict[str, Any]:
|
|
39
|
+
resp = self.session.post(BASE + path, data=form, timeout=self.timeout)
|
|
40
|
+
try:
|
|
41
|
+
body = resp.json()
|
|
42
|
+
except ValueError:
|
|
43
|
+
raise CookielessAudiencesError(resp.status_code, "Response was not JSON")
|
|
44
|
+
status = body.get("status", 200) if isinstance(body, dict) else 200
|
|
45
|
+
if status != 200:
|
|
46
|
+
raise CookielessAudiencesError(status, STATUS_TEXT.get(status, "API error"), body)
|
|
47
|
+
return body
|
|
48
|
+
|
|
49
|
+
def segment(self, url: str, structured: bool = True) -> Dict[str, Any]:
|
|
50
|
+
"""Page-level audience segmentation. structured=False returns the legacy free-text shape."""
|
|
51
|
+
form = {"query": url, "api_key": self.api_key}
|
|
52
|
+
if structured:
|
|
53
|
+
form["format"] = "structured"
|
|
54
|
+
return self._post("/api/audience/segment.php", form)
|
|
55
|
+
|
|
56
|
+
def categorize(self, url: str, confidence: bool = True, root_fallback: bool = False) -> Dict[str, Any]:
|
|
57
|
+
"""IAB content categorization (v3 and v2) of a URL."""
|
|
58
|
+
form = {"query": url, "api_key": self.api_key, "data_type": "url"}
|
|
59
|
+
if confidence:
|
|
60
|
+
form["confidence"] = "1"
|
|
61
|
+
if root_fallback:
|
|
62
|
+
form["use_domain_as_basis_of_categorization_for_insufficient_subdomain_content"] = "1"
|
|
63
|
+
return self._post("/api/iab/iab_web_content_filtering.php", form)
|
|
64
|
+
|
|
65
|
+
def categorize_text(self, text: str, confidence: bool = True) -> Dict[str, Any]:
|
|
66
|
+
"""IAB content categorization of plain text."""
|
|
67
|
+
form = {"query": text, "api_key": self.api_key, "data_type": "text"}
|
|
68
|
+
if confidence:
|
|
69
|
+
form["confidence"] = "1"
|
|
70
|
+
return self._post("/api/iab/iab_content_filtering.php", form)
|
|
71
|
+
|
|
72
|
+
def segment_many(self, urls: Iterable[str], workers: int = 8, structured: bool = True) -> Dict[str, Any]:
|
|
73
|
+
"""Segment several URLs in parallel. Returns {url: result or CookielessAudiencesError}."""
|
|
74
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
75
|
+
|
|
76
|
+
urls = list(urls)
|
|
77
|
+
|
|
78
|
+
def one(u: str):
|
|
79
|
+
try:
|
|
80
|
+
return self.segment(u, structured=structured)
|
|
81
|
+
except Exception as exc: # keep the batch going
|
|
82
|
+
return exc
|
|
83
|
+
|
|
84
|
+
with ThreadPoolExecutor(max_workers=max(1, workers)) as pool:
|
|
85
|
+
return dict(zip(urls, pool.map(one, urls)))
|
|
86
|
+
|
|
87
|
+
@staticmethod
|
|
88
|
+
def vocabularies(timeout: float = 60.0) -> Dict[str, Any]:
|
|
89
|
+
"""Controlled vocabularies, public, no key required."""
|
|
90
|
+
resp = requests.get(BASE + "/api/audience/filters.php", timeout=timeout)
|
|
91
|
+
resp.raise_for_status()
|
|
92
|
+
return resp.json()
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def labels_for(result: Dict[str, Any]) -> Dict[str, List[str]]:
|
|
96
|
+
"""Readable labels for the INT.* and PI.* codes of a structured response."""
|
|
97
|
+
names = result.get("labels", {})
|
|
98
|
+
out = {}
|
|
99
|
+
for group in ("interests", "purchase_intent"):
|
|
100
|
+
block = result.get(group) or {}
|
|
101
|
+
codes = list(block.get("tier1", [])) + list(block.get("tier2", [])) + list(block.get("codes", []))
|
|
102
|
+
out[group] = [names.get(c, c) for c in codes]
|
|
103
|
+
return out
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: cookielessaudiences
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Python client for the Cookieless Audiences API: page-level audience segmentation (demographics, interests, purchase intent, B2B firmographics, personas) and IAB content categorization for any URL, with no cookies and no PII.
|
|
5
|
+
Home-page: https://www.cookielessaudiences.com
|
|
6
|
+
Author: Alpha Quantum
|
|
7
|
+
Author-email: info@alpha-quantum.com
|
|
8
|
+
License: MIT
|
|
9
|
+
Project-URL: Homepage, https://www.cookielessaudiences.com
|
|
10
|
+
Project-URL: Documentation, https://www.cookielessaudiences.com/api.php
|
|
11
|
+
Project-URL: Source, https://github.com/explainableaixai/cookielessaudiences
|
|
12
|
+
Project-URL: Mirror, https://gitlab.com/url-classifications/cookielessaudiences
|
|
13
|
+
Project-URL: Tracker, https://www.cookielessaudiences.com/contact.php
|
|
14
|
+
Project-URL: Pricing, https://www.cookielessaudiences.com/pricing.php
|
|
15
|
+
Description: # cookielessaudiences (Python)
|
|
16
|
+
|
|
17
|
+
Cookieless audience data for a URL, as a Python call. You pass a page and receive demographics, interests, purchase intent, B2B firmographics and personas, with no cookies and no personal data involved.
|
|
18
|
+
|
|
19
|
+
Built on `requests`, works on Python 3.7 and later.
|
|
20
|
+
|
|
21
|
+
## Setup
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install cookielessaudiences
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
An API key comes with any plan. If you are still comparing options, read how [cookieless audience segmentation](https://www.cookielessaudiences.com/features/cookieless-audience-segmentation.php) works before you buy.
|
|
28
|
+
|
|
29
|
+
## Quick start
|
|
30
|
+
|
|
31
|
+
```python
|
|
32
|
+
from cookielessaudiences import CookielessAudiences, labels_for
|
|
33
|
+
|
|
34
|
+
client = CookielessAudiences("YOUR_API_KEY")
|
|
35
|
+
result = client.segment("https://example.com/blog")
|
|
36
|
+
|
|
37
|
+
print(result["audience_type"])
|
|
38
|
+
print(result["demographics"]["income_level"])
|
|
39
|
+
print(labels_for(result))
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Reading the response
|
|
43
|
+
|
|
44
|
+
Think of the structured result as five questions about the reader of a page:
|
|
45
|
+
|
|
46
|
+
1. Who are they? `demographics` and `b2b`
|
|
47
|
+
2. What do they like? `interests`, two tiers of `INT.*` codes
|
|
48
|
+
3. What might they buy? `purchase_intent`, `PI.*` codes
|
|
49
|
+
4. Which named audiences fit? `personas`
|
|
50
|
+
5. What kind of page is it? `content_context`
|
|
51
|
+
|
|
52
|
+
`labels_for(result)` turns the codes in questions 2 and 3 into names you can show to a person.
|
|
53
|
+
|
|
54
|
+
## API surface
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
client.segment(url, structured=True)
|
|
58
|
+
client.categorize(url, confidence=True, root_fallback=False)
|
|
59
|
+
client.categorize_text(text, confidence=True)
|
|
60
|
+
client.segment_many(urls, workers=8)
|
|
61
|
+
CookielessAudiences.vocabularies()
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
`segment_many` returns a dict of url to result. A URL that failed holds its exception instead, so one bad page never stops a batch.
|
|
65
|
+
|
|
66
|
+
## Errors
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
from cookielessaudiences import CookielessAudiencesError
|
|
70
|
+
|
|
71
|
+
try:
|
|
72
|
+
client.segment(url)
|
|
73
|
+
except CookielessAudiencesError as exc:
|
|
74
|
+
print(exc.status, exc.body)
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
The HTTP layer is plain; the signal lives in `status` inside the JSON. 401 is a bad key, 403 means the key is inactive or credits are gone, 410 and 411 mean the page could not be read.
|
|
78
|
+
|
|
79
|
+
## Example: audience report for a media plan
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
import csv
|
|
83
|
+
from cookielessaudiences import CookielessAudiences, labels_for
|
|
84
|
+
|
|
85
|
+
client = CookielessAudiences("YOUR_API_KEY")
|
|
86
|
+
sites = [line.strip() for line in open("sites.txt") if line.strip()]
|
|
87
|
+
results = client.segment_many(sites, workers=10)
|
|
88
|
+
|
|
89
|
+
with open("plan.csv", "w", newline="") as fh:
|
|
90
|
+
out = csv.writer(fh)
|
|
91
|
+
out.writerow(["site", "type", "age", "income", "top_interest", "top_intent"])
|
|
92
|
+
for site, res in results.items():
|
|
93
|
+
if isinstance(res, Exception):
|
|
94
|
+
out.writerow([site, "error", "", "", "", ""])
|
|
95
|
+
continue
|
|
96
|
+
names = labels_for(res)
|
|
97
|
+
out.writerow([
|
|
98
|
+
site,
|
|
99
|
+
res.get("audience_type"),
|
|
100
|
+
"|".join(res["demographics"].get("age_bracket", [])),
|
|
101
|
+
res["demographics"].get("income_level"),
|
|
102
|
+
(names["interests"] or [""])[0],
|
|
103
|
+
(names["purchase_intent"] or [""])[0],
|
|
104
|
+
])
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
This is the shape most [website audience demographics](https://www.cookielessaudiences.com/features/website-audience-demographics.php) work takes: many sites in, one row per site out.
|
|
108
|
+
|
|
109
|
+
## Example: pandas enrichment
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
import pandas as pd
|
|
113
|
+
|
|
114
|
+
df = pd.read_csv("accounts.csv") # column: url
|
|
115
|
+
found = client.segment_many(df["url"].tolist())
|
|
116
|
+
df["audience_type"] = [r.get("audience_type") if isinstance(r, dict) else None for r in found.values()]
|
|
117
|
+
df["seniority"] = [
|
|
118
|
+
",".join(r.get("b2b", {}).get("seniority", [])) if isinstance(r, dict) else None
|
|
119
|
+
for r in found.values()
|
|
120
|
+
]
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
## Example: IAB categories for a text snippet
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
snippet = "Best budget laptops for students heading to university"
|
|
127
|
+
cats = client.categorize_text(snippet)
|
|
128
|
+
for name, score in cats["iab_classification"]:
|
|
129
|
+
print(name, score)
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
## Frequently asked
|
|
133
|
+
|
|
134
|
+
**Can I get the full word list for a field?** Yes. `CookielessAudiences.vocabularies()` returns every enumerated value and needs no key.
|
|
135
|
+
|
|
136
|
+
**What about scale?** The service accepts up to 50 parallel threads by default, about 270 URLs a minute.
|
|
137
|
+
|
|
138
|
+
**Which fields are free text?** None in the structured shape. Small fields are closed lists; open-ended signals are mapped to canonical codes or dropped.
|
|
139
|
+
|
|
140
|
+
## Package notes
|
|
141
|
+
|
|
142
|
+
MIT licensed. Questions go to info@alpha-quantum.com. Version 1.0.0.
|
|
143
|
+
|
|
144
|
+
Keywords: cookieless,audience segmentation,iab audience taxonomy,iab categorization,contextual targeting,seller defined audiences,purchase intent,adtech,media planning,inventory curation,audience data,privacy-safe advertising,domain audience data
|
|
145
|
+
Platform: UNKNOWN
|
|
146
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
147
|
+
Classifier: Intended Audience :: Developers
|
|
148
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
149
|
+
Classifier: Programming Language :: Python :: 3
|
|
150
|
+
Classifier: Programming Language :: Python :: 3.7
|
|
151
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
152
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
153
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
154
|
+
Requires-Python: >=3.7
|
|
155
|
+
Description-Content-Type: text/markdown
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
MANIFEST.in
|
|
3
|
+
README.md
|
|
4
|
+
setup.py
|
|
5
|
+
cookielessaudiences/__init__.py
|
|
6
|
+
cookielessaudiences/client.py
|
|
7
|
+
cookielessaudiences.egg-info/PKG-INFO
|
|
8
|
+
cookielessaudiences.egg-info/SOURCES.txt
|
|
9
|
+
cookielessaudiences.egg-info/dependency_links.txt
|
|
10
|
+
cookielessaudiences.egg-info/requires.txt
|
|
11
|
+
cookielessaudiences.egg-info/top_level.txt
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
requests>=2.20.0
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
cookielessaudiences
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import pathlib
|
|
2
|
+
from setuptools import setup, find_packages
|
|
3
|
+
|
|
4
|
+
README = (pathlib.Path(__file__).parent / "README.md").read_text(encoding="utf-8")
|
|
5
|
+
|
|
6
|
+
setup(
|
|
7
|
+
name="cookielessaudiences",
|
|
8
|
+
version="1.0.0",
|
|
9
|
+
description="Python client for the Cookieless Audiences API: page-level audience segmentation (demographics, interests, purchase intent, B2B firmographics, personas) and IAB content categorization for any URL, with no cookies and no PII.",
|
|
10
|
+
long_description=README,
|
|
11
|
+
long_description_content_type="text/markdown",
|
|
12
|
+
author="Alpha Quantum",
|
|
13
|
+
author_email="info@alpha-quantum.com",
|
|
14
|
+
url="https://www.cookielessaudiences.com",
|
|
15
|
+
project_urls={
|
|
16
|
+
"Homepage": "https://www.cookielessaudiences.com",
|
|
17
|
+
"Documentation": "https://www.cookielessaudiences.com/api.php",
|
|
18
|
+
"Source": "https://github.com/explainableaixai/cookielessaudiences",
|
|
19
|
+
"Mirror": "https://gitlab.com/url-classifications/cookielessaudiences",
|
|
20
|
+
"Tracker": "https://www.cookielessaudiences.com/contact.php",
|
|
21
|
+
"Pricing": "https://www.cookielessaudiences.com/pricing.php",
|
|
22
|
+
},
|
|
23
|
+
license="MIT",
|
|
24
|
+
packages=find_packages(exclude=("tests", "test")),
|
|
25
|
+
python_requires=">=3.7",
|
|
26
|
+
install_requires=["requests>=2.20.0"],
|
|
27
|
+
keywords=["cookieless", "audience segmentation", "iab audience taxonomy", "iab categorization", "contextual targeting",
|
|
28
|
+
"seller defined audiences", "purchase intent", "adtech", "media planning", "inventory curation",
|
|
29
|
+
"audience data", "privacy-safe advertising", "domain audience data"],
|
|
30
|
+
classifiers=[
|
|
31
|
+
"Development Status :: 5 - Production/Stable",
|
|
32
|
+
"Intended Audience :: Developers",
|
|
33
|
+
"License :: OSI Approved :: MIT License",
|
|
34
|
+
"Programming Language :: Python :: 3",
|
|
35
|
+
"Programming Language :: Python :: 3.7",
|
|
36
|
+
"Programming Language :: Python :: 3.12",
|
|
37
|
+
"Topic :: Internet :: WWW/HTTP",
|
|
38
|
+
"Topic :: Scientific/Engineering :: Information Analysis",
|
|
39
|
+
],
|
|
40
|
+
)
|