quanticdata 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quanticdata/__init__.py +32 -0
- quanticdata/_client.py +559 -0
- quanticdata/py.typed +0 -0
- quanticdata-0.1.0.dist-info/METADATA +134 -0
- quanticdata-0.1.0.dist-info/RECORD +7 -0
- quanticdata-0.1.0.dist-info/WHEEL +4 -0
- quanticdata-0.1.0.dist-info/licenses/LICENSE +21 -0
quanticdata/__init__.py
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Official Python SDK for the QuanticData API — scrape, search, map and
|
|
2
|
+
crawl the web through residential proxies, run the 74 ready-made Collectors,
|
|
3
|
+
build datasets from a prompt, and generate proxy endpoints of every type.
|
|
4
|
+
|
|
5
|
+
Quickstart::
|
|
6
|
+
|
|
7
|
+
from quanticdata import QuanticData
|
|
8
|
+
|
|
9
|
+
client = QuanticData() # reads QUANTICDATA_API_KEY
|
|
10
|
+
page = client.scrape("https://example.com")
|
|
11
|
+
print(page["content"])
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
__version__ = "0.1.0"
|
|
15
|
+
|
|
16
|
+
from ._client import ( # noqa: E402
|
|
17
|
+
DEFAULT_BASE_URL,
|
|
18
|
+
QuanticData,
|
|
19
|
+
QuanticDataError,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
Client = QuanticData
|
|
23
|
+
ApiError = QuanticDataError
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"QuanticData",
|
|
27
|
+
"QuanticDataError",
|
|
28
|
+
"Client",
|
|
29
|
+
"ApiError",
|
|
30
|
+
"DEFAULT_BASE_URL",
|
|
31
|
+
"__version__",
|
|
32
|
+
]
|
quanticdata/_client.py
ADDED
|
@@ -0,0 +1,559 @@
|
|
|
1
|
+
"""HTTP client for the QuanticData API.
|
|
2
|
+
|
|
3
|
+
Thin by design: every method maps 1:1 onto a REST endpoint, request bodies are
|
|
4
|
+
plain dicts, responses are the API envelope's ``payload`` already unwrapped.
|
|
5
|
+
The only sugar is ``wait=True`` on the async job starters, which polls the
|
|
6
|
+
job's status endpoint until it settles.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import os
|
|
12
|
+
import time
|
|
13
|
+
from typing import Any, Dict, List, Optional, Union
|
|
14
|
+
|
|
15
|
+
import requests
|
|
16
|
+
|
|
17
|
+
from . import __version__
|
|
18
|
+
|
|
19
|
+
DEFAULT_BASE_URL = "https://api.quanticdata.io/v1"
|
|
20
|
+
ENV_API_KEY = "QUANTICDATA_API_KEY"
|
|
21
|
+
ENV_API_BASE = "QUANTICDATA_API_BASE"
|
|
22
|
+
|
|
23
|
+
# Job statuses after which polling stops (jobs report e.g. done/completed on
|
|
24
|
+
# success, failed/error on failure; partial results are still in the payload).
|
|
25
|
+
_TERMINAL_STATUSES = {"done", "completed", "failed", "error", "cancelled"}
|
|
26
|
+
|
|
27
|
+
_JOB_ID_FIELDS = ("jobId", "job_id", "run_id", "runId", "id")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class QuanticDataError(Exception):
|
|
31
|
+
"""Raised when the API answers with the error envelope or transport fails.
|
|
32
|
+
|
|
33
|
+
Attributes:
|
|
34
|
+
message: human-readable reason from the API's ``message`` field.
|
|
35
|
+
status: HTTP status code, when a response was received.
|
|
36
|
+
payload: the error envelope's ``payload``, when present.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
def __init__(self, message: str, status: Optional[int] = None, payload: Any = None):
|
|
40
|
+
super().__init__(message)
|
|
41
|
+
self.message = message
|
|
42
|
+
self.status = status
|
|
43
|
+
self.payload = payload
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class QuanticData:
|
|
47
|
+
"""Client for the QuanticData API.
|
|
48
|
+
|
|
49
|
+
Args:
|
|
50
|
+
api_key: your ``qd_live_…`` key; falls back to ``QUANTICDATA_API_KEY``.
|
|
51
|
+
base_url: API base, default ``https://api.quanticdata.io/v1``
|
|
52
|
+
(override with ``QUANTICDATA_API_BASE``).
|
|
53
|
+
timeout: per-request timeout in seconds (default 90, matching the
|
|
54
|
+
slowest render-tier scrapes).
|
|
55
|
+
max_retries: extra attempts on connection errors and HTTP 429. Billable
|
|
56
|
+
calls are never retried after a response was received, so a
|
|
57
|
+
successful-but-lost call is not silently re-billed.
|
|
58
|
+
session: optional ``requests.Session`` to reuse connections.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
def __init__(
|
|
62
|
+
self,
|
|
63
|
+
api_key: Optional[str] = None,
|
|
64
|
+
*,
|
|
65
|
+
base_url: Optional[str] = None,
|
|
66
|
+
timeout: float = 90.0,
|
|
67
|
+
max_retries: int = 2,
|
|
68
|
+
session: Optional[requests.Session] = None,
|
|
69
|
+
):
|
|
70
|
+
self.api_key = api_key or os.environ.get(ENV_API_KEY, "")
|
|
71
|
+
if not self.api_key:
|
|
72
|
+
raise QuanticDataError(
|
|
73
|
+
f"No API key. Pass api_key=... or set {ENV_API_KEY}. "
|
|
74
|
+
"Get a free key at https://quanticdata.io"
|
|
75
|
+
)
|
|
76
|
+
self.base_url = (base_url or os.environ.get(ENV_API_BASE, DEFAULT_BASE_URL)).rstrip("/")
|
|
77
|
+
self.timeout = timeout
|
|
78
|
+
self.max_retries = max_retries
|
|
79
|
+
self._session = session or requests.Session()
|
|
80
|
+
self._session.headers.update(
|
|
81
|
+
{
|
|
82
|
+
"Authorization": f"Bearer {self.api_key}",
|
|
83
|
+
"User-Agent": f"quanticdata-python/{__version__}",
|
|
84
|
+
}
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
# ── transport ───────────────────────────────────────────────────────────
|
|
88
|
+
|
|
89
|
+
def _request(
|
|
90
|
+
self,
|
|
91
|
+
method: str,
|
|
92
|
+
path: str,
|
|
93
|
+
body: Optional[Dict[str, Any]] = None,
|
|
94
|
+
params: Optional[Dict[str, Any]] = None,
|
|
95
|
+
) -> Any:
|
|
96
|
+
url = f"{self.base_url}{path}"
|
|
97
|
+
attempt = 0
|
|
98
|
+
while True:
|
|
99
|
+
try:
|
|
100
|
+
resp = self._session.request(
|
|
101
|
+
method,
|
|
102
|
+
url,
|
|
103
|
+
json=body if method != "GET" else None,
|
|
104
|
+
params=params,
|
|
105
|
+
timeout=self.timeout,
|
|
106
|
+
)
|
|
107
|
+
except requests.RequestException as exc:
|
|
108
|
+
if attempt < self.max_retries:
|
|
109
|
+
attempt += 1
|
|
110
|
+
time.sleep(min(2**attempt, 8))
|
|
111
|
+
continue
|
|
112
|
+
raise QuanticDataError(f"Cannot reach the QuanticData API at {url}: {exc}") from exc
|
|
113
|
+
|
|
114
|
+
if resp.status_code == 429 and attempt < self.max_retries:
|
|
115
|
+
attempt += 1
|
|
116
|
+
retry_after = resp.headers.get("Retry-After")
|
|
117
|
+
time.sleep(float(retry_after) if retry_after else min(2**attempt, 8))
|
|
118
|
+
continue
|
|
119
|
+
|
|
120
|
+
ctype = resp.headers.get("content-type", "")
|
|
121
|
+
if "json" not in ctype:
|
|
122
|
+
# CSV exports and other text endpoints pass through verbatim.
|
|
123
|
+
if resp.ok:
|
|
124
|
+
return resp.text
|
|
125
|
+
raise QuanticDataError(
|
|
126
|
+
resp.text[:500] or f"Request failed (HTTP {resp.status_code})",
|
|
127
|
+
status=resp.status_code,
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
try:
|
|
131
|
+
doc = resp.json()
|
|
132
|
+
except ValueError:
|
|
133
|
+
raise QuanticDataError(
|
|
134
|
+
f"Non-JSON response (HTTP {resp.status_code})", status=resp.status_code
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
if not resp.ok or (isinstance(doc, dict) and doc.get("type") == "error"):
|
|
138
|
+
message = doc.get("message") if isinstance(doc, dict) else None
|
|
139
|
+
raise QuanticDataError(
|
|
140
|
+
message or f"Request failed (HTTP {resp.status_code})",
|
|
141
|
+
status=resp.status_code,
|
|
142
|
+
payload=doc.get("payload") if isinstance(doc, dict) else None,
|
|
143
|
+
)
|
|
144
|
+
if isinstance(doc, dict) and "payload" in doc:
|
|
145
|
+
return doc["payload"]
|
|
146
|
+
return doc
|
|
147
|
+
|
|
148
|
+
@staticmethod
|
|
149
|
+
def _clean(mapping: Dict[str, Any]) -> Dict[str, Any]:
|
|
150
|
+
return {k: v for k, v in mapping.items() if v is not None}
|
|
151
|
+
|
|
152
|
+
# ── jobs: shared polling ───────────────────────────────────────────────
|
|
153
|
+
|
|
154
|
+
@staticmethod
|
|
155
|
+
def _job_id(payload: Any) -> Optional[str]:
|
|
156
|
+
if not isinstance(payload, dict):
|
|
157
|
+
return None
|
|
158
|
+
for field in _JOB_ID_FIELDS:
|
|
159
|
+
value = payload.get(field)
|
|
160
|
+
if isinstance(value, str) and value:
|
|
161
|
+
return value
|
|
162
|
+
return None
|
|
163
|
+
|
|
164
|
+
def _wait(self, poll, job_id: str, poll_interval: float, timeout: float, final=None) -> Any:
|
|
165
|
+
deadline = time.monotonic() + timeout
|
|
166
|
+
while True:
|
|
167
|
+
payload = poll(job_id)
|
|
168
|
+
status = payload.get("status") if isinstance(payload, dict) else None
|
|
169
|
+
if status in _TERMINAL_STATUSES:
|
|
170
|
+
return final(job_id) if final is not None else payload
|
|
171
|
+
if time.monotonic() >= deadline:
|
|
172
|
+
raise QuanticDataError(
|
|
173
|
+
f"Job {job_id} still '{status}' after {timeout:.0f}s — poll it yourself later",
|
|
174
|
+
payload=payload,
|
|
175
|
+
)
|
|
176
|
+
time.sleep(poll_interval)
|
|
177
|
+
|
|
178
|
+
# ── scraping ────────────────────────────────────────────────────────────
|
|
179
|
+
|
|
180
|
+
def scrape(self, url: Optional[str] = None, **options: Any) -> Any:
|
|
181
|
+
"""Scrape one page to Markdown/HTML/text through a residential proxy.
|
|
182
|
+
|
|
183
|
+
Accepts every parameter of ``POST /scraper/extract`` as keyword
|
|
184
|
+
arguments (``format``, ``content_mode``, ``engine``, ``render``,
|
|
185
|
+
``country``, ``extract``, ``ai_prompt``, ``actions``, ``cookies``,
|
|
186
|
+
``query``, ``max_tokens`` …). Two ergonomic aliases:
|
|
187
|
+
|
|
188
|
+
- ``preset_id="…"`` runs a stored parser preset.
|
|
189
|
+
- ``fetch_resource="regex"`` returns the first matching network
|
|
190
|
+
response's body instead of the page.
|
|
191
|
+
"""
|
|
192
|
+
body = self._clean({"url": url, **options})
|
|
193
|
+
preset_id = body.pop("preset_id", None)
|
|
194
|
+
if preset_id:
|
|
195
|
+
body["presetId"] = preset_id
|
|
196
|
+
fetch_resource = body.pop("fetch_resource", None)
|
|
197
|
+
if fetch_resource:
|
|
198
|
+
actions = list(body.get("actions") or [])
|
|
199
|
+
actions.append({"fetchResource": {"pattern": fetch_resource}})
|
|
200
|
+
body["actions"] = actions
|
|
201
|
+
return self._request("POST", "/scraper/extract", body)
|
|
202
|
+
|
|
203
|
+
def batch(
|
|
204
|
+
self,
|
|
205
|
+
urls: List[str],
|
|
206
|
+
*,
|
|
207
|
+
wait: bool = False,
|
|
208
|
+
poll_interval: float = 3.0,
|
|
209
|
+
wait_timeout: float = 600.0,
|
|
210
|
+
**options: Any,
|
|
211
|
+
) -> Any:
|
|
212
|
+
"""Scrape many URLs asynchronously (``POST /scraper/batch``).
|
|
213
|
+
|
|
214
|
+
Returns the job payload (with its job id). With ``wait=True`` polls
|
|
215
|
+
until the job settles and returns the finished job including content.
|
|
216
|
+
"""
|
|
217
|
+
payload = self._request("POST", "/scraper/batch", self._clean({"urls": urls, **options}))
|
|
218
|
+
if not wait:
|
|
219
|
+
return payload
|
|
220
|
+
job_id = self._job_id(payload)
|
|
221
|
+
if not job_id:
|
|
222
|
+
return payload
|
|
223
|
+
return self._wait(
|
|
224
|
+
lambda j: self.batch_status(j),
|
|
225
|
+
job_id,
|
|
226
|
+
poll_interval,
|
|
227
|
+
wait_timeout,
|
|
228
|
+
final=lambda j: self.batch_status(j, include_content=True),
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
def batch_status(
|
|
232
|
+
self,
|
|
233
|
+
job_id: str,
|
|
234
|
+
*,
|
|
235
|
+
since: Optional[int] = None,
|
|
236
|
+
include_content: bool = False,
|
|
237
|
+
) -> Any:
|
|
238
|
+
"""Poll a batch job; pass the previous ``nextCursor`` as ``since``."""
|
|
239
|
+
params = self._clean({"since": since, "include_content": "true" if include_content else None})
|
|
240
|
+
return self._request("GET", f"/scraper/batch/{job_id}", params=params)
|
|
241
|
+
|
|
242
|
+
def crawl(
|
|
243
|
+
self,
|
|
244
|
+
url: str,
|
|
245
|
+
*,
|
|
246
|
+
wait: bool = False,
|
|
247
|
+
poll_interval: float = 3.0,
|
|
248
|
+
wait_timeout: float = 600.0,
|
|
249
|
+
**options: Any,
|
|
250
|
+
) -> Any:
|
|
251
|
+
"""Start an async BFS crawl (``POST /scraper/crawl``); options include
|
|
252
|
+
``limit``, ``depth``, ``content_mode``, ``include``, ``exclude``,
|
|
253
|
+
``country``. With ``wait=True`` returns the finished job with content.
|
|
254
|
+
"""
|
|
255
|
+
payload = self._request("POST", "/scraper/crawl", self._clean({"url": url, **options}))
|
|
256
|
+
if not wait:
|
|
257
|
+
return payload
|
|
258
|
+
job_id = self._job_id(payload)
|
|
259
|
+
if not job_id:
|
|
260
|
+
return payload
|
|
261
|
+
return self._wait(
|
|
262
|
+
lambda j: self.crawl_status(j),
|
|
263
|
+
job_id,
|
|
264
|
+
poll_interval,
|
|
265
|
+
wait_timeout,
|
|
266
|
+
final=lambda j: self.crawl_status(j, include_content=True),
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
def crawl_status(
|
|
270
|
+
self,
|
|
271
|
+
job_id: str,
|
|
272
|
+
*,
|
|
273
|
+
since: Optional[int] = None,
|
|
274
|
+
include_content: bool = False,
|
|
275
|
+
) -> Any:
|
|
276
|
+
"""Poll a crawl job; content is omitted unless ``include_content=True``."""
|
|
277
|
+
params = {"include_content": "true" if include_content else "false"}
|
|
278
|
+
if since is not None:
|
|
279
|
+
params["since"] = since
|
|
280
|
+
return self._request("GET", f"/scraper/crawl/{job_id}", params=params)
|
|
281
|
+
|
|
282
|
+
def map(self, url: str, **options: Any) -> Any:
|
|
283
|
+
"""Discover a site's URLs fast (``POST /scraper/map``): sitemaps +
|
|
284
|
+
homepage links, with ``search`` substring filter and ``group_by="path"``.
|
|
285
|
+
"""
|
|
286
|
+
return self._request("POST", "/scraper/map", self._clean({"url": url, **options}))
|
|
287
|
+
|
|
288
|
+
def seo_audit(self, url: str, **options: Any) -> Any:
|
|
289
|
+
"""Audit a URL twice — no-JS bot view and rendered view — plus the diff
|
|
290
|
+
(``POST /scraper/seo-audit``). ``no_render=True`` for the cheap pass.
|
|
291
|
+
"""
|
|
292
|
+
return self._request("POST", "/scraper/seo-audit", self._clean({"url": url, **options}))
|
|
293
|
+
|
|
294
|
+
# ── search ──────────────────────────────────────────────────────────────
|
|
295
|
+
|
|
296
|
+
def search(self, query: Optional[str] = None, **options: Any) -> Any:
|
|
297
|
+
"""Structured Google/Bing/DuckDuckGo search (``POST /scraper/serp``).
|
|
298
|
+
|
|
299
|
+
Every SERP API parameter is accepted as a keyword argument:
|
|
300
|
+
``engine``, ``search_type`` (17 verticals), ``country``, ``lang``,
|
|
301
|
+
``num``, ``page``, ``location``, ``uule``, ``device`` … ID-addressed
|
|
302
|
+
verticals (``place_details``, ``product``, ``flights``, ``reviews``)
|
|
303
|
+
take their ids the same way.
|
|
304
|
+
"""
|
|
305
|
+
return self._request("POST", "/scraper/serp", self._clean({"query": query, **options}))
|
|
306
|
+
|
|
307
|
+
def search_and_read(self, query: str, **options: Any) -> Any:
|
|
308
|
+
"""Search the web and return the top pages as citation-ready Markdown
|
|
309
|
+
context (``POST /ai/search``). Options: ``top_n``, ``max_tokens``,
|
|
310
|
+
``engine``, ``country``, ``lang``, ``fetch_content``.
|
|
311
|
+
"""
|
|
312
|
+
return self._request("POST", "/ai/search", self._clean({"query": query, **options}))
|
|
313
|
+
|
|
314
|
+
def search_bulk(
|
|
315
|
+
self,
|
|
316
|
+
query: str,
|
|
317
|
+
*,
|
|
318
|
+
wait: bool = False,
|
|
319
|
+
poll_interval: float = 3.0,
|
|
320
|
+
wait_timeout: float = 600.0,
|
|
321
|
+
**options: Any,
|
|
322
|
+
) -> Any:
|
|
323
|
+
"""Paginate one query across many result pages asynchronously
|
|
324
|
+
(``POST /scraper/serp/bulk``). Options: ``max_pages``, ``engine``,
|
|
325
|
+
``search_type``, ``country``, ``render``, ``webhook`` …
|
|
326
|
+
"""
|
|
327
|
+
payload = self._request("POST", "/scraper/serp/bulk", self._clean({"query": query, **options}))
|
|
328
|
+
if not wait:
|
|
329
|
+
return payload
|
|
330
|
+
job_id = self._job_id(payload)
|
|
331
|
+
if not job_id:
|
|
332
|
+
return payload
|
|
333
|
+
return self._wait(lambda j: self.search_bulk_status(j), job_id, poll_interval, wait_timeout)
|
|
334
|
+
|
|
335
|
+
def search_bulk_status(self, job_id: str, *, since: Optional[int] = None) -> Any:
|
|
336
|
+
"""Poll a bulk search job; pass the previous ``nextCursor`` as ``since``."""
|
|
337
|
+
return self._request("GET", f"/scraper/serp/bulk/{job_id}", params=self._clean({"since": since}))
|
|
338
|
+
|
|
339
|
+
# ── parsers ─────────────────────────────────────────────────────────────
|
|
340
|
+
|
|
341
|
+
def generate_parser(self, url: Optional[str] = None, **options: Any) -> Any:
|
|
342
|
+
"""Learn CSS selectors for a page layout once with an LLM
|
|
343
|
+
(``POST /scraper/parser/generate``), then reuse them for free via
|
|
344
|
+
``scrape(extract=…)`` or a saved preset. Options: ``fields``,
|
|
345
|
+
``prompt``, ``html``, ``render``, ``country``.
|
|
346
|
+
"""
|
|
347
|
+
return self._request("POST", "/scraper/parser/generate", self._clean({"url": url, **options}))
|
|
348
|
+
|
|
349
|
+
def save_parser_preset(
|
|
350
|
+
self,
|
|
351
|
+
name: str,
|
|
352
|
+
parser: Dict[str, Any],
|
|
353
|
+
*,
|
|
354
|
+
source_url: Optional[str] = None,
|
|
355
|
+
fields: Optional[Dict[str, str]] = None,
|
|
356
|
+
render: Optional[bool] = None,
|
|
357
|
+
auto_heal: Optional[bool] = None,
|
|
358
|
+
) -> Any:
|
|
359
|
+
"""Store a parser under a name (``POST /scraper/parser/presets``);
|
|
360
|
+
give ``source_url`` so it can self-heal when the site changes.
|
|
361
|
+
"""
|
|
362
|
+
return self._request(
|
|
363
|
+
"POST",
|
|
364
|
+
"/scraper/parser/presets",
|
|
365
|
+
self._clean(
|
|
366
|
+
{
|
|
367
|
+
"name": name,
|
|
368
|
+
"parser": parser,
|
|
369
|
+
"sourceUrl": source_url,
|
|
370
|
+
"fields": fields,
|
|
371
|
+
"render": render,
|
|
372
|
+
"autoHeal": auto_heal,
|
|
373
|
+
}
|
|
374
|
+
),
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
def list_parser_presets(self) -> Any:
|
|
378
|
+
"""List stored parser presets with version, health and changelog."""
|
|
379
|
+
return self._request("GET", "/scraper/parser/presets")
|
|
380
|
+
|
|
381
|
+
def parser_preset_stats(self, preset_id: str) -> Any:
|
|
382
|
+
"""Per-field success rate and decay verdict for one preset."""
|
|
383
|
+
return self._request("GET", f"/scraper/parser/presets/{preset_id}/stats")
|
|
384
|
+
|
|
385
|
+
def heal_parser_preset(self, preset_id: str, *, force: bool = False) -> Any:
|
|
386
|
+
"""Regenerate a preset's selectors now; adopted only if they extract
|
|
387
|
+
more than the current ones (a no-better heal is not billed).
|
|
388
|
+
"""
|
|
389
|
+
return self._request("POST", f"/scraper/parser/presets/{preset_id}/heal", {"force": force})
|
|
390
|
+
|
|
391
|
+
# ── datasets ────────────────────────────────────────────────────────────
|
|
392
|
+
|
|
393
|
+
def create_dataset(
|
|
394
|
+
self,
|
|
395
|
+
prompt: str,
|
|
396
|
+
*,
|
|
397
|
+
wait: bool = False,
|
|
398
|
+
poll_interval: float = 5.0,
|
|
399
|
+
wait_timeout: float = 1800.0,
|
|
400
|
+
**options: Any,
|
|
401
|
+
) -> Any:
|
|
402
|
+
"""Build a validated dataset from a plain-language prompt
|
|
403
|
+
(``POST /scraper/datasets``). Options: ``columns``, ``country``,
|
|
404
|
+
``sources``, ``limits`` (incl. ``max_cost_usd``), ``webhook``.
|
|
405
|
+
"""
|
|
406
|
+
payload = self._request("POST", "/scraper/datasets", self._clean({"prompt": prompt, **options}))
|
|
407
|
+
if not wait:
|
|
408
|
+
return payload
|
|
409
|
+
job_id = self._job_id(payload)
|
|
410
|
+
if not job_id:
|
|
411
|
+
return payload
|
|
412
|
+
return self._wait(
|
|
413
|
+
lambda j: self.dataset_status(j, mode="summary"),
|
|
414
|
+
job_id,
|
|
415
|
+
poll_interval,
|
|
416
|
+
wait_timeout,
|
|
417
|
+
final=lambda j: self.dataset_status(j),
|
|
418
|
+
)
|
|
419
|
+
|
|
420
|
+
def dataset_status(
|
|
421
|
+
self,
|
|
422
|
+
job_id: str,
|
|
423
|
+
*,
|
|
424
|
+
since: Optional[int] = None,
|
|
425
|
+
mode: Optional[str] = None,
|
|
426
|
+
) -> Any:
|
|
427
|
+
"""Poll a dataset job. ``mode="summary"`` omits rows (light poll);
|
|
428
|
+
a completed job includes signed CSV/JSON download URLs.
|
|
429
|
+
"""
|
|
430
|
+
return self._request(
|
|
431
|
+
"GET", f"/scraper/datasets/{job_id}", params=self._clean({"since": since, "mode": mode})
|
|
432
|
+
)
|
|
433
|
+
|
|
434
|
+
# ── collectors ──────────────────────────────────────────────────────────
|
|
435
|
+
|
|
436
|
+
def list_collectors(self, category: Optional[str] = None) -> Any:
|
|
437
|
+
"""Catalog of the 74 ready-made Collectors with schemas, examples,
|
|
438
|
+
prices and health (``GET /scraper/collectors``, free).
|
|
439
|
+
"""
|
|
440
|
+
return self._request("GET", "/scraper/collectors", params=self._clean({"category": category}))
|
|
441
|
+
|
|
442
|
+
def run_collector(
|
|
443
|
+
self,
|
|
444
|
+
slug: str,
|
|
445
|
+
input: Optional[Dict[str, Any]] = None,
|
|
446
|
+
*,
|
|
447
|
+
wait: bool = False,
|
|
448
|
+
poll_interval: float = 3.0,
|
|
449
|
+
wait_timeout: float = 900.0,
|
|
450
|
+
**input_kwargs: Any,
|
|
451
|
+
) -> Any:
|
|
452
|
+
"""Run a Collector by slug with its semantic input, e.g.::
|
|
453
|
+
|
|
454
|
+
client.run_collector("google_maps_places",
|
|
455
|
+
keyword="dentist", location="Austin, TX")
|
|
456
|
+
|
|
457
|
+
Short runs return rows inline; long runs return a ``run_id`` — with
|
|
458
|
+
``wait=True`` the client polls it for you. Billed per delivered row.
|
|
459
|
+
"""
|
|
460
|
+
body = dict(input or {})
|
|
461
|
+
body.update(input_kwargs)
|
|
462
|
+
payload = self._request("POST", f"/scraper/collectors/{slug}/run", body)
|
|
463
|
+
run_id = self._job_id(payload) if isinstance(payload, dict) else None
|
|
464
|
+
if wait and run_id and isinstance(payload, dict) and payload.get("status") not in _TERMINAL_STATUSES:
|
|
465
|
+
return self._wait(lambda r: self.collector_run_status(r), run_id, poll_interval, wait_timeout)
|
|
466
|
+
return payload
|
|
467
|
+
|
|
468
|
+
def collector_run_status(self, run_id: str, *, format: Optional[str] = None) -> Any:
|
|
469
|
+
"""Fetch a Collector run: status, cost and rows. ``format="csv"``
|
|
470
|
+
returns the rows as CSV text.
|
|
471
|
+
"""
|
|
472
|
+
return self._request(
|
|
473
|
+
"GET", f"/scraper/collectors/runs/{run_id}", params=self._clean({"format": format})
|
|
474
|
+
)
|
|
475
|
+
|
|
476
|
+
# ── proxies ─────────────────────────────────────────────────────────────
|
|
477
|
+
|
|
478
|
+
def list_proxies(
|
|
479
|
+
self,
|
|
480
|
+
*,
|
|
481
|
+
active: Optional[bool] = None,
|
|
482
|
+
plan_type: Optional[str] = None,
|
|
483
|
+
limit: Optional[int] = None,
|
|
484
|
+
offset: Optional[int] = None,
|
|
485
|
+
) -> Any:
|
|
486
|
+
"""List the account's proxy services of every type, with the
|
|
487
|
+
``orderId`` to pass to :meth:`generate_proxies`.
|
|
488
|
+
"""
|
|
489
|
+
return self._request(
|
|
490
|
+
"GET",
|
|
491
|
+
"/public/proxies",
|
|
492
|
+
params=self._clean(
|
|
493
|
+
{
|
|
494
|
+
"active": None if active is None else str(active).lower(),
|
|
495
|
+
"planType": plan_type,
|
|
496
|
+
"limit": limit,
|
|
497
|
+
"offset": offset,
|
|
498
|
+
}
|
|
499
|
+
),
|
|
500
|
+
)
|
|
501
|
+
|
|
502
|
+
def generate_proxies(self, order_id: str, **options: Any) -> Any:
|
|
503
|
+
"""Generate ready-to-use proxy endpoint strings from an active service
|
|
504
|
+
(``POST /public/proxies/generate``). Options: ``protocol``, ``format``,
|
|
505
|
+
``quantity``, ``country``/``state``/``city``, ``rotation``,
|
|
506
|
+
``sessionTime``, ``isp``, ``asn`` … The strings plug straight into any
|
|
507
|
+
HTTP client (``curl -x``, ``requests``' ``proxies=``).
|
|
508
|
+
"""
|
|
509
|
+
return self._request(
|
|
510
|
+
"POST", "/public/proxies/generate", self._clean({"orderId": order_id, **options})
|
|
511
|
+
)
|
|
512
|
+
|
|
513
|
+
def proxy_locations(
|
|
514
|
+
self,
|
|
515
|
+
plan_type: str,
|
|
516
|
+
*,
|
|
517
|
+
level: str = "countries",
|
|
518
|
+
country: Optional[str] = None,
|
|
519
|
+
state: Optional[str] = None,
|
|
520
|
+
) -> Any:
|
|
521
|
+
"""Valid geo-targeting values for a plan type: ``countries`` (default),
|
|
522
|
+
``states``, ``cities``, ``asns``, or ``tree`` for the full
|
|
523
|
+
country→region→city→ISP tree with slugs.
|
|
524
|
+
"""
|
|
525
|
+
if level == "tree":
|
|
526
|
+
if plan_type in ("residentialpremium", "resiprivate"):
|
|
527
|
+
path = "/public/generator/residential-premium/targeting-options"
|
|
528
|
+
elif plan_type in ("mobile", "mobile_v2"):
|
|
529
|
+
path = "/public/generator/mobile/targeting-options"
|
|
530
|
+
else:
|
|
531
|
+
path = "/public/generator/datacenter/targeting-options"
|
|
532
|
+
return self._request("GET", path)
|
|
533
|
+
segment = "cities" if level == "cities" else level
|
|
534
|
+
return self._request(
|
|
535
|
+
"GET",
|
|
536
|
+
f"/public/geo/{segment}",
|
|
537
|
+
params=self._clean({"planType": plan_type, "country": country, "state": state}),
|
|
538
|
+
)
|
|
539
|
+
|
|
540
|
+
def whitelist_ip(
|
|
541
|
+
self,
|
|
542
|
+
action: str,
|
|
543
|
+
order_id: str,
|
|
544
|
+
ip: Optional[str] = None,
|
|
545
|
+
**options: Any,
|
|
546
|
+
) -> Any:
|
|
547
|
+
"""Manage IP-auth whitelisting on a proxy service: ``action`` is
|
|
548
|
+
``"list"``, ``"add"`` or ``"remove"`` (mobile ``add`` also accepts
|
|
549
|
+
``ports_count``, ``protocol``, ``country``, ``sticky``, ``ttl`` …).
|
|
550
|
+
"""
|
|
551
|
+
if action == "list":
|
|
552
|
+
return self._request("GET", "/public/proxies/whitelist-ip", params={"orderId": order_id})
|
|
553
|
+
if not ip:
|
|
554
|
+
raise QuanticDataError("`ip` is required for add/remove")
|
|
555
|
+
if action == "remove":
|
|
556
|
+
return self._request("DELETE", "/public/proxies/whitelist-ip", {"orderId": order_id, "ip": ip})
|
|
557
|
+
return self._request(
|
|
558
|
+
"POST", "/public/proxies/whitelist-ip", self._clean({"orderId": order_id, "ip": ip, **options})
|
|
559
|
+
)
|
quanticdata/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: quanticdata
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Official Python SDK for the QuanticData API — web scraping to Markdown, SERP search, crawl/map, 74 ready-made Collectors and residential/mobile/datacenter proxy generation.
|
|
5
|
+
Project-URL: Homepage, https://quanticdata.io
|
|
6
|
+
Project-URL: Documentation, https://quanticdata.io/docs
|
|
7
|
+
Project-URL: Repository, https://github.com/quanticdata/quanticdata-python
|
|
8
|
+
Author-email: QuanticData <support@quanticdata.io>
|
|
9
|
+
License: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: ai-agents,crawler,data-extraction,google-search,proxy,quanticdata,residential-proxies,scraper,serp,web-scraping
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
22
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
23
|
+
Requires-Python: >=3.9
|
|
24
|
+
Requires-Dist: requests>=2.25
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
# quanticdata — Python SDK for the QuanticData API
|
|
28
|
+
|
|
29
|
+
Scrape any page to clean Markdown, run structured Google/Bing/DuckDuckGo
|
|
30
|
+
searches, crawl and map whole sites, run 74 ready-made Collectors
|
|
31
|
+
(Amazon, Google Maps, LinkedIn jobs, app stores…), build datasets from a
|
|
32
|
+
plain-language prompt — everything through QuanticData' residential
|
|
33
|
+
proxy network with real-browser TLS fingerprints. Pay per successful call;
|
|
34
|
+
blocked pages cost nothing.
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
pip install quanticdata
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## Quickstart
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
from quanticdata import QuanticData
|
|
44
|
+
|
|
45
|
+
client = QuanticData() # reads QUANTICDATA_API_KEY from the environment
|
|
46
|
+
|
|
47
|
+
page = client.scrape("https://example.com")
|
|
48
|
+
print(page["title"], page["engine"])
|
|
49
|
+
print(page["content"]) # the page as clean Markdown
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Get a free API key at [quanticdata.io](https://quanticdata.io) — every
|
|
53
|
+
account includes free monthly usage, no card required. Set it once:
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
export QUANTICDATA_API_KEY=qd_live_your_key_here
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## What's in the box
|
|
60
|
+
|
|
61
|
+
Every REST endpoint, one method each — responses come back with the API
|
|
62
|
+
envelope already unwrapped:
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
# Structured search — 3 engines, 17 verticals, SerpApi-compatible JSON
|
|
66
|
+
serp = client.search("best espresso machine", country="us", num=20)
|
|
67
|
+
for r in serp["organic"]:
|
|
68
|
+
print(r["rank"], r["title"], r["link"])
|
|
69
|
+
|
|
70
|
+
# SERP → citation-ready Markdown context for an AI prompt
|
|
71
|
+
ctx = client.search_and_read("latest EU AI act status", top_n=3)
|
|
72
|
+
|
|
73
|
+
# Map a site's URLs in seconds (sitemaps + homepage links)
|
|
74
|
+
urls = client.map("https://stripe.com", search="/blog")
|
|
75
|
+
|
|
76
|
+
# Async crawl — wait=True polls until it settles and returns the pages
|
|
77
|
+
job = client.crawl("https://docs.python.org", limit=30, depth=2, wait=True)
|
|
78
|
+
|
|
79
|
+
# Batch-scrape known URLs
|
|
80
|
+
job = client.batch(["https://a.example", "https://b.example"], wait=True)
|
|
81
|
+
|
|
82
|
+
# CSS/AI extraction on one page
|
|
83
|
+
data = client.scrape(
|
|
84
|
+
"https://books.toscrape.com",
|
|
85
|
+
extract={"titles": {"selector": "h3 a", "attr": "title", "all": True}},
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
# Learn selectors once with an LLM, then scrape the same layout for free
|
|
89
|
+
parser = client.generate_parser(
|
|
90
|
+
"https://news.ycombinator.com",
|
|
91
|
+
fields={"titles": "every story title, as a list"},
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
# 74 ready-made Collectors — semantic input instead of URLs
|
|
95
|
+
places = client.run_collector(
|
|
96
|
+
"google_maps_places", keyword="dentist", location="Austin, TX", max_results=20
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
# Dataset from a prompt (validated rows, budget-capped)
|
|
100
|
+
ds = client.create_dataset(
|
|
101
|
+
"coffee roasters in Portland with email and phone",
|
|
102
|
+
limits={"max_rows": 50, "max_cost_usd": 2},
|
|
103
|
+
wait=True,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
# Proxy endpoints of every type — residential, mobile, datacenter, ISP, IPv6
|
|
107
|
+
plans = client.list_proxies(active=True)
|
|
108
|
+
proxies = client.generate_proxies(plans["proxies"][0]["orderId"], country="us", quantity=5)
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
## Errors and retries
|
|
112
|
+
|
|
113
|
+
Failures raise `QuanticDataError` with `.status`, `.message` and
|
|
114
|
+
`.payload`. Connection errors and HTTP 429 are retried with backoff;
|
|
115
|
+
billable calls are never re-sent after a response was received, so nothing
|
|
116
|
+
gets double-billed behind your back.
|
|
117
|
+
|
|
118
|
+
```python
|
|
119
|
+
from quanticdata import QuanticData, QuanticDataError
|
|
120
|
+
|
|
121
|
+
try:
|
|
122
|
+
QuanticData(api_key="qd_live_wrong").scrape("https://example.com")
|
|
123
|
+
except QuanticDataError as err:
|
|
124
|
+
print(err.status, err.message)
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
## Also available
|
|
128
|
+
|
|
129
|
+
- **MCP server** for Claude, Cursor and any MCP client:
|
|
130
|
+
[`npx -y quanticdata-mcp`](https://www.npmjs.com/package/quanticdata-mcp)
|
|
131
|
+
exposes the same 25 tools to AI agents.
|
|
132
|
+
- **REST reference**: [quanticdata.io/docs](https://quanticdata.io/docs)
|
|
133
|
+
|
|
134
|
+
MIT licensed.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
quanticdata/__init__.py,sha256=_fanPbyQVtqVfEQvVt3Fj3sP--oR0s439c7Dz0Olm4c,722
|
|
2
|
+
quanticdata/_client.py,sha256=ZqD_6KfCwRXadYgaGN78dd8i66WVhLAwd6t0bPgwNa4,23092
|
|
3
|
+
quanticdata/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
4
|
+
quanticdata-0.1.0.dist-info/METADATA,sha256=734ROiFG4q6gvLANHcDbXDZ7I1sr4nW3EeJryGt542I,4812
|
|
5
|
+
quanticdata-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
6
|
+
quanticdata-0.1.0.dist-info/licenses/LICENSE,sha256=eh5KMOSslIV7vffJUKHKyHYJO0QoeV7irDuRzgjrcsY,1068
|
|
7
|
+
quanticdata-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 QuanticData
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|