quantumproxies 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,34 @@
1
+ # Build artifacts / archives — never commit to git
2
+ landing.tar.gz
3
+ chrome-extension.zip
4
+ paypal-update.tar.gz
5
+ *.tar.gz
6
+
7
+ # OS / editor
8
+ .DS_Store
9
+ .cursor/
10
+ .claude/
11
+
12
+ # Upstream study clone — do not commit (see OpenSERP local study plan)
13
+ openserp/
14
+
15
+ # SERP cookie-farm runtime state — real Google cookie jars, never commit
16
+ scraper-service/.jar-store/
17
+ scraper-service/.jars/
18
+
19
+ # Dipendenze installate — mai in git (2.302 file al 2026-08-21)
20
+ node_modules/
21
+
22
+ # Segreti reali. Gli .env.example restano tracciati apposta.
23
+ .env
24
+ .env.*
25
+ !.env.example
26
+ !.env.*.example
27
+ blog-tools/.env.neon
28
+
29
+ # Materiale CA del web unlocker: chiave privata + certificato, mai in git
30
+ scraper-service/.unlock-ca/
31
+
32
+ # Stato runtime dei job/dataset del servizio
33
+ scraper-service/.job-store/
34
+ scraper-service/.dataset-files/
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 QuantumProxies
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,134 @@
1
+ Metadata-Version: 2.5
2
+ Name: quantumproxies
3
+ Version: 0.1.0
4
+ Summary: Official Python SDK for the QuantumProxies API — web scraping to Markdown, SERP search, crawl/map, 74 ready-made Collectors and residential/mobile/datacenter proxy generation.
5
+ Project-URL: Homepage, https://quantumproxies.io
6
+ Project-URL: Documentation, https://quantumproxies.io/data-api
7
+ Project-URL: Repository, https://github.com/quantumproxies/quantumproxies-python
8
+ Author-email: QuantumProxies <support@quantumproxies.io>
9
+ License: MIT
10
+ License-File: LICENSE
11
+ Keywords: ai-agents,crawler,data-extraction,google-search,proxy,quantumproxies,residential-proxies,scraper,serp,web-scraping
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Internet :: WWW/HTTP
22
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
23
+ Requires-Python: >=3.9
24
+ Requires-Dist: requests>=2.25
25
+ Description-Content-Type: text/markdown
26
+
27
+ # quantumproxies — Python SDK for the QuantumProxies API
28
+
29
+ Scrape any page to clean Markdown, run structured Google/Bing/DuckDuckGo
30
+ searches, crawl and map whole sites, run 74 ready-made Collectors
31
+ (Amazon, Google Maps, LinkedIn jobs, app stores…), build datasets from a
32
+ plain-language prompt — everything through QuantumProxies' residential
33
+ proxy network with real-browser TLS fingerprints. Pay per successful call;
34
+ blocked pages cost nothing.
35
+
36
+ ```bash
37
+ pip install quantumproxies
38
+ ```
39
+
40
+ ## Quickstart
41
+
42
+ ```python
43
+ from quantumproxies import QuantumProxies
44
+
45
+ client = QuantumProxies() # reads QUANTUMPROXIES_API_KEY from the environment
46
+
47
+ page = client.scrape("https://example.com")
48
+ print(page["title"], page["engine"])
49
+ print(page["content"]) # the page as clean Markdown
50
+ ```
51
+
52
+ Get a free API key at [quantumproxies.io](https://quantumproxies.io) — every
53
+ account includes free monthly usage, no card required. Set it once:
54
+
55
+ ```bash
56
+ export QUANTUMPROXIES_API_KEY=qp_live_your_key_here
57
+ ```
58
+
59
+ ## What's in the box
60
+
61
+ Every REST endpoint, one method each — responses come back with the API
62
+ envelope already unwrapped:
63
+
64
+ ```python
65
+ # Structured search — 3 engines, 17 verticals, SerpApi-compatible JSON
66
+ serp = client.search("best espresso machine", country="us", num=20)
67
+ for r in serp["organic"]:
68
+ print(r["rank"], r["title"], r["link"])
69
+
70
+ # SERP → citation-ready Markdown context for an AI prompt
71
+ ctx = client.search_and_read("latest EU AI act status", top_n=3)
72
+
73
+ # Map a site's URLs in seconds (sitemaps + homepage links)
74
+ urls = client.map("https://stripe.com", search="/blog")
75
+
76
+ # Async crawl — wait=True polls until it settles and returns the pages
77
+ job = client.crawl("https://docs.python.org", limit=30, depth=2, wait=True)
78
+
79
+ # Batch-scrape known URLs
80
+ job = client.batch(["https://a.example", "https://b.example"], wait=True)
81
+
82
+ # CSS/AI extraction on one page
83
+ data = client.scrape(
84
+ "https://books.toscrape.com",
85
+ extract={"titles": {"selector": "h3 a", "attr": "title", "all": True}},
86
+ )
87
+
88
+ # Learn selectors once with an LLM, then scrape the same layout for free
89
+ parser = client.generate_parser(
90
+ "https://news.ycombinator.com",
91
+ fields={"titles": "every story title, as a list"},
92
+ )
93
+
94
+ # 74 ready-made Collectors — semantic input instead of URLs
95
+ places = client.run_collector(
96
+ "google_maps_places", keyword="dentist", location="Austin, TX", max_results=20
97
+ )
98
+
99
+ # Dataset from a prompt (validated rows, budget-capped)
100
+ ds = client.create_dataset(
101
+ "coffee roasters in Portland with email and phone",
102
+ limits={"max_rows": 50, "max_cost_usd": 2},
103
+ wait=True,
104
+ )
105
+
106
+ # Proxy endpoints of every type — residential, mobile, datacenter, ISP, IPv6
107
+ plans = client.list_proxies(active=True)
108
+ proxies = client.generate_proxies(plans["proxies"][0]["orderId"], country="us", quantity=5)
109
+ ```
110
+
111
+ ## Errors and retries
112
+
113
+ Failures raise `QuantumProxiesError` with `.status`, `.message` and
114
+ `.payload`. Connection errors and HTTP 429 are retried with backoff;
115
+ billable calls are never re-sent after a response was received, so nothing
116
+ gets double-billed behind your back.
117
+
118
+ ```python
119
+ from quantumproxies import QuantumProxies, QuantumProxiesError
120
+
121
+ try:
122
+ QuantumProxies(api_key="qp_live_wrong").scrape("https://example.com")
123
+ except QuantumProxiesError as err:
124
+ print(err.status, err.message)
125
+ ```
126
+
127
+ ## Also available
128
+
129
+ - **MCP server** for Claude, Cursor and any MCP client:
130
+ [`npx -y quantumproxies-mcp`](https://www.npmjs.com/package/quantumproxies-mcp)
131
+ exposes the same 25 tools to AI agents.
132
+ - **REST reference**: [quantumproxies.io/data-api](https://quantumproxies.io/data-api)
133
+
134
+ MIT licensed.
@@ -0,0 +1,108 @@
1
+ # quantumproxies — Python SDK for the QuantumProxies API
2
+
3
+ Scrape any page to clean Markdown, run structured Google/Bing/DuckDuckGo
4
+ searches, crawl and map whole sites, run 74 ready-made Collectors
5
+ (Amazon, Google Maps, LinkedIn jobs, app stores…), build datasets from a
6
+ plain-language prompt — everything through QuantumProxies' residential
7
+ proxy network with real-browser TLS fingerprints. Pay per successful call;
8
+ blocked pages cost nothing.
9
+
10
+ ```bash
11
+ pip install quantumproxies
12
+ ```
13
+
14
+ ## Quickstart
15
+
16
+ ```python
17
+ from quantumproxies import QuantumProxies
18
+
19
+ client = QuantumProxies() # reads QUANTUMPROXIES_API_KEY from the environment
20
+
21
+ page = client.scrape("https://example.com")
22
+ print(page["title"], page["engine"])
23
+ print(page["content"]) # the page as clean Markdown
24
+ ```
25
+
26
+ Get a free API key at [quantumproxies.io](https://quantumproxies.io) — every
27
+ account includes free monthly usage, no card required. Set it once:
28
+
29
+ ```bash
30
+ export QUANTUMPROXIES_API_KEY=qp_live_your_key_here
31
+ ```
32
+
33
+ ## What's in the box
34
+
35
+ Every REST endpoint, one method each — responses come back with the API
36
+ envelope already unwrapped:
37
+
38
+ ```python
39
+ # Structured search — 3 engines, 17 verticals, SerpApi-compatible JSON
40
+ serp = client.search("best espresso machine", country="us", num=20)
41
+ for r in serp["organic"]:
42
+ print(r["rank"], r["title"], r["link"])
43
+
44
+ # SERP → citation-ready Markdown context for an AI prompt
45
+ ctx = client.search_and_read("latest EU AI act status", top_n=3)
46
+
47
+ # Map a site's URLs in seconds (sitemaps + homepage links)
48
+ urls = client.map("https://stripe.com", search="/blog")
49
+
50
+ # Async crawl — wait=True polls until it settles and returns the pages
51
+ job = client.crawl("https://docs.python.org", limit=30, depth=2, wait=True)
52
+
53
+ # Batch-scrape known URLs
54
+ job = client.batch(["https://a.example", "https://b.example"], wait=True)
55
+
56
+ # CSS/AI extraction on one page
57
+ data = client.scrape(
58
+ "https://books.toscrape.com",
59
+ extract={"titles": {"selector": "h3 a", "attr": "title", "all": True}},
60
+ )
61
+
62
+ # Learn selectors once with an LLM, then scrape the same layout for free
63
+ parser = client.generate_parser(
64
+ "https://news.ycombinator.com",
65
+ fields={"titles": "every story title, as a list"},
66
+ )
67
+
68
+ # 74 ready-made Collectors — semantic input instead of URLs
69
+ places = client.run_collector(
70
+ "google_maps_places", keyword="dentist", location="Austin, TX", max_results=20
71
+ )
72
+
73
+ # Dataset from a prompt (validated rows, budget-capped)
74
+ ds = client.create_dataset(
75
+ "coffee roasters in Portland with email and phone",
76
+ limits={"max_rows": 50, "max_cost_usd": 2},
77
+ wait=True,
78
+ )
79
+
80
+ # Proxy endpoints of every type — residential, mobile, datacenter, ISP, IPv6
81
+ plans = client.list_proxies(active=True)
82
+ proxies = client.generate_proxies(plans["proxies"][0]["orderId"], country="us", quantity=5)
83
+ ```
84
+
85
+ ## Errors and retries
86
+
87
+ Failures raise `QuantumProxiesError` with `.status`, `.message` and
88
+ `.payload`. Connection errors and HTTP 429 are retried with backoff;
89
+ billable calls are never re-sent after a response was received, so nothing
90
+ gets double-billed behind your back.
91
+
92
+ ```python
93
+ from quantumproxies import QuantumProxies, QuantumProxiesError
94
+
95
+ try:
96
+ QuantumProxies(api_key="qp_live_wrong").scrape("https://example.com")
97
+ except QuantumProxiesError as err:
98
+ print(err.status, err.message)
99
+ ```
100
+
101
+ ## Also available
102
+
103
+ - **MCP server** for Claude, Cursor and any MCP client:
104
+ [`npx -y quantumproxies-mcp`](https://www.npmjs.com/package/quantumproxies-mcp)
105
+ exposes the same 25 tools to AI agents.
106
+ - **REST reference**: [quantumproxies.io/data-api](https://quantumproxies.io/data-api)
107
+
108
+ MIT licensed.
@@ -0,0 +1,46 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "quantumproxies"
7
+ version = "0.1.0"
8
+ description = "Official Python SDK for the QuantumProxies API — web scraping to Markdown, SERP search, crawl/map, 74 ready-made Collectors and residential/mobile/datacenter proxy generation."
9
+ readme = "README.md"
10
+ license = { text = "MIT" }
11
+ authors = [{ name = "QuantumProxies", email = "support@quantumproxies.io" }]
12
+ requires-python = ">=3.9"
13
+ dependencies = ["requests>=2.25"]
14
+ keywords = [
15
+ "web-scraping",
16
+ "scraper",
17
+ "serp",
18
+ "proxy",
19
+ "residential-proxies",
20
+ "crawler",
21
+ "google-search",
22
+ "quantumproxies",
23
+ "data-extraction",
24
+ "ai-agents",
25
+ ]
26
+ classifiers = [
27
+ "Development Status :: 4 - Beta",
28
+ "Intended Audience :: Developers",
29
+ "License :: OSI Approved :: MIT License",
30
+ "Programming Language :: Python :: 3",
31
+ "Programming Language :: Python :: 3.9",
32
+ "Programming Language :: Python :: 3.10",
33
+ "Programming Language :: Python :: 3.11",
34
+ "Programming Language :: Python :: 3.12",
35
+ "Programming Language :: Python :: 3.13",
36
+ "Topic :: Internet :: WWW/HTTP",
37
+ "Topic :: Software Development :: Libraries :: Python Modules",
38
+ ]
39
+
40
+ [project.urls]
41
+ Homepage = "https://quantumproxies.io"
42
+ Documentation = "https://quantumproxies.io/data-api"
43
+ Repository = "https://github.com/quantumproxies/quantumproxies-python"
44
+
45
+ [tool.hatch.build.targets.wheel]
46
+ packages = ["src/quantumproxies"]
@@ -0,0 +1,32 @@
1
+ """Official Python SDK for the QuantumProxies API — scrape, search, map and
2
+ crawl the web through residential proxies, run the 74 ready-made Collectors,
3
+ build datasets from a prompt, and generate proxy endpoints of every type.
4
+
5
+ Quickstart::
6
+
7
+ from quantumproxies import QuantumProxies
8
+
9
+ client = QuantumProxies() # reads QUANTUMPROXIES_API_KEY
10
+ page = client.scrape("https://example.com")
11
+ print(page["content"])
12
+ """
13
+
14
+ __version__ = "0.1.0"
15
+
16
+ from ._client import ( # noqa: E402
17
+ DEFAULT_BASE_URL,
18
+ QuantumProxies,
19
+ QuantumProxiesError,
20
+ )
21
+
22
+ Client = QuantumProxies
23
+ ApiError = QuantumProxiesError
24
+
25
+ __all__ = [
26
+ "QuantumProxies",
27
+ "QuantumProxiesError",
28
+ "Client",
29
+ "ApiError",
30
+ "DEFAULT_BASE_URL",
31
+ "__version__",
32
+ ]
@@ -0,0 +1,559 @@
1
+ """HTTP client for the QuantumProxies API.
2
+
3
+ Thin by design: every method maps 1:1 onto a REST endpoint, request bodies are
4
+ plain dicts, responses are the API envelope's ``payload`` already unwrapped.
5
+ The only sugar is ``wait=True`` on the async job starters, which polls the
6
+ job's status endpoint until it settles.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import os
12
+ import time
13
+ from typing import Any, Dict, List, Optional, Union
14
+
15
+ import requests
16
+
17
+ from . import __version__
18
+
19
+ DEFAULT_BASE_URL = "https://api.quantumproxies.io/v1"
20
+ ENV_API_KEY = "QUANTUMPROXIES_API_KEY"
21
+ ENV_API_BASE = "QUANTUMPROXIES_API_BASE"
22
+
23
+ # Job statuses after which polling stops (jobs report e.g. done/completed on
24
+ # success, failed/error on failure; partial results are still in the payload).
25
+ _TERMINAL_STATUSES = {"done", "completed", "failed", "error", "cancelled"}
26
+
27
+ _JOB_ID_FIELDS = ("jobId", "job_id", "run_id", "runId", "id")
28
+
29
+
30
+ class QuantumProxiesError(Exception):
31
+ """Raised when the API answers with the error envelope or transport fails.
32
+
33
+ Attributes:
34
+ message: human-readable reason from the API's ``message`` field.
35
+ status: HTTP status code, when a response was received.
36
+ payload: the error envelope's ``payload``, when present.
37
+ """
38
+
39
+ def __init__(self, message: str, status: Optional[int] = None, payload: Any = None):
40
+ super().__init__(message)
41
+ self.message = message
42
+ self.status = status
43
+ self.payload = payload
44
+
45
+
46
+ class QuantumProxies:
47
+ """Client for the QuantumProxies API.
48
+
49
+ Args:
50
+ api_key: your ``qp_live_…`` key; falls back to ``QUANTUMPROXIES_API_KEY``.
51
+ base_url: API base, default ``https://api.quantumproxies.io/v1``
52
+ (override with ``QUANTUMPROXIES_API_BASE``).
53
+ timeout: per-request timeout in seconds (default 90, matching the
54
+ slowest render-tier scrapes).
55
+ max_retries: extra attempts on connection errors and HTTP 429. Billable
56
+ calls are never retried after a response was received, so a
57
+ successful-but-lost call is not silently re-billed.
58
+ session: optional ``requests.Session`` to reuse connections.
59
+ """
60
+
61
+ def __init__(
62
+ self,
63
+ api_key: Optional[str] = None,
64
+ *,
65
+ base_url: Optional[str] = None,
66
+ timeout: float = 90.0,
67
+ max_retries: int = 2,
68
+ session: Optional[requests.Session] = None,
69
+ ):
70
+ self.api_key = api_key or os.environ.get(ENV_API_KEY, "")
71
+ if not self.api_key:
72
+ raise QuantumProxiesError(
73
+ f"No API key. Pass api_key=... or set {ENV_API_KEY}. "
74
+ "Get a free key at https://quantumproxies.io"
75
+ )
76
+ self.base_url = (base_url or os.environ.get(ENV_API_BASE, DEFAULT_BASE_URL)).rstrip("/")
77
+ self.timeout = timeout
78
+ self.max_retries = max_retries
79
+ self._session = session or requests.Session()
80
+ self._session.headers.update(
81
+ {
82
+ "Authorization": f"Bearer {self.api_key}",
83
+ "User-Agent": f"quantumproxies-python/{__version__}",
84
+ }
85
+ )
86
+
87
+ # ── transport ───────────────────────────────────────────────────────────
88
+
89
+ def _request(
90
+ self,
91
+ method: str,
92
+ path: str,
93
+ body: Optional[Dict[str, Any]] = None,
94
+ params: Optional[Dict[str, Any]] = None,
95
+ ) -> Any:
96
+ url = f"{self.base_url}{path}"
97
+ attempt = 0
98
+ while True:
99
+ try:
100
+ resp = self._session.request(
101
+ method,
102
+ url,
103
+ json=body if method != "GET" else None,
104
+ params=params,
105
+ timeout=self.timeout,
106
+ )
107
+ except requests.RequestException as exc:
108
+ if attempt < self.max_retries:
109
+ attempt += 1
110
+ time.sleep(min(2**attempt, 8))
111
+ continue
112
+ raise QuantumProxiesError(f"Cannot reach the QuantumProxies API at {url}: {exc}") from exc
113
+
114
+ if resp.status_code == 429 and attempt < self.max_retries:
115
+ attempt += 1
116
+ retry_after = resp.headers.get("Retry-After")
117
+ time.sleep(float(retry_after) if retry_after else min(2**attempt, 8))
118
+ continue
119
+
120
+ ctype = resp.headers.get("content-type", "")
121
+ if "json" not in ctype:
122
+ # CSV exports and other text endpoints pass through verbatim.
123
+ if resp.ok:
124
+ return resp.text
125
+ raise QuantumProxiesError(
126
+ resp.text[:500] or f"Request failed (HTTP {resp.status_code})",
127
+ status=resp.status_code,
128
+ )
129
+
130
+ try:
131
+ doc = resp.json()
132
+ except ValueError:
133
+ raise QuantumProxiesError(
134
+ f"Non-JSON response (HTTP {resp.status_code})", status=resp.status_code
135
+ )
136
+
137
+ if not resp.ok or (isinstance(doc, dict) and doc.get("type") == "error"):
138
+ message = doc.get("message") if isinstance(doc, dict) else None
139
+ raise QuantumProxiesError(
140
+ message or f"Request failed (HTTP {resp.status_code})",
141
+ status=resp.status_code,
142
+ payload=doc.get("payload") if isinstance(doc, dict) else None,
143
+ )
144
+ if isinstance(doc, dict) and "payload" in doc:
145
+ return doc["payload"]
146
+ return doc
147
+
148
+ @staticmethod
149
+ def _clean(mapping: Dict[str, Any]) -> Dict[str, Any]:
150
+ return {k: v for k, v in mapping.items() if v is not None}
151
+
152
+ # ── jobs: shared polling ───────────────────────────────────────────────
153
+
154
+ @staticmethod
155
+ def _job_id(payload: Any) -> Optional[str]:
156
+ if not isinstance(payload, dict):
157
+ return None
158
+ for field in _JOB_ID_FIELDS:
159
+ value = payload.get(field)
160
+ if isinstance(value, str) and value:
161
+ return value
162
+ return None
163
+
164
+ def _wait(self, poll, job_id: str, poll_interval: float, timeout: float, final=None) -> Any:
165
+ deadline = time.monotonic() + timeout
166
+ while True:
167
+ payload = poll(job_id)
168
+ status = payload.get("status") if isinstance(payload, dict) else None
169
+ if status in _TERMINAL_STATUSES:
170
+ return final(job_id) if final is not None else payload
171
+ if time.monotonic() >= deadline:
172
+ raise QuantumProxiesError(
173
+ f"Job {job_id} still '{status}' after {timeout:.0f}s — poll it yourself later",
174
+ payload=payload,
175
+ )
176
+ time.sleep(poll_interval)
177
+
178
+ # ── scraping ────────────────────────────────────────────────────────────
179
+
180
+ def scrape(self, url: Optional[str] = None, **options: Any) -> Any:
181
+ """Scrape one page to Markdown/HTML/text through a residential proxy.
182
+
183
+ Accepts every parameter of ``POST /scraper/extract`` as keyword
184
+ arguments (``format``, ``content_mode``, ``engine``, ``render``,
185
+ ``country``, ``extract``, ``ai_prompt``, ``actions``, ``cookies``,
186
+ ``query``, ``max_tokens`` …). Two ergonomic aliases:
187
+
188
+ - ``preset_id="…"`` runs a stored parser preset.
189
+ - ``fetch_resource="regex"`` returns the first matching network
190
+ response's body instead of the page.
191
+ """
192
+ body = self._clean({"url": url, **options})
193
+ preset_id = body.pop("preset_id", None)
194
+ if preset_id:
195
+ body["presetId"] = preset_id
196
+ fetch_resource = body.pop("fetch_resource", None)
197
+ if fetch_resource:
198
+ actions = list(body.get("actions") or [])
199
+ actions.append({"fetchResource": {"pattern": fetch_resource}})
200
+ body["actions"] = actions
201
+ return self._request("POST", "/scraper/extract", body)
202
+
203
+ def batch(
204
+ self,
205
+ urls: List[str],
206
+ *,
207
+ wait: bool = False,
208
+ poll_interval: float = 3.0,
209
+ wait_timeout: float = 600.0,
210
+ **options: Any,
211
+ ) -> Any:
212
+ """Scrape many URLs asynchronously (``POST /scraper/batch``).
213
+
214
+ Returns the job payload (with its job id). With ``wait=True`` polls
215
+ until the job settles and returns the finished job including content.
216
+ """
217
+ payload = self._request("POST", "/scraper/batch", self._clean({"urls": urls, **options}))
218
+ if not wait:
219
+ return payload
220
+ job_id = self._job_id(payload)
221
+ if not job_id:
222
+ return payload
223
+ return self._wait(
224
+ lambda j: self.batch_status(j),
225
+ job_id,
226
+ poll_interval,
227
+ wait_timeout,
228
+ final=lambda j: self.batch_status(j, include_content=True),
229
+ )
230
+
231
+ def batch_status(
232
+ self,
233
+ job_id: str,
234
+ *,
235
+ since: Optional[int] = None,
236
+ include_content: bool = False,
237
+ ) -> Any:
238
+ """Poll a batch job; pass the previous ``nextCursor`` as ``since``."""
239
+ params = self._clean({"since": since, "include_content": "true" if include_content else None})
240
+ return self._request("GET", f"/scraper/batch/{job_id}", params=params)
241
+
242
+ def crawl(
243
+ self,
244
+ url: str,
245
+ *,
246
+ wait: bool = False,
247
+ poll_interval: float = 3.0,
248
+ wait_timeout: float = 600.0,
249
+ **options: Any,
250
+ ) -> Any:
251
+ """Start an async BFS crawl (``POST /scraper/crawl``); options include
252
+ ``limit``, ``depth``, ``content_mode``, ``include``, ``exclude``,
253
+ ``country``. With ``wait=True`` returns the finished job with content.
254
+ """
255
+ payload = self._request("POST", "/scraper/crawl", self._clean({"url": url, **options}))
256
+ if not wait:
257
+ return payload
258
+ job_id = self._job_id(payload)
259
+ if not job_id:
260
+ return payload
261
+ return self._wait(
262
+ lambda j: self.crawl_status(j),
263
+ job_id,
264
+ poll_interval,
265
+ wait_timeout,
266
+ final=lambda j: self.crawl_status(j, include_content=True),
267
+ )
268
+
269
+ def crawl_status(
270
+ self,
271
+ job_id: str,
272
+ *,
273
+ since: Optional[int] = None,
274
+ include_content: bool = False,
275
+ ) -> Any:
276
+ """Poll a crawl job; content is omitted unless ``include_content=True``."""
277
+ params = {"include_content": "true" if include_content else "false"}
278
+ if since is not None:
279
+ params["since"] = since
280
+ return self._request("GET", f"/scraper/crawl/{job_id}", params=params)
281
+
282
+ def map(self, url: str, **options: Any) -> Any:
283
+ """Discover a site's URLs fast (``POST /scraper/map``): sitemaps +
284
+ homepage links, with ``search`` substring filter and ``group_by="path"``.
285
+ """
286
+ return self._request("POST", "/scraper/map", self._clean({"url": url, **options}))
287
+
288
+ def seo_audit(self, url: str, **options: Any) -> Any:
289
+ """Audit a URL twice — no-JS bot view and rendered view — plus the diff
290
+ (``POST /scraper/seo-audit``). ``no_render=True`` for the cheap pass.
291
+ """
292
+ return self._request("POST", "/scraper/seo-audit", self._clean({"url": url, **options}))
293
+
294
+ # ── search ──────────────────────────────────────────────────────────────
295
+
296
+ def search(self, query: Optional[str] = None, **options: Any) -> Any:
297
+ """Structured Google/Bing/DuckDuckGo search (``POST /scraper/serp``).
298
+
299
+ Every SERP API parameter is accepted as a keyword argument:
300
+ ``engine``, ``search_type`` (17 verticals), ``country``, ``lang``,
301
+ ``num``, ``page``, ``location``, ``uule``, ``device`` … ID-addressed
302
+ verticals (``place_details``, ``product``, ``flights``, ``reviews``)
303
+ take their ids the same way.
304
+ """
305
+ return self._request("POST", "/scraper/serp", self._clean({"query": query, **options}))
306
+
307
+ def search_and_read(self, query: str, **options: Any) -> Any:
308
+ """Search the web and return the top pages as citation-ready Markdown
309
+ context (``POST /ai/search``). Options: ``top_n``, ``max_tokens``,
310
+ ``engine``, ``country``, ``lang``, ``fetch_content``.
311
+ """
312
+ return self._request("POST", "/ai/search", self._clean({"query": query, **options}))
313
+
314
+ def search_bulk(
315
+ self,
316
+ query: str,
317
+ *,
318
+ wait: bool = False,
319
+ poll_interval: float = 3.0,
320
+ wait_timeout: float = 600.0,
321
+ **options: Any,
322
+ ) -> Any:
323
+ """Paginate one query across many result pages asynchronously
324
+ (``POST /scraper/serp/bulk``). Options: ``max_pages``, ``engine``,
325
+ ``search_type``, ``country``, ``render``, ``webhook`` …
326
+ """
327
+ payload = self._request("POST", "/scraper/serp/bulk", self._clean({"query": query, **options}))
328
+ if not wait:
329
+ return payload
330
+ job_id = self._job_id(payload)
331
+ if not job_id:
332
+ return payload
333
+ return self._wait(lambda j: self.search_bulk_status(j), job_id, poll_interval, wait_timeout)
334
+
335
+ def search_bulk_status(self, job_id: str, *, since: Optional[int] = None) -> Any:
336
+ """Poll a bulk search job; pass the previous ``nextCursor`` as ``since``."""
337
+ return self._request("GET", f"/scraper/serp/bulk/{job_id}", params=self._clean({"since": since}))
338
+
339
+ # ── parsers ─────────────────────────────────────────────────────────────
340
+
341
+ def generate_parser(self, url: Optional[str] = None, **options: Any) -> Any:
342
+ """Learn CSS selectors for a page layout once with an LLM
343
+ (``POST /scraper/parser/generate``), then reuse them for free via
344
+ ``scrape(extract=…)`` or a saved preset. Options: ``fields``,
345
+ ``prompt``, ``html``, ``render``, ``country``.
346
+ """
347
+ return self._request("POST", "/scraper/parser/generate", self._clean({"url": url, **options}))
348
+
349
+ def save_parser_preset(
350
+ self,
351
+ name: str,
352
+ parser: Dict[str, Any],
353
+ *,
354
+ source_url: Optional[str] = None,
355
+ fields: Optional[Dict[str, str]] = None,
356
+ render: Optional[bool] = None,
357
+ auto_heal: Optional[bool] = None,
358
+ ) -> Any:
359
+ """Store a parser under a name (``POST /scraper/parser/presets``);
360
+ give ``source_url`` so it can self-heal when the site changes.
361
+ """
362
+ return self._request(
363
+ "POST",
364
+ "/scraper/parser/presets",
365
+ self._clean(
366
+ {
367
+ "name": name,
368
+ "parser": parser,
369
+ "sourceUrl": source_url,
370
+ "fields": fields,
371
+ "render": render,
372
+ "autoHeal": auto_heal,
373
+ }
374
+ ),
375
+ )
376
+
377
+ def list_parser_presets(self) -> Any:
378
+ """List stored parser presets with version, health and changelog."""
379
+ return self._request("GET", "/scraper/parser/presets")
380
+
381
+ def parser_preset_stats(self, preset_id: str) -> Any:
382
+ """Per-field success rate and decay verdict for one preset."""
383
+ return self._request("GET", f"/scraper/parser/presets/{preset_id}/stats")
384
+
385
+ def heal_parser_preset(self, preset_id: str, *, force: bool = False) -> Any:
386
+ """Regenerate a preset's selectors now; adopted only if they extract
387
+ more than the current ones (a no-better heal is not billed).
388
+ """
389
+ return self._request("POST", f"/scraper/parser/presets/{preset_id}/heal", {"force": force})
390
+
391
+ # ── datasets ────────────────────────────────────────────────────────────
392
+
393
+ def create_dataset(
394
+ self,
395
+ prompt: str,
396
+ *,
397
+ wait: bool = False,
398
+ poll_interval: float = 5.0,
399
+ wait_timeout: float = 1800.0,
400
+ **options: Any,
401
+ ) -> Any:
402
+ """Build a validated dataset from a plain-language prompt
403
+ (``POST /scraper/datasets``). Options: ``columns``, ``country``,
404
+ ``sources``, ``limits`` (incl. ``max_cost_usd``), ``webhook``.
405
+ """
406
+ payload = self._request("POST", "/scraper/datasets", self._clean({"prompt": prompt, **options}))
407
+ if not wait:
408
+ return payload
409
+ job_id = self._job_id(payload)
410
+ if not job_id:
411
+ return payload
412
+ return self._wait(
413
+ lambda j: self.dataset_status(j, mode="summary"),
414
+ job_id,
415
+ poll_interval,
416
+ wait_timeout,
417
+ final=lambda j: self.dataset_status(j),
418
+ )
419
+
420
+ def dataset_status(
421
+ self,
422
+ job_id: str,
423
+ *,
424
+ since: Optional[int] = None,
425
+ mode: Optional[str] = None,
426
+ ) -> Any:
427
+ """Poll a dataset job. ``mode="summary"`` omits rows (light poll);
428
+ a completed job includes signed CSV/JSON download URLs.
429
+ """
430
+ return self._request(
431
+ "GET", f"/scraper/datasets/{job_id}", params=self._clean({"since": since, "mode": mode})
432
+ )
433
+
434
+ # ── collectors ──────────────────────────────────────────────────────────
435
+
436
+ def list_collectors(self, category: Optional[str] = None) -> Any:
437
+ """Catalog of the 74 ready-made Collectors with schemas, examples,
438
+ prices and health (``GET /scraper/collectors``, free).
439
+ """
440
+ return self._request("GET", "/scraper/collectors", params=self._clean({"category": category}))
441
+
442
+ def run_collector(
443
+ self,
444
+ slug: str,
445
+ input: Optional[Dict[str, Any]] = None,
446
+ *,
447
+ wait: bool = False,
448
+ poll_interval: float = 3.0,
449
+ wait_timeout: float = 900.0,
450
+ **input_kwargs: Any,
451
+ ) -> Any:
452
+ """Run a Collector by slug with its semantic input, e.g.::
453
+
454
+ client.run_collector("google_maps_places",
455
+ keyword="dentist", location="Austin, TX")
456
+
457
+ Short runs return rows inline; long runs return a ``run_id`` — with
458
+ ``wait=True`` the client polls it for you. Billed per delivered row.
459
+ """
460
+ body = dict(input or {})
461
+ body.update(input_kwargs)
462
+ payload = self._request("POST", f"/scraper/collectors/{slug}/run", body)
463
+ run_id = self._job_id(payload) if isinstance(payload, dict) else None
464
+ if wait and run_id and isinstance(payload, dict) and payload.get("status") not in _TERMINAL_STATUSES:
465
+ return self._wait(lambda r: self.collector_run_status(r), run_id, poll_interval, wait_timeout)
466
+ return payload
467
+
468
+ def collector_run_status(self, run_id: str, *, format: Optional[str] = None) -> Any:
469
+ """Fetch a Collector run: status, cost and rows. ``format="csv"``
470
+ returns the rows as CSV text.
471
+ """
472
+ return self._request(
473
+ "GET", f"/scraper/collectors/runs/{run_id}", params=self._clean({"format": format})
474
+ )
475
+
476
+ # ── proxies ─────────────────────────────────────────────────────────────
477
+
478
+ def list_proxies(
479
+ self,
480
+ *,
481
+ active: Optional[bool] = None,
482
+ plan_type: Optional[str] = None,
483
+ limit: Optional[int] = None,
484
+ offset: Optional[int] = None,
485
+ ) -> Any:
486
+ """List the account's proxy services of every type, with the
487
+ ``orderId`` to pass to :meth:`generate_proxies`.
488
+ """
489
+ return self._request(
490
+ "GET",
491
+ "/public/proxies",
492
+ params=self._clean(
493
+ {
494
+ "active": None if active is None else str(active).lower(),
495
+ "planType": plan_type,
496
+ "limit": limit,
497
+ "offset": offset,
498
+ }
499
+ ),
500
+ )
501
+
502
+ def generate_proxies(self, order_id: str, **options: Any) -> Any:
503
+ """Generate ready-to-use proxy endpoint strings from an active service
504
+ (``POST /public/proxies/generate``). Options: ``protocol``, ``format``,
505
+ ``quantity``, ``country``/``state``/``city``, ``rotation``,
506
+ ``sessionTime``, ``isp``, ``asn`` … The strings plug straight into any
507
+ HTTP client (``curl -x``, ``requests``' ``proxies=``).
508
+ """
509
+ return self._request(
510
+ "POST", "/public/proxies/generate", self._clean({"orderId": order_id, **options})
511
+ )
512
+
513
+ def proxy_locations(
514
+ self,
515
+ plan_type: str,
516
+ *,
517
+ level: str = "countries",
518
+ country: Optional[str] = None,
519
+ state: Optional[str] = None,
520
+ ) -> Any:
521
+ """Valid geo-targeting values for a plan type: ``countries`` (default),
522
+ ``states``, ``cities``, ``asns``, or ``tree`` for the full
523
+ country→region→city→ISP tree with slugs.
524
+ """
525
+ if level == "tree":
526
+ if plan_type in ("residentialpremium", "resiprivate"):
527
+ path = "/public/generator/residential-premium/targeting-options"
528
+ elif plan_type in ("mobile", "mobile_v2"):
529
+ path = "/public/generator/mobile/targeting-options"
530
+ else:
531
+ path = "/public/generator/datacenter/targeting-options"
532
+ return self._request("GET", path)
533
+ segment = "cities" if level == "cities" else level
534
+ return self._request(
535
+ "GET",
536
+ f"/public/geo/{segment}",
537
+ params=self._clean({"planType": plan_type, "country": country, "state": state}),
538
+ )
539
+
540
+ def whitelist_ip(
541
+ self,
542
+ action: str,
543
+ order_id: str,
544
+ ip: Optional[str] = None,
545
+ **options: Any,
546
+ ) -> Any:
547
+ """Manage IP-auth whitelisting on a proxy service: ``action`` is
548
+ ``"list"``, ``"add"`` or ``"remove"`` (mobile ``add`` also accepts
549
+ ``ports_count``, ``protocol``, ``country``, ``sticky``, ``ttl`` …).
550
+ """
551
+ if action == "list":
552
+ return self._request("GET", "/public/proxies/whitelist-ip", params={"orderId": order_id})
553
+ if not ip:
554
+ raise QuantumProxiesError("`ip` is required for add/remove")
555
+ if action == "remove":
556
+ return self._request("DELETE", "/public/proxies/whitelist-ip", {"orderId": order_id, "ip": ip})
557
+ return self._request(
558
+ "POST", "/public/proxies/whitelist-ip", self._clean({"orderId": order_id, "ip": ip, **options})
559
+ )
File without changes