scrapeunblocker 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,47 @@
1
+ """Official Python client for the ScrapeUnblocker web scraping API.
2
+
3
+ Basic usage::
4
+
5
+ from scrapeunblocker import Client
6
+
7
+ su = Client(api_key="su_live_...") # or set SCRAPEUNBLOCKER_KEY
8
+ html = su.get_page_source("https://example.com")
9
+ product = su.get_parsed("https://www.amazon.com/dp/B08N5WRWNW")
10
+
11
+ See https://developers.scrapeunblocker.com for the full API reference.
12
+ """
13
+
14
+ from ._async_client import AsyncClient
15
+ from ._client import Client
16
+ from .exceptions import (
17
+ APIError,
18
+ AuthenticationError,
19
+ BlockedError,
20
+ ConnectionError,
21
+ InvalidRequestError,
22
+ RateLimitError,
23
+ ScrapeTimeoutError,
24
+ ScrapeUnblockerError,
25
+ ServerError,
26
+ UpstreamOutageError,
27
+ )
28
+ from .models import PageResult, ParsedPage
29
+ from .version import __version__
30
+
31
+ __all__ = [
32
+ "Client",
33
+ "AsyncClient",
34
+ "ParsedPage",
35
+ "PageResult",
36
+ "ScrapeUnblockerError",
37
+ "APIError",
38
+ "AuthenticationError",
39
+ "InvalidRequestError",
40
+ "BlockedError",
41
+ "RateLimitError",
42
+ "UpstreamOutageError",
43
+ "ServerError",
44
+ "ScrapeTimeoutError",
45
+ "ConnectionError",
46
+ "__version__",
47
+ ]
@@ -0,0 +1,201 @@
1
+ """Asynchronous ScrapeUnblocker client (mirror of the sync client)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ from typing import Any, Dict, Optional
7
+
8
+ import httpx
9
+
10
+ from . import _base
11
+ from .exceptions import ConnectionError as SUConnectionError
12
+ from .exceptions import ScrapeTimeoutError
13
+ from .models import PageResult, ParsedPage
14
+ from .version import __version__
15
+
16
+
17
+ class _AsyncSkyscannerNamespace:
18
+ """Skyscanner plugin endpoints (flights, hotels, car hire)."""
19
+
20
+ def __init__(self, client: "AsyncClient"):
21
+ self._c = client
22
+
23
+ async def flight_locations(self, q: str, **params: Any) -> Any:
24
+ return await self._c._post_json("/flights/skyscanner-locations", q=q, **params)
25
+
26
+ async def flights(self, **params: Any) -> Any:
27
+ return await self._c._post_json("/flights/skyscanner-quotes", **params)
28
+
29
+ async def hotel_locations(self, q: str, **params: Any) -> Any:
30
+ return await self._c._post_json("/hotels/skyscanner-locations", q=q, **params)
31
+
32
+ async def hotels(self, **params: Any) -> Any:
33
+ return await self._c._post_json("/hotels/skyscanner-quotes", **params)
34
+
35
+ async def carhire_locations(self, q: str, **params: Any) -> Any:
36
+ return await self._c._post_json("/carhire/skyscanner-locations", q=q, **params)
37
+
38
+ async def carhire(self, **params: Any) -> Any:
39
+ return await self._c._post_json("/carhire/skyscanner-quotes", **params)
40
+
41
+
42
+ class AsyncClient:
43
+ """Async client for the ScrapeUnblocker API.
44
+
45
+ Mirrors :class:`~scrapeunblocker.Client`; every method is a coroutine.
46
+
47
+ Example:
48
+ >>> import asyncio
49
+ >>> from scrapeunblocker import AsyncClient
50
+ >>> async def main():
51
+ ... async with AsyncClient(api_key="su_live_...") as su:
52
+ ... html = await su.get_page_source("https://example.com")
53
+ >>> asyncio.run(main())
54
+ """
55
+
56
+ def __init__(
57
+ self,
58
+ api_key: Optional[str] = None,
59
+ *,
60
+ base_url: str = _base.DEFAULT_BASE_URL,
61
+ timeout: float = _base.DEFAULT_TIMEOUT,
62
+ max_retries: int = _base.DEFAULT_MAX_RETRIES,
63
+ ):
64
+ self._api_key = _base.resolve_api_key(api_key)
65
+ self._max_retries = max_retries
66
+ self._http = httpx.AsyncClient(
67
+ base_url=base_url.rstrip("/"),
68
+ timeout=timeout,
69
+ headers={
70
+ _base.API_KEY_HEADER: self._api_key,
71
+ "User-Agent": _base.user_agent(__version__),
72
+ "Accept": "*/*",
73
+ },
74
+ )
75
+ self.skyscanner = _AsyncSkyscannerNamespace(self)
76
+
77
+ async def __aenter__(self) -> "AsyncClient":
78
+ return self
79
+
80
+ async def __aexit__(self, *exc: Any) -> None:
81
+ await self.aclose()
82
+
83
+ async def aclose(self) -> None:
84
+ await self._http.aclose()
85
+
86
+ async def _request(self, path: str, params: Dict[str, Any]) -> httpx.Response:
87
+ attempt = 0
88
+ while True:
89
+ try:
90
+ response = await self._http.post(path, params=params)
91
+ except httpx.TimeoutException as exc:
92
+ raise ScrapeTimeoutError(str(exc)) from exc
93
+ except httpx.TransportError as exc:
94
+ if attempt < self._max_retries:
95
+ await self._sleep(attempt)
96
+ attempt += 1
97
+ continue
98
+ raise SUConnectionError(str(exc)) from exc
99
+
100
+ if _base.is_retryable(response.status_code) and attempt < self._max_retries:
101
+ await self._sleep(attempt)
102
+ attempt += 1
103
+ continue
104
+
105
+ _base.raise_for_status(response)
106
+ return response
107
+
108
+ @staticmethod
109
+ async def _sleep(attempt: int) -> None:
110
+ await asyncio.sleep(min(0.5 * (2 ** attempt), 8.0))
111
+
112
+ async def _post_json(self, path: str, **params: Any) -> Any:
113
+ response = await self._request(path, _base.build_params(**params))
114
+ return response.json()
115
+
116
+ async def get_page_source(
117
+ self,
118
+ url: str,
119
+ *,
120
+ proxy_country: Optional[str] = None,
121
+ time_sleep: Optional[int] = None,
122
+ method: Optional[str] = None,
123
+ value: Optional[str] = None,
124
+ method_timeout: Optional[int] = None,
125
+ ) -> str:
126
+ """Fetch a URL and return the fully rendered HTML."""
127
+ params = _base.build_params(
128
+ url=url,
129
+ proxy_country=proxy_country,
130
+ time_sleep=time_sleep,
131
+ method=method,
132
+ value=value,
133
+ method_timeout=method_timeout,
134
+ )
135
+ response = await self._request("/getPageSource", params)
136
+ return response.text
137
+
138
+ async def get_parsed(
139
+ self,
140
+ url: str,
141
+ *,
142
+ refresh_rules: bool = False,
143
+ rules_hint: Optional[str] = None,
144
+ proxy_country: Optional[str] = None,
145
+ time_sleep: Optional[int] = None,
146
+ ) -> ParsedPage:
147
+ """Fetch a URL and return structured JSON instead of HTML."""
148
+ params = _base.build_params(
149
+ url=url,
150
+ parsed_data=True,
151
+ refresh_rules=refresh_rules or None,
152
+ rules_hint=rules_hint,
153
+ proxy_country=proxy_country,
154
+ time_sleep=time_sleep,
155
+ )
156
+ response = await self._request("/getPageSource", params)
157
+ return ParsedPage.from_response(response.json())
158
+
159
+ async def get_page_with_cookies(
160
+ self,
161
+ url: str,
162
+ *,
163
+ proxy_country: Optional[str] = None,
164
+ time_sleep: Optional[int] = None,
165
+ ) -> PageResult:
166
+ """Fetch a URL and also return the cookies and proxy that served it."""
167
+ params = _base.build_params(
168
+ url=url,
169
+ get_cookies=True,
170
+ proxy_country=proxy_country,
171
+ time_sleep=time_sleep,
172
+ )
173
+ response = await self._request("/getPageSource", params)
174
+ return PageResult.from_response(response.json())
175
+
176
+ async def serp(
177
+ self,
178
+ keyword: str,
179
+ *,
180
+ proxy_country: Optional[str] = None,
181
+ pages_to_check: int = 1,
182
+ wait_after_load: int = 0,
183
+ captcha_pause: int = 0,
184
+ ) -> Any:
185
+ """Run a Google search and return the parsed SERP as JSON."""
186
+ return await self._post_json(
187
+ "/serp",
188
+ keyword=keyword,
189
+ proxy_country=proxy_country,
190
+ pages_to_check=pages_to_check,
191
+ wait_after_load=wait_after_load or None,
192
+ captcha_pause=captcha_pause or None,
193
+ )
194
+
195
+ async def get_image(
196
+ self, url: str, *, proxy_country: Optional[str] = None
197
+ ) -> bytes:
198
+ """Fetch an image URL through the bypass chain and return its bytes."""
199
+ params = _base.build_params(url=url, proxy_country=proxy_country)
200
+ response = await self._request("/getImage", params)
201
+ return response.content
@@ -0,0 +1,109 @@
1
+ """Shared logic between the sync and async clients.
2
+
3
+ Keeps request construction, response decoding, error mapping and retry
4
+ decisions in one place so the two client classes stay thin and cannot drift
5
+ apart.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import os
11
+ from typing import Any, Dict, Optional
12
+
13
+ import httpx
14
+
15
+ from .exceptions import (
16
+ APIError,
17
+ AuthenticationError,
18
+ BlockedError,
19
+ InvalidRequestError,
20
+ RateLimitError,
21
+ ScrapeUnblockerError,
22
+ ServerError,
23
+ UpstreamOutageError,
24
+ )
25
+
26
+ DEFAULT_BASE_URL = "https://api.scrapeunblocker.com"
27
+ DEFAULT_TIMEOUT = 180.0
28
+ DEFAULT_MAX_RETRIES = 2
29
+ API_KEY_HEADER = "x-scrapeunblocker-key"
30
+
31
+ # Status codes worth retrying: transient upstream outage, rate limiting, and
32
+ # generic 5xx. A 400/403 is deterministic - retrying only wastes time.
33
+ _RETRYABLE_STATUS = frozenset({429, 502, 503, 504})
34
+
35
+
36
+ def resolve_api_key(api_key: Optional[str]) -> str:
37
+ """Return the API key, falling back to the SCRAPEUNBLOCKER_KEY env var."""
38
+ key = api_key or os.environ.get("SCRAPEUNBLOCKER_KEY")
39
+ if not key:
40
+ raise ScrapeUnblockerError(
41
+ "No API key provided. Pass api_key=... or set the "
42
+ "SCRAPEUNBLOCKER_KEY environment variable. Get your key at "
43
+ "https://app.scrapeunblocker.com"
44
+ )
45
+ return key
46
+
47
+
48
+ def build_params(**kwargs: Any) -> Dict[str, Any]:
49
+ """Drop None values so optional params are simply omitted from the query."""
50
+ return {k: v for k, v in kwargs.items() if v is not None}
51
+
52
+
53
+ def user_agent(version: str) -> str:
54
+ return f"scrapeunblocker-python/{version}"
55
+
56
+
57
+ def is_retryable(status_code: int) -> bool:
58
+ return status_code in _RETRYABLE_STATUS
59
+
60
+
61
+ def raise_for_status(response: httpx.Response) -> None:
62
+ """Map a non-2xx response to the matching typed exception."""
63
+ if response.is_success:
64
+ return
65
+
66
+ status = response.status_code
67
+ try:
68
+ body = response.text
69
+ except Exception: # pragma: no cover - defensive
70
+ body = None
71
+
72
+ message = _message_for(status, body)
73
+
74
+ if status == 400:
75
+ raise InvalidRequestError(message, status_code=status, body=body)
76
+ if status == 401:
77
+ raise AuthenticationError(message, status_code=status, body=body)
78
+ if status == 403:
79
+ raise BlockedError(message, status_code=status, body=body)
80
+ if status == 429:
81
+ raise RateLimitError(message, status_code=status, body=body)
82
+ if status == 503:
83
+ raise UpstreamOutageError(message, status_code=status, body=body)
84
+ if status >= 500:
85
+ raise ServerError(message, status_code=status, body=body)
86
+ raise APIError(message, status_code=status, body=body)
87
+
88
+
89
+ def _message_for(status: int, body: Optional[str]) -> str:
90
+ snippet = (body or "").strip().replace("\n", " ")
91
+ if len(snippet) > 200:
92
+ snippet = snippet[:200] + "..."
93
+ base = {
94
+ 400: "Invalid request (bad URL or unsupported scheme)",
95
+ 401: "Authentication failed - check your API key",
96
+ 403: "Target blocked by bot protection on every bypass path",
97
+ 429: "Rate limited - too many requests",
98
+ 503: "Upstream origin returned a server-side outage page",
99
+ }.get(status, f"API returned HTTP {status}")
100
+ return f"{base}: {snippet}" if snippet else base
101
+
102
+
103
+ def decode_page_response(
104
+ response: httpx.Response, *, want_json: bool
105
+ ) -> Any:
106
+ """Return parsed JSON when the caller asked for it, else the text body."""
107
+ if want_json:
108
+ return response.json()
109
+ return response.text
@@ -0,0 +1,248 @@
1
+ """Synchronous ScrapeUnblocker client."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import time
6
+ from typing import Any, Dict, Optional
7
+
8
+ import httpx
9
+
10
+ from . import _base
11
+ from .exceptions import ConnectionError as SUConnectionError
12
+ from .exceptions import ScrapeTimeoutError
13
+ from .models import PageResult, ParsedPage
14
+ from .version import __version__
15
+
16
+
17
+ class _SkyscannerNamespace:
18
+ """Skyscanner plugin endpoints (flights, hotels, car hire)."""
19
+
20
+ def __init__(self, client: "Client"):
21
+ self._c = client
22
+
23
+ def flight_locations(self, q: str, **params: Any) -> Any:
24
+ return self._c._post_json("/flights/skyscanner-locations", q=q, **params)
25
+
26
+ def flights(self, **params: Any) -> Any:
27
+ return self._c._post_json("/flights/skyscanner-quotes", **params)
28
+
29
+ def hotel_locations(self, q: str, **params: Any) -> Any:
30
+ return self._c._post_json("/hotels/skyscanner-locations", q=q, **params)
31
+
32
+ def hotels(self, **params: Any) -> Any:
33
+ return self._c._post_json("/hotels/skyscanner-quotes", **params)
34
+
35
+ def carhire_locations(self, q: str, **params: Any) -> Any:
36
+ return self._c._post_json("/carhire/skyscanner-locations", q=q, **params)
37
+
38
+ def carhire(self, **params: Any) -> Any:
39
+ return self._c._post_json("/carhire/skyscanner-quotes", **params)
40
+
41
+
42
+ class Client:
43
+ """Client for the ScrapeUnblocker API.
44
+
45
+ Args:
46
+ api_key: Your API key. Falls back to the ``SCRAPEUNBLOCKER_KEY``
47
+ environment variable when omitted.
48
+ base_url: Override the API host (rarely needed).
49
+ timeout: Per-request timeout in seconds. Rendering heavy, protected
50
+ pages can take a while, so the default is generous (180s).
51
+ max_retries: How many times to retry transient failures (429/5xx and
52
+ network errors) with exponential backoff.
53
+
54
+ Example:
55
+ >>> from scrapeunblocker import Client
56
+ >>> su = Client(api_key="su_live_...")
57
+ >>> html = su.get_page_source("https://example.com")
58
+ """
59
+
60
+ def __init__(
61
+ self,
62
+ api_key: Optional[str] = None,
63
+ *,
64
+ base_url: str = _base.DEFAULT_BASE_URL,
65
+ timeout: float = _base.DEFAULT_TIMEOUT,
66
+ max_retries: int = _base.DEFAULT_MAX_RETRIES,
67
+ ):
68
+ self._api_key = _base.resolve_api_key(api_key)
69
+ self._max_retries = max_retries
70
+ self._http = httpx.Client(
71
+ base_url=base_url.rstrip("/"),
72
+ timeout=timeout,
73
+ headers={
74
+ _base.API_KEY_HEADER: self._api_key,
75
+ "User-Agent": _base.user_agent(__version__),
76
+ "Accept": "*/*",
77
+ },
78
+ )
79
+ self.skyscanner = _SkyscannerNamespace(self)
80
+
81
+ # -- context manager ------------------------------------------------
82
+ def __enter__(self) -> "Client":
83
+ return self
84
+
85
+ def __exit__(self, *exc: Any) -> None:
86
+ self.close()
87
+
88
+ def close(self) -> None:
89
+ self._http.close()
90
+
91
+ # -- low-level request with retry -----------------------------------
92
+ def _request(self, path: str, params: Dict[str, Any]) -> httpx.Response:
93
+ attempt = 0
94
+ while True:
95
+ try:
96
+ response = self._http.post(path, params=params)
97
+ except httpx.TimeoutException as exc:
98
+ raise ScrapeTimeoutError(str(exc)) from exc
99
+ except httpx.TransportError as exc:
100
+ if attempt < self._max_retries:
101
+ self._sleep(attempt)
102
+ attempt += 1
103
+ continue
104
+ raise SUConnectionError(str(exc)) from exc
105
+
106
+ if _base.is_retryable(response.status_code) and attempt < self._max_retries:
107
+ self._sleep(attempt)
108
+ attempt += 1
109
+ continue
110
+
111
+ _base.raise_for_status(response)
112
+ return response
113
+
114
+ @staticmethod
115
+ def _sleep(attempt: int) -> None:
116
+ time.sleep(min(0.5 * (2 ** attempt), 8.0))
117
+
118
+ def _post_json(self, path: str, **params: Any) -> Any:
119
+ response = self._request(path, _base.build_params(**params))
120
+ return response.json()
121
+
122
+ # -- public API -----------------------------------------------------
123
+ def get_page_source(
124
+ self,
125
+ url: str,
126
+ *,
127
+ proxy_country: Optional[str] = None,
128
+ time_sleep: Optional[int] = None,
129
+ method: Optional[str] = None,
130
+ value: Optional[str] = None,
131
+ method_timeout: Optional[int] = None,
132
+ ) -> str:
133
+ """Fetch a URL and return the fully rendered HTML.
134
+
135
+ The page is loaded in a real browser behind the appropriate anti-bot
136
+ bypass chain, so JavaScript-heavy and protected sites come back
137
+ rendered.
138
+
139
+ Args:
140
+ url: The page to fetch.
141
+ proxy_country: ISO country code to route through (e.g. ``"US"``).
142
+ time_sleep: Extra seconds to wait after load before capture.
143
+ method: Advanced render-wait method (``"css"``, ``"js"``, ...).
144
+ value: The selector/expression paired with ``method``.
145
+ method_timeout: Cap in seconds for the render-wait method.
146
+
147
+ Returns:
148
+ The page HTML as a string.
149
+ """
150
+ params = _base.build_params(
151
+ url=url,
152
+ proxy_country=proxy_country,
153
+ time_sleep=time_sleep,
154
+ method=method,
155
+ value=value,
156
+ method_timeout=method_timeout,
157
+ )
158
+ response = self._request("/getPageSource", params)
159
+ return response.text
160
+
161
+ def get_parsed(
162
+ self,
163
+ url: str,
164
+ *,
165
+ refresh_rules: bool = False,
166
+ rules_hint: Optional[str] = None,
167
+ proxy_country: Optional[str] = None,
168
+ time_sleep: Optional[int] = None,
169
+ ) -> ParsedPage:
170
+ """Fetch a URL and return structured JSON instead of HTML.
171
+
172
+ The API extracts fields using Schema.org, ``__NEXT_DATA__`` or
173
+ AI-generated rules - no per-site parser to maintain.
174
+
175
+ Args:
176
+ url: The page to parse.
177
+ refresh_rules: Force-regenerate the cached extraction rules for
178
+ this domain (use when a parse comes back clearly wrong).
179
+ rules_hint: Free-text steer for regeneration, e.g. ``"price is
180
+ missing"``.
181
+ proxy_country: ISO country code to route through.
182
+ time_sleep: Extra seconds to wait after load before capture.
183
+
184
+ Returns:
185
+ A :class:`ParsedPage` with ``page_type``, ``source`` and ``data``.
186
+ """
187
+ params = _base.build_params(
188
+ url=url,
189
+ parsed_data=True,
190
+ refresh_rules=refresh_rules or None,
191
+ rules_hint=rules_hint,
192
+ proxy_country=proxy_country,
193
+ time_sleep=time_sleep,
194
+ )
195
+ response = self._request("/getPageSource", params)
196
+ return ParsedPage.from_response(response.json())
197
+
198
+ def get_page_with_cookies(
199
+ self,
200
+ url: str,
201
+ *,
202
+ proxy_country: Optional[str] = None,
203
+ time_sleep: Optional[int] = None,
204
+ ) -> PageResult:
205
+ """Fetch a URL and also return the cookies and proxy that served it."""
206
+ params = _base.build_params(
207
+ url=url,
208
+ get_cookies=True,
209
+ proxy_country=proxy_country,
210
+ time_sleep=time_sleep,
211
+ )
212
+ response = self._request("/getPageSource", params)
213
+ return PageResult.from_response(response.json())
214
+
215
+ def serp(
216
+ self,
217
+ keyword: str,
218
+ *,
219
+ proxy_country: Optional[str] = None,
220
+ pages_to_check: int = 1,
221
+ wait_after_load: int = 0,
222
+ captcha_pause: int = 0,
223
+ ) -> Any:
224
+ """Run a Google search and return the parsed SERP as JSON.
225
+
226
+ Args:
227
+ keyword: The search query.
228
+ proxy_country: ISO country code to search from.
229
+ pages_to_check: How many result pages to fetch (1-10).
230
+ wait_after_load: Seconds to wait after the page loads.
231
+ captcha_pause: Seconds to pause if a captcha appears.
232
+ """
233
+ return self._post_json(
234
+ "/serp",
235
+ keyword=keyword,
236
+ proxy_country=proxy_country,
237
+ pages_to_check=pages_to_check,
238
+ wait_after_load=wait_after_load or None,
239
+ captcha_pause=captcha_pause or None,
240
+ )
241
+
242
+ def get_image(
243
+ self, url: str, *, proxy_country: Optional[str] = None
244
+ ) -> bytes:
245
+ """Fetch an image URL through the bypass chain and return its bytes."""
246
+ params = _base.build_params(url=url, proxy_country=proxy_country)
247
+ response = self._request("/getImage", params)
248
+ return response.content
@@ -0,0 +1,70 @@
1
+ """Exception hierarchy for the ScrapeUnblocker client.
2
+
3
+ Every error raised by the client derives from :class:`ScrapeUnblockerError`,
4
+ so ``except ScrapeUnblockerError`` catches everything. API responses map to
5
+ typed subclasses by status code, letting callers react to a hard block
6
+ differently than a transient upstream outage without parsing status codes by
7
+ hand.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from typing import Optional
13
+
14
+
15
+ class ScrapeUnblockerError(Exception):
16
+ """Base class for every error raised by this library."""
17
+
18
+
19
+ class APIError(ScrapeUnblockerError):
20
+ """An error response returned by the ScrapeUnblocker API.
21
+
22
+ Attributes:
23
+ status_code: The HTTP status code of the response.
24
+ body: The raw response body (text), useful for debugging.
25
+ """
26
+
27
+ def __init__(self, message: str, *, status_code: int, body: Optional[str] = None):
28
+ super().__init__(message)
29
+ self.status_code = status_code
30
+ self.body = body
31
+
32
+
33
+ class AuthenticationError(APIError):
34
+ """The API key is missing, malformed, or not recognised (HTTP 401)."""
35
+
36
+
37
+ class InvalidRequestError(APIError):
38
+ """The request was rejected as invalid, e.g. a malformed URL (HTTP 400)."""
39
+
40
+
41
+ class BlockedError(APIError):
42
+ """The target site blocked every available bypass path (HTTP 403).
43
+
44
+ This is the target's anti-bot protection winning, not a problem with your
45
+ request. Blocked calls are not billed.
46
+ """
47
+
48
+
49
+ class RateLimitError(APIError):
50
+ """Too many requests against your account in a short window (HTTP 429)."""
51
+
52
+
53
+ class UpstreamOutageError(APIError):
54
+ """The origin site returned a server-side outage page (HTTP 503).
55
+
56
+ This means the *target* is down, not ScrapeUnblocker. Retrying later
57
+ usually succeeds.
58
+ """
59
+
60
+
61
+ class ServerError(APIError):
62
+ """ScrapeUnblocker returned an unexpected 5xx error."""
63
+
64
+
65
+ class ScrapeTimeoutError(ScrapeUnblockerError):
66
+ """The request did not complete within the configured timeout."""
67
+
68
+
69
+ class ConnectionError(ScrapeUnblockerError):
70
+ """The client could not reach the ScrapeUnblocker API."""
@@ -0,0 +1,52 @@
1
+ """Lightweight return types for structured responses."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from typing import Any, Dict, List, Optional
7
+
8
+
9
+ @dataclass
10
+ class ParsedPage:
11
+ """Structured data extracted from a page (``parsed_data=True``).
12
+
13
+ Attributes:
14
+ page_type: What the API classified the page as, e.g. ``"product"``.
15
+ source: How the data was extracted (Schema.org, __NEXT_DATA__, AI rules).
16
+ data: The extracted fields as a plain dict.
17
+ raw: The full JSON payload as returned by the API.
18
+ """
19
+
20
+ page_type: Optional[str]
21
+ source: Optional[str]
22
+ data: Any
23
+ raw: Dict[str, Any]
24
+
25
+ @classmethod
26
+ def from_response(cls, payload: Dict[str, Any]) -> "ParsedPage":
27
+ inner = payload.get("data", payload) or {}
28
+ return cls(
29
+ page_type=inner.get("page_type"),
30
+ source=inner.get("source"),
31
+ data=inner.get("data"),
32
+ raw=payload,
33
+ )
34
+
35
+
36
+ @dataclass
37
+ class PageResult:
38
+ """HTML plus the cookies and proxy that served it (``get_cookies=True``)."""
39
+
40
+ html: Optional[str]
41
+ cookies: Any
42
+ proxy: Optional[str]
43
+ raw: Dict[str, Any]
44
+
45
+ @classmethod
46
+ def from_response(cls, payload: Dict[str, Any]) -> "PageResult":
47
+ return cls(
48
+ html=payload.get("html") or payload.get("page_source") or payload.get("content"),
49
+ cookies=payload.get("cookies"),
50
+ proxy=payload.get("proxy") or payload.get("proxy_address"),
51
+ raw=payload,
52
+ )
File without changes
@@ -0,0 +1 @@
1
+ __version__ = "0.1.0"
@@ -0,0 +1,208 @@
1
+ Metadata-Version: 2.4
2
+ Name: scrapeunblocker
3
+ Version: 0.1.0
4
+ Summary: Official Python client for the ScrapeUnblocker web scraping API - JavaScript-rendered pages that bypass Cloudflare, DataDome, PerimeterX, Akamai and more.
5
+ Project-URL: Homepage, https://scrapeunblocker.com
6
+ Project-URL: Documentation, https://developers.scrapeunblocker.com
7
+ Project-URL: Source, https://github.com/ScrapeUnblocker/scrapeunblocker-python
8
+ Project-URL: Changelog, https://github.com/ScrapeUnblocker/scrapeunblocker-python/blob/main/CHANGELOG.md
9
+ Author-email: ScrapeUnblocker <support@scrapeunblocker.com>
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: anti-bot,captcha,cloudflare,crawler,datadome,proxy,scraping api,web scraping
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.8
19
+ Classifier: Programming Language :: Python :: 3.9
20
+ Classifier: Programming Language :: Python :: 3.10
21
+ Classifier: Programming Language :: Python :: 3.11
22
+ Classifier: Programming Language :: Python :: 3.12
23
+ Classifier: Programming Language :: Python :: 3.13
24
+ Classifier: Topic :: Internet :: WWW/HTTP
25
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
26
+ Classifier: Typing :: Typed
27
+ Requires-Python: >=3.8
28
+ Requires-Dist: httpx>=0.23
29
+ Provides-Extra: dev
30
+ Requires-Dist: mypy>=1.0; extra == 'dev'
31
+ Requires-Dist: pytest-asyncio>=0.21; extra == 'dev'
32
+ Requires-Dist: pytest>=7; extra == 'dev'
33
+ Requires-Dist: respx>=0.20; extra == 'dev'
34
+ Description-Content-Type: text/markdown
35
+
36
+ # ScrapeUnblocker Python client
37
+
38
+ Official Python client for the [ScrapeUnblocker](https://scrapeunblocker.com) web scraping API.
39
+
40
+ Every request is fully JavaScript-rendered in a real browser and routed through premium proxies, so it bypasses Cloudflare, DataDome, PerimeterX, Akamai, Kasada and similar anti-bot systems - from one simple call. You are only billed for successful requests.
41
+
42
+ - **Highest success rate on the market** (95%+ on live production traffic)
43
+ - **Rendered HTML or parsed JSON** - no per-site parsers to maintain
44
+ - Sync **and** async clients, fully type-hinted
45
+
46
+ ## Install
47
+
48
+ ```bash
49
+ pip install scrapeunblocker
50
+ ```
51
+
52
+ Requires Python 3.8+.
53
+
54
+ ## Quickstart
55
+
56
+ ```python
57
+ from scrapeunblocker import Client
58
+
59
+ su = Client(api_key="su_live_...") # or set the SCRAPEUNBLOCKER_KEY env var
60
+
61
+ # Rendered HTML for any URL
62
+ html = su.get_page_source("https://example.com")
63
+
64
+ # Structured JSON instead of HTML (products, listings, search results, ...)
65
+ product = su.get_parsed("https://www.amazon.com/dp/B08N5WRWNW")
66
+ print(product.page_type) # "product"
67
+ print(product.data) # {...}
68
+ ```
69
+
70
+ Get your API key at [app.scrapeunblocker.com](https://app.scrapeunblocker.com). The free trial does not require a credit card.
71
+
72
+ ## Authentication
73
+
74
+ Pass the key directly, or set an environment variable and omit it:
75
+
76
+ ```bash
77
+ export SCRAPEUNBLOCKER_KEY="su_live_..."
78
+ ```
79
+
80
+ ```python
81
+ from scrapeunblocker import Client
82
+ su = Client() # reads SCRAPEUNBLOCKER_KEY
83
+ ```
84
+
85
+ ## Fetch rendered HTML
86
+
87
+ ```python
88
+ html = su.get_page_source(
89
+ "https://www.nordstrom.com/browse/women/clothing/dresses",
90
+ proxy_country="US", # route through a specific country
91
+ time_sleep=3, # wait extra seconds after load
92
+ )
93
+ ```
94
+
95
+ ## Get parsed JSON
96
+
97
+ Pass a URL and get back structured data extracted via Schema.org, `__NEXT_DATA__` or AI-generated rules:
98
+
99
+ ```python
100
+ result = su.get_parsed("https://www.walmart.com/ip/12345")
101
+ print(result.page_type) # e.g. "product"
102
+ print(result.source) # how it was extracted
103
+ print(result.data) # the fields
104
+
105
+ # If a parse ever comes back wrong, force a fresh set of rules:
106
+ result = su.get_parsed(url, refresh_rules=True, rules_hint="price is missing")
107
+ ```
108
+
109
+ ## Google search (SERP)
110
+
111
+ ```python
112
+ serp = su.serp("web scraping api", pages_to_check=2, proxy_country="US")
113
+ ```
114
+
115
+ ## Cookies and the serving proxy
116
+
117
+ ```python
118
+ page = su.get_page_with_cookies("https://example.com")
119
+ print(page.html, page.cookies, page.proxy)
120
+ ```
121
+
122
+ ## Images
123
+
124
+ ```python
125
+ data = su.get_image("https://example.com/photo.jpg")
126
+ open("photo.jpg", "wb").write(data)
127
+ ```
128
+
129
+ ## Skyscanner plugins
130
+
131
+ ```python
132
+ # Resolve a place name to entity IDs, then search
133
+ locs = su.skyscanner.flight_locations("London")
134
+ flights = su.skyscanner.flights(
135
+ origin="London", dest="New York",
136
+ depart_date="2026-09-01", adults=1, currency="USD",
137
+ )
138
+
139
+ hotels = su.skyscanner.hotels(destination="Madrid", checkin="2026-09-01", checkout="2026-09-03")
140
+ cars = su.skyscanner.carhire(pickup="Madrid", pickup_datetime="2026-09-01T10:00", dropoff_datetime="2026-09-03T10:00")
141
+ ```
142
+
143
+ ## Async
144
+
145
+ Every method has an async twin on `AsyncClient`:
146
+
147
+ ```python
148
+ import asyncio
149
+ from scrapeunblocker import AsyncClient
150
+
151
+ async def main():
152
+ async with AsyncClient(api_key="su_live_...") as su:
153
+ html = await su.get_page_source("https://example.com")
154
+
155
+ asyncio.run(main())
156
+ ```
157
+
158
+ ## Error handling
159
+
160
+ Non-2xx responses raise typed exceptions, all subclasses of `ScrapeUnblockerError`:
161
+
162
+ ```python
163
+ from scrapeunblocker import Client, BlockedError, RateLimitError, UpstreamOutageError
164
+
165
+ su = Client()
166
+ try:
167
+ html = su.get_page_source("https://example.com")
168
+ except BlockedError:
169
+ ... # 403: the target blocked every bypass path (not billed)
170
+ except RateLimitError:
171
+ ... # 429: slow down
172
+ except UpstreamOutageError:
173
+ ... # 503: the target site itself is down - retry later
174
+ ```
175
+
176
+ | Exception | Status | Meaning |
177
+ |---|---|---|
178
+ | `InvalidRequestError` | 400 | Bad URL or unsupported scheme |
179
+ | `AuthenticationError` | 401 | Missing or invalid API key |
180
+ | `BlockedError` | 403 | Blocked by bot protection on every path |
181
+ | `RateLimitError` | 429 | Too many requests |
182
+ | `UpstreamOutageError` | 503 | The target origin is down |
183
+ | `ServerError` | 5xx | Unexpected server error |
184
+ | `ScrapeTimeoutError` | - | Request exceeded the timeout |
185
+ | `ConnectionError` | - | Could not reach the API |
186
+
187
+ Transient failures (429, 502, 503, 504 and network errors) are retried automatically with exponential backoff; tune with `Client(max_retries=...)`.
188
+
189
+ ## Configuration
190
+
191
+ ```python
192
+ Client(
193
+ api_key=None, # or SCRAPEUNBLOCKER_KEY env var
194
+ base_url="https://api.scrapeunblocker.com",
195
+ timeout=180.0, # seconds; protected pages can be slow
196
+ max_retries=2,
197
+ )
198
+ ```
199
+
200
+ ## Links
201
+
202
+ - Documentation: https://developers.scrapeunblocker.com
203
+ - Website: https://scrapeunblocker.com
204
+ - Dashboard: https://app.scrapeunblocker.com
205
+
206
+ ## License
207
+
208
+ MIT
@@ -0,0 +1,12 @@
1
+ scrapeunblocker/__init__.py,sha256=JicnXv9l6hOuXPyyhDOIcPw9q1iIEILS3t5X99B7Smk,1123
2
+ scrapeunblocker/_async_client.py,sha256=jbT5bn6vYgRSXQ0PwHK-YD_ro8mH6bOzX3ledBjCaSU,6824
3
+ scrapeunblocker/_base.py,sha256=hs8c06Wynf_48kiXKeyJbZZy0IpG02KtqziYZ7sSkic,3496
4
+ scrapeunblocker/_client.py,sha256=HC-lNzwytLnI_Xl9uAkB5IDW0QCEddCQVWnWt-A0iz8,8670
5
+ scrapeunblocker/exceptions.py,sha256=bDVgiVxtAWQNt3hmkt9_WCN_iucbVxPI1eyYMPB4cjE,2084
6
+ scrapeunblocker/models.py,sha256=DZqdwHs6llqd0kTVue9wWqSiL5-m_4LXm4Jlx8ffV4c,1531
7
+ scrapeunblocker/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
8
+ scrapeunblocker/version.py,sha256=kUR5RAFc7HCeiqdlX36dZOHkUI5wI6V_43RpEcD8b-0,22
9
+ scrapeunblocker-0.1.0.dist-info/METADATA,sha256=VESjtiBYUhpcDdgJk6Wr0U4rq6kor0LI1gmh6niwISU,6591
10
+ scrapeunblocker-0.1.0.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
11
+ scrapeunblocker-0.1.0.dist-info/licenses/LICENSE,sha256=-Qs4w8IjJt3RWr3it4xjyous5qjVyQKBbmaxNP4Mlmo,1072
12
+ scrapeunblocker-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.31.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 ScrapeUnblocker
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.