scrapeunblocker 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scrapeunblocker/__init__.py +47 -0
- scrapeunblocker/_async_client.py +201 -0
- scrapeunblocker/_base.py +109 -0
- scrapeunblocker/_client.py +248 -0
- scrapeunblocker/exceptions.py +70 -0
- scrapeunblocker/models.py +52 -0
- scrapeunblocker/py.typed +0 -0
- scrapeunblocker/version.py +1 -0
- scrapeunblocker-0.1.0.dist-info/METADATA +208 -0
- scrapeunblocker-0.1.0.dist-info/RECORD +12 -0
- scrapeunblocker-0.1.0.dist-info/WHEEL +4 -0
- scrapeunblocker-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Official Python client for the ScrapeUnblocker web scraping API.
|
|
2
|
+
|
|
3
|
+
Basic usage::
|
|
4
|
+
|
|
5
|
+
from scrapeunblocker import Client
|
|
6
|
+
|
|
7
|
+
su = Client(api_key="su_live_...") # or set SCRAPEUNBLOCKER_KEY
|
|
8
|
+
html = su.get_page_source("https://example.com")
|
|
9
|
+
product = su.get_parsed("https://www.amazon.com/dp/B08N5WRWNW")
|
|
10
|
+
|
|
11
|
+
See https://developers.scrapeunblocker.com for the full API reference.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from ._async_client import AsyncClient
|
|
15
|
+
from ._client import Client
|
|
16
|
+
from .exceptions import (
|
|
17
|
+
APIError,
|
|
18
|
+
AuthenticationError,
|
|
19
|
+
BlockedError,
|
|
20
|
+
ConnectionError,
|
|
21
|
+
InvalidRequestError,
|
|
22
|
+
RateLimitError,
|
|
23
|
+
ScrapeTimeoutError,
|
|
24
|
+
ScrapeUnblockerError,
|
|
25
|
+
ServerError,
|
|
26
|
+
UpstreamOutageError,
|
|
27
|
+
)
|
|
28
|
+
from .models import PageResult, ParsedPage
|
|
29
|
+
from .version import __version__
|
|
30
|
+
|
|
31
|
+
__all__ = [
|
|
32
|
+
"Client",
|
|
33
|
+
"AsyncClient",
|
|
34
|
+
"ParsedPage",
|
|
35
|
+
"PageResult",
|
|
36
|
+
"ScrapeUnblockerError",
|
|
37
|
+
"APIError",
|
|
38
|
+
"AuthenticationError",
|
|
39
|
+
"InvalidRequestError",
|
|
40
|
+
"BlockedError",
|
|
41
|
+
"RateLimitError",
|
|
42
|
+
"UpstreamOutageError",
|
|
43
|
+
"ServerError",
|
|
44
|
+
"ScrapeTimeoutError",
|
|
45
|
+
"ConnectionError",
|
|
46
|
+
"__version__",
|
|
47
|
+
]
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""Asynchronous ScrapeUnblocker client (mirror of the sync client)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
from typing import Any, Dict, Optional
|
|
7
|
+
|
|
8
|
+
import httpx
|
|
9
|
+
|
|
10
|
+
from . import _base
|
|
11
|
+
from .exceptions import ConnectionError as SUConnectionError
|
|
12
|
+
from .exceptions import ScrapeTimeoutError
|
|
13
|
+
from .models import PageResult, ParsedPage
|
|
14
|
+
from .version import __version__
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class _AsyncSkyscannerNamespace:
|
|
18
|
+
"""Skyscanner plugin endpoints (flights, hotels, car hire)."""
|
|
19
|
+
|
|
20
|
+
def __init__(self, client: "AsyncClient"):
|
|
21
|
+
self._c = client
|
|
22
|
+
|
|
23
|
+
async def flight_locations(self, q: str, **params: Any) -> Any:
|
|
24
|
+
return await self._c._post_json("/flights/skyscanner-locations", q=q, **params)
|
|
25
|
+
|
|
26
|
+
async def flights(self, **params: Any) -> Any:
|
|
27
|
+
return await self._c._post_json("/flights/skyscanner-quotes", **params)
|
|
28
|
+
|
|
29
|
+
async def hotel_locations(self, q: str, **params: Any) -> Any:
|
|
30
|
+
return await self._c._post_json("/hotels/skyscanner-locations", q=q, **params)
|
|
31
|
+
|
|
32
|
+
async def hotels(self, **params: Any) -> Any:
|
|
33
|
+
return await self._c._post_json("/hotels/skyscanner-quotes", **params)
|
|
34
|
+
|
|
35
|
+
async def carhire_locations(self, q: str, **params: Any) -> Any:
|
|
36
|
+
return await self._c._post_json("/carhire/skyscanner-locations", q=q, **params)
|
|
37
|
+
|
|
38
|
+
async def carhire(self, **params: Any) -> Any:
|
|
39
|
+
return await self._c._post_json("/carhire/skyscanner-quotes", **params)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class AsyncClient:
|
|
43
|
+
"""Async client for the ScrapeUnblocker API.
|
|
44
|
+
|
|
45
|
+
Mirrors :class:`~scrapeunblocker.Client`; every method is a coroutine.
|
|
46
|
+
|
|
47
|
+
Example:
|
|
48
|
+
>>> import asyncio
|
|
49
|
+
>>> from scrapeunblocker import AsyncClient
|
|
50
|
+
>>> async def main():
|
|
51
|
+
... async with AsyncClient(api_key="su_live_...") as su:
|
|
52
|
+
... html = await su.get_page_source("https://example.com")
|
|
53
|
+
>>> asyncio.run(main())
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
def __init__(
|
|
57
|
+
self,
|
|
58
|
+
api_key: Optional[str] = None,
|
|
59
|
+
*,
|
|
60
|
+
base_url: str = _base.DEFAULT_BASE_URL,
|
|
61
|
+
timeout: float = _base.DEFAULT_TIMEOUT,
|
|
62
|
+
max_retries: int = _base.DEFAULT_MAX_RETRIES,
|
|
63
|
+
):
|
|
64
|
+
self._api_key = _base.resolve_api_key(api_key)
|
|
65
|
+
self._max_retries = max_retries
|
|
66
|
+
self._http = httpx.AsyncClient(
|
|
67
|
+
base_url=base_url.rstrip("/"),
|
|
68
|
+
timeout=timeout,
|
|
69
|
+
headers={
|
|
70
|
+
_base.API_KEY_HEADER: self._api_key,
|
|
71
|
+
"User-Agent": _base.user_agent(__version__),
|
|
72
|
+
"Accept": "*/*",
|
|
73
|
+
},
|
|
74
|
+
)
|
|
75
|
+
self.skyscanner = _AsyncSkyscannerNamespace(self)
|
|
76
|
+
|
|
77
|
+
async def __aenter__(self) -> "AsyncClient":
|
|
78
|
+
return self
|
|
79
|
+
|
|
80
|
+
async def __aexit__(self, *exc: Any) -> None:
|
|
81
|
+
await self.aclose()
|
|
82
|
+
|
|
83
|
+
async def aclose(self) -> None:
|
|
84
|
+
await self._http.aclose()
|
|
85
|
+
|
|
86
|
+
async def _request(self, path: str, params: Dict[str, Any]) -> httpx.Response:
|
|
87
|
+
attempt = 0
|
|
88
|
+
while True:
|
|
89
|
+
try:
|
|
90
|
+
response = await self._http.post(path, params=params)
|
|
91
|
+
except httpx.TimeoutException as exc:
|
|
92
|
+
raise ScrapeTimeoutError(str(exc)) from exc
|
|
93
|
+
except httpx.TransportError as exc:
|
|
94
|
+
if attempt < self._max_retries:
|
|
95
|
+
await self._sleep(attempt)
|
|
96
|
+
attempt += 1
|
|
97
|
+
continue
|
|
98
|
+
raise SUConnectionError(str(exc)) from exc
|
|
99
|
+
|
|
100
|
+
if _base.is_retryable(response.status_code) and attempt < self._max_retries:
|
|
101
|
+
await self._sleep(attempt)
|
|
102
|
+
attempt += 1
|
|
103
|
+
continue
|
|
104
|
+
|
|
105
|
+
_base.raise_for_status(response)
|
|
106
|
+
return response
|
|
107
|
+
|
|
108
|
+
@staticmethod
|
|
109
|
+
async def _sleep(attempt: int) -> None:
|
|
110
|
+
await asyncio.sleep(min(0.5 * (2 ** attempt), 8.0))
|
|
111
|
+
|
|
112
|
+
async def _post_json(self, path: str, **params: Any) -> Any:
|
|
113
|
+
response = await self._request(path, _base.build_params(**params))
|
|
114
|
+
return response.json()
|
|
115
|
+
|
|
116
|
+
async def get_page_source(
|
|
117
|
+
self,
|
|
118
|
+
url: str,
|
|
119
|
+
*,
|
|
120
|
+
proxy_country: Optional[str] = None,
|
|
121
|
+
time_sleep: Optional[int] = None,
|
|
122
|
+
method: Optional[str] = None,
|
|
123
|
+
value: Optional[str] = None,
|
|
124
|
+
method_timeout: Optional[int] = None,
|
|
125
|
+
) -> str:
|
|
126
|
+
"""Fetch a URL and return the fully rendered HTML."""
|
|
127
|
+
params = _base.build_params(
|
|
128
|
+
url=url,
|
|
129
|
+
proxy_country=proxy_country,
|
|
130
|
+
time_sleep=time_sleep,
|
|
131
|
+
method=method,
|
|
132
|
+
value=value,
|
|
133
|
+
method_timeout=method_timeout,
|
|
134
|
+
)
|
|
135
|
+
response = await self._request("/getPageSource", params)
|
|
136
|
+
return response.text
|
|
137
|
+
|
|
138
|
+
async def get_parsed(
|
|
139
|
+
self,
|
|
140
|
+
url: str,
|
|
141
|
+
*,
|
|
142
|
+
refresh_rules: bool = False,
|
|
143
|
+
rules_hint: Optional[str] = None,
|
|
144
|
+
proxy_country: Optional[str] = None,
|
|
145
|
+
time_sleep: Optional[int] = None,
|
|
146
|
+
) -> ParsedPage:
|
|
147
|
+
"""Fetch a URL and return structured JSON instead of HTML."""
|
|
148
|
+
params = _base.build_params(
|
|
149
|
+
url=url,
|
|
150
|
+
parsed_data=True,
|
|
151
|
+
refresh_rules=refresh_rules or None,
|
|
152
|
+
rules_hint=rules_hint,
|
|
153
|
+
proxy_country=proxy_country,
|
|
154
|
+
time_sleep=time_sleep,
|
|
155
|
+
)
|
|
156
|
+
response = await self._request("/getPageSource", params)
|
|
157
|
+
return ParsedPage.from_response(response.json())
|
|
158
|
+
|
|
159
|
+
async def get_page_with_cookies(
|
|
160
|
+
self,
|
|
161
|
+
url: str,
|
|
162
|
+
*,
|
|
163
|
+
proxy_country: Optional[str] = None,
|
|
164
|
+
time_sleep: Optional[int] = None,
|
|
165
|
+
) -> PageResult:
|
|
166
|
+
"""Fetch a URL and also return the cookies and proxy that served it."""
|
|
167
|
+
params = _base.build_params(
|
|
168
|
+
url=url,
|
|
169
|
+
get_cookies=True,
|
|
170
|
+
proxy_country=proxy_country,
|
|
171
|
+
time_sleep=time_sleep,
|
|
172
|
+
)
|
|
173
|
+
response = await self._request("/getPageSource", params)
|
|
174
|
+
return PageResult.from_response(response.json())
|
|
175
|
+
|
|
176
|
+
async def serp(
|
|
177
|
+
self,
|
|
178
|
+
keyword: str,
|
|
179
|
+
*,
|
|
180
|
+
proxy_country: Optional[str] = None,
|
|
181
|
+
pages_to_check: int = 1,
|
|
182
|
+
wait_after_load: int = 0,
|
|
183
|
+
captcha_pause: int = 0,
|
|
184
|
+
) -> Any:
|
|
185
|
+
"""Run a Google search and return the parsed SERP as JSON."""
|
|
186
|
+
return await self._post_json(
|
|
187
|
+
"/serp",
|
|
188
|
+
keyword=keyword,
|
|
189
|
+
proxy_country=proxy_country,
|
|
190
|
+
pages_to_check=pages_to_check,
|
|
191
|
+
wait_after_load=wait_after_load or None,
|
|
192
|
+
captcha_pause=captcha_pause or None,
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
async def get_image(
|
|
196
|
+
self, url: str, *, proxy_country: Optional[str] = None
|
|
197
|
+
) -> bytes:
|
|
198
|
+
"""Fetch an image URL through the bypass chain and return its bytes."""
|
|
199
|
+
params = _base.build_params(url=url, proxy_country=proxy_country)
|
|
200
|
+
response = await self._request("/getImage", params)
|
|
201
|
+
return response.content
|
scrapeunblocker/_base.py
ADDED
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Shared logic between the sync and async clients.
|
|
2
|
+
|
|
3
|
+
Keeps request construction, response decoding, error mapping and retry
|
|
4
|
+
decisions in one place so the two client classes stay thin and cannot drift
|
|
5
|
+
apart.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import os
|
|
11
|
+
from typing import Any, Dict, Optional
|
|
12
|
+
|
|
13
|
+
import httpx
|
|
14
|
+
|
|
15
|
+
from .exceptions import (
|
|
16
|
+
APIError,
|
|
17
|
+
AuthenticationError,
|
|
18
|
+
BlockedError,
|
|
19
|
+
InvalidRequestError,
|
|
20
|
+
RateLimitError,
|
|
21
|
+
ScrapeUnblockerError,
|
|
22
|
+
ServerError,
|
|
23
|
+
UpstreamOutageError,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
DEFAULT_BASE_URL = "https://api.scrapeunblocker.com"
|
|
27
|
+
DEFAULT_TIMEOUT = 180.0
|
|
28
|
+
DEFAULT_MAX_RETRIES = 2
|
|
29
|
+
API_KEY_HEADER = "x-scrapeunblocker-key"
|
|
30
|
+
|
|
31
|
+
# Status codes worth retrying: transient upstream outage, rate limiting, and
|
|
32
|
+
# generic 5xx. A 400/403 is deterministic - retrying only wastes time.
|
|
33
|
+
_RETRYABLE_STATUS = frozenset({429, 502, 503, 504})
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def resolve_api_key(api_key: Optional[str]) -> str:
|
|
37
|
+
"""Return the API key, falling back to the SCRAPEUNBLOCKER_KEY env var."""
|
|
38
|
+
key = api_key or os.environ.get("SCRAPEUNBLOCKER_KEY")
|
|
39
|
+
if not key:
|
|
40
|
+
raise ScrapeUnblockerError(
|
|
41
|
+
"No API key provided. Pass api_key=... or set the "
|
|
42
|
+
"SCRAPEUNBLOCKER_KEY environment variable. Get your key at "
|
|
43
|
+
"https://app.scrapeunblocker.com"
|
|
44
|
+
)
|
|
45
|
+
return key
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def build_params(**kwargs: Any) -> Dict[str, Any]:
|
|
49
|
+
"""Drop None values so optional params are simply omitted from the query."""
|
|
50
|
+
return {k: v for k, v in kwargs.items() if v is not None}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def user_agent(version: str) -> str:
|
|
54
|
+
return f"scrapeunblocker-python/{version}"
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def is_retryable(status_code: int) -> bool:
|
|
58
|
+
return status_code in _RETRYABLE_STATUS
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def raise_for_status(response: httpx.Response) -> None:
|
|
62
|
+
"""Map a non-2xx response to the matching typed exception."""
|
|
63
|
+
if response.is_success:
|
|
64
|
+
return
|
|
65
|
+
|
|
66
|
+
status = response.status_code
|
|
67
|
+
try:
|
|
68
|
+
body = response.text
|
|
69
|
+
except Exception: # pragma: no cover - defensive
|
|
70
|
+
body = None
|
|
71
|
+
|
|
72
|
+
message = _message_for(status, body)
|
|
73
|
+
|
|
74
|
+
if status == 400:
|
|
75
|
+
raise InvalidRequestError(message, status_code=status, body=body)
|
|
76
|
+
if status == 401:
|
|
77
|
+
raise AuthenticationError(message, status_code=status, body=body)
|
|
78
|
+
if status == 403:
|
|
79
|
+
raise BlockedError(message, status_code=status, body=body)
|
|
80
|
+
if status == 429:
|
|
81
|
+
raise RateLimitError(message, status_code=status, body=body)
|
|
82
|
+
if status == 503:
|
|
83
|
+
raise UpstreamOutageError(message, status_code=status, body=body)
|
|
84
|
+
if status >= 500:
|
|
85
|
+
raise ServerError(message, status_code=status, body=body)
|
|
86
|
+
raise APIError(message, status_code=status, body=body)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _message_for(status: int, body: Optional[str]) -> str:
|
|
90
|
+
snippet = (body or "").strip().replace("\n", " ")
|
|
91
|
+
if len(snippet) > 200:
|
|
92
|
+
snippet = snippet[:200] + "..."
|
|
93
|
+
base = {
|
|
94
|
+
400: "Invalid request (bad URL or unsupported scheme)",
|
|
95
|
+
401: "Authentication failed - check your API key",
|
|
96
|
+
403: "Target blocked by bot protection on every bypass path",
|
|
97
|
+
429: "Rate limited - too many requests",
|
|
98
|
+
503: "Upstream origin returned a server-side outage page",
|
|
99
|
+
}.get(status, f"API returned HTTP {status}")
|
|
100
|
+
return f"{base}: {snippet}" if snippet else base
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def decode_page_response(
|
|
104
|
+
response: httpx.Response, *, want_json: bool
|
|
105
|
+
) -> Any:
|
|
106
|
+
"""Return parsed JSON when the caller asked for it, else the text body."""
|
|
107
|
+
if want_json:
|
|
108
|
+
return response.json()
|
|
109
|
+
return response.text
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
"""Synchronous ScrapeUnblocker client."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import time
|
|
6
|
+
from typing import Any, Dict, Optional
|
|
7
|
+
|
|
8
|
+
import httpx
|
|
9
|
+
|
|
10
|
+
from . import _base
|
|
11
|
+
from .exceptions import ConnectionError as SUConnectionError
|
|
12
|
+
from .exceptions import ScrapeTimeoutError
|
|
13
|
+
from .models import PageResult, ParsedPage
|
|
14
|
+
from .version import __version__
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class _SkyscannerNamespace:
|
|
18
|
+
"""Skyscanner plugin endpoints (flights, hotels, car hire)."""
|
|
19
|
+
|
|
20
|
+
def __init__(self, client: "Client"):
|
|
21
|
+
self._c = client
|
|
22
|
+
|
|
23
|
+
def flight_locations(self, q: str, **params: Any) -> Any:
|
|
24
|
+
return self._c._post_json("/flights/skyscanner-locations", q=q, **params)
|
|
25
|
+
|
|
26
|
+
def flights(self, **params: Any) -> Any:
|
|
27
|
+
return self._c._post_json("/flights/skyscanner-quotes", **params)
|
|
28
|
+
|
|
29
|
+
def hotel_locations(self, q: str, **params: Any) -> Any:
|
|
30
|
+
return self._c._post_json("/hotels/skyscanner-locations", q=q, **params)
|
|
31
|
+
|
|
32
|
+
def hotels(self, **params: Any) -> Any:
|
|
33
|
+
return self._c._post_json("/hotels/skyscanner-quotes", **params)
|
|
34
|
+
|
|
35
|
+
def carhire_locations(self, q: str, **params: Any) -> Any:
|
|
36
|
+
return self._c._post_json("/carhire/skyscanner-locations", q=q, **params)
|
|
37
|
+
|
|
38
|
+
def carhire(self, **params: Any) -> Any:
|
|
39
|
+
return self._c._post_json("/carhire/skyscanner-quotes", **params)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class Client:
|
|
43
|
+
"""Client for the ScrapeUnblocker API.
|
|
44
|
+
|
|
45
|
+
Args:
|
|
46
|
+
api_key: Your API key. Falls back to the ``SCRAPEUNBLOCKER_KEY``
|
|
47
|
+
environment variable when omitted.
|
|
48
|
+
base_url: Override the API host (rarely needed).
|
|
49
|
+
timeout: Per-request timeout in seconds. Rendering heavy, protected
|
|
50
|
+
pages can take a while, so the default is generous (180s).
|
|
51
|
+
max_retries: How many times to retry transient failures (429/5xx and
|
|
52
|
+
network errors) with exponential backoff.
|
|
53
|
+
|
|
54
|
+
Example:
|
|
55
|
+
>>> from scrapeunblocker import Client
|
|
56
|
+
>>> su = Client(api_key="su_live_...")
|
|
57
|
+
>>> html = su.get_page_source("https://example.com")
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
def __init__(
|
|
61
|
+
self,
|
|
62
|
+
api_key: Optional[str] = None,
|
|
63
|
+
*,
|
|
64
|
+
base_url: str = _base.DEFAULT_BASE_URL,
|
|
65
|
+
timeout: float = _base.DEFAULT_TIMEOUT,
|
|
66
|
+
max_retries: int = _base.DEFAULT_MAX_RETRIES,
|
|
67
|
+
):
|
|
68
|
+
self._api_key = _base.resolve_api_key(api_key)
|
|
69
|
+
self._max_retries = max_retries
|
|
70
|
+
self._http = httpx.Client(
|
|
71
|
+
base_url=base_url.rstrip("/"),
|
|
72
|
+
timeout=timeout,
|
|
73
|
+
headers={
|
|
74
|
+
_base.API_KEY_HEADER: self._api_key,
|
|
75
|
+
"User-Agent": _base.user_agent(__version__),
|
|
76
|
+
"Accept": "*/*",
|
|
77
|
+
},
|
|
78
|
+
)
|
|
79
|
+
self.skyscanner = _SkyscannerNamespace(self)
|
|
80
|
+
|
|
81
|
+
# -- context manager ------------------------------------------------
|
|
82
|
+
def __enter__(self) -> "Client":
|
|
83
|
+
return self
|
|
84
|
+
|
|
85
|
+
def __exit__(self, *exc: Any) -> None:
|
|
86
|
+
self.close()
|
|
87
|
+
|
|
88
|
+
def close(self) -> None:
|
|
89
|
+
self._http.close()
|
|
90
|
+
|
|
91
|
+
# -- low-level request with retry -----------------------------------
|
|
92
|
+
def _request(self, path: str, params: Dict[str, Any]) -> httpx.Response:
|
|
93
|
+
attempt = 0
|
|
94
|
+
while True:
|
|
95
|
+
try:
|
|
96
|
+
response = self._http.post(path, params=params)
|
|
97
|
+
except httpx.TimeoutException as exc:
|
|
98
|
+
raise ScrapeTimeoutError(str(exc)) from exc
|
|
99
|
+
except httpx.TransportError as exc:
|
|
100
|
+
if attempt < self._max_retries:
|
|
101
|
+
self._sleep(attempt)
|
|
102
|
+
attempt += 1
|
|
103
|
+
continue
|
|
104
|
+
raise SUConnectionError(str(exc)) from exc
|
|
105
|
+
|
|
106
|
+
if _base.is_retryable(response.status_code) and attempt < self._max_retries:
|
|
107
|
+
self._sleep(attempt)
|
|
108
|
+
attempt += 1
|
|
109
|
+
continue
|
|
110
|
+
|
|
111
|
+
_base.raise_for_status(response)
|
|
112
|
+
return response
|
|
113
|
+
|
|
114
|
+
@staticmethod
|
|
115
|
+
def _sleep(attempt: int) -> None:
|
|
116
|
+
time.sleep(min(0.5 * (2 ** attempt), 8.0))
|
|
117
|
+
|
|
118
|
+
def _post_json(self, path: str, **params: Any) -> Any:
|
|
119
|
+
response = self._request(path, _base.build_params(**params))
|
|
120
|
+
return response.json()
|
|
121
|
+
|
|
122
|
+
# -- public API -----------------------------------------------------
|
|
123
|
+
def get_page_source(
|
|
124
|
+
self,
|
|
125
|
+
url: str,
|
|
126
|
+
*,
|
|
127
|
+
proxy_country: Optional[str] = None,
|
|
128
|
+
time_sleep: Optional[int] = None,
|
|
129
|
+
method: Optional[str] = None,
|
|
130
|
+
value: Optional[str] = None,
|
|
131
|
+
method_timeout: Optional[int] = None,
|
|
132
|
+
) -> str:
|
|
133
|
+
"""Fetch a URL and return the fully rendered HTML.
|
|
134
|
+
|
|
135
|
+
The page is loaded in a real browser behind the appropriate anti-bot
|
|
136
|
+
bypass chain, so JavaScript-heavy and protected sites come back
|
|
137
|
+
rendered.
|
|
138
|
+
|
|
139
|
+
Args:
|
|
140
|
+
url: The page to fetch.
|
|
141
|
+
proxy_country: ISO country code to route through (e.g. ``"US"``).
|
|
142
|
+
time_sleep: Extra seconds to wait after load before capture.
|
|
143
|
+
method: Advanced render-wait method (``"css"``, ``"js"``, ...).
|
|
144
|
+
value: The selector/expression paired with ``method``.
|
|
145
|
+
method_timeout: Cap in seconds for the render-wait method.
|
|
146
|
+
|
|
147
|
+
Returns:
|
|
148
|
+
The page HTML as a string.
|
|
149
|
+
"""
|
|
150
|
+
params = _base.build_params(
|
|
151
|
+
url=url,
|
|
152
|
+
proxy_country=proxy_country,
|
|
153
|
+
time_sleep=time_sleep,
|
|
154
|
+
method=method,
|
|
155
|
+
value=value,
|
|
156
|
+
method_timeout=method_timeout,
|
|
157
|
+
)
|
|
158
|
+
response = self._request("/getPageSource", params)
|
|
159
|
+
return response.text
|
|
160
|
+
|
|
161
|
+
def get_parsed(
|
|
162
|
+
self,
|
|
163
|
+
url: str,
|
|
164
|
+
*,
|
|
165
|
+
refresh_rules: bool = False,
|
|
166
|
+
rules_hint: Optional[str] = None,
|
|
167
|
+
proxy_country: Optional[str] = None,
|
|
168
|
+
time_sleep: Optional[int] = None,
|
|
169
|
+
) -> ParsedPage:
|
|
170
|
+
"""Fetch a URL and return structured JSON instead of HTML.
|
|
171
|
+
|
|
172
|
+
The API extracts fields using Schema.org, ``__NEXT_DATA__`` or
|
|
173
|
+
AI-generated rules - no per-site parser to maintain.
|
|
174
|
+
|
|
175
|
+
Args:
|
|
176
|
+
url: The page to parse.
|
|
177
|
+
refresh_rules: Force-regenerate the cached extraction rules for
|
|
178
|
+
this domain (use when a parse comes back clearly wrong).
|
|
179
|
+
rules_hint: Free-text steer for regeneration, e.g. ``"price is
|
|
180
|
+
missing"``.
|
|
181
|
+
proxy_country: ISO country code to route through.
|
|
182
|
+
time_sleep: Extra seconds to wait after load before capture.
|
|
183
|
+
|
|
184
|
+
Returns:
|
|
185
|
+
A :class:`ParsedPage` with ``page_type``, ``source`` and ``data``.
|
|
186
|
+
"""
|
|
187
|
+
params = _base.build_params(
|
|
188
|
+
url=url,
|
|
189
|
+
parsed_data=True,
|
|
190
|
+
refresh_rules=refresh_rules or None,
|
|
191
|
+
rules_hint=rules_hint,
|
|
192
|
+
proxy_country=proxy_country,
|
|
193
|
+
time_sleep=time_sleep,
|
|
194
|
+
)
|
|
195
|
+
response = self._request("/getPageSource", params)
|
|
196
|
+
return ParsedPage.from_response(response.json())
|
|
197
|
+
|
|
198
|
+
def get_page_with_cookies(
|
|
199
|
+
self,
|
|
200
|
+
url: str,
|
|
201
|
+
*,
|
|
202
|
+
proxy_country: Optional[str] = None,
|
|
203
|
+
time_sleep: Optional[int] = None,
|
|
204
|
+
) -> PageResult:
|
|
205
|
+
"""Fetch a URL and also return the cookies and proxy that served it."""
|
|
206
|
+
params = _base.build_params(
|
|
207
|
+
url=url,
|
|
208
|
+
get_cookies=True,
|
|
209
|
+
proxy_country=proxy_country,
|
|
210
|
+
time_sleep=time_sleep,
|
|
211
|
+
)
|
|
212
|
+
response = self._request("/getPageSource", params)
|
|
213
|
+
return PageResult.from_response(response.json())
|
|
214
|
+
|
|
215
|
+
def serp(
|
|
216
|
+
self,
|
|
217
|
+
keyword: str,
|
|
218
|
+
*,
|
|
219
|
+
proxy_country: Optional[str] = None,
|
|
220
|
+
pages_to_check: int = 1,
|
|
221
|
+
wait_after_load: int = 0,
|
|
222
|
+
captcha_pause: int = 0,
|
|
223
|
+
) -> Any:
|
|
224
|
+
"""Run a Google search and return the parsed SERP as JSON.
|
|
225
|
+
|
|
226
|
+
Args:
|
|
227
|
+
keyword: The search query.
|
|
228
|
+
proxy_country: ISO country code to search from.
|
|
229
|
+
pages_to_check: How many result pages to fetch (1-10).
|
|
230
|
+
wait_after_load: Seconds to wait after the page loads.
|
|
231
|
+
captcha_pause: Seconds to pause if a captcha appears.
|
|
232
|
+
"""
|
|
233
|
+
return self._post_json(
|
|
234
|
+
"/serp",
|
|
235
|
+
keyword=keyword,
|
|
236
|
+
proxy_country=proxy_country,
|
|
237
|
+
pages_to_check=pages_to_check,
|
|
238
|
+
wait_after_load=wait_after_load or None,
|
|
239
|
+
captcha_pause=captcha_pause or None,
|
|
240
|
+
)
|
|
241
|
+
|
|
242
|
+
def get_image(
|
|
243
|
+
self, url: str, *, proxy_country: Optional[str] = None
|
|
244
|
+
) -> bytes:
|
|
245
|
+
"""Fetch an image URL through the bypass chain and return its bytes."""
|
|
246
|
+
params = _base.build_params(url=url, proxy_country=proxy_country)
|
|
247
|
+
response = self._request("/getImage", params)
|
|
248
|
+
return response.content
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Exception hierarchy for the ScrapeUnblocker client.
|
|
2
|
+
|
|
3
|
+
Every error raised by the client derives from :class:`ScrapeUnblockerError`,
|
|
4
|
+
so ``except ScrapeUnblockerError`` catches everything. API responses map to
|
|
5
|
+
typed subclasses by status code, letting callers react to a hard block
|
|
6
|
+
differently than a transient upstream outage without parsing status codes by
|
|
7
|
+
hand.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Optional
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class ScrapeUnblockerError(Exception):
|
|
16
|
+
"""Base class for every error raised by this library."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class APIError(ScrapeUnblockerError):
|
|
20
|
+
"""An error response returned by the ScrapeUnblocker API.
|
|
21
|
+
|
|
22
|
+
Attributes:
|
|
23
|
+
status_code: The HTTP status code of the response.
|
|
24
|
+
body: The raw response body (text), useful for debugging.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
def __init__(self, message: str, *, status_code: int, body: Optional[str] = None):
|
|
28
|
+
super().__init__(message)
|
|
29
|
+
self.status_code = status_code
|
|
30
|
+
self.body = body
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class AuthenticationError(APIError):
|
|
34
|
+
"""The API key is missing, malformed, or not recognised (HTTP 401)."""
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class InvalidRequestError(APIError):
|
|
38
|
+
"""The request was rejected as invalid, e.g. a malformed URL (HTTP 400)."""
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class BlockedError(APIError):
|
|
42
|
+
"""The target site blocked every available bypass path (HTTP 403).
|
|
43
|
+
|
|
44
|
+
This is the target's anti-bot protection winning, not a problem with your
|
|
45
|
+
request. Blocked calls are not billed.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class RateLimitError(APIError):
|
|
50
|
+
"""Too many requests against your account in a short window (HTTP 429)."""
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class UpstreamOutageError(APIError):
|
|
54
|
+
"""The origin site returned a server-side outage page (HTTP 503).
|
|
55
|
+
|
|
56
|
+
This means the *target* is down, not ScrapeUnblocker. Retrying later
|
|
57
|
+
usually succeeds.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class ServerError(APIError):
|
|
62
|
+
"""ScrapeUnblocker returned an unexpected 5xx error."""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class ScrapeTimeoutError(ScrapeUnblockerError):
|
|
66
|
+
"""The request did not complete within the configured timeout."""
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class ConnectionError(ScrapeUnblockerError):
|
|
70
|
+
"""The client could not reach the ScrapeUnblocker API."""
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""Lightweight return types for structured responses."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import Any, Dict, List, Optional
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass
|
|
10
|
+
class ParsedPage:
|
|
11
|
+
"""Structured data extracted from a page (``parsed_data=True``).
|
|
12
|
+
|
|
13
|
+
Attributes:
|
|
14
|
+
page_type: What the API classified the page as, e.g. ``"product"``.
|
|
15
|
+
source: How the data was extracted (Schema.org, __NEXT_DATA__, AI rules).
|
|
16
|
+
data: The extracted fields as a plain dict.
|
|
17
|
+
raw: The full JSON payload as returned by the API.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
page_type: Optional[str]
|
|
21
|
+
source: Optional[str]
|
|
22
|
+
data: Any
|
|
23
|
+
raw: Dict[str, Any]
|
|
24
|
+
|
|
25
|
+
@classmethod
|
|
26
|
+
def from_response(cls, payload: Dict[str, Any]) -> "ParsedPage":
|
|
27
|
+
inner = payload.get("data", payload) or {}
|
|
28
|
+
return cls(
|
|
29
|
+
page_type=inner.get("page_type"),
|
|
30
|
+
source=inner.get("source"),
|
|
31
|
+
data=inner.get("data"),
|
|
32
|
+
raw=payload,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class PageResult:
|
|
38
|
+
"""HTML plus the cookies and proxy that served it (``get_cookies=True``)."""
|
|
39
|
+
|
|
40
|
+
html: Optional[str]
|
|
41
|
+
cookies: Any
|
|
42
|
+
proxy: Optional[str]
|
|
43
|
+
raw: Dict[str, Any]
|
|
44
|
+
|
|
45
|
+
@classmethod
|
|
46
|
+
def from_response(cls, payload: Dict[str, Any]) -> "PageResult":
|
|
47
|
+
return cls(
|
|
48
|
+
html=payload.get("html") or payload.get("page_source") or payload.get("content"),
|
|
49
|
+
cookies=payload.get("cookies"),
|
|
50
|
+
proxy=payload.get("proxy") or payload.get("proxy_address"),
|
|
51
|
+
raw=payload,
|
|
52
|
+
)
|
scrapeunblocker/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: scrapeunblocker
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Official Python client for the ScrapeUnblocker web scraping API - JavaScript-rendered pages that bypass Cloudflare, DataDome, PerimeterX, Akamai and more.
|
|
5
|
+
Project-URL: Homepage, https://scrapeunblocker.com
|
|
6
|
+
Project-URL: Documentation, https://developers.scrapeunblocker.com
|
|
7
|
+
Project-URL: Source, https://github.com/ScrapeUnblocker/scrapeunblocker-python
|
|
8
|
+
Project-URL: Changelog, https://github.com/ScrapeUnblocker/scrapeunblocker-python/blob/main/CHANGELOG.md
|
|
9
|
+
Author-email: ScrapeUnblocker <support@scrapeunblocker.com>
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: anti-bot,captcha,cloudflare,crawler,datadome,proxy,scraping api,web scraping
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
24
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
25
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
26
|
+
Classifier: Typing :: Typed
|
|
27
|
+
Requires-Python: >=3.8
|
|
28
|
+
Requires-Dist: httpx>=0.23
|
|
29
|
+
Provides-Extra: dev
|
|
30
|
+
Requires-Dist: mypy>=1.0; extra == 'dev'
|
|
31
|
+
Requires-Dist: pytest-asyncio>=0.21; extra == 'dev'
|
|
32
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
33
|
+
Requires-Dist: respx>=0.20; extra == 'dev'
|
|
34
|
+
Description-Content-Type: text/markdown
|
|
35
|
+
|
|
36
|
+
# ScrapeUnblocker Python client
|
|
37
|
+
|
|
38
|
+
Official Python client for the [ScrapeUnblocker](https://scrapeunblocker.com) web scraping API.
|
|
39
|
+
|
|
40
|
+
Every request is fully JavaScript-rendered in a real browser and routed through premium proxies, so it bypasses Cloudflare, DataDome, PerimeterX, Akamai, Kasada and similar anti-bot systems - from one simple call. You are only billed for successful requests.
|
|
41
|
+
|
|
42
|
+
- **Highest success rate on the market** (95%+ on live production traffic)
|
|
43
|
+
- **Rendered HTML or parsed JSON** - no per-site parsers to maintain
|
|
44
|
+
- Sync **and** async clients, fully type-hinted
|
|
45
|
+
|
|
46
|
+
## Install
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install scrapeunblocker
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Requires Python 3.8+.
|
|
53
|
+
|
|
54
|
+
## Quickstart
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
from scrapeunblocker import Client
|
|
58
|
+
|
|
59
|
+
su = Client(api_key="su_live_...") # or set the SCRAPEUNBLOCKER_KEY env var
|
|
60
|
+
|
|
61
|
+
# Rendered HTML for any URL
|
|
62
|
+
html = su.get_page_source("https://example.com")
|
|
63
|
+
|
|
64
|
+
# Structured JSON instead of HTML (products, listings, search results, ...)
|
|
65
|
+
product = su.get_parsed("https://www.amazon.com/dp/B08N5WRWNW")
|
|
66
|
+
print(product.page_type) # "product"
|
|
67
|
+
print(product.data) # {...}
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
Get your API key at [app.scrapeunblocker.com](https://app.scrapeunblocker.com). The free trial does not require a credit card.
|
|
71
|
+
|
|
72
|
+
## Authentication
|
|
73
|
+
|
|
74
|
+
Pass the key directly, or set an environment variable and omit it:
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
export SCRAPEUNBLOCKER_KEY="su_live_..."
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
from scrapeunblocker import Client
|
|
82
|
+
su = Client() # reads SCRAPEUNBLOCKER_KEY
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
## Fetch rendered HTML
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
html = su.get_page_source(
|
|
89
|
+
"https://www.nordstrom.com/browse/women/clothing/dresses",
|
|
90
|
+
proxy_country="US", # route through a specific country
|
|
91
|
+
time_sleep=3, # wait extra seconds after load
|
|
92
|
+
)
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## Get parsed JSON
|
|
96
|
+
|
|
97
|
+
Pass a URL and get back structured data extracted via Schema.org, `__NEXT_DATA__` or AI-generated rules:
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
result = su.get_parsed("https://www.walmart.com/ip/12345")
|
|
101
|
+
print(result.page_type) # e.g. "product"
|
|
102
|
+
print(result.source) # how it was extracted
|
|
103
|
+
print(result.data) # the fields
|
|
104
|
+
|
|
105
|
+
# If a parse ever comes back wrong, force a fresh set of rules:
|
|
106
|
+
result = su.get_parsed(url, refresh_rules=True, rules_hint="price is missing")
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## Google search (SERP)
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
serp = su.serp("web scraping api", pages_to_check=2, proxy_country="US")
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
## Cookies and the serving proxy
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
page = su.get_page_with_cookies("https://example.com")
|
|
119
|
+
print(page.html, page.cookies, page.proxy)
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
## Images
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
data = su.get_image("https://example.com/photo.jpg")
|
|
126
|
+
open("photo.jpg", "wb").write(data)
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
## Skyscanner plugins
|
|
130
|
+
|
|
131
|
+
```python
|
|
132
|
+
# Resolve a place name to entity IDs, then search
|
|
133
|
+
locs = su.skyscanner.flight_locations("London")
|
|
134
|
+
flights = su.skyscanner.flights(
|
|
135
|
+
origin="London", dest="New York",
|
|
136
|
+
depart_date="2026-09-01", adults=1, currency="USD",
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
hotels = su.skyscanner.hotels(destination="Madrid", checkin="2026-09-01", checkout="2026-09-03")
|
|
140
|
+
cars = su.skyscanner.carhire(pickup="Madrid", pickup_datetime="2026-09-01T10:00", dropoff_datetime="2026-09-03T10:00")
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
## Async
|
|
144
|
+
|
|
145
|
+
Every method has an async twin on `AsyncClient`:
|
|
146
|
+
|
|
147
|
+
```python
|
|
148
|
+
import asyncio
|
|
149
|
+
from scrapeunblocker import AsyncClient
|
|
150
|
+
|
|
151
|
+
async def main():
|
|
152
|
+
async with AsyncClient(api_key="su_live_...") as su:
|
|
153
|
+
html = await su.get_page_source("https://example.com")
|
|
154
|
+
|
|
155
|
+
asyncio.run(main())
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
## Error handling
|
|
159
|
+
|
|
160
|
+
Non-2xx responses raise typed exceptions, all subclasses of `ScrapeUnblockerError`:
|
|
161
|
+
|
|
162
|
+
```python
|
|
163
|
+
from scrapeunblocker import Client, BlockedError, RateLimitError, UpstreamOutageError
|
|
164
|
+
|
|
165
|
+
su = Client()
|
|
166
|
+
try:
|
|
167
|
+
html = su.get_page_source("https://example.com")
|
|
168
|
+
except BlockedError:
|
|
169
|
+
... # 403: the target blocked every bypass path (not billed)
|
|
170
|
+
except RateLimitError:
|
|
171
|
+
... # 429: slow down
|
|
172
|
+
except UpstreamOutageError:
|
|
173
|
+
... # 503: the target site itself is down - retry later
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
| Exception | Status | Meaning |
|
|
177
|
+
|---|---|---|
|
|
178
|
+
| `InvalidRequestError` | 400 | Bad URL or unsupported scheme |
|
|
179
|
+
| `AuthenticationError` | 401 | Missing or invalid API key |
|
|
180
|
+
| `BlockedError` | 403 | Blocked by bot protection on every path |
|
|
181
|
+
| `RateLimitError` | 429 | Too many requests |
|
|
182
|
+
| `UpstreamOutageError` | 503 | The target origin is down |
|
|
183
|
+
| `ServerError` | 5xx | Unexpected server error |
|
|
184
|
+
| `ScrapeTimeoutError` | - | Request exceeded the timeout |
|
|
185
|
+
| `ConnectionError` | - | Could not reach the API |
|
|
186
|
+
|
|
187
|
+
Transient failures (429, 502, 503, 504 and network errors) are retried automatically with exponential backoff; tune with `Client(max_retries=...)`.
|
|
188
|
+
|
|
189
|
+
## Configuration
|
|
190
|
+
|
|
191
|
+
```python
|
|
192
|
+
Client(
|
|
193
|
+
api_key=None, # or SCRAPEUNBLOCKER_KEY env var
|
|
194
|
+
base_url="https://api.scrapeunblocker.com",
|
|
195
|
+
timeout=180.0, # seconds; protected pages can be slow
|
|
196
|
+
max_retries=2,
|
|
197
|
+
)
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
## Links
|
|
201
|
+
|
|
202
|
+
- Documentation: https://developers.scrapeunblocker.com
|
|
203
|
+
- Website: https://scrapeunblocker.com
|
|
204
|
+
- Dashboard: https://app.scrapeunblocker.com
|
|
205
|
+
|
|
206
|
+
## License
|
|
207
|
+
|
|
208
|
+
MIT
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
scrapeunblocker/__init__.py,sha256=JicnXv9l6hOuXPyyhDOIcPw9q1iIEILS3t5X99B7Smk,1123
|
|
2
|
+
scrapeunblocker/_async_client.py,sha256=jbT5bn6vYgRSXQ0PwHK-YD_ro8mH6bOzX3ledBjCaSU,6824
|
|
3
|
+
scrapeunblocker/_base.py,sha256=hs8c06Wynf_48kiXKeyJbZZy0IpG02KtqziYZ7sSkic,3496
|
|
4
|
+
scrapeunblocker/_client.py,sha256=HC-lNzwytLnI_Xl9uAkB5IDW0QCEddCQVWnWt-A0iz8,8670
|
|
5
|
+
scrapeunblocker/exceptions.py,sha256=bDVgiVxtAWQNt3hmkt9_WCN_iucbVxPI1eyYMPB4cjE,2084
|
|
6
|
+
scrapeunblocker/models.py,sha256=DZqdwHs6llqd0kTVue9wWqSiL5-m_4LXm4Jlx8ffV4c,1531
|
|
7
|
+
scrapeunblocker/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
8
|
+
scrapeunblocker/version.py,sha256=kUR5RAFc7HCeiqdlX36dZOHkUI5wI6V_43RpEcD8b-0,22
|
|
9
|
+
scrapeunblocker-0.1.0.dist-info/METADATA,sha256=VESjtiBYUhpcDdgJk6Wr0U4rq6kor0LI1gmh6niwISU,6591
|
|
10
|
+
scrapeunblocker-0.1.0.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
11
|
+
scrapeunblocker-0.1.0.dist-info/licenses/LICENSE,sha256=-Qs4w8IjJt3RWr3it4xjyous5qjVyQKBbmaxNP4Mlmo,1072
|
|
12
|
+
scrapeunblocker-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 ScrapeUnblocker
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|