litescrape-sdk 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- litescrape_sdk/__init__.py +91 -0
- litescrape_sdk/_runtime.py +201 -0
- litescrape_sdk/_version.py +1 -0
- litescrape_sdk/client.py +331 -0
- litescrape_sdk/errors.py +129 -0
- litescrape_sdk/models.py +621 -0
- litescrape_sdk/py.typed +0 -0
- litescrape_sdk-0.1.0.dist-info/METADATA +50 -0
- litescrape_sdk-0.1.0.dist-info/RECORD +11 -0
- litescrape_sdk-0.1.0.dist-info/WHEEL +4 -0
- litescrape_sdk-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Python SDK for the Litescrape API. Start with ``help(litescrape_sdk.scrape)``."""
|
|
2
|
+
|
|
3
|
+
from ._version import __version__
|
|
4
|
+
from .client import Result, akey_status, ascrape, key_status, scrape
|
|
5
|
+
from .errors import (
|
|
6
|
+
APIError,
|
|
7
|
+
AuthenticationError,
|
|
8
|
+
LitescrapeError,
|
|
9
|
+
NotFoundError,
|
|
10
|
+
PaymentRequiredError,
|
|
11
|
+
RateLimitError,
|
|
12
|
+
TransportError,
|
|
13
|
+
ValidationError,
|
|
14
|
+
)
|
|
15
|
+
from .models import (
|
|
16
|
+
REQUEST_TYPES,
|
|
17
|
+
AnyRequest,
|
|
18
|
+
AppleMapsPlaces,
|
|
19
|
+
AppleMapsReviews,
|
|
20
|
+
BingMaps,
|
|
21
|
+
BingSearch,
|
|
22
|
+
DuckDuckGoMaps,
|
|
23
|
+
DuckDuckGoSearch,
|
|
24
|
+
GoogleAds,
|
|
25
|
+
GoogleAiMode,
|
|
26
|
+
GoogleAiOverview,
|
|
27
|
+
GoogleContributorReviews,
|
|
28
|
+
GoogleLocal,
|
|
29
|
+
GoogleMaps,
|
|
30
|
+
GoogleMapsLiveFootTraffic,
|
|
31
|
+
GoogleMapsPhoto,
|
|
32
|
+
GoogleMapsPosts,
|
|
33
|
+
GoogleMapsWebResults,
|
|
34
|
+
GoogleReviews,
|
|
35
|
+
GoogleSearch,
|
|
36
|
+
GoogleShopping,
|
|
37
|
+
GoogleShoppingProduct,
|
|
38
|
+
KeyStatus,
|
|
39
|
+
ScrapeRequest,
|
|
40
|
+
TripadvisorPlace,
|
|
41
|
+
TripadvisorReviews,
|
|
42
|
+
TripadvisorSearch,
|
|
43
|
+
YelpReviews,
|
|
44
|
+
YelpSearch,
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
__all__ = [
|
|
48
|
+
"__version__",
|
|
49
|
+
"scrape",
|
|
50
|
+
"ascrape",
|
|
51
|
+
"key_status",
|
|
52
|
+
"akey_status",
|
|
53
|
+
"Result",
|
|
54
|
+
"KeyStatus",
|
|
55
|
+
"ScrapeRequest",
|
|
56
|
+
"AnyRequest",
|
|
57
|
+
"REQUEST_TYPES",
|
|
58
|
+
"LitescrapeError",
|
|
59
|
+
"ValidationError",
|
|
60
|
+
"TransportError",
|
|
61
|
+
"APIError",
|
|
62
|
+
"AuthenticationError",
|
|
63
|
+
"PaymentRequiredError",
|
|
64
|
+
"NotFoundError",
|
|
65
|
+
"RateLimitError",
|
|
66
|
+
"GoogleSearch",
|
|
67
|
+
"GoogleAiOverview",
|
|
68
|
+
"GoogleAiMode",
|
|
69
|
+
"GoogleAds",
|
|
70
|
+
"GoogleShopping",
|
|
71
|
+
"GoogleShoppingProduct",
|
|
72
|
+
"GoogleLocal",
|
|
73
|
+
"GoogleMaps",
|
|
74
|
+
"GoogleMapsLiveFootTraffic",
|
|
75
|
+
"GoogleMapsPosts",
|
|
76
|
+
"GoogleMapsPhoto",
|
|
77
|
+
"GoogleMapsWebResults",
|
|
78
|
+
"GoogleReviews",
|
|
79
|
+
"GoogleContributorReviews",
|
|
80
|
+
"BingSearch",
|
|
81
|
+
"BingMaps",
|
|
82
|
+
"DuckDuckGoSearch",
|
|
83
|
+
"DuckDuckGoMaps",
|
|
84
|
+
"YelpSearch",
|
|
85
|
+
"YelpReviews",
|
|
86
|
+
"TripadvisorSearch",
|
|
87
|
+
"TripadvisorPlace",
|
|
88
|
+
"TripadvisorReviews",
|
|
89
|
+
"AppleMapsPlaces",
|
|
90
|
+
"AppleMapsReviews",
|
|
91
|
+
]
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""Event loop, HTTP client, concurrency gate, and retry loop behind the public functions."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import concurrent.futures
|
|
7
|
+
import contextlib
|
|
8
|
+
import random
|
|
9
|
+
import sys
|
|
10
|
+
import threading
|
|
11
|
+
from collections.abc import Callable, Coroutine
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from typing import Any, TypeVar
|
|
14
|
+
|
|
15
|
+
import httpx
|
|
16
|
+
|
|
17
|
+
from ._version import __version__
|
|
18
|
+
from .errors import APIError, LitescrapeError, TransportError, api_error_from_response
|
|
19
|
+
|
|
20
|
+
DEFAULT_BASE_URL = "https://api.litescrape.com"
|
|
21
|
+
DEFAULT_CONCURRENCY = 25
|
|
22
|
+
SELECTOR_LOOP_CAP = 500
|
|
23
|
+
BACKOFF_CAP = 30.0
|
|
24
|
+
CONNECT_TIMEOUT = 10.0
|
|
25
|
+
|
|
26
|
+
_RETRY_CODES = frozenset(
|
|
27
|
+
{"proxy_capacity_unavailable", "upstream_session_unavailable", "service_unavailable"}
|
|
28
|
+
)
|
|
29
|
+
_RETRY_TRANSPORT = (httpx.TimeoutException, httpx.NetworkError, httpx.RemoteProtocolError)
|
|
30
|
+
|
|
31
|
+
T = TypeVar("T")
|
|
32
|
+
_LoopEntry = tuple[asyncio.AbstractEventLoop, dict[Any, Any]]
|
|
33
|
+
|
|
34
|
+
_clients: dict[int, _LoopEntry] = {}
|
|
35
|
+
_semaphores: dict[int, _LoopEntry] = {}
|
|
36
|
+
_sleep = asyncio.sleep
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(slots=True)
|
|
40
|
+
class Outcome:
|
|
41
|
+
data: dict[str, Any] | None = None
|
|
42
|
+
error: LitescrapeError | None = None
|
|
43
|
+
status_code: int | None = None
|
|
44
|
+
request_id: str = ""
|
|
45
|
+
attempts: int = 0
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _per_loop(registry: dict[int, _LoopEntry]) -> dict[Any, Any]:
|
|
49
|
+
loop = asyncio.get_running_loop()
|
|
50
|
+
for key, (known, _) in list(registry.items()):
|
|
51
|
+
if known.is_closed():
|
|
52
|
+
del registry[key]
|
|
53
|
+
return registry.setdefault(id(loop), (loop, {}))[1]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _make_client(base_url: str) -> httpx.AsyncClient:
|
|
57
|
+
return httpx.AsyncClient(
|
|
58
|
+
base_url=base_url,
|
|
59
|
+
headers={"Accept": "application/json", "User-Agent": f"litescrape-sdk/{__version__}"},
|
|
60
|
+
limits=httpx.Limits(max_connections=None, max_keepalive_connections=None),
|
|
61
|
+
timeout=httpx.Timeout(120.0, connect=CONNECT_TIMEOUT),
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def client_for(base_url: str) -> httpx.AsyncClient:
|
|
66
|
+
clients = _per_loop(_clients)
|
|
67
|
+
client = clients.get(base_url)
|
|
68
|
+
if client is None:
|
|
69
|
+
client = clients[base_url] = _make_client(base_url)
|
|
70
|
+
return client
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def semaphore_for(api_key: str, base_url: str, limit: int) -> asyncio.Semaphore:
|
|
74
|
+
semaphores = _per_loop(_semaphores)
|
|
75
|
+
entry = semaphores.get((api_key, base_url))
|
|
76
|
+
if entry is None or entry[0] != limit:
|
|
77
|
+
entry = semaphores[(api_key, base_url)] = (limit, asyncio.Semaphore(limit))
|
|
78
|
+
return entry[1]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def is_selector_limited(loop: asyncio.AbstractEventLoop) -> bool:
|
|
82
|
+
return sys.platform == "win32" and isinstance(loop, asyncio.SelectorEventLoop)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def effective_limit(status_limit: int | None, override: int | None, *, selector_limited: bool) -> int:
|
|
86
|
+
limit = status_limit or DEFAULT_CONCURRENCY
|
|
87
|
+
if override is not None:
|
|
88
|
+
limit = min(limit, override)
|
|
89
|
+
if selector_limited:
|
|
90
|
+
limit = min(limit, SELECTOR_LOOP_CAP)
|
|
91
|
+
return max(1, limit)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _should_retry(error: LitescrapeError) -> bool:
|
|
95
|
+
if isinstance(error, TransportError):
|
|
96
|
+
return error.retryable
|
|
97
|
+
if isinstance(error, APIError):
|
|
98
|
+
status = error.status_code or 0
|
|
99
|
+
return error.retryable or status == 429 or status >= 500 or error.error_code in _RETRY_CODES
|
|
100
|
+
return False
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _delay(attempt: int, retry_after: float | None) -> float:
|
|
104
|
+
if retry_after is not None:
|
|
105
|
+
return min(max(retry_after, 0.0), BACKOFF_CAP)
|
|
106
|
+
ceiling = min(BACKOFF_CAP, float(2 ** (attempt - 1)))
|
|
107
|
+
return random.uniform(ceiling / 2, ceiling)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
async def _attempt(
|
|
111
|
+
client: httpx.AsyncClient,
|
|
112
|
+
path: str,
|
|
113
|
+
params: dict[str, str],
|
|
114
|
+
headers: dict[str, str],
|
|
115
|
+
timeout: float,
|
|
116
|
+
outcome: Outcome,
|
|
117
|
+
) -> LitescrapeError | None:
|
|
118
|
+
outcome.status_code, outcome.request_id = None, ""
|
|
119
|
+
try:
|
|
120
|
+
response = await client.get(
|
|
121
|
+
path,
|
|
122
|
+
params=params,
|
|
123
|
+
headers=headers,
|
|
124
|
+
timeout=httpx.Timeout(timeout, connect=min(CONNECT_TIMEOUT, timeout)),
|
|
125
|
+
)
|
|
126
|
+
except _RETRY_TRANSPORT as exc:
|
|
127
|
+
return TransportError(f"{type(exc).__name__}: {exc}", retryable=True, cause=exc)
|
|
128
|
+
except Exception as exc:
|
|
129
|
+
return TransportError(f"{type(exc).__name__}: {exc}", retryable=False, cause=exc)
|
|
130
|
+
outcome.status_code = response.status_code
|
|
131
|
+
outcome.request_id = response.headers.get("x-request-id", "")
|
|
132
|
+
if response.status_code != 200:
|
|
133
|
+
error = api_error_from_response(response)
|
|
134
|
+
outcome.request_id = error.request_id or outcome.request_id
|
|
135
|
+
return error
|
|
136
|
+
try:
|
|
137
|
+
data = response.json()
|
|
138
|
+
except ValueError as exc:
|
|
139
|
+
return TransportError("Response body was not JSON", retryable=True, cause=exc)
|
|
140
|
+
if not isinstance(data, dict):
|
|
141
|
+
return TransportError("Response body was not a JSON object", retryable=True)
|
|
142
|
+
outcome.data = data
|
|
143
|
+
return None
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
async def request_with_retries(
|
|
147
|
+
client: httpx.AsyncClient,
|
|
148
|
+
path: str,
|
|
149
|
+
params: dict[str, str],
|
|
150
|
+
*,
|
|
151
|
+
headers: dict[str, str],
|
|
152
|
+
attempts: int,
|
|
153
|
+
timeout: float,
|
|
154
|
+
semaphore: asyncio.Semaphore | None = None,
|
|
155
|
+
gate: Callable[[], LitescrapeError | None] | None = None,
|
|
156
|
+
) -> Outcome:
|
|
157
|
+
outcome = Outcome()
|
|
158
|
+
slot = semaphore if semaphore is not None else contextlib.nullcontext()
|
|
159
|
+
for attempt in range(1, attempts + 1):
|
|
160
|
+
async with slot:
|
|
161
|
+
blocked = gate() if gate is not None else None
|
|
162
|
+
if blocked is not None:
|
|
163
|
+
outcome.error = blocked
|
|
164
|
+
outcome.status_code = getattr(blocked, "status_code", None)
|
|
165
|
+
return outcome
|
|
166
|
+
outcome.attempts = attempt
|
|
167
|
+
error = await _attempt(client, path, params, headers, timeout, outcome)
|
|
168
|
+
outcome.error = error
|
|
169
|
+
if error is None:
|
|
170
|
+
return outcome
|
|
171
|
+
if attempt == attempts or not _should_retry(error):
|
|
172
|
+
return outcome
|
|
173
|
+
await _sleep(_delay(attempt, getattr(error, "retry_after", None)))
|
|
174
|
+
return outcome
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
_loop_lock = threading.Lock()
|
|
178
|
+
_loop: asyncio.AbstractEventLoop | None = None
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def background_loop() -> asyncio.AbstractEventLoop:
|
|
182
|
+
global _loop
|
|
183
|
+
with _loop_lock:
|
|
184
|
+
if _loop is None or _loop.is_closed():
|
|
185
|
+
loop = asyncio.ProactorEventLoop() if sys.platform == "win32" else asyncio.new_event_loop()
|
|
186
|
+
threading.Thread(target=loop.run_forever, name="litescrape-sdk", daemon=True).start()
|
|
187
|
+
_loop = loop
|
|
188
|
+
return _loop
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def run_sync(coro: Coroutine[Any, Any, T]) -> T:
|
|
192
|
+
future = asyncio.run_coroutine_threadsafe(coro, background_loop())
|
|
193
|
+
try:
|
|
194
|
+
while True:
|
|
195
|
+
try:
|
|
196
|
+
return future.result(timeout=0.25)
|
|
197
|
+
except concurrent.futures.TimeoutError:
|
|
198
|
+
continue
|
|
199
|
+
except BaseException:
|
|
200
|
+
future.cancel()
|
|
201
|
+
raise
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
litescrape_sdk/client.py
ADDED
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
"""Public entry points: scrape / ascrape and key_status / akey_status."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import os
|
|
7
|
+
import time
|
|
8
|
+
from collections.abc import Mapping, Sequence
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
import httpx
|
|
13
|
+
import pydantic
|
|
14
|
+
from tqdm import tqdm
|
|
15
|
+
|
|
16
|
+
from . import _runtime
|
|
17
|
+
from .errors import (
|
|
18
|
+
APIError,
|
|
19
|
+
AuthenticationError,
|
|
20
|
+
LitescrapeError,
|
|
21
|
+
PaymentRequiredError,
|
|
22
|
+
TransportError,
|
|
23
|
+
ValidationError,
|
|
24
|
+
)
|
|
25
|
+
from .models import REQUEST_ADAPTER, REQUEST_TYPES, KeyStatus, ScrapeRequest
|
|
26
|
+
|
|
27
|
+
RequestItem = Mapping[str, Any] | ScrapeRequest
|
|
28
|
+
_FATAL_CODES = frozenset({"api_key_disabled"})
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True, slots=True)
|
|
32
|
+
class Result:
|
|
33
|
+
"""Outcome of one request in a batch; ``index`` is its position in the input list."""
|
|
34
|
+
|
|
35
|
+
index: int
|
|
36
|
+
request: ScrapeRequest
|
|
37
|
+
data: dict[str, Any] | None
|
|
38
|
+
error: LitescrapeError | None
|
|
39
|
+
status_code: int | None
|
|
40
|
+
request_id: str
|
|
41
|
+
attempts: int
|
|
42
|
+
elapsed: float
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def ok(self) -> bool:
|
|
46
|
+
return self.error is None
|
|
47
|
+
|
|
48
|
+
def raise_for_error(self) -> dict[str, Any]:
|
|
49
|
+
if self.error is not None:
|
|
50
|
+
raise self.error
|
|
51
|
+
return self.data if self.data is not None else {}
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class _Batch:
|
|
55
|
+
__slots__ = ("fatal",)
|
|
56
|
+
|
|
57
|
+
def __init__(self) -> None:
|
|
58
|
+
self.fatal: APIError | None = None
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _describe(exc: pydantic.ValidationError) -> str:
|
|
62
|
+
lines = []
|
|
63
|
+
for error in exc.errors():
|
|
64
|
+
location = ".".join(str(part) for part in error["loc"] if part not in REQUEST_TYPES)
|
|
65
|
+
message = error["msg"]
|
|
66
|
+
if error["type"] == "union_tag_not_found":
|
|
67
|
+
message += f" (valid endpoint values: {', '.join(sorted(REQUEST_TYPES))})"
|
|
68
|
+
lines.append(f"{location}: {message}" if location else message)
|
|
69
|
+
return "; ".join(lines)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _validate(items: Sequence[RequestItem]) -> list[ScrapeRequest]:
|
|
73
|
+
validated: list[ScrapeRequest] = []
|
|
74
|
+
problems: list[tuple[int, str]] = []
|
|
75
|
+
for index, item in enumerate(items):
|
|
76
|
+
try:
|
|
77
|
+
validated.append(REQUEST_ADAPTER.validate_python(item))
|
|
78
|
+
except pydantic.ValidationError as exc:
|
|
79
|
+
problems.append((index, _describe(exc)))
|
|
80
|
+
if problems:
|
|
81
|
+
raise ValidationError(problems)
|
|
82
|
+
return validated
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _check_args(attempts: int, concurrency: int | None, timeout: float) -> None:
|
|
86
|
+
if attempts < 1:
|
|
87
|
+
raise ValueError("attempts must be at least 1")
|
|
88
|
+
if concurrency is not None and concurrency < 1:
|
|
89
|
+
raise ValueError("concurrency must be at least 1")
|
|
90
|
+
if timeout <= 0:
|
|
91
|
+
raise ValueError("timeout must be positive")
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _resolve_key(api_key: str | None) -> str:
|
|
95
|
+
key = (api_key or os.environ.get("LITESCRAPE_API_KEY") or "").strip()
|
|
96
|
+
if not key:
|
|
97
|
+
raise AuthenticationError(
|
|
98
|
+
"Pass api_key or set the LITESCRAPE_API_KEY environment variable.",
|
|
99
|
+
status_code=None,
|
|
100
|
+
error_code="missing_api_key",
|
|
101
|
+
)
|
|
102
|
+
return key
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _resolve_base_url(base_url: str | None) -> str:
|
|
106
|
+
return (base_url or os.environ.get("LITESCRAPE_API_URL") or _runtime.DEFAULT_BASE_URL).rstrip("/")
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
async def _fetch_status(
|
|
110
|
+
client: httpx.AsyncClient, headers: dict[str, str], attempts: int, timeout: float
|
|
111
|
+
) -> KeyStatus:
|
|
112
|
+
outcome = await _runtime.request_with_retries(
|
|
113
|
+
client, "/api/keys/status", {}, headers=headers, attempts=attempts, timeout=timeout
|
|
114
|
+
)
|
|
115
|
+
if outcome.error is not None:
|
|
116
|
+
raise outcome.error
|
|
117
|
+
try:
|
|
118
|
+
return KeyStatus.model_validate({**(outcome.data or {}), "request_id": outcome.request_id})
|
|
119
|
+
except pydantic.ValidationError as exc:
|
|
120
|
+
raise TransportError(f"Unexpected key status body: {exc}", retryable=False, cause=exc) from exc
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
async def _run_one(
|
|
124
|
+
client: httpx.AsyncClient,
|
|
125
|
+
index: int,
|
|
126
|
+
request: ScrapeRequest,
|
|
127
|
+
headers: dict[str, str],
|
|
128
|
+
attempts: int,
|
|
129
|
+
timeout: float,
|
|
130
|
+
semaphore: asyncio.Semaphore,
|
|
131
|
+
batch: _Batch,
|
|
132
|
+
) -> Result:
|
|
133
|
+
started = time.monotonic()
|
|
134
|
+
try:
|
|
135
|
+
outcome = await _runtime.request_with_retries(
|
|
136
|
+
client,
|
|
137
|
+
request.path,
|
|
138
|
+
request.query_params(),
|
|
139
|
+
headers=headers,
|
|
140
|
+
attempts=attempts,
|
|
141
|
+
timeout=timeout,
|
|
142
|
+
semaphore=semaphore,
|
|
143
|
+
gate=lambda: batch.fatal,
|
|
144
|
+
)
|
|
145
|
+
except asyncio.CancelledError:
|
|
146
|
+
raise
|
|
147
|
+
except Exception as exc:
|
|
148
|
+
outcome = _runtime.Outcome(
|
|
149
|
+
error=TransportError(f"Unexpected {type(exc).__name__}: {exc}", retryable=False, cause=exc)
|
|
150
|
+
)
|
|
151
|
+
error = outcome.error
|
|
152
|
+
if isinstance(error, APIError) and (error.status_code in (401, 402) or error.error_code in _FATAL_CODES):
|
|
153
|
+
batch.fatal = error
|
|
154
|
+
return Result(
|
|
155
|
+
index=index,
|
|
156
|
+
request=request,
|
|
157
|
+
data=outcome.data,
|
|
158
|
+
error=error,
|
|
159
|
+
status_code=outcome.status_code,
|
|
160
|
+
request_id=outcome.request_id,
|
|
161
|
+
attempts=outcome.attempts,
|
|
162
|
+
elapsed=time.monotonic() - started,
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
async def ascrape(
|
|
167
|
+
requests: Sequence[RequestItem],
|
|
168
|
+
*,
|
|
169
|
+
api_key: str | None = None,
|
|
170
|
+
attempts: int = 5,
|
|
171
|
+
concurrency: int | None = None,
|
|
172
|
+
timeout: float = 120.0,
|
|
173
|
+
tqdm_disable: bool = False,
|
|
174
|
+
base_url: str | None = None,
|
|
175
|
+
) -> list[Result]:
|
|
176
|
+
"""Async form of :func:`scrape`: same arguments, results, and errors, run on the current event loop.
|
|
177
|
+
|
|
178
|
+
``scrape`` runs this coroutine on a private background loop so it can be called from anywhere,
|
|
179
|
+
including inside a running loop; call ``ascrape`` directly from asyncio code.
|
|
180
|
+
"""
|
|
181
|
+
_check_args(attempts, concurrency, timeout)
|
|
182
|
+
items = _validate(list(requests))
|
|
183
|
+
if not items:
|
|
184
|
+
return []
|
|
185
|
+
key = _resolve_key(api_key)
|
|
186
|
+
url = _resolve_base_url(base_url)
|
|
187
|
+
headers = {"Authorization": f"Bearer {key}"}
|
|
188
|
+
client = _runtime.client_for(url)
|
|
189
|
+
status = await _fetch_status(client, headers, attempts, timeout)
|
|
190
|
+
if status.remaining_calls < len(items):
|
|
191
|
+
raise PaymentRequiredError(
|
|
192
|
+
f"This batch needs {len(items)} calls but the key has {status.remaining_calls} remaining.",
|
|
193
|
+
status_code=402,
|
|
194
|
+
error_code="payment_required",
|
|
195
|
+
request_id=status.request_id,
|
|
196
|
+
)
|
|
197
|
+
limit = _runtime.effective_limit(
|
|
198
|
+
status.concurrency_limit,
|
|
199
|
+
concurrency,
|
|
200
|
+
selector_limited=_runtime.is_selector_limited(asyncio.get_running_loop()),
|
|
201
|
+
)
|
|
202
|
+
semaphore = _runtime.semaphore_for(key, url, limit)
|
|
203
|
+
batch = _Batch()
|
|
204
|
+
tasks = [
|
|
205
|
+
asyncio.create_task(_run_one(client, index, item, headers, attempts, timeout, semaphore, batch))
|
|
206
|
+
for index, item in enumerate(items)
|
|
207
|
+
]
|
|
208
|
+
results: list[Result | None] = [None] * len(items)
|
|
209
|
+
bar = tqdm(total=len(items), disable=tqdm_disable or len(items) == 1, unit="req")
|
|
210
|
+
try:
|
|
211
|
+
for completed in asyncio.as_completed(tasks):
|
|
212
|
+
result = await completed
|
|
213
|
+
results[result.index] = result
|
|
214
|
+
bar.update(1)
|
|
215
|
+
except BaseException:
|
|
216
|
+
for task in tasks:
|
|
217
|
+
task.cancel()
|
|
218
|
+
raise
|
|
219
|
+
finally:
|
|
220
|
+
bar.close()
|
|
221
|
+
return [result for result in results if result is not None]
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def scrape(
|
|
225
|
+
requests: Sequence[RequestItem],
|
|
226
|
+
*,
|
|
227
|
+
api_key: str | None = None,
|
|
228
|
+
attempts: int = 5,
|
|
229
|
+
concurrency: int | None = None,
|
|
230
|
+
timeout: float = 120.0,
|
|
231
|
+
tqdm_disable: bool = False,
|
|
232
|
+
base_url: str | None = None,
|
|
233
|
+
) -> list[Result]:
|
|
234
|
+
"""Run every item in ``requests`` against the Litescrape API; one ``Result`` per item, in input order.
|
|
235
|
+
|
|
236
|
+
Each item is a dict with an ``endpoint`` key, or a typed request object::
|
|
237
|
+
|
|
238
|
+
from litescrape_sdk import GoogleMaps, GoogleSearch, scrape
|
|
239
|
+
|
|
240
|
+
results = scrape([
|
|
241
|
+
{"endpoint": "google_search", "q": "coffee grinders", "gl": "us"},
|
|
242
|
+
GoogleMaps(q="coffee", type="search", ll="@40.745,-73.988,14z"),
|
|
243
|
+
])
|
|
244
|
+
for result in results:
|
|
245
|
+
if result.ok:
|
|
246
|
+
print(result.data["search_metadata"]["id"])
|
|
247
|
+
else:
|
|
248
|
+
print(result.error)
|
|
249
|
+
|
|
250
|
+
``endpoint`` values are the slugs in ``REQUEST_TYPES`` (``google_search``, ``google_maps``,
|
|
251
|
+
``google_reviews``, ``bing_search``, ``yelp_reviews``, ...); each maps to a request class whose fields
|
|
252
|
+
are that endpoint's query parameters and whose ``path`` is its route. Dicts are validated into the same
|
|
253
|
+
classes, so both forms send identical requests. Unknown parameter names, wrong types, and missing
|
|
254
|
+
required parameters are rejected locally.
|
|
255
|
+
|
|
256
|
+
Before any request is sent:
|
|
257
|
+
|
|
258
|
+
1. Every item is validated. If any item is invalid, ``ValidationError`` names each bad index and
|
|
259
|
+
nothing is sent.
|
|
260
|
+
2. The API key is resolved from ``api_key`` or the ``LITESCRAPE_API_KEY`` environment variable
|
|
261
|
+
(``AuthenticationError`` if neither is set).
|
|
262
|
+
3. ``GET /api/keys/status`` is fetched; it is not billed. ``PaymentRequiredError`` is raised if the key
|
|
263
|
+
has fewer calls remaining than the batch needs. The status also supplies the key's concurrency limit.
|
|
264
|
+
|
|
265
|
+
Requests run concurrently up to the key's ``concurrency_limit`` (25 when the key reports none), shared by
|
|
266
|
+
every call in this process for the same key. ``concurrency`` can lower that cap, never raise it. On a
|
|
267
|
+
Windows selector event loop the cap is also limited to 500. A progress bar advances as results arrive,
|
|
268
|
+
in completion order; the returned list is always in input order. ``tqdm_disable=True`` hides the bar,
|
|
269
|
+
and a single-item call shows none.
|
|
270
|
+
|
|
271
|
+
Each item is tried up to ``attempts`` times (``attempts=1`` disables retries). Retried: connection
|
|
272
|
+
errors and timeouts, HTTP 429 and 5xx, and any error the API marks ``retryable``. Not retried: 400,
|
|
273
|
+
401, 402, 403, 404, 422. Waits grow exponentially with jitter from about 1 s, capped at 30 s, or follow
|
|
274
|
+
the API's ``Retry-After``. Only a 200 response is billed; a retry is a new call, so if an attempt
|
|
275
|
+
succeeded server-side but its response was lost in transit, both calls are billed. ``timeout`` is per
|
|
276
|
+
attempt in seconds and defaults to 120, above the API's own 90 s deadline.
|
|
277
|
+
|
|
278
|
+
Per-item failures never raise. ``Result.ok`` is False and ``Result.error`` holds the exception: an
|
|
279
|
+
``APIError`` subclass with ``status_code``, ``error_code``, and ``request_id`` for API envelopes
|
|
280
|
+
(``NotFoundError`` for a 404, for example Google AI Overview when no overview exists), or
|
|
281
|
+
``TransportError`` when no usable response arrived. ``Result.raise_for_error()`` raises it or returns
|
|
282
|
+
``Result.data``. A 401, 402, or disabled-key error seen mid-batch is copied to every item not yet
|
|
283
|
+
started, without further requests.
|
|
284
|
+
|
|
285
|
+
Ctrl-C (or cancelling the task) cancels in-flight requests; a request the API had already completed
|
|
286
|
+
is still billed.
|
|
287
|
+
|
|
288
|
+
Args:
|
|
289
|
+
requests: dicts with an ``endpoint`` key, or request objects such as ``GoogleSearch(q=...)``.
|
|
290
|
+
api_key: bearer key; defaults to ``LITESCRAPE_API_KEY``.
|
|
291
|
+
attempts: total tries per item, at least 1.
|
|
292
|
+
concurrency: optional lower cap on in-flight requests, at least 1.
|
|
293
|
+
timeout: seconds per attempt.
|
|
294
|
+
tqdm_disable: hide the progress bar.
|
|
295
|
+
base_url: API origin; defaults to ``LITESCRAPE_API_URL`` or ``https://api.litescrape.com``.
|
|
296
|
+
|
|
297
|
+
Returns:
|
|
298
|
+
``list[Result]`` aligned with ``requests``.
|
|
299
|
+
|
|
300
|
+
Raises:
|
|
301
|
+
ValidationError, AuthenticationError, PaymentRequiredError, APIError, TransportError: before any
|
|
302
|
+
scrape request is sent, as described above. ``ValueError`` for out-of-range arguments.
|
|
303
|
+
"""
|
|
304
|
+
return _runtime.run_sync(
|
|
305
|
+
ascrape(
|
|
306
|
+
requests,
|
|
307
|
+
api_key=api_key,
|
|
308
|
+
attempts=attempts,
|
|
309
|
+
concurrency=concurrency,
|
|
310
|
+
timeout=timeout,
|
|
311
|
+
tqdm_disable=tqdm_disable,
|
|
312
|
+
base_url=base_url,
|
|
313
|
+
)
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
async def akey_status(api_key: str | None = None, *, base_url: str | None = None) -> KeyStatus:
|
|
318
|
+
"""Async form of :func:`key_status`."""
|
|
319
|
+
key = _resolve_key(api_key)
|
|
320
|
+
url = _resolve_base_url(base_url)
|
|
321
|
+
client = _runtime.client_for(url)
|
|
322
|
+
return await _fetch_status(client, {"Authorization": f"Bearer {key}"}, attempts=5, timeout=120.0)
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def key_status(api_key: str | None = None, *, base_url: str | None = None) -> KeyStatus:
|
|
326
|
+
"""Read the key's balance and limits from ``GET /api/keys/status``; the call is not billed.
|
|
327
|
+
|
|
328
|
+
Returns a ``KeyStatus`` with ``remaining_calls``, ``concurrency_limit``, ``status``, and the other
|
|
329
|
+
fields the API reports. Raises ``AuthenticationError`` for a missing or invalid key.
|
|
330
|
+
"""
|
|
331
|
+
return _runtime.run_sync(akey_status(api_key, base_url=base_url))
|