search1api 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- search1api/__init__.py +38 -0
- search1api/client.py +686 -0
- search1api/errors.py +107 -0
- search1api/py.typed +1 -0
- search1api/types.py +180 -0
- search1api-0.1.0.dist-info/METADATA +131 -0
- search1api-0.1.0.dist-info/RECORD +9 -0
- search1api-0.1.0.dist-info/WHEEL +4 -0
- search1api-0.1.0.dist-info/licenses/LICENSE +22 -0
search1api/__init__.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
from .client import AsyncSearch1API, Search1API
|
|
2
|
+
from .errors import (
|
|
3
|
+
APIConnectionError,
|
|
4
|
+
APIStatusError,
|
|
5
|
+
APITimeoutError,
|
|
6
|
+
AuthenticationError,
|
|
7
|
+
BadRequestError,
|
|
8
|
+
DeepcrawlFailedError,
|
|
9
|
+
DeepcrawlTimeoutError,
|
|
10
|
+
InternalServerError,
|
|
11
|
+
NotFoundError,
|
|
12
|
+
PaymentRequiredError,
|
|
13
|
+
RateLimitError,
|
|
14
|
+
Search1APIConfigurationError,
|
|
15
|
+
Search1APIError,
|
|
16
|
+
UnprocessableEntityError,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"APIConnectionError",
|
|
21
|
+
"APIStatusError",
|
|
22
|
+
"APITimeoutError",
|
|
23
|
+
"AsyncSearch1API",
|
|
24
|
+
"AuthenticationError",
|
|
25
|
+
"BadRequestError",
|
|
26
|
+
"DeepcrawlFailedError",
|
|
27
|
+
"DeepcrawlTimeoutError",
|
|
28
|
+
"InternalServerError",
|
|
29
|
+
"NotFoundError",
|
|
30
|
+
"PaymentRequiredError",
|
|
31
|
+
"RateLimitError",
|
|
32
|
+
"Search1API",
|
|
33
|
+
"Search1APIConfigurationError",
|
|
34
|
+
"Search1APIError",
|
|
35
|
+
"UnprocessableEntityError",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
__version__ = "0.1.0"
|
search1api/client.py
ADDED
|
@@ -0,0 +1,686 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import random
|
|
7
|
+
import time
|
|
8
|
+
from email.utils import parsedate_to_datetime
|
|
9
|
+
from typing import Any, Dict, List, Mapping, Optional, cast
|
|
10
|
+
from urllib.parse import quote
|
|
11
|
+
|
|
12
|
+
import httpx
|
|
13
|
+
|
|
14
|
+
from .errors import (
|
|
15
|
+
APIConnectionError,
|
|
16
|
+
APITimeoutError,
|
|
17
|
+
DeepcrawlFailedError,
|
|
18
|
+
DeepcrawlTimeoutError,
|
|
19
|
+
Search1APIConfigurationError,
|
|
20
|
+
Search1APIError,
|
|
21
|
+
api_status_error,
|
|
22
|
+
)
|
|
23
|
+
from .types import (
|
|
24
|
+
BatchResponse,
|
|
25
|
+
CrawlRequest,
|
|
26
|
+
CrawlResponse,
|
|
27
|
+
CrawlType,
|
|
28
|
+
DeepcrawlAcceptedResponse,
|
|
29
|
+
DeepcrawlStatusResponse,
|
|
30
|
+
ExtractResponse,
|
|
31
|
+
HealthResponse,
|
|
32
|
+
NewsEngine,
|
|
33
|
+
NewsRequest,
|
|
34
|
+
NewsResponse,
|
|
35
|
+
SearchEngine,
|
|
36
|
+
SearchRequest,
|
|
37
|
+
SearchResponse,
|
|
38
|
+
SitemapResponse,
|
|
39
|
+
TimeRange,
|
|
40
|
+
TrendingResponse,
|
|
41
|
+
UsageResponse,
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
DEFAULT_BASE_URL = "https://api.search1api.com"
|
|
45
|
+
DEFAULT_TIMEOUT = 30.0
|
|
46
|
+
DEFAULT_MAX_RETRIES = 2
|
|
47
|
+
DEFAULT_RETRY_DELAY = 0.5
|
|
48
|
+
DEFAULT_DEEPCRAWL_POLL_INTERVAL = 2.0
|
|
49
|
+
DEFAULT_DEEPCRAWL_TIMEOUT = 300.0
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _compact(payload: Mapping[str, Any]) -> Dict[str, Any]:
|
|
53
|
+
return {key: value for key, value in payload.items() if value is not None}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _search_payload(
|
|
57
|
+
query: str,
|
|
58
|
+
*,
|
|
59
|
+
search_service: Optional[str] = None,
|
|
60
|
+
max_results: Optional[int] = None,
|
|
61
|
+
crawl_results: Optional[int] = None,
|
|
62
|
+
image: Optional[bool] = None,
|
|
63
|
+
include_sites: Optional[List[str]] = None,
|
|
64
|
+
exclude_sites: Optional[List[str]] = None,
|
|
65
|
+
language: Optional[str] = None,
|
|
66
|
+
time_range: Optional[TimeRange] = None,
|
|
67
|
+
) -> Dict[str, Any]:
|
|
68
|
+
return _compact(
|
|
69
|
+
{
|
|
70
|
+
"query": query,
|
|
71
|
+
"search_service": search_service,
|
|
72
|
+
"max_results": max_results,
|
|
73
|
+
"crawl_results": crawl_results,
|
|
74
|
+
"image": image,
|
|
75
|
+
"include_sites": include_sites,
|
|
76
|
+
"exclude_sites": exclude_sites,
|
|
77
|
+
"language": language,
|
|
78
|
+
"time_range": time_range,
|
|
79
|
+
}
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _retry_after(response: httpx.Response) -> Optional[float]:
|
|
84
|
+
value = response.headers.get("retry-after")
|
|
85
|
+
if not value:
|
|
86
|
+
return None
|
|
87
|
+
try:
|
|
88
|
+
return max(0.0, float(value))
|
|
89
|
+
except ValueError:
|
|
90
|
+
try:
|
|
91
|
+
return max(0.0, parsedate_to_datetime(value).timestamp() - time.time())
|
|
92
|
+
except (TypeError, ValueError, OverflowError):
|
|
93
|
+
return None
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _should_retry(status_code: int) -> bool:
|
|
97
|
+
return status_code == 429 or status_code >= 500
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _error_body(response: httpx.Response) -> Mapping[str, Any]:
|
|
101
|
+
try:
|
|
102
|
+
body = response.json()
|
|
103
|
+
except (json.JSONDecodeError, ValueError):
|
|
104
|
+
return {"message": response.text}
|
|
105
|
+
return body if isinstance(body, dict) else {"message": response.text}
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class _ClientConfig:
|
|
109
|
+
def __init__(
|
|
110
|
+
self,
|
|
111
|
+
api_key: Optional[str],
|
|
112
|
+
*,
|
|
113
|
+
base_url: str,
|
|
114
|
+
timeout: float,
|
|
115
|
+
max_retries: int,
|
|
116
|
+
retry_delay: float,
|
|
117
|
+
headers: Optional[Mapping[str, str]],
|
|
118
|
+
) -> None:
|
|
119
|
+
resolved_key = api_key or os.getenv("SEARCH1API_API_KEY")
|
|
120
|
+
if not resolved_key:
|
|
121
|
+
raise Search1APIConfigurationError(
|
|
122
|
+
"Missing API key. Pass api_key or set SEARCH1API_API_KEY."
|
|
123
|
+
)
|
|
124
|
+
if max_retries < 0:
|
|
125
|
+
raise Search1APIConfigurationError("max_retries cannot be negative")
|
|
126
|
+
if timeout <= 0:
|
|
127
|
+
raise Search1APIConfigurationError("timeout must be positive")
|
|
128
|
+
if retry_delay < 0:
|
|
129
|
+
raise Search1APIConfigurationError("retry_delay cannot be negative")
|
|
130
|
+
|
|
131
|
+
self.api_key: str = resolved_key
|
|
132
|
+
self.base_url: str = base_url.rstrip("/")
|
|
133
|
+
self.timeout: float = timeout
|
|
134
|
+
self.max_retries: int = max_retries
|
|
135
|
+
self.retry_delay: float = retry_delay
|
|
136
|
+
self.headers: Dict[str, str] = dict(headers or {})
|
|
137
|
+
|
|
138
|
+
def _request_headers(self) -> Dict[str, str]:
|
|
139
|
+
return {
|
|
140
|
+
"Accept": "application/json",
|
|
141
|
+
"Authorization": f"Bearer {self.api_key}",
|
|
142
|
+
"X-Search1API-Client": "python/0.1.0",
|
|
143
|
+
**self.headers,
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
def _retry_sleep(self, attempt: int) -> float:
|
|
147
|
+
base = self.retry_delay * (2**attempt)
|
|
148
|
+
return float(base + random.random() * base * 0.25)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
class Search1API(_ClientConfig):
|
|
152
|
+
"""Synchronous Search1API client."""
|
|
153
|
+
|
|
154
|
+
def __init__(
|
|
155
|
+
self,
|
|
156
|
+
api_key: Optional[str] = None,
|
|
157
|
+
*,
|
|
158
|
+
base_url: str = DEFAULT_BASE_URL,
|
|
159
|
+
timeout: float = DEFAULT_TIMEOUT,
|
|
160
|
+
max_retries: int = DEFAULT_MAX_RETRIES,
|
|
161
|
+
retry_delay: float = DEFAULT_RETRY_DELAY,
|
|
162
|
+
headers: Optional[Mapping[str, str]] = None,
|
|
163
|
+
client: Optional[httpx.Client] = None,
|
|
164
|
+
) -> None:
|
|
165
|
+
super().__init__(
|
|
166
|
+
api_key,
|
|
167
|
+
base_url=base_url,
|
|
168
|
+
timeout=timeout,
|
|
169
|
+
max_retries=max_retries,
|
|
170
|
+
retry_delay=retry_delay,
|
|
171
|
+
headers=headers,
|
|
172
|
+
)
|
|
173
|
+
self._client = client or httpx.Client()
|
|
174
|
+
self._owns_client = client is None
|
|
175
|
+
|
|
176
|
+
def close(self) -> None:
|
|
177
|
+
if self._owns_client:
|
|
178
|
+
self._client.close()
|
|
179
|
+
|
|
180
|
+
def __enter__(self) -> "Search1API":
|
|
181
|
+
return self
|
|
182
|
+
|
|
183
|
+
def __exit__(self, *_: object) -> None:
|
|
184
|
+
self.close()
|
|
185
|
+
|
|
186
|
+
def search(
|
|
187
|
+
self,
|
|
188
|
+
query: str,
|
|
189
|
+
*,
|
|
190
|
+
search_service: Optional[SearchEngine] = None,
|
|
191
|
+
max_results: Optional[int] = None,
|
|
192
|
+
crawl_results: Optional[int] = None,
|
|
193
|
+
image: Optional[bool] = None,
|
|
194
|
+
include_sites: Optional[List[str]] = None,
|
|
195
|
+
exclude_sites: Optional[List[str]] = None,
|
|
196
|
+
language: Optional[str] = None,
|
|
197
|
+
time_range: Optional[TimeRange] = None,
|
|
198
|
+
) -> SearchResponse:
|
|
199
|
+
payload = _search_payload(
|
|
200
|
+
query,
|
|
201
|
+
search_service=search_service,
|
|
202
|
+
max_results=max_results,
|
|
203
|
+
crawl_results=crawl_results,
|
|
204
|
+
image=image,
|
|
205
|
+
include_sites=include_sites,
|
|
206
|
+
exclude_sites=exclude_sites,
|
|
207
|
+
language=language,
|
|
208
|
+
time_range=time_range,
|
|
209
|
+
)
|
|
210
|
+
return cast(SearchResponse, self._request_json("POST", "/search", json=payload))
|
|
211
|
+
|
|
212
|
+
def search_batch(self, requests: List[SearchRequest]) -> BatchResponse:
|
|
213
|
+
return cast(BatchResponse, self._request_json("POST", "/search", json=requests))
|
|
214
|
+
|
|
215
|
+
def news(
|
|
216
|
+
self,
|
|
217
|
+
query: str,
|
|
218
|
+
*,
|
|
219
|
+
search_service: Optional[NewsEngine] = None,
|
|
220
|
+
max_results: Optional[int] = None,
|
|
221
|
+
crawl_results: Optional[int] = None,
|
|
222
|
+
image: Optional[bool] = None,
|
|
223
|
+
include_sites: Optional[List[str]] = None,
|
|
224
|
+
exclude_sites: Optional[List[str]] = None,
|
|
225
|
+
language: Optional[str] = None,
|
|
226
|
+
time_range: Optional[TimeRange] = None,
|
|
227
|
+
) -> NewsResponse:
|
|
228
|
+
payload = _search_payload(
|
|
229
|
+
query,
|
|
230
|
+
search_service=search_service,
|
|
231
|
+
max_results=max_results,
|
|
232
|
+
crawl_results=crawl_results,
|
|
233
|
+
image=image,
|
|
234
|
+
include_sites=include_sites,
|
|
235
|
+
exclude_sites=exclude_sites,
|
|
236
|
+
language=language,
|
|
237
|
+
time_range=time_range,
|
|
238
|
+
)
|
|
239
|
+
return cast(NewsResponse, self._request_json("POST", "/news", json=payload))
|
|
240
|
+
|
|
241
|
+
def news_batch(self, requests: List[NewsRequest]) -> BatchResponse:
|
|
242
|
+
return cast(BatchResponse, self._request_json("POST", "/news", json=requests))
|
|
243
|
+
|
|
244
|
+
def crawl(
|
|
245
|
+
self, url: str, *, enable_fallback: Optional[bool] = None
|
|
246
|
+
) -> CrawlResponse:
|
|
247
|
+
payload = _compact({"url": url, "enableFallback": enable_fallback})
|
|
248
|
+
return cast(CrawlResponse, self._request_json("POST", "/crawl", json=payload))
|
|
249
|
+
|
|
250
|
+
def crawl_batch(self, requests: List[CrawlRequest]) -> List[CrawlResponse]:
|
|
251
|
+
return cast(
|
|
252
|
+
List[CrawlResponse], self._request_json("POST", "/crawl", json=requests)
|
|
253
|
+
)
|
|
254
|
+
|
|
255
|
+
def sitemap(self, url: str, *, type: Optional[CrawlType] = None) -> SitemapResponse:
|
|
256
|
+
return cast(
|
|
257
|
+
SitemapResponse,
|
|
258
|
+
self._request_json(
|
|
259
|
+
"POST", "/sitemap", json=_compact({"url": url, "type": type})
|
|
260
|
+
),
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
def trending(
|
|
264
|
+
self, search_service: str, *, max_results: Optional[int] = None
|
|
265
|
+
) -> TrendingResponse:
|
|
266
|
+
return cast(
|
|
267
|
+
TrendingResponse,
|
|
268
|
+
self._request_json(
|
|
269
|
+
"POST",
|
|
270
|
+
"/trending",
|
|
271
|
+
json=_compact(
|
|
272
|
+
{"search_service": search_service, "max_results": max_results}
|
|
273
|
+
),
|
|
274
|
+
),
|
|
275
|
+
)
|
|
276
|
+
|
|
277
|
+
def extract(
|
|
278
|
+
self,
|
|
279
|
+
url: str,
|
|
280
|
+
*,
|
|
281
|
+
prompt: Optional[str] = None,
|
|
282
|
+
response_format: Optional[Mapping[str, Any]] = None,
|
|
283
|
+
) -> ExtractResponse:
|
|
284
|
+
return cast(
|
|
285
|
+
ExtractResponse,
|
|
286
|
+
self._request_json(
|
|
287
|
+
"POST",
|
|
288
|
+
"/extract",
|
|
289
|
+
json=_compact(
|
|
290
|
+
{"url": url, "prompt": prompt, "response_format": response_format}
|
|
291
|
+
),
|
|
292
|
+
),
|
|
293
|
+
)
|
|
294
|
+
|
|
295
|
+
def start_deepcrawl(
|
|
296
|
+
self, url: str, *, type: Optional[CrawlType] = None
|
|
297
|
+
) -> DeepcrawlAcceptedResponse:
|
|
298
|
+
return cast(
|
|
299
|
+
DeepcrawlAcceptedResponse,
|
|
300
|
+
self._request_json(
|
|
301
|
+
"POST",
|
|
302
|
+
"/deepcrawl",
|
|
303
|
+
json=_compact({"url": url, "type": type}),
|
|
304
|
+
retryable=False,
|
|
305
|
+
),
|
|
306
|
+
)
|
|
307
|
+
|
|
308
|
+
def get_deepcrawl_status(self, task_id: str) -> DeepcrawlStatusResponse:
|
|
309
|
+
return cast(
|
|
310
|
+
DeepcrawlStatusResponse,
|
|
311
|
+
self._request_json("GET", f"/deepcrawl/status/{quote(task_id, safe='')}"),
|
|
312
|
+
)
|
|
313
|
+
|
|
314
|
+
def wait_for_deepcrawl(
|
|
315
|
+
self,
|
|
316
|
+
task_id: str,
|
|
317
|
+
*,
|
|
318
|
+
poll_interval: float = DEFAULT_DEEPCRAWL_POLL_INTERVAL,
|
|
319
|
+
timeout: float = DEFAULT_DEEPCRAWL_TIMEOUT,
|
|
320
|
+
) -> DeepcrawlStatusResponse:
|
|
321
|
+
deadline = time.monotonic() + timeout
|
|
322
|
+
while True:
|
|
323
|
+
response = self.get_deepcrawl_status(task_id)
|
|
324
|
+
if response.get("success") is True or response.get("status") == "completed":
|
|
325
|
+
return response
|
|
326
|
+
if response.get("success") is False or response.get("status") in {
|
|
327
|
+
"failed",
|
|
328
|
+
"not_found",
|
|
329
|
+
}:
|
|
330
|
+
raise DeepcrawlFailedError(response)
|
|
331
|
+
if time.monotonic() + poll_interval > deadline:
|
|
332
|
+
raise DeepcrawlTimeoutError(task_id, timeout)
|
|
333
|
+
time.sleep(poll_interval)
|
|
334
|
+
|
|
335
|
+
def deepcrawl(
|
|
336
|
+
self,
|
|
337
|
+
url: str,
|
|
338
|
+
*,
|
|
339
|
+
type: Optional[CrawlType] = None,
|
|
340
|
+
poll_interval: float = DEFAULT_DEEPCRAWL_POLL_INTERVAL,
|
|
341
|
+
timeout: float = DEFAULT_DEEPCRAWL_TIMEOUT,
|
|
342
|
+
) -> DeepcrawlStatusResponse:
|
|
343
|
+
task = self.start_deepcrawl(url, type=type)
|
|
344
|
+
return self.wait_for_deepcrawl(
|
|
345
|
+
task["taskId"], poll_interval=poll_interval, timeout=timeout
|
|
346
|
+
)
|
|
347
|
+
|
|
348
|
+
def usage(self, period: Optional[TimeRange] = None) -> UsageResponse:
|
|
349
|
+
return cast(
|
|
350
|
+
UsageResponse,
|
|
351
|
+
self._request_json("GET", "/usage", params=_compact({"period": period})),
|
|
352
|
+
)
|
|
353
|
+
|
|
354
|
+
def health(self) -> HealthResponse:
|
|
355
|
+
return cast(HealthResponse, self._request_json("GET", "/health"))
|
|
356
|
+
|
|
357
|
+
def _request_json(self, method: str, path: str, **kwargs: Any) -> Any:
|
|
358
|
+
response = self._request(method, path, **kwargs)
|
|
359
|
+
try:
|
|
360
|
+
return response.json()
|
|
361
|
+
except (json.JSONDecodeError, ValueError) as exc:
|
|
362
|
+
raise Search1APIError(
|
|
363
|
+
f"Search1API returned invalid JSON for {path}"
|
|
364
|
+
) from exc
|
|
365
|
+
|
|
366
|
+
def _request(self, method: str, path: str, **kwargs: Any) -> httpx.Response:
|
|
367
|
+
retryable = bool(kwargs.pop("retryable", True))
|
|
368
|
+
max_retries = self.max_retries if retryable else 0
|
|
369
|
+
for attempt in range(max_retries + 1):
|
|
370
|
+
try:
|
|
371
|
+
response = self._client.request(
|
|
372
|
+
method,
|
|
373
|
+
f"{self.base_url}{path}",
|
|
374
|
+
headers=self._request_headers(),
|
|
375
|
+
timeout=self.timeout,
|
|
376
|
+
**kwargs,
|
|
377
|
+
)
|
|
378
|
+
except httpx.TimeoutException as exc:
|
|
379
|
+
if attempt < max_retries:
|
|
380
|
+
time.sleep(self._retry_sleep(attempt))
|
|
381
|
+
continue
|
|
382
|
+
raise APITimeoutError(
|
|
383
|
+
f"Search1API request timed out after {self.timeout:g}s"
|
|
384
|
+
) from exc
|
|
385
|
+
except httpx.RequestError as exc:
|
|
386
|
+
if attempt < max_retries:
|
|
387
|
+
time.sleep(self._retry_sleep(attempt))
|
|
388
|
+
continue
|
|
389
|
+
raise APIConnectionError("Unable to connect to Search1API") from exc
|
|
390
|
+
|
|
391
|
+
if response.is_success:
|
|
392
|
+
return response
|
|
393
|
+
if attempt < max_retries and _should_retry(response.status_code):
|
|
394
|
+
delay = _retry_after(response)
|
|
395
|
+
time.sleep(delay if delay is not None else self._retry_sleep(attempt))
|
|
396
|
+
continue
|
|
397
|
+
raise api_status_error(
|
|
398
|
+
response.status_code, _error_body(response), response.headers
|
|
399
|
+
)
|
|
400
|
+
raise Search1APIError("Search1API request exhausted its retry budget")
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
class AsyncSearch1API(_ClientConfig):
|
|
404
|
+
"""Asynchronous Search1API client."""
|
|
405
|
+
|
|
406
|
+
def __init__(
|
|
407
|
+
self,
|
|
408
|
+
api_key: Optional[str] = None,
|
|
409
|
+
*,
|
|
410
|
+
base_url: str = DEFAULT_BASE_URL,
|
|
411
|
+
timeout: float = DEFAULT_TIMEOUT,
|
|
412
|
+
max_retries: int = DEFAULT_MAX_RETRIES,
|
|
413
|
+
retry_delay: float = DEFAULT_RETRY_DELAY,
|
|
414
|
+
headers: Optional[Mapping[str, str]] = None,
|
|
415
|
+
client: Optional[httpx.AsyncClient] = None,
|
|
416
|
+
) -> None:
|
|
417
|
+
super().__init__(
|
|
418
|
+
api_key,
|
|
419
|
+
base_url=base_url,
|
|
420
|
+
timeout=timeout,
|
|
421
|
+
max_retries=max_retries,
|
|
422
|
+
retry_delay=retry_delay,
|
|
423
|
+
headers=headers,
|
|
424
|
+
)
|
|
425
|
+
self._client = client or httpx.AsyncClient()
|
|
426
|
+
self._owns_client = client is None
|
|
427
|
+
|
|
428
|
+
async def close(self) -> None:
|
|
429
|
+
if self._owns_client:
|
|
430
|
+
await self._client.aclose()
|
|
431
|
+
|
|
432
|
+
async def __aenter__(self) -> "AsyncSearch1API":
|
|
433
|
+
return self
|
|
434
|
+
|
|
435
|
+
async def __aexit__(self, *_: object) -> None:
|
|
436
|
+
await self.close()
|
|
437
|
+
|
|
438
|
+
async def search(
|
|
439
|
+
self,
|
|
440
|
+
query: str,
|
|
441
|
+
*,
|
|
442
|
+
search_service: Optional[SearchEngine] = None,
|
|
443
|
+
max_results: Optional[int] = None,
|
|
444
|
+
crawl_results: Optional[int] = None,
|
|
445
|
+
image: Optional[bool] = None,
|
|
446
|
+
include_sites: Optional[List[str]] = None,
|
|
447
|
+
exclude_sites: Optional[List[str]] = None,
|
|
448
|
+
language: Optional[str] = None,
|
|
449
|
+
time_range: Optional[TimeRange] = None,
|
|
450
|
+
) -> SearchResponse:
|
|
451
|
+
return cast(
|
|
452
|
+
SearchResponse,
|
|
453
|
+
await self._request_json(
|
|
454
|
+
"POST",
|
|
455
|
+
"/search",
|
|
456
|
+
json=_search_payload(
|
|
457
|
+
query,
|
|
458
|
+
search_service=search_service,
|
|
459
|
+
max_results=max_results,
|
|
460
|
+
crawl_results=crawl_results,
|
|
461
|
+
image=image,
|
|
462
|
+
include_sites=include_sites,
|
|
463
|
+
exclude_sites=exclude_sites,
|
|
464
|
+
language=language,
|
|
465
|
+
time_range=time_range,
|
|
466
|
+
),
|
|
467
|
+
),
|
|
468
|
+
)
|
|
469
|
+
|
|
470
|
+
async def search_batch(self, requests: List[SearchRequest]) -> BatchResponse:
|
|
471
|
+
return cast(
|
|
472
|
+
BatchResponse,
|
|
473
|
+
await self._request_json("POST", "/search", json=requests),
|
|
474
|
+
)
|
|
475
|
+
|
|
476
|
+
async def news(
|
|
477
|
+
self,
|
|
478
|
+
query: str,
|
|
479
|
+
*,
|
|
480
|
+
search_service: Optional[NewsEngine] = None,
|
|
481
|
+
max_results: Optional[int] = None,
|
|
482
|
+
crawl_results: Optional[int] = None,
|
|
483
|
+
image: Optional[bool] = None,
|
|
484
|
+
include_sites: Optional[List[str]] = None,
|
|
485
|
+
exclude_sites: Optional[List[str]] = None,
|
|
486
|
+
language: Optional[str] = None,
|
|
487
|
+
time_range: Optional[TimeRange] = None,
|
|
488
|
+
) -> NewsResponse:
|
|
489
|
+
return cast(
|
|
490
|
+
NewsResponse,
|
|
491
|
+
await self._request_json(
|
|
492
|
+
"POST",
|
|
493
|
+
"/news",
|
|
494
|
+
json=_search_payload(
|
|
495
|
+
query,
|
|
496
|
+
search_service=search_service,
|
|
497
|
+
max_results=max_results,
|
|
498
|
+
crawl_results=crawl_results,
|
|
499
|
+
image=image,
|
|
500
|
+
include_sites=include_sites,
|
|
501
|
+
exclude_sites=exclude_sites,
|
|
502
|
+
language=language,
|
|
503
|
+
time_range=time_range,
|
|
504
|
+
),
|
|
505
|
+
),
|
|
506
|
+
)
|
|
507
|
+
|
|
508
|
+
async def news_batch(self, requests: List[NewsRequest]) -> BatchResponse:
|
|
509
|
+
return cast(
|
|
510
|
+
BatchResponse,
|
|
511
|
+
await self._request_json("POST", "/news", json=requests),
|
|
512
|
+
)
|
|
513
|
+
|
|
514
|
+
async def crawl(
|
|
515
|
+
self, url: str, *, enable_fallback: Optional[bool] = None
|
|
516
|
+
) -> CrawlResponse:
|
|
517
|
+
return cast(
|
|
518
|
+
CrawlResponse,
|
|
519
|
+
await self._request_json(
|
|
520
|
+
"POST",
|
|
521
|
+
"/crawl",
|
|
522
|
+
json=_compact({"url": url, "enableFallback": enable_fallback}),
|
|
523
|
+
),
|
|
524
|
+
)
|
|
525
|
+
|
|
526
|
+
async def crawl_batch(self, requests: List[CrawlRequest]) -> List[CrawlResponse]:
|
|
527
|
+
return cast(
|
|
528
|
+
List[CrawlResponse],
|
|
529
|
+
await self._request_json("POST", "/crawl", json=requests),
|
|
530
|
+
)
|
|
531
|
+
|
|
532
|
+
async def sitemap(
|
|
533
|
+
self, url: str, *, type: Optional[CrawlType] = None
|
|
534
|
+
) -> SitemapResponse:
|
|
535
|
+
return cast(
|
|
536
|
+
SitemapResponse,
|
|
537
|
+
await self._request_json(
|
|
538
|
+
"POST", "/sitemap", json=_compact({"url": url, "type": type})
|
|
539
|
+
),
|
|
540
|
+
)
|
|
541
|
+
|
|
542
|
+
async def trending(
|
|
543
|
+
self, search_service: str, *, max_results: Optional[int] = None
|
|
544
|
+
) -> TrendingResponse:
|
|
545
|
+
return cast(
|
|
546
|
+
TrendingResponse,
|
|
547
|
+
await self._request_json(
|
|
548
|
+
"POST",
|
|
549
|
+
"/trending",
|
|
550
|
+
json=_compact(
|
|
551
|
+
{"search_service": search_service, "max_results": max_results}
|
|
552
|
+
),
|
|
553
|
+
),
|
|
554
|
+
)
|
|
555
|
+
|
|
556
|
+
async def extract(
|
|
557
|
+
self,
|
|
558
|
+
url: str,
|
|
559
|
+
*,
|
|
560
|
+
prompt: Optional[str] = None,
|
|
561
|
+
response_format: Optional[Mapping[str, Any]] = None,
|
|
562
|
+
) -> ExtractResponse:
|
|
563
|
+
return cast(
|
|
564
|
+
ExtractResponse,
|
|
565
|
+
await self._request_json(
|
|
566
|
+
"POST",
|
|
567
|
+
"/extract",
|
|
568
|
+
json=_compact(
|
|
569
|
+
{"url": url, "prompt": prompt, "response_format": response_format}
|
|
570
|
+
),
|
|
571
|
+
),
|
|
572
|
+
)
|
|
573
|
+
|
|
574
|
+
async def start_deepcrawl(
|
|
575
|
+
self, url: str, *, type: Optional[CrawlType] = None
|
|
576
|
+
) -> DeepcrawlAcceptedResponse:
|
|
577
|
+
return cast(
|
|
578
|
+
DeepcrawlAcceptedResponse,
|
|
579
|
+
await self._request_json(
|
|
580
|
+
"POST",
|
|
581
|
+
"/deepcrawl",
|
|
582
|
+
json=_compact({"url": url, "type": type}),
|
|
583
|
+
retryable=False,
|
|
584
|
+
),
|
|
585
|
+
)
|
|
586
|
+
|
|
587
|
+
async def get_deepcrawl_status(self, task_id: str) -> DeepcrawlStatusResponse:
|
|
588
|
+
return cast(
|
|
589
|
+
DeepcrawlStatusResponse,
|
|
590
|
+
await self._request_json(
|
|
591
|
+
"GET", f"/deepcrawl/status/{quote(task_id, safe='')}"
|
|
592
|
+
),
|
|
593
|
+
)
|
|
594
|
+
|
|
595
|
+
async def wait_for_deepcrawl(
|
|
596
|
+
self,
|
|
597
|
+
task_id: str,
|
|
598
|
+
*,
|
|
599
|
+
poll_interval: float = DEFAULT_DEEPCRAWL_POLL_INTERVAL,
|
|
600
|
+
timeout: float = DEFAULT_DEEPCRAWL_TIMEOUT,
|
|
601
|
+
) -> DeepcrawlStatusResponse:
|
|
602
|
+
loop = asyncio.get_running_loop()
|
|
603
|
+
deadline = loop.time() + timeout
|
|
604
|
+
while True:
|
|
605
|
+
response = await self.get_deepcrawl_status(task_id)
|
|
606
|
+
if response.get("success") is True or response.get("status") == "completed":
|
|
607
|
+
return response
|
|
608
|
+
if response.get("success") is False or response.get("status") in {
|
|
609
|
+
"failed",
|
|
610
|
+
"not_found",
|
|
611
|
+
}:
|
|
612
|
+
raise DeepcrawlFailedError(response)
|
|
613
|
+
if loop.time() + poll_interval > deadline:
|
|
614
|
+
raise DeepcrawlTimeoutError(task_id, timeout)
|
|
615
|
+
await asyncio.sleep(poll_interval)
|
|
616
|
+
|
|
617
|
+
async def deepcrawl(
|
|
618
|
+
self,
|
|
619
|
+
url: str,
|
|
620
|
+
*,
|
|
621
|
+
type: Optional[CrawlType] = None,
|
|
622
|
+
poll_interval: float = DEFAULT_DEEPCRAWL_POLL_INTERVAL,
|
|
623
|
+
timeout: float = DEFAULT_DEEPCRAWL_TIMEOUT,
|
|
624
|
+
) -> DeepcrawlStatusResponse:
|
|
625
|
+
task = await self.start_deepcrawl(url, type=type)
|
|
626
|
+
return await self.wait_for_deepcrawl(
|
|
627
|
+
task["taskId"], poll_interval=poll_interval, timeout=timeout
|
|
628
|
+
)
|
|
629
|
+
|
|
630
|
+
async def usage(self, period: Optional[TimeRange] = None) -> UsageResponse:
|
|
631
|
+
return cast(
|
|
632
|
+
UsageResponse,
|
|
633
|
+
await self._request_json(
|
|
634
|
+
"GET", "/usage", params=_compact({"period": period})
|
|
635
|
+
),
|
|
636
|
+
)
|
|
637
|
+
|
|
638
|
+
async def health(self) -> HealthResponse:
|
|
639
|
+
return cast(HealthResponse, await self._request_json("GET", "/health"))
|
|
640
|
+
|
|
641
|
+
async def _request_json(self, method: str, path: str, **kwargs: Any) -> Any:
|
|
642
|
+
response = await self._request(method, path, **kwargs)
|
|
643
|
+
try:
|
|
644
|
+
return response.json()
|
|
645
|
+
except (json.JSONDecodeError, ValueError) as exc:
|
|
646
|
+
raise Search1APIError(
|
|
647
|
+
f"Search1API returned invalid JSON for {path}"
|
|
648
|
+
) from exc
|
|
649
|
+
|
|
650
|
+
async def _request(self, method: str, path: str, **kwargs: Any) -> httpx.Response:
|
|
651
|
+
retryable = bool(kwargs.pop("retryable", True))
|
|
652
|
+
max_retries = self.max_retries if retryable else 0
|
|
653
|
+
for attempt in range(max_retries + 1):
|
|
654
|
+
try:
|
|
655
|
+
response = await self._client.request(
|
|
656
|
+
method,
|
|
657
|
+
f"{self.base_url}{path}",
|
|
658
|
+
headers=self._request_headers(),
|
|
659
|
+
timeout=self.timeout,
|
|
660
|
+
**kwargs,
|
|
661
|
+
)
|
|
662
|
+
except httpx.TimeoutException as exc:
|
|
663
|
+
if attempt < max_retries:
|
|
664
|
+
await asyncio.sleep(self._retry_sleep(attempt))
|
|
665
|
+
continue
|
|
666
|
+
raise APITimeoutError(
|
|
667
|
+
f"Search1API request timed out after {self.timeout:g}s"
|
|
668
|
+
) from exc
|
|
669
|
+
except httpx.RequestError as exc:
|
|
670
|
+
if attempt < max_retries:
|
|
671
|
+
await asyncio.sleep(self._retry_sleep(attempt))
|
|
672
|
+
continue
|
|
673
|
+
raise APIConnectionError("Unable to connect to Search1API") from exc
|
|
674
|
+
|
|
675
|
+
if response.is_success:
|
|
676
|
+
return response
|
|
677
|
+
if attempt < max_retries and _should_retry(response.status_code):
|
|
678
|
+
delay = _retry_after(response)
|
|
679
|
+
await asyncio.sleep(
|
|
680
|
+
delay if delay is not None else self._retry_sleep(attempt)
|
|
681
|
+
)
|
|
682
|
+
continue
|
|
683
|
+
raise api_status_error(
|
|
684
|
+
response.status_code, _error_body(response), response.headers
|
|
685
|
+
)
|
|
686
|
+
raise Search1APIError("Search1API request exhausted its retry budget")
|
search1api/errors.py
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any, Mapping, Optional
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class Search1APIError(Exception):
|
|
7
|
+
"""Base exception for the Search1API SDK."""
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class Search1APIConfigurationError(Search1APIError):
|
|
11
|
+
pass
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class APIConnectionError(Search1APIError):
|
|
15
|
+
pass
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class APITimeoutError(APIConnectionError):
|
|
19
|
+
pass
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class APIStatusError(Search1APIError):
|
|
23
|
+
def __init__(
|
|
24
|
+
self,
|
|
25
|
+
message: str,
|
|
26
|
+
*,
|
|
27
|
+
status_code: int,
|
|
28
|
+
body: Optional[Mapping[str, Any]] = None,
|
|
29
|
+
headers: Optional[Mapping[str, str]] = None,
|
|
30
|
+
) -> None:
|
|
31
|
+
super().__init__(message)
|
|
32
|
+
self.status_code = status_code
|
|
33
|
+
self.body = body
|
|
34
|
+
self.headers = headers or {}
|
|
35
|
+
self.request_id = self.headers.get("x-request-id")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class BadRequestError(APIStatusError):
|
|
39
|
+
pass
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class AuthenticationError(APIStatusError):
|
|
43
|
+
pass
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class PaymentRequiredError(APIStatusError):
|
|
47
|
+
pass
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class NotFoundError(APIStatusError):
|
|
51
|
+
pass
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class UnprocessableEntityError(APIStatusError):
|
|
55
|
+
pass
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class RateLimitError(APIStatusError):
|
|
59
|
+
pass
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class InternalServerError(APIStatusError):
|
|
63
|
+
pass
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class DeepcrawlFailedError(Search1APIError):
|
|
67
|
+
def __init__(self, response: Mapping[str, Any]) -> None:
|
|
68
|
+
task_id = response.get("taskId", "unknown")
|
|
69
|
+
message = (
|
|
70
|
+
response.get("error")
|
|
71
|
+
or response.get("message")
|
|
72
|
+
or f"Deepcrawl task {task_id} failed"
|
|
73
|
+
)
|
|
74
|
+
super().__init__(str(message))
|
|
75
|
+
self.response = response
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class DeepcrawlTimeoutError(Search1APIError):
|
|
79
|
+
def __init__(self, task_id: str, timeout: float) -> None:
|
|
80
|
+
super().__init__(f"Deepcrawl task {task_id} did not finish within {timeout:g}s")
|
|
81
|
+
self.task_id = task_id
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def api_status_error(
|
|
85
|
+
status_code: int,
|
|
86
|
+
body: Optional[Mapping[str, Any]],
|
|
87
|
+
headers: Mapping[str, str],
|
|
88
|
+
) -> APIStatusError:
|
|
89
|
+
body = body or {}
|
|
90
|
+
message = (
|
|
91
|
+
body.get("message")
|
|
92
|
+
or body.get("detail")
|
|
93
|
+
or body.get("error")
|
|
94
|
+
or body.get("title")
|
|
95
|
+
or f"Search1API request failed with status {status_code}"
|
|
96
|
+
)
|
|
97
|
+
error_class = {
|
|
98
|
+
400: BadRequestError,
|
|
99
|
+
401: AuthenticationError,
|
|
100
|
+
402: PaymentRequiredError,
|
|
101
|
+
404: NotFoundError,
|
|
102
|
+
422: UnprocessableEntityError,
|
|
103
|
+
429: RateLimitError,
|
|
104
|
+
}.get(status_code, InternalServerError if status_code >= 500 else APIStatusError)
|
|
105
|
+
return error_class(
|
|
106
|
+
str(message), status_code=status_code, body=body, headers=headers
|
|
107
|
+
)
|
search1api/py.typed
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
search1api/types.py
ADDED
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any, Dict, List, Literal, Optional, TypedDict, Union
|
|
4
|
+
|
|
5
|
+
JsonPrimitive = Union[bool, float, int, str, None]
|
|
6
|
+
JsonValue = Union[JsonPrimitive, List["JsonValue"], Dict[str, "JsonValue"]]
|
|
7
|
+
TimeRange = Literal["day", "week", "month", "year"]
|
|
8
|
+
SearchEngine = Literal[
|
|
9
|
+
"google",
|
|
10
|
+
"bing",
|
|
11
|
+
"duckduckgo",
|
|
12
|
+
"yahoo",
|
|
13
|
+
"youtube",
|
|
14
|
+
"x",
|
|
15
|
+
"reddit",
|
|
16
|
+
"github",
|
|
17
|
+
"arxiv",
|
|
18
|
+
"wechat",
|
|
19
|
+
"bilibili",
|
|
20
|
+
"imdb",
|
|
21
|
+
"wikipedia",
|
|
22
|
+
"sogou",
|
|
23
|
+
"baidu",
|
|
24
|
+
"360",
|
|
25
|
+
"quark",
|
|
26
|
+
]
|
|
27
|
+
NewsEngine = Literal["google", "bing", "duckduckgo", "yahoo", "hackernews", "reuters"]
|
|
28
|
+
CrawlType = Literal["sitemap", "all"]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class SearchRequestRequired(TypedDict):
|
|
32
|
+
query: str
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class SearchRequest(SearchRequestRequired, total=False):
|
|
36
|
+
search_service: SearchEngine
|
|
37
|
+
max_results: int
|
|
38
|
+
crawl_results: int
|
|
39
|
+
image: bool
|
|
40
|
+
include_sites: List[str]
|
|
41
|
+
exclude_sites: List[str]
|
|
42
|
+
language: str
|
|
43
|
+
time_range: TimeRange
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class SearchResultRequired(TypedDict):
|
|
47
|
+
title: str
|
|
48
|
+
link: str
|
|
49
|
+
snippet: str
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class SearchResult(SearchResultRequired, total=False):
|
|
53
|
+
content: str
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class SearchResponseRequired(TypedDict):
|
|
57
|
+
searchParameters: SearchRequest
|
|
58
|
+
results: List[SearchResult]
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class SearchResponse(SearchResponseRequired, total=False):
|
|
62
|
+
images: List[str]
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
NewsResponse = SearchResponse
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class NewsRequestRequired(TypedDict):
|
|
69
|
+
query: str
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class NewsRequest(NewsRequestRequired, total=False):
|
|
73
|
+
search_service: NewsEngine
|
|
74
|
+
max_results: int
|
|
75
|
+
crawl_results: int
|
|
76
|
+
image: bool
|
|
77
|
+
include_sites: List[str]
|
|
78
|
+
exclude_sites: List[str]
|
|
79
|
+
language: str
|
|
80
|
+
time_range: TimeRange
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class BatchItemRequired(TypedDict):
|
|
84
|
+
success: bool
|
|
85
|
+
cost: int
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class BatchItem(BatchItemRequired, total=False):
|
|
89
|
+
data: SearchResponse
|
|
90
|
+
error: Dict[str, Any]
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
class BatchSummary(TypedDict):
|
|
94
|
+
total: int
|
|
95
|
+
successful: int
|
|
96
|
+
failed: int
|
|
97
|
+
totalCost: int
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class BatchResponse(TypedDict):
|
|
101
|
+
results: List[BatchItem]
|
|
102
|
+
summary: BatchSummary
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class CrawlRequestRequired(TypedDict):
|
|
106
|
+
url: str
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class CrawlRequest(CrawlRequestRequired, total=False):
|
|
110
|
+
enableFallback: bool
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class CrawlResultRequired(TypedDict):
|
|
114
|
+
title: str
|
|
115
|
+
link: str
|
|
116
|
+
content: str
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class CrawlResult(CrawlResultRequired, total=False):
|
|
120
|
+
metadata: Dict[str, Any]
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
class CrawlResponse(TypedDict):
|
|
124
|
+
crawlParameters: Dict[str, str]
|
|
125
|
+
results: CrawlResult
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
class SitemapResponse(TypedDict):
|
|
129
|
+
links: List[str]
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
class TrendingResult(TypedDict, total=False):
|
|
133
|
+
title: str
|
|
134
|
+
url: str
|
|
135
|
+
description: Optional[str]
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
class TrendingResponse(TypedDict):
|
|
139
|
+
trendingParameters: Dict[str, Any]
|
|
140
|
+
results: List[TrendingResult]
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
class ExtractResponse(TypedDict):
|
|
144
|
+
success: bool
|
|
145
|
+
extractParameters: Dict[str, str]
|
|
146
|
+
results: Any
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
class DeepcrawlAcceptedResponse(TypedDict):
|
|
150
|
+
taskId: str
|
|
151
|
+
status: str
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
class DeepcrawlStatusResponseRequired(TypedDict):
|
|
155
|
+
taskId: str
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
class DeepcrawlStatusResponse(DeepcrawlStatusResponseRequired, total=False):
|
|
159
|
+
status: str
|
|
160
|
+
success: bool
|
|
161
|
+
message: str
|
|
162
|
+
error: Optional[str]
|
|
163
|
+
r2Key: str
|
|
164
|
+
zipUrl: Optional[str]
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
class UsageResponse(TypedDict):
|
|
168
|
+
usage: int
|
|
169
|
+
user_id: Optional[str]
|
|
170
|
+
credential_type: str
|
|
171
|
+
client_id: Optional[str]
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
class HealthResponseRequired(TypedDict):
|
|
175
|
+
status: str
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
class HealthResponse(HealthResponseRequired, total=False):
|
|
179
|
+
timestamp: str
|
|
180
|
+
version: str
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: search1api
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Official Python client for Search1API
|
|
5
|
+
Project-URL: Documentation, https://www.search1api.com/docs/integrations/sdks
|
|
6
|
+
Project-URL: Repository, https://github.com/superagents-lab/search1api-python
|
|
7
|
+
Project-URL: Issues, https://github.com/superagents-lab/search1api-python/issues
|
|
8
|
+
Author: Search1API
|
|
9
|
+
License: MIT License
|
|
10
|
+
|
|
11
|
+
Copyright (c) 2026 Search1API
|
|
12
|
+
|
|
13
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
14
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
15
|
+
in the Software without restriction, including without limitation the rights
|
|
16
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
17
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
18
|
+
furnished to do so, subject to the following conditions:
|
|
19
|
+
|
|
20
|
+
The above copyright notice and this permission notice shall be included in all
|
|
21
|
+
copies or substantial portions of the Software.
|
|
22
|
+
|
|
23
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
24
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
25
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
26
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
27
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
28
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
29
|
+
SOFTWARE.
|
|
30
|
+
|
|
31
|
+
License-File: LICENSE
|
|
32
|
+
Keywords: ai,api,crawl,sdk,search
|
|
33
|
+
Classifier: Development Status :: 3 - Alpha
|
|
34
|
+
Classifier: Intended Audience :: Developers
|
|
35
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
36
|
+
Classifier: Programming Language :: Python :: 3
|
|
37
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
38
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
39
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
40
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
41
|
+
Classifier: Typing :: Typed
|
|
42
|
+
Requires-Python: >=3.9
|
|
43
|
+
Requires-Dist: httpx<1,>=0.27
|
|
44
|
+
Provides-Extra: dev
|
|
45
|
+
Requires-Dist: build<2,>=1.2; extra == 'dev'
|
|
46
|
+
Requires-Dist: mypy<2,>=1.14; extra == 'dev'
|
|
47
|
+
Requires-Dist: pytest<9,>=8; extra == 'dev'
|
|
48
|
+
Requires-Dist: ruff<1,>=0.9; extra == 'dev'
|
|
49
|
+
Description-Content-Type: text/markdown
|
|
50
|
+
|
|
51
|
+
# Search1API Python SDK
|
|
52
|
+
|
|
53
|
+
Official synchronous and asynchronous Python clients for Search1API.
|
|
54
|
+
|
|
55
|
+
API documentation: [search1api.com/docs](https://www.search1api.com/docs)
|
|
56
|
+
|
|
57
|
+
## Install
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
pip install search1api
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
## Search
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
from search1api import Search1API
|
|
67
|
+
|
|
68
|
+
client = Search1API() # reads SEARCH1API_API_KEY
|
|
69
|
+
response = client.search(
|
|
70
|
+
"latest AI agent frameworks",
|
|
71
|
+
max_results=10,
|
|
72
|
+
crawl_results=3,
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
for result in response["results"]:
|
|
76
|
+
print(result["title"], result["link"])
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Use the client as a context manager when it owns the HTTP connection pool:
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
with Search1API("your-api-key") as client:
|
|
83
|
+
print(client.usage())
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
## Async
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
from search1api import AsyncSearch1API
|
|
90
|
+
|
|
91
|
+
async with AsyncSearch1API() as client:
|
|
92
|
+
response = await client.search("latest AI agent frameworks")
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## Deepcrawl
|
|
96
|
+
|
|
97
|
+
`deepcrawl` starts a task and waits for it to finish:
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
result = client.deepcrawl("https://example.com", type="all")
|
|
101
|
+
print(result["zipUrl"])
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Use `start_deepcrawl`, `get_deepcrawl_status`, and `wait_for_deepcrawl` when
|
|
105
|
+
the application needs to control persistence or polling itself.
|
|
106
|
+
|
|
107
|
+
The clients also support news, crawl, sitemap, trending, extract, usage, and
|
|
108
|
+
batch operations exposed by the Search1API HTTP API. Requests time out after
|
|
109
|
+
30 seconds and retry `429` and transient `5xx` responses twice by default.
|
|
110
|
+
Authentication, payment, and validation errors are never retried. Deepcrawl
|
|
111
|
+
task creation is not retried automatically because it is not idempotent.
|
|
112
|
+
|
|
113
|
+
## Development
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
python -m venv .venv
|
|
117
|
+
. .venv/bin/activate
|
|
118
|
+
python -m pip install -e ".[dev]"
|
|
119
|
+
ruff format --check .
|
|
120
|
+
ruff check .
|
|
121
|
+
mypy src/search1api
|
|
122
|
+
pytest
|
|
123
|
+
python -m build
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
The checked-in OpenAPI snapshot is used to verify that the client covers every
|
|
127
|
+
public operation.
|
|
128
|
+
|
|
129
|
+
## License
|
|
130
|
+
|
|
131
|
+
MIT
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
search1api/__init__.py,sha256=QrwqbTlbyGV3VDjCcCjEHdNJm_sY_n1NTDsUK9EcSrE,857
|
|
2
|
+
search1api/client.py,sha256=LqICwme7-h9uCn8dcPfMTEw4Sc0JQiHnU68Rq1dc5kA,23018
|
|
3
|
+
search1api/errors.py,sha256=pkrzglctP5nS5xi1WVPoCD2XfJ383p1Voae817Wm-QI,2532
|
|
4
|
+
search1api/py.typed,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
|
|
5
|
+
search1api/types.py,sha256=xHCnykgpuejv33bwrw-h9AooC_NdGAjD4cdo_RbPcdY,3447
|
|
6
|
+
search1api-0.1.0.dist-info/METADATA,sha256=e98Fq5KRJH0Ffi_2SVAbfylQWwGDfhvzfsDO8HRnf54,4196
|
|
7
|
+
search1api-0.1.0.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
8
|
+
search1api-0.1.0.dist-info/licenses/LICENSE,sha256=fOUw4rbziXxGWo6j89XSpLgYuXQPAQw3WjZpG1ezHyc,1068
|
|
9
|
+
search1api-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Search1API
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
22
|
+
|