reqstorm 2.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- reqstorm-2.0.1/LICENSE +21 -0
- reqstorm-2.0.1/PKG-INFO +271 -0
- reqstorm-2.0.1/README.md +234 -0
- reqstorm-2.0.1/pyproject.toml +74 -0
- reqstorm-2.0.1/reqstorm/__init__.py +53 -0
- reqstorm-2.0.1/reqstorm/_client.py +560 -0
- reqstorm-2.0.1/reqstorm/_files.py +548 -0
- reqstorm-2.0.1/reqstorm/_legacy.py +95 -0
- reqstorm-2.0.1/reqstorm/_limits.py +73 -0
- reqstorm-2.0.1/reqstorm/_plan.py +60 -0
- reqstorm-2.0.1/reqstorm/_progress.py +82 -0
- reqstorm-2.0.1/reqstorm/_sync.py +135 -0
- reqstorm-2.0.1/reqstorm/py.typed +0 -0
- reqstorm-2.0.1/reqstorm.egg-info/PKG-INFO +271 -0
- reqstorm-2.0.1/reqstorm.egg-info/SOURCES.txt +26 -0
- reqstorm-2.0.1/reqstorm.egg-info/dependency_links.txt +1 -0
- reqstorm-2.0.1/reqstorm.egg-info/requires.txt +10 -0
- reqstorm-2.0.1/reqstorm.egg-info/top_level.txt +1 -0
- reqstorm-2.0.1/setup.cfg +4 -0
- reqstorm-2.0.1/tests/test_fetch_all.py +153 -0
- reqstorm-2.0.1/tests/test_files.py +113 -0
- reqstorm-2.0.1/tests/test_legacy.py +63 -0
- reqstorm-2.0.1/tests/test_limits.py +61 -0
- reqstorm-2.0.1/tests/test_ordered_and_db.py +156 -0
- reqstorm-2.0.1/tests/test_report.py +120 -0
- reqstorm-2.0.1/tests/test_stream.py +41 -0
- reqstorm-2.0.1/tests/test_sync.py +59 -0
- reqstorm-2.0.1/tests/test_tls.py +21 -0
reqstorm-2.0.1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2021-2026 Melih Colpan
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
reqstorm-2.0.1/PKG-INFO
ADDED
|
@@ -0,0 +1,271 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: reqstorm
|
|
3
|
+
Version: 2.0.1
|
|
4
|
+
Summary: Send large numbers of HTTP requests concurrently with asyncio, with retries, timeouts and per-request results.
|
|
5
|
+
Author-email: Melih Colpan <melihcolpan1@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://reqstorm.github.io
|
|
8
|
+
Project-URL: Source, https://github.com/melihcolpan/reqstorm
|
|
9
|
+
Project-URL: Changelog, https://github.com/melihcolpan/reqstorm/blob/master/CHANGELOG.md
|
|
10
|
+
Project-URL: Issues, https://github.com/melihcolpan/reqstorm/issues
|
|
11
|
+
Keywords: asyncio,aiohttp,http,requests,concurrent,bulk,rate-limit,retry,crawler,scraping
|
|
12
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
13
|
+
Classifier: Framework :: AsyncIO
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
24
|
+
Classifier: Typing :: Typed
|
|
25
|
+
Requires-Python: >=3.9
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
License-File: LICENSE
|
|
28
|
+
Requires-Dist: aiohttp<4,>=3.9
|
|
29
|
+
Provides-Extra: test
|
|
30
|
+
Requires-Dist: pytest>=8; extra == "test"
|
|
31
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == "test"
|
|
32
|
+
Requires-Dist: trustme>=1.1; extra == "test"
|
|
33
|
+
Provides-Extra: lint
|
|
34
|
+
Requires-Dist: ruff>=0.6; extra == "lint"
|
|
35
|
+
Requires-Dist: mypy>=1.10; extra == "lint"
|
|
36
|
+
Dynamic: license-file
|
|
37
|
+
|
|
38
|
+
# reqstorm
|
|
39
|
+
|
|
40
|
+
[](https://pypi.org/project/reqstorm/)
|
|
41
|
+
[](https://pypi.org/project/reqstorm/)
|
|
42
|
+
[](https://github.com/melihcolpan/reqstorm/actions/workflows/ci.yml)
|
|
43
|
+
[](LICENSE)
|
|
44
|
+
|
|
45
|
+
**Documentation: [reqstorm.github.io](https://reqstorm.github.io)**
|
|
46
|
+
|
|
47
|
+
> **reqstorm** is the new name of **reqt**. If you used reqt, see [Coming from reqt](#coming-from-reqt).
|
|
48
|
+
|
|
49
|
+
**reqstorm** sends large numbers of HTTP requests concurrently and gives you one result per request, with rate limits, retries, progress and output to files or databases built in.
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
import reqstorm
|
|
53
|
+
|
|
54
|
+
urls = [f"https://api.example.com/items/{i}" for i in range(7000)]
|
|
55
|
+
|
|
56
|
+
print(reqstorm.estimate(len(urls), rate_limit="100/min"))
|
|
57
|
+
# 7000 requests: about 1h 10m (limited by rate_limit)
|
|
58
|
+
|
|
59
|
+
results = reqstorm.fetch_all_sync(urls, rate_limit="100/min", retries=2, progress=True)
|
|
60
|
+
|
|
61
|
+
print(results.summary())
|
|
62
|
+
# {'total': 7000, 'ok': 6987, 'failed': 13, 'failures': {'HTTP 404': 9, 'TimeoutError': 4}}
|
|
63
|
+
for error in results.errors():
|
|
64
|
+
print(error["url"], error["error"] or error["status"])
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
- **Every request gets a result.** A timeout, a dropped connection or an invalid URL is recorded on that request and never stops the others. Each result keeps the history of its attempts.
|
|
68
|
+
- **Rate limits per host** in any unit (`"10/s"`, `"100/min"`, `"1000/h"`), plus overall and per-host concurrency limits.
|
|
69
|
+
- **Timeouts and retries**: immediate retries with exponential backoff and `Retry-After` support, and retry rounds that resend what still failed at the end.
|
|
70
|
+
- **Large batches**: stream results as they finish, or write them straight to JSONL, CSV, SQLite, PostgreSQL or MySQL, and resume an interrupted run where it stopped.
|
|
71
|
+
- **Works with or without asyncio**, and inside Jupyter notebooks.
|
|
72
|
+
- **TLS certificates are verified** by default. Fully typed, with a single dependency: [aiohttp](https://docs.aiohttp.org).
|
|
73
|
+
|
|
74
|
+
## Installation
|
|
75
|
+
|
|
76
|
+
```console
|
|
77
|
+
$ python -m pip install reqstorm
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
reqstorm supports Python 3.9 and newer.
|
|
81
|
+
|
|
82
|
+
## Usage
|
|
83
|
+
|
|
84
|
+
Every function has an async version for code that already runs an event loop: `fetch_all`, `stream`, `fetch_to_file`, `fetch_to_db`. The blocking versions end in `_sync`. They take the same options.
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
results = reqstorm.fetch_all_sync(urls) # in a script or a notebook
|
|
88
|
+
results = await reqstorm.fetch_all(urls) # inside async code
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
### Results
|
|
92
|
+
|
|
93
|
+
`fetch_all` returns a `Results` list with one `Result` per request, in input order.
|
|
94
|
+
|
|
95
|
+
| `Result` attribute | Meaning |
|
|
96
|
+
|---|---|
|
|
97
|
+
| `ok` | `True` when a response arrived with a status below 400 |
|
|
98
|
+
| `status`, `headers`, `body` | The response (`status` is `None` if none arrived) |
|
|
99
|
+
| `text()`, `json()` | The body decoded as text (using the response charset) or JSON |
|
|
100
|
+
| `error` | The exception that prevented a response, or `None` |
|
|
101
|
+
| `attempts`, `elapsed`, `history` | Number of attempts, total seconds, and each attempt's status, error and duration |
|
|
102
|
+
| `url`, `method`, `final_url`, `index` | What was requested, where redirects ended, and its position in the input |
|
|
103
|
+
|
|
104
|
+
`result.raise_for_error()` raises the request's error, or `reqstorm.HTTPStatusError` for a status of 400 or above. `result.to_dict()` gives the same fields as plain data.
|
|
105
|
+
|
|
106
|
+
The `Results` list has reporting helpers:
|
|
107
|
+
|
|
108
|
+
```python
|
|
109
|
+
results.succeeded # list of successful results
|
|
110
|
+
results.failed # list of failed results
|
|
111
|
+
results.summary() # {'total': ..., 'ok': ..., 'failed': ..., 'failures': {'HTTP 404': 9, ...}}
|
|
112
|
+
results.errors() # failed requests as dicts with index, url, status, error, attempts and history
|
|
113
|
+
results.to_dicts() # every result as a dict, ready for json.dump or a DataFrame
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
### Rate limits and concurrency
|
|
117
|
+
|
|
118
|
+
```python
|
|
119
|
+
results = reqstorm.fetch_all_sync(
|
|
120
|
+
urls,
|
|
121
|
+
rate_limit="100/min", # per host; also 5 (per second), "10/s", "30/5min", "1000/h", (100, 60)
|
|
122
|
+
concurrency=50, # at most 50 requests in flight in total
|
|
123
|
+
concurrency_per_host=10, # and at most 10 to the same host
|
|
124
|
+
)
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
The rate limit applies to each host (`host:port`) separately and counts retries too. `reqstorm.estimate(requests, rate_limit=..., hosts=..., concurrency=..., latency=...)` tells you how long a batch should take before you send it.
|
|
128
|
+
|
|
129
|
+
### Timeouts and retries
|
|
130
|
+
|
|
131
|
+
```python
|
|
132
|
+
results = reqstorm.fetch_all_sync(
|
|
133
|
+
urls,
|
|
134
|
+
timeout=10, # seconds per attempt, including the body; None disables it
|
|
135
|
+
retries=3, # retry right away: 0.5 s, 1 s, 2 s apart (Retry-After takes precedence)
|
|
136
|
+
backoff=0.5,
|
|
137
|
+
retry_rounds=2, # then resend whatever still failed, after everything else...
|
|
138
|
+
retry_round_delay=30, # ...waiting 30 s before each round
|
|
139
|
+
)
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
Connection errors, timeouts and the statuses in `retry_statuses` (default: 429, 500, 502, 503, 504) are retried. Invalid URLs and other statuses such as 404 are not.
|
|
143
|
+
|
|
144
|
+
### Writing results to a file
|
|
145
|
+
|
|
146
|
+
`fetch_to_file` writes each result as soon as it is final, so results never pile up in memory:
|
|
147
|
+
|
|
148
|
+
```python
|
|
149
|
+
summary = reqstorm.fetch_to_file_sync(urls, "results.jsonl", rate_limit="10/s", progress=True)
|
|
150
|
+
print(summary.ok, summary.failed, summary.errors[:5])
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
- **Format** follows the file name: `.jsonl` (JSON Lines), `.csv`, or `.db` / `.sqlite` for SQLite. Pass `format=` to override it.
|
|
154
|
+
- **Order:** records are written as requests complete. With `ordered=True` they are written in input order instead; finished records then wait in memory until the ones before them are done.
|
|
155
|
+
- **Resume:** with `resume=True`, requests already recorded as successful are skipped, and the others, including earlier failures, are sent again and appended. Without it a JSONL or CSV file is overwritten.
|
|
156
|
+
- **Fields:** each record has `index`, `method`, `url`, `status`, `ok`, `error`, `attempts`, `elapsed`, `final_url`, `history` and `body`. Use `body="none"` to leave the body out, `body="base64"` for binary responses, and `include_headers=True` to add the response headers.
|
|
157
|
+
|
|
158
|
+
`progress=True` prints a line such as `reqstorm: 3500/7000 (50%) ok 3493 failed 7 1.7 req/s ETA 35m 00s` to stderr.
|
|
159
|
+
|
|
160
|
+
### Writing results to a database
|
|
161
|
+
|
|
162
|
+
`fetch_to_db` inserts one row per request through a database connection you already have:
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
import psycopg # or sqlite3, psycopg2, pymysql, MySQLdb, mysql.connector
|
|
166
|
+
|
|
167
|
+
with psycopg.connect("dbname=crawl") as connection:
|
|
168
|
+
summary = reqstorm.fetch_to_db_sync(urls, connection, table="api_results", resume=True)
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
reqstorm creates the table if it does not exist. The fields you filter on are real columns, and the variable parts are JSON:
|
|
172
|
+
|
|
173
|
+
| Column | PostgreSQL | MySQL | SQLite |
|
|
174
|
+
|---|---|---|---|
|
|
175
|
+
| `id` | `BIGSERIAL` | `BIGINT AUTO_INCREMENT` | `INTEGER` |
|
|
176
|
+
| `request_index`, `status`, `attempts` | `INTEGER` | `INT` | `INTEGER` |
|
|
177
|
+
| `method`, `url`, `error`, `final_url` | `TEXT` | `VARCHAR` / `LONGTEXT` | `TEXT` |
|
|
178
|
+
| `ok` | `BOOLEAN` | `BOOLEAN` | `INTEGER` |
|
|
179
|
+
| `elapsed` | `DOUBLE PRECISION` | `DOUBLE` | `REAL` |
|
|
180
|
+
| `history`, `headers` | `JSONB` | `JSON` | `TEXT` (JSON) |
|
|
181
|
+
| `body` | `TEXT`, or `BYTEA` with `body="bytes"` | `LONGTEXT` / `LONGBLOB` | `TEXT` / `BLOB` |
|
|
182
|
+
| `created_at` | `TIMESTAMPTZ` | `TIMESTAMP` | `TEXT` |
|
|
183
|
+
|
|
184
|
+
So `SELECT url, error FROM api_results WHERE NOT ok` works directly. Rows are inserted in batches on a background thread, so a remote database does not slow down the requests. Existing rows are never deleted; with `resume=True`, requests that already have a successful row are skipped.
|
|
185
|
+
|
|
186
|
+
### Methods, headers and bodies
|
|
187
|
+
|
|
188
|
+
Options given to `fetch_all` apply to every request. Use `reqstorm.Request` to vary them per request:
|
|
189
|
+
|
|
190
|
+
```python
|
|
191
|
+
results = reqstorm.fetch_all_sync(
|
|
192
|
+
[
|
|
193
|
+
"https://api.example.com/items/1",
|
|
194
|
+
reqstorm.Request("https://api.example.com/items", method="POST", json={"name": "new"}),
|
|
195
|
+
],
|
|
196
|
+
headers={"Authorization": "Bearer ..."}, # merged with each Request's own headers
|
|
197
|
+
)
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
`fetch_all_sync(urls, "POST", json=...)` sends every plain URL as a POST. `params`, `json` and `data` work the same way.
|
|
201
|
+
|
|
202
|
+
### Streaming results
|
|
203
|
+
|
|
204
|
+
`stream` (or `stream_sync`) yields each result as soon as it completes. `urls` can be a generator, which is read lazily, so millions of URLs never need to be in memory at once:
|
|
205
|
+
|
|
206
|
+
```python
|
|
207
|
+
for result in reqstorm.stream_sync(read_urls_from_file(), concurrency=100):
|
|
208
|
+
save(result.url, result.status, result.body)
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
`fetch_all` also accepts `callback=`, a function or coroutine function called with each result as it completes.
|
|
212
|
+
|
|
213
|
+
### TLS
|
|
214
|
+
|
|
215
|
+
Certificates are verified by default. To trust a private certificate authority, pass an `ssl.SSLContext`:
|
|
216
|
+
|
|
217
|
+
```python
|
|
218
|
+
import ssl
|
|
219
|
+
|
|
220
|
+
context = ssl.create_default_context(cafile="internal-ca.pem")
|
|
221
|
+
results = reqstorm.fetch_all_sync(urls, ssl=context)
|
|
222
|
+
```
|
|
223
|
+
|
|
224
|
+
`verify_ssl=False` turns verification off. Only use it for hosts you control.
|
|
225
|
+
|
|
226
|
+
### Using your own session
|
|
227
|
+
|
|
228
|
+
In async code, pass an existing `aiohttp.ClientSession` with `session=` to share cookies, connection pools or proxy settings. reqstorm will not close it.
|
|
229
|
+
|
|
230
|
+
## Coming from reqt
|
|
231
|
+
|
|
232
|
+
reqstorm 2.0 is the next version of reqt, renamed because another project already uses that name. Install `reqstorm` and replace `import reqt` with `import reqstorm`. The `reqt` package on PyPI now just installs reqstorm and shows a deprecation warning, so existing code keeps working in the meantime.
|
|
233
|
+
|
|
234
|
+
reqt 1.x called a function with each raw response and returned nothing. That style still works, with a `DeprecationWarning`:
|
|
235
|
+
|
|
236
|
+
```python
|
|
237
|
+
async def handle(response): # reqt 1.x
|
|
238
|
+
print(response.status)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
await reqstorm.fetch_all(urls=urls, method=handle)
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
Two things changed even in the 1.x style:
|
|
245
|
+
|
|
246
|
+
- **TLS certificates are now verified.** 1.x silently skipped verification, which let anyone in the network path impersonate the server. Pass `verify_ssl=False` only if you really need the old behaviour for hosts you control.
|
|
247
|
+
- **Every failed request is logged** to the `reqstorm` logger instead of only connection errors; other errors no longer abort the whole batch.
|
|
248
|
+
|
|
249
|
+
The 2.0 equivalent of the example above is:
|
|
250
|
+
|
|
251
|
+
```python
|
|
252
|
+
results = await reqstorm.fetch_all(urls)
|
|
253
|
+
for result in results:
|
|
254
|
+
print(result.status)
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
The reqt 1.x style will be removed in reqstorm 3.0.
|
|
258
|
+
|
|
259
|
+
## Development
|
|
260
|
+
|
|
261
|
+
```console
|
|
262
|
+
$ python -m pip install -e ".[test,lint]"
|
|
263
|
+
$ pytest
|
|
264
|
+
$ ruff check . && ruff format --check . && mypy reqstorm
|
|
265
|
+
```
|
|
266
|
+
|
|
267
|
+
The tests run against a local server and need no network access. The PostgreSQL and MySQL tests run when `REQSTORM_TEST_POSTGRES` (a libpq connection string) and `REQSTORM_TEST_MYSQL` (`host:port:user:password:database`) are set.
|
|
268
|
+
|
|
269
|
+
## License
|
|
270
|
+
|
|
271
|
+
MIT, see [LICENSE](LICENSE).
|
reqstorm-2.0.1/README.md
ADDED
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
# reqstorm
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/reqstorm/)
|
|
4
|
+
[](https://pypi.org/project/reqstorm/)
|
|
5
|
+
[](https://github.com/melihcolpan/reqstorm/actions/workflows/ci.yml)
|
|
6
|
+
[](LICENSE)
|
|
7
|
+
|
|
8
|
+
**Documentation: [reqstorm.github.io](https://reqstorm.github.io)**
|
|
9
|
+
|
|
10
|
+
> **reqstorm** is the new name of **reqt**. If you used reqt, see [Coming from reqt](#coming-from-reqt).
|
|
11
|
+
|
|
12
|
+
**reqstorm** sends large numbers of HTTP requests concurrently and gives you one result per request, with rate limits, retries, progress and output to files or databases built in.
|
|
13
|
+
|
|
14
|
+
```python
|
|
15
|
+
import reqstorm
|
|
16
|
+
|
|
17
|
+
urls = [f"https://api.example.com/items/{i}" for i in range(7000)]
|
|
18
|
+
|
|
19
|
+
print(reqstorm.estimate(len(urls), rate_limit="100/min"))
|
|
20
|
+
# 7000 requests: about 1h 10m (limited by rate_limit)
|
|
21
|
+
|
|
22
|
+
results = reqstorm.fetch_all_sync(urls, rate_limit="100/min", retries=2, progress=True)
|
|
23
|
+
|
|
24
|
+
print(results.summary())
|
|
25
|
+
# {'total': 7000, 'ok': 6987, 'failed': 13, 'failures': {'HTTP 404': 9, 'TimeoutError': 4}}
|
|
26
|
+
for error in results.errors():
|
|
27
|
+
print(error["url"], error["error"] or error["status"])
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
- **Every request gets a result.** A timeout, a dropped connection or an invalid URL is recorded on that request and never stops the others. Each result keeps the history of its attempts.
|
|
31
|
+
- **Rate limits per host** in any unit (`"10/s"`, `"100/min"`, `"1000/h"`), plus overall and per-host concurrency limits.
|
|
32
|
+
- **Timeouts and retries**: immediate retries with exponential backoff and `Retry-After` support, and retry rounds that resend what still failed at the end.
|
|
33
|
+
- **Large batches**: stream results as they finish, or write them straight to JSONL, CSV, SQLite, PostgreSQL or MySQL, and resume an interrupted run where it stopped.
|
|
34
|
+
- **Works with or without asyncio**, and inside Jupyter notebooks.
|
|
35
|
+
- **TLS certificates are verified** by default. Fully typed, with a single dependency: [aiohttp](https://docs.aiohttp.org).
|
|
36
|
+
|
|
37
|
+
## Installation
|
|
38
|
+
|
|
39
|
+
```console
|
|
40
|
+
$ python -m pip install reqstorm
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
reqstorm supports Python 3.9 and newer.
|
|
44
|
+
|
|
45
|
+
## Usage
|
|
46
|
+
|
|
47
|
+
Every function has an async version for code that already runs an event loop: `fetch_all`, `stream`, `fetch_to_file`, `fetch_to_db`. The blocking versions end in `_sync`. They take the same options.
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
results = reqstorm.fetch_all_sync(urls) # in a script or a notebook
|
|
51
|
+
results = await reqstorm.fetch_all(urls) # inside async code
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
### Results
|
|
55
|
+
|
|
56
|
+
`fetch_all` returns a `Results` list with one `Result` per request, in input order.
|
|
57
|
+
|
|
58
|
+
| `Result` attribute | Meaning |
|
|
59
|
+
|---|---|
|
|
60
|
+
| `ok` | `True` when a response arrived with a status below 400 |
|
|
61
|
+
| `status`, `headers`, `body` | The response (`status` is `None` if none arrived) |
|
|
62
|
+
| `text()`, `json()` | The body decoded as text (using the response charset) or JSON |
|
|
63
|
+
| `error` | The exception that prevented a response, or `None` |
|
|
64
|
+
| `attempts`, `elapsed`, `history` | Number of attempts, total seconds, and each attempt's status, error and duration |
|
|
65
|
+
| `url`, `method`, `final_url`, `index` | What was requested, where redirects ended, and its position in the input |
|
|
66
|
+
|
|
67
|
+
`result.raise_for_error()` raises the request's error, or `reqstorm.HTTPStatusError` for a status of 400 or above. `result.to_dict()` gives the same fields as plain data.
|
|
68
|
+
|
|
69
|
+
The `Results` list has reporting helpers:
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
results.succeeded # list of successful results
|
|
73
|
+
results.failed # list of failed results
|
|
74
|
+
results.summary() # {'total': ..., 'ok': ..., 'failed': ..., 'failures': {'HTTP 404': 9, ...}}
|
|
75
|
+
results.errors() # failed requests as dicts with index, url, status, error, attempts and history
|
|
76
|
+
results.to_dicts() # every result as a dict, ready for json.dump or a DataFrame
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
### Rate limits and concurrency
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
results = reqstorm.fetch_all_sync(
|
|
83
|
+
urls,
|
|
84
|
+
rate_limit="100/min", # per host; also 5 (per second), "10/s", "30/5min", "1000/h", (100, 60)
|
|
85
|
+
concurrency=50, # at most 50 requests in flight in total
|
|
86
|
+
concurrency_per_host=10, # and at most 10 to the same host
|
|
87
|
+
)
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
The rate limit applies to each host (`host:port`) separately and counts retries too. `reqstorm.estimate(requests, rate_limit=..., hosts=..., concurrency=..., latency=...)` tells you how long a batch should take before you send it.
|
|
91
|
+
|
|
92
|
+
### Timeouts and retries
|
|
93
|
+
|
|
94
|
+
```python
|
|
95
|
+
results = reqstorm.fetch_all_sync(
|
|
96
|
+
urls,
|
|
97
|
+
timeout=10, # seconds per attempt, including the body; None disables it
|
|
98
|
+
retries=3, # retry right away: 0.5 s, 1 s, 2 s apart (Retry-After takes precedence)
|
|
99
|
+
backoff=0.5,
|
|
100
|
+
retry_rounds=2, # then resend whatever still failed, after everything else...
|
|
101
|
+
retry_round_delay=30, # ...waiting 30 s before each round
|
|
102
|
+
)
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Connection errors, timeouts and the statuses in `retry_statuses` (default: 429, 500, 502, 503, 504) are retried. Invalid URLs and other statuses such as 404 are not.
|
|
106
|
+
|
|
107
|
+
### Writing results to a file
|
|
108
|
+
|
|
109
|
+
`fetch_to_file` writes each result as soon as it is final, so results never pile up in memory:
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
summary = reqstorm.fetch_to_file_sync(urls, "results.jsonl", rate_limit="10/s", progress=True)
|
|
113
|
+
print(summary.ok, summary.failed, summary.errors[:5])
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
- **Format** follows the file name: `.jsonl` (JSON Lines), `.csv`, or `.db` / `.sqlite` for SQLite. Pass `format=` to override it.
|
|
117
|
+
- **Order:** records are written as requests complete. With `ordered=True` they are written in input order instead; finished records then wait in memory until the ones before them are done.
|
|
118
|
+
- **Resume:** with `resume=True`, requests already recorded as successful are skipped, and the others, including earlier failures, are sent again and appended. Without it a JSONL or CSV file is overwritten.
|
|
119
|
+
- **Fields:** each record has `index`, `method`, `url`, `status`, `ok`, `error`, `attempts`, `elapsed`, `final_url`, `history` and `body`. Use `body="none"` to leave the body out, `body="base64"` for binary responses, and `include_headers=True` to add the response headers.
|
|
120
|
+
|
|
121
|
+
`progress=True` prints a line such as `reqstorm: 3500/7000 (50%) ok 3493 failed 7 1.7 req/s ETA 35m 00s` to stderr.
|
|
122
|
+
|
|
123
|
+
### Writing results to a database
|
|
124
|
+
|
|
125
|
+
`fetch_to_db` inserts one row per request through a database connection you already have:
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
import psycopg # or sqlite3, psycopg2, pymysql, MySQLdb, mysql.connector
|
|
129
|
+
|
|
130
|
+
with psycopg.connect("dbname=crawl") as connection:
|
|
131
|
+
summary = reqstorm.fetch_to_db_sync(urls, connection, table="api_results", resume=True)
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
reqstorm creates the table if it does not exist. The fields you filter on are real columns, and the variable parts are JSON:
|
|
135
|
+
|
|
136
|
+
| Column | PostgreSQL | MySQL | SQLite |
|
|
137
|
+
|---|---|---|---|
|
|
138
|
+
| `id` | `BIGSERIAL` | `BIGINT AUTO_INCREMENT` | `INTEGER` |
|
|
139
|
+
| `request_index`, `status`, `attempts` | `INTEGER` | `INT` | `INTEGER` |
|
|
140
|
+
| `method`, `url`, `error`, `final_url` | `TEXT` | `VARCHAR` / `LONGTEXT` | `TEXT` |
|
|
141
|
+
| `ok` | `BOOLEAN` | `BOOLEAN` | `INTEGER` |
|
|
142
|
+
| `elapsed` | `DOUBLE PRECISION` | `DOUBLE` | `REAL` |
|
|
143
|
+
| `history`, `headers` | `JSONB` | `JSON` | `TEXT` (JSON) |
|
|
144
|
+
| `body` | `TEXT`, or `BYTEA` with `body="bytes"` | `LONGTEXT` / `LONGBLOB` | `TEXT` / `BLOB` |
|
|
145
|
+
| `created_at` | `TIMESTAMPTZ` | `TIMESTAMP` | `TEXT` |
|
|
146
|
+
|
|
147
|
+
So `SELECT url, error FROM api_results WHERE NOT ok` works directly. Rows are inserted in batches on a background thread, so a remote database does not slow down the requests. Existing rows are never deleted; with `resume=True`, requests that already have a successful row are skipped.
|
|
148
|
+
|
|
149
|
+
### Methods, headers and bodies
|
|
150
|
+
|
|
151
|
+
Options given to `fetch_all` apply to every request. Use `reqstorm.Request` to vary them per request:
|
|
152
|
+
|
|
153
|
+
```python
|
|
154
|
+
results = reqstorm.fetch_all_sync(
|
|
155
|
+
[
|
|
156
|
+
"https://api.example.com/items/1",
|
|
157
|
+
reqstorm.Request("https://api.example.com/items", method="POST", json={"name": "new"}),
|
|
158
|
+
],
|
|
159
|
+
headers={"Authorization": "Bearer ..."}, # merged with each Request's own headers
|
|
160
|
+
)
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
`fetch_all_sync(urls, "POST", json=...)` sends every plain URL as a POST. `params`, `json` and `data` work the same way.
|
|
164
|
+
|
|
165
|
+
### Streaming results
|
|
166
|
+
|
|
167
|
+
`stream` (or `stream_sync`) yields each result as soon as it completes. `urls` can be a generator, which is read lazily, so millions of URLs never need to be in memory at once:
|
|
168
|
+
|
|
169
|
+
```python
|
|
170
|
+
for result in reqstorm.stream_sync(read_urls_from_file(), concurrency=100):
|
|
171
|
+
save(result.url, result.status, result.body)
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
`fetch_all` also accepts `callback=`, a function or coroutine function called with each result as it completes.
|
|
175
|
+
|
|
176
|
+
### TLS
|
|
177
|
+
|
|
178
|
+
Certificates are verified by default. To trust a private certificate authority, pass an `ssl.SSLContext`:
|
|
179
|
+
|
|
180
|
+
```python
|
|
181
|
+
import ssl
|
|
182
|
+
|
|
183
|
+
context = ssl.create_default_context(cafile="internal-ca.pem")
|
|
184
|
+
results = reqstorm.fetch_all_sync(urls, ssl=context)
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
`verify_ssl=False` turns verification off. Only use it for hosts you control.
|
|
188
|
+
|
|
189
|
+
### Using your own session
|
|
190
|
+
|
|
191
|
+
In async code, pass an existing `aiohttp.ClientSession` with `session=` to share cookies, connection pools or proxy settings. reqstorm will not close it.
|
|
192
|
+
|
|
193
|
+
## Coming from reqt
|
|
194
|
+
|
|
195
|
+
reqstorm 2.0 is the next version of reqt, renamed because another project already uses that name. Install `reqstorm` and replace `import reqt` with `import reqstorm`. The `reqt` package on PyPI now just installs reqstorm and shows a deprecation warning, so existing code keeps working in the meantime.
|
|
196
|
+
|
|
197
|
+
reqt 1.x called a function with each raw response and returned nothing. That style still works, with a `DeprecationWarning`:
|
|
198
|
+
|
|
199
|
+
```python
|
|
200
|
+
async def handle(response): # reqt 1.x
|
|
201
|
+
print(response.status)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
await reqstorm.fetch_all(urls=urls, method=handle)
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
Two things changed even in the 1.x style:
|
|
208
|
+
|
|
209
|
+
- **TLS certificates are now verified.** 1.x silently skipped verification, which let anyone in the network path impersonate the server. Pass `verify_ssl=False` only if you really need the old behaviour for hosts you control.
|
|
210
|
+
- **Every failed request is logged** to the `reqstorm` logger instead of only connection errors; other errors no longer abort the whole batch.
|
|
211
|
+
|
|
212
|
+
The 2.0 equivalent of the example above is:
|
|
213
|
+
|
|
214
|
+
```python
|
|
215
|
+
results = await reqstorm.fetch_all(urls)
|
|
216
|
+
for result in results:
|
|
217
|
+
print(result.status)
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
The reqt 1.x style will be removed in reqstorm 3.0.
|
|
221
|
+
|
|
222
|
+
## Development
|
|
223
|
+
|
|
224
|
+
```console
|
|
225
|
+
$ python -m pip install -e ".[test,lint]"
|
|
226
|
+
$ pytest
|
|
227
|
+
$ ruff check . && ruff format --check . && mypy reqstorm
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
The tests run against a local server and need no network access. The PostgreSQL and MySQL tests run when `REQSTORM_TEST_POSTGRES` (a libpq connection string) and `REQSTORM_TEST_MYSQL` (`host:port:user:password:database`) are set.
|
|
231
|
+
|
|
232
|
+
## License
|
|
233
|
+
|
|
234
|
+
MIT, see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "reqstorm"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "Send large numbers of HTTP requests concurrently with asyncio, with retries, timeouts and per-request results."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
authors = [{ name = "Melih Colpan", email = "melihcolpan1@gmail.com" }]
|
|
13
|
+
requires-python = ">=3.9"
|
|
14
|
+
dependencies = ["aiohttp>=3.9,<4"]
|
|
15
|
+
keywords = ["asyncio", "aiohttp", "http", "requests", "concurrent", "bulk", "rate-limit", "retry", "crawler", "scraping"]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Development Status :: 5 - Production/Stable",
|
|
18
|
+
"Framework :: AsyncIO",
|
|
19
|
+
"Intended Audience :: Developers",
|
|
20
|
+
"Operating System :: OS Independent",
|
|
21
|
+
"Programming Language :: Python :: 3",
|
|
22
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
23
|
+
"Programming Language :: Python :: 3.9",
|
|
24
|
+
"Programming Language :: Python :: 3.10",
|
|
25
|
+
"Programming Language :: Python :: 3.11",
|
|
26
|
+
"Programming Language :: Python :: 3.12",
|
|
27
|
+
"Programming Language :: Python :: 3.13",
|
|
28
|
+
"Topic :: Internet :: WWW/HTTP",
|
|
29
|
+
"Typing :: Typed",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
[project.optional-dependencies]
|
|
33
|
+
test = ["pytest>=8", "pytest-asyncio>=0.23", "trustme>=1.1"]
|
|
34
|
+
lint = ["ruff>=0.6", "mypy>=1.10"]
|
|
35
|
+
|
|
36
|
+
[project.urls]
|
|
37
|
+
Homepage = "https://reqstorm.github.io"
|
|
38
|
+
Source = "https://github.com/melihcolpan/reqstorm"
|
|
39
|
+
Changelog = "https://github.com/melihcolpan/reqstorm/blob/master/CHANGELOG.md"
|
|
40
|
+
Issues = "https://github.com/melihcolpan/reqstorm/issues"
|
|
41
|
+
|
|
42
|
+
[tool.setuptools]
|
|
43
|
+
packages = ["reqstorm"]
|
|
44
|
+
|
|
45
|
+
[tool.setuptools.dynamic]
|
|
46
|
+
version = { attr = "reqstorm.__version__" }
|
|
47
|
+
|
|
48
|
+
[tool.setuptools.package-data]
|
|
49
|
+
reqstorm = ["py.typed"]
|
|
50
|
+
|
|
51
|
+
[tool.pytest.ini_options]
|
|
52
|
+
asyncio_mode = "auto"
|
|
53
|
+
filterwarnings = [
|
|
54
|
+
"error",
|
|
55
|
+
# On Python 3.9, pytest-asyncio creates an event loop through asyncio.get_event_loop()
|
|
56
|
+
# after a test has used asyncio.run() (the blocking API does) and never closes it.
|
|
57
|
+
"ignore:unclosed event loop:ResourceWarning",
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
[tool.ruff]
|
|
61
|
+
line-length = 110
|
|
62
|
+
target-version = "py39"
|
|
63
|
+
# Code samples in the docs and README are aligned by hand for reading
|
|
64
|
+
extend-exclude = ["*.md"]
|
|
65
|
+
|
|
66
|
+
[tool.ruff.lint]
|
|
67
|
+
select = ["E", "F", "W", "I", "B", "UP", "ASYNC"]
|
|
68
|
+
# Optional[...] / List[...] are kept for Python 3.9 support
|
|
69
|
+
ignore = ["UP006", "UP007", "UP035", "UP045"]
|
|
70
|
+
|
|
71
|
+
[tool.mypy]
|
|
72
|
+
python_version = "3.10"
|
|
73
|
+
strict = false
|
|
74
|
+
warn_unused_ignores = true
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""reqstorm: send large numbers of HTTP requests concurrently with asyncio.
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import reqstorm
|
|
5
|
+
|
|
6
|
+
async def main():
|
|
7
|
+
results = await reqstorm.fetch_all(["https://example.com", "https://example.org"])
|
|
8
|
+
for result in results:
|
|
9
|
+
print(result.url, result.status if result.ok else result.error)
|
|
10
|
+
|
|
11
|
+
asyncio.run(main())
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from ._client import (
|
|
15
|
+
DEFAULT_RETRY_STATUSES,
|
|
16
|
+
Attempt,
|
|
17
|
+
HTTPStatusError,
|
|
18
|
+
Request,
|
|
19
|
+
Result,
|
|
20
|
+
Results,
|
|
21
|
+
fetch_all,
|
|
22
|
+
stream,
|
|
23
|
+
)
|
|
24
|
+
from ._files import Summary, fetch_to_db, fetch_to_file
|
|
25
|
+
from ._legacy import Reqt
|
|
26
|
+
from ._limits import parse_rate
|
|
27
|
+
from ._plan import Estimate, estimate
|
|
28
|
+
from ._sync import fetch_all_sync, fetch_to_db_sync, fetch_to_file_sync, stream_sync
|
|
29
|
+
|
|
30
|
+
__version__ = "2.0.1"
|
|
31
|
+
|
|
32
|
+
__all__ = [
|
|
33
|
+
"DEFAULT_RETRY_STATUSES",
|
|
34
|
+
"Attempt",
|
|
35
|
+
"Estimate",
|
|
36
|
+
"HTTPStatusError",
|
|
37
|
+
"Request",
|
|
38
|
+
"Reqt",
|
|
39
|
+
"Result",
|
|
40
|
+
"Results",
|
|
41
|
+
"Summary",
|
|
42
|
+
"estimate",
|
|
43
|
+
"fetch_all",
|
|
44
|
+
"fetch_all_sync",
|
|
45
|
+
"fetch_to_db",
|
|
46
|
+
"fetch_to_db_sync",
|
|
47
|
+
"fetch_to_file",
|
|
48
|
+
"fetch_to_file_sync",
|
|
49
|
+
"stream",
|
|
50
|
+
"parse_rate",
|
|
51
|
+
"stream_sync",
|
|
52
|
+
"__version__",
|
|
53
|
+
]
|