reqstorm 2.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
reqstorm-2.0.1/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2021-2026 Melih Colpan
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,271 @@
1
+ Metadata-Version: 2.4
2
+ Name: reqstorm
3
+ Version: 2.0.1
4
+ Summary: Send large numbers of HTTP requests concurrently with asyncio, with retries, timeouts and per-request results.
5
+ Author-email: Melih Colpan <melihcolpan1@gmail.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://reqstorm.github.io
8
+ Project-URL: Source, https://github.com/melihcolpan/reqstorm
9
+ Project-URL: Changelog, https://github.com/melihcolpan/reqstorm/blob/master/CHANGELOG.md
10
+ Project-URL: Issues, https://github.com/melihcolpan/reqstorm/issues
11
+ Keywords: asyncio,aiohttp,http,requests,concurrent,bulk,rate-limit,retry,crawler,scraping
12
+ Classifier: Development Status :: 5 - Production/Stable
13
+ Classifier: Framework :: AsyncIO
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3 :: Only
18
+ Classifier: Programming Language :: Python :: 3.9
19
+ Classifier: Programming Language :: Python :: 3.10
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Programming Language :: Python :: 3.13
23
+ Classifier: Topic :: Internet :: WWW/HTTP
24
+ Classifier: Typing :: Typed
25
+ Requires-Python: >=3.9
26
+ Description-Content-Type: text/markdown
27
+ License-File: LICENSE
28
+ Requires-Dist: aiohttp<4,>=3.9
29
+ Provides-Extra: test
30
+ Requires-Dist: pytest>=8; extra == "test"
31
+ Requires-Dist: pytest-asyncio>=0.23; extra == "test"
32
+ Requires-Dist: trustme>=1.1; extra == "test"
33
+ Provides-Extra: lint
34
+ Requires-Dist: ruff>=0.6; extra == "lint"
35
+ Requires-Dist: mypy>=1.10; extra == "lint"
36
+ Dynamic: license-file
37
+
38
+ # reqstorm
39
+
40
+ [![PyPI](https://img.shields.io/pypi/v/reqstorm)](https://pypi.org/project/reqstorm/)
41
+ [![Python](https://img.shields.io/pypi/pyversions/reqstorm)](https://pypi.org/project/reqstorm/)
42
+ [![CI](https://github.com/melihcolpan/reqstorm/actions/workflows/ci.yml/badge.svg)](https://github.com/melihcolpan/reqstorm/actions/workflows/ci.yml)
43
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue)](LICENSE)
44
+
45
+ **Documentation: [reqstorm.github.io](https://reqstorm.github.io)**
46
+
47
+ > **reqstorm** is the new name of **reqt**. If you used reqt, see [Coming from reqt](#coming-from-reqt).
48
+
49
+ **reqstorm** sends large numbers of HTTP requests concurrently and gives you one result per request, with rate limits, retries, progress and output to files or databases built in.
50
+
51
+ ```python
52
+ import reqstorm
53
+
54
+ urls = [f"https://api.example.com/items/{i}" for i in range(7000)]
55
+
56
+ print(reqstorm.estimate(len(urls), rate_limit="100/min"))
57
+ # 7000 requests: about 1h 10m (limited by rate_limit)
58
+
59
+ results = reqstorm.fetch_all_sync(urls, rate_limit="100/min", retries=2, progress=True)
60
+
61
+ print(results.summary())
62
+ # {'total': 7000, 'ok': 6987, 'failed': 13, 'failures': {'HTTP 404': 9, 'TimeoutError': 4}}
63
+ for error in results.errors():
64
+ print(error["url"], error["error"] or error["status"])
65
+ ```
66
+
67
+ - **Every request gets a result.** A timeout, a dropped connection or an invalid URL is recorded on that request and never stops the others. Each result keeps the history of its attempts.
68
+ - **Rate limits per host** in any unit (`"10/s"`, `"100/min"`, `"1000/h"`), plus overall and per-host concurrency limits.
69
+ - **Timeouts and retries**: immediate retries with exponential backoff and `Retry-After` support, and retry rounds that resend what still failed at the end.
70
+ - **Large batches**: stream results as they finish, or write them straight to JSONL, CSV, SQLite, PostgreSQL or MySQL, and resume an interrupted run where it stopped.
71
+ - **Works with or without asyncio**, and inside Jupyter notebooks.
72
+ - **TLS certificates are verified** by default. Fully typed, with a single dependency: [aiohttp](https://docs.aiohttp.org).
73
+
74
+ ## Installation
75
+
76
+ ```console
77
+ $ python -m pip install reqstorm
78
+ ```
79
+
80
+ reqstorm supports Python 3.9 and newer.
81
+
82
+ ## Usage
83
+
84
+ Every function has an async version for code that already runs an event loop: `fetch_all`, `stream`, `fetch_to_file`, `fetch_to_db`. The blocking versions end in `_sync`. They take the same options.
85
+
86
+ ```python
87
+ results = reqstorm.fetch_all_sync(urls) # in a script or a notebook
88
+ results = await reqstorm.fetch_all(urls) # inside async code
89
+ ```
90
+
91
+ ### Results
92
+
93
+ `fetch_all` returns a `Results` list with one `Result` per request, in input order.
94
+
95
+ | `Result` attribute | Meaning |
96
+ |---|---|
97
+ | `ok` | `True` when a response arrived with a status below 400 |
98
+ | `status`, `headers`, `body` | The response (`status` is `None` if none arrived) |
99
+ | `text()`, `json()` | The body decoded as text (using the response charset) or JSON |
100
+ | `error` | The exception that prevented a response, or `None` |
101
+ | `attempts`, `elapsed`, `history` | Number of attempts, total seconds, and each attempt's status, error and duration |
102
+ | `url`, `method`, `final_url`, `index` | What was requested, where redirects ended, and its position in the input |
103
+
104
+ `result.raise_for_error()` raises the request's error, or `reqstorm.HTTPStatusError` for a status of 400 or above. `result.to_dict()` gives the same fields as plain data.
105
+
106
+ The `Results` list has reporting helpers:
107
+
108
+ ```python
109
+ results.succeeded # list of successful results
110
+ results.failed # list of failed results
111
+ results.summary() # {'total': ..., 'ok': ..., 'failed': ..., 'failures': {'HTTP 404': 9, ...}}
112
+ results.errors() # failed requests as dicts with index, url, status, error, attempts and history
113
+ results.to_dicts() # every result as a dict, ready for json.dump or a DataFrame
114
+ ```
115
+
116
+ ### Rate limits and concurrency
117
+
118
+ ```python
119
+ results = reqstorm.fetch_all_sync(
120
+ urls,
121
+ rate_limit="100/min", # per host; also 5 (per second), "10/s", "30/5min", "1000/h", (100, 60)
122
+ concurrency=50, # at most 50 requests in flight in total
123
+ concurrency_per_host=10, # and at most 10 to the same host
124
+ )
125
+ ```
126
+
127
+ The rate limit applies to each host (`host:port`) separately and counts retries too. `reqstorm.estimate(requests, rate_limit=..., hosts=..., concurrency=..., latency=...)` tells you how long a batch should take before you send it.
128
+
129
+ ### Timeouts and retries
130
+
131
+ ```python
132
+ results = reqstorm.fetch_all_sync(
133
+ urls,
134
+ timeout=10, # seconds per attempt, including the body; None disables it
135
+ retries=3, # retry right away: 0.5 s, 1 s, 2 s apart (Retry-After takes precedence)
136
+ backoff=0.5,
137
+ retry_rounds=2, # then resend whatever still failed, after everything else...
138
+ retry_round_delay=30, # ...waiting 30 s before each round
139
+ )
140
+ ```
141
+
142
+ Connection errors, timeouts and the statuses in `retry_statuses` (default: 429, 500, 502, 503, 504) are retried. Invalid URLs and other statuses such as 404 are not.
143
+
144
+ ### Writing results to a file
145
+
146
+ `fetch_to_file` writes each result as soon as it is final, so results never pile up in memory:
147
+
148
+ ```python
149
+ summary = reqstorm.fetch_to_file_sync(urls, "results.jsonl", rate_limit="10/s", progress=True)
150
+ print(summary.ok, summary.failed, summary.errors[:5])
151
+ ```
152
+
153
+ - **Format** follows the file name: `.jsonl` (JSON Lines), `.csv`, or `.db` / `.sqlite` for SQLite. Pass `format=` to override it.
154
+ - **Order:** records are written as requests complete. With `ordered=True` they are written in input order instead; finished records then wait in memory until the ones before them are done.
155
+ - **Resume:** with `resume=True`, requests already recorded as successful are skipped, and the others, including earlier failures, are sent again and appended. Without it a JSONL or CSV file is overwritten.
156
+ - **Fields:** each record has `index`, `method`, `url`, `status`, `ok`, `error`, `attempts`, `elapsed`, `final_url`, `history` and `body`. Use `body="none"` to leave the body out, `body="base64"` for binary responses, and `include_headers=True` to add the response headers.
157
+
158
+ `progress=True` prints a line such as `reqstorm: 3500/7000 (50%) ok 3493 failed 7 1.7 req/s ETA 35m 00s` to stderr.
159
+
160
+ ### Writing results to a database
161
+
162
+ `fetch_to_db` inserts one row per request through a database connection you already have:
163
+
164
+ ```python
165
+ import psycopg # or sqlite3, psycopg2, pymysql, MySQLdb, mysql.connector
166
+
167
+ with psycopg.connect("dbname=crawl") as connection:
168
+ summary = reqstorm.fetch_to_db_sync(urls, connection, table="api_results", resume=True)
169
+ ```
170
+
171
+ reqstorm creates the table if it does not exist. The fields you filter on are real columns, and the variable parts are JSON:
172
+
173
+ | Column | PostgreSQL | MySQL | SQLite |
174
+ |---|---|---|---|
175
+ | `id` | `BIGSERIAL` | `BIGINT AUTO_INCREMENT` | `INTEGER` |
176
+ | `request_index`, `status`, `attempts` | `INTEGER` | `INT` | `INTEGER` |
177
+ | `method`, `url`, `error`, `final_url` | `TEXT` | `VARCHAR` / `LONGTEXT` | `TEXT` |
178
+ | `ok` | `BOOLEAN` | `BOOLEAN` | `INTEGER` |
179
+ | `elapsed` | `DOUBLE PRECISION` | `DOUBLE` | `REAL` |
180
+ | `history`, `headers` | `JSONB` | `JSON` | `TEXT` (JSON) |
181
+ | `body` | `TEXT`, or `BYTEA` with `body="bytes"` | `LONGTEXT` / `LONGBLOB` | `TEXT` / `BLOB` |
182
+ | `created_at` | `TIMESTAMPTZ` | `TIMESTAMP` | `TEXT` |
183
+
184
+ So `SELECT url, error FROM api_results WHERE NOT ok` works directly. Rows are inserted in batches on a background thread, so a remote database does not slow down the requests. Existing rows are never deleted; with `resume=True`, requests that already have a successful row are skipped.
185
+
186
+ ### Methods, headers and bodies
187
+
188
+ Options given to `fetch_all` apply to every request. Use `reqstorm.Request` to vary them per request:
189
+
190
+ ```python
191
+ results = reqstorm.fetch_all_sync(
192
+ [
193
+ "https://api.example.com/items/1",
194
+ reqstorm.Request("https://api.example.com/items", method="POST", json={"name": "new"}),
195
+ ],
196
+ headers={"Authorization": "Bearer ..."}, # merged with each Request's own headers
197
+ )
198
+ ```
199
+
200
+ `fetch_all_sync(urls, "POST", json=...)` sends every plain URL as a POST. `params`, `json` and `data` work the same way.
201
+
202
+ ### Streaming results
203
+
204
+ `stream` (or `stream_sync`) yields each result as soon as it completes. `urls` can be a generator, which is read lazily, so millions of URLs never need to be in memory at once:
205
+
206
+ ```python
207
+ for result in reqstorm.stream_sync(read_urls_from_file(), concurrency=100):
208
+ save(result.url, result.status, result.body)
209
+ ```
210
+
211
+ `fetch_all` also accepts `callback=`, a function or coroutine function called with each result as it completes.
212
+
213
+ ### TLS
214
+
215
+ Certificates are verified by default. To trust a private certificate authority, pass an `ssl.SSLContext`:
216
+
217
+ ```python
218
+ import ssl
219
+
220
+ context = ssl.create_default_context(cafile="internal-ca.pem")
221
+ results = reqstorm.fetch_all_sync(urls, ssl=context)
222
+ ```
223
+
224
+ `verify_ssl=False` turns verification off. Only use it for hosts you control.
225
+
226
+ ### Using your own session
227
+
228
+ In async code, pass an existing `aiohttp.ClientSession` with `session=` to share cookies, connection pools or proxy settings. reqstorm will not close it.
229
+
230
+ ## Coming from reqt
231
+
232
+ reqstorm 2.0 is the next version of reqt, renamed because another project already uses that name. Install `reqstorm` and replace `import reqt` with `import reqstorm`. The `reqt` package on PyPI now just installs reqstorm and shows a deprecation warning, so existing code keeps working in the meantime.
233
+
234
+ reqt 1.x called a function with each raw response and returned nothing. That style still works, with a `DeprecationWarning`:
235
+
236
+ ```python
237
+ async def handle(response): # reqt 1.x
238
+ print(response.status)
239
+
240
+
241
+ await reqstorm.fetch_all(urls=urls, method=handle)
242
+ ```
243
+
244
+ Two things changed even in the 1.x style:
245
+
246
+ - **TLS certificates are now verified.** 1.x silently skipped verification, which let anyone in the network path impersonate the server. Pass `verify_ssl=False` only if you really need the old behaviour for hosts you control.
247
+ - **Every failed request is logged** to the `reqstorm` logger instead of only connection errors; other errors no longer abort the whole batch.
248
+
249
+ The 2.0 equivalent of the example above is:
250
+
251
+ ```python
252
+ results = await reqstorm.fetch_all(urls)
253
+ for result in results:
254
+ print(result.status)
255
+ ```
256
+
257
+ The reqt 1.x style will be removed in reqstorm 3.0.
258
+
259
+ ## Development
260
+
261
+ ```console
262
+ $ python -m pip install -e ".[test,lint]"
263
+ $ pytest
264
+ $ ruff check . && ruff format --check . && mypy reqstorm
265
+ ```
266
+
267
+ The tests run against a local server and need no network access. The PostgreSQL and MySQL tests run when `REQSTORM_TEST_POSTGRES` (a libpq connection string) and `REQSTORM_TEST_MYSQL` (`host:port:user:password:database`) are set.
268
+
269
+ ## License
270
+
271
+ MIT, see [LICENSE](LICENSE).
@@ -0,0 +1,234 @@
1
+ # reqstorm
2
+
3
+ [![PyPI](https://img.shields.io/pypi/v/reqstorm)](https://pypi.org/project/reqstorm/)
4
+ [![Python](https://img.shields.io/pypi/pyversions/reqstorm)](https://pypi.org/project/reqstorm/)
5
+ [![CI](https://github.com/melihcolpan/reqstorm/actions/workflows/ci.yml/badge.svg)](https://github.com/melihcolpan/reqstorm/actions/workflows/ci.yml)
6
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue)](LICENSE)
7
+
8
+ **Documentation: [reqstorm.github.io](https://reqstorm.github.io)**
9
+
10
+ > **reqstorm** is the new name of **reqt**. If you used reqt, see [Coming from reqt](#coming-from-reqt).
11
+
12
+ **reqstorm** sends large numbers of HTTP requests concurrently and gives you one result per request, with rate limits, retries, progress and output to files or databases built in.
13
+
14
+ ```python
15
+ import reqstorm
16
+
17
+ urls = [f"https://api.example.com/items/{i}" for i in range(7000)]
18
+
19
+ print(reqstorm.estimate(len(urls), rate_limit="100/min"))
20
+ # 7000 requests: about 1h 10m (limited by rate_limit)
21
+
22
+ results = reqstorm.fetch_all_sync(urls, rate_limit="100/min", retries=2, progress=True)
23
+
24
+ print(results.summary())
25
+ # {'total': 7000, 'ok': 6987, 'failed': 13, 'failures': {'HTTP 404': 9, 'TimeoutError': 4}}
26
+ for error in results.errors():
27
+ print(error["url"], error["error"] or error["status"])
28
+ ```
29
+
30
+ - **Every request gets a result.** A timeout, a dropped connection or an invalid URL is recorded on that request and never stops the others. Each result keeps the history of its attempts.
31
+ - **Rate limits per host** in any unit (`"10/s"`, `"100/min"`, `"1000/h"`), plus overall and per-host concurrency limits.
32
+ - **Timeouts and retries**: immediate retries with exponential backoff and `Retry-After` support, and retry rounds that resend what still failed at the end.
33
+ - **Large batches**: stream results as they finish, or write them straight to JSONL, CSV, SQLite, PostgreSQL or MySQL, and resume an interrupted run where it stopped.
34
+ - **Works with or without asyncio**, and inside Jupyter notebooks.
35
+ - **TLS certificates are verified** by default. Fully typed, with a single dependency: [aiohttp](https://docs.aiohttp.org).
36
+
37
+ ## Installation
38
+
39
+ ```console
40
+ $ python -m pip install reqstorm
41
+ ```
42
+
43
+ reqstorm supports Python 3.9 and newer.
44
+
45
+ ## Usage
46
+
47
+ Every function has an async version for code that already runs an event loop: `fetch_all`, `stream`, `fetch_to_file`, `fetch_to_db`. The blocking versions end in `_sync`. They take the same options.
48
+
49
+ ```python
50
+ results = reqstorm.fetch_all_sync(urls) # in a script or a notebook
51
+ results = await reqstorm.fetch_all(urls) # inside async code
52
+ ```
53
+
54
+ ### Results
55
+
56
+ `fetch_all` returns a `Results` list with one `Result` per request, in input order.
57
+
58
+ | `Result` attribute | Meaning |
59
+ |---|---|
60
+ | `ok` | `True` when a response arrived with a status below 400 |
61
+ | `status`, `headers`, `body` | The response (`status` is `None` if none arrived) |
62
+ | `text()`, `json()` | The body decoded as text (using the response charset) or JSON |
63
+ | `error` | The exception that prevented a response, or `None` |
64
+ | `attempts`, `elapsed`, `history` | Number of attempts, total seconds, and each attempt's status, error and duration |
65
+ | `url`, `method`, `final_url`, `index` | What was requested, where redirects ended, and its position in the input |
66
+
67
+ `result.raise_for_error()` raises the request's error, or `reqstorm.HTTPStatusError` for a status of 400 or above. `result.to_dict()` gives the same fields as plain data.
68
+
69
+ The `Results` list has reporting helpers:
70
+
71
+ ```python
72
+ results.succeeded # list of successful results
73
+ results.failed # list of failed results
74
+ results.summary() # {'total': ..., 'ok': ..., 'failed': ..., 'failures': {'HTTP 404': 9, ...}}
75
+ results.errors() # failed requests as dicts with index, url, status, error, attempts and history
76
+ results.to_dicts() # every result as a dict, ready for json.dump or a DataFrame
77
+ ```
78
+
79
+ ### Rate limits and concurrency
80
+
81
+ ```python
82
+ results = reqstorm.fetch_all_sync(
83
+ urls,
84
+ rate_limit="100/min", # per host; also 5 (per second), "10/s", "30/5min", "1000/h", (100, 60)
85
+ concurrency=50, # at most 50 requests in flight in total
86
+ concurrency_per_host=10, # and at most 10 to the same host
87
+ )
88
+ ```
89
+
90
+ The rate limit applies to each host (`host:port`) separately and counts retries too. `reqstorm.estimate(requests, rate_limit=..., hosts=..., concurrency=..., latency=...)` tells you how long a batch should take before you send it.
91
+
92
+ ### Timeouts and retries
93
+
94
+ ```python
95
+ results = reqstorm.fetch_all_sync(
96
+ urls,
97
+ timeout=10, # seconds per attempt, including the body; None disables it
98
+ retries=3, # retry right away: 0.5 s, 1 s, 2 s apart (Retry-After takes precedence)
99
+ backoff=0.5,
100
+ retry_rounds=2, # then resend whatever still failed, after everything else...
101
+ retry_round_delay=30, # ...waiting 30 s before each round
102
+ )
103
+ ```
104
+
105
+ Connection errors, timeouts and the statuses in `retry_statuses` (default: 429, 500, 502, 503, 504) are retried. Invalid URLs and other statuses such as 404 are not.
106
+
107
+ ### Writing results to a file
108
+
109
+ `fetch_to_file` writes each result as soon as it is final, so results never pile up in memory:
110
+
111
+ ```python
112
+ summary = reqstorm.fetch_to_file_sync(urls, "results.jsonl", rate_limit="10/s", progress=True)
113
+ print(summary.ok, summary.failed, summary.errors[:5])
114
+ ```
115
+
116
+ - **Format** follows the file name: `.jsonl` (JSON Lines), `.csv`, or `.db` / `.sqlite` for SQLite. Pass `format=` to override it.
117
+ - **Order:** records are written as requests complete. With `ordered=True` they are written in input order instead; finished records then wait in memory until the ones before them are done.
118
+ - **Resume:** with `resume=True`, requests already recorded as successful are skipped, and the others, including earlier failures, are sent again and appended. Without it a JSONL or CSV file is overwritten.
119
+ - **Fields:** each record has `index`, `method`, `url`, `status`, `ok`, `error`, `attempts`, `elapsed`, `final_url`, `history` and `body`. Use `body="none"` to leave the body out, `body="base64"` for binary responses, and `include_headers=True` to add the response headers.
120
+
121
+ `progress=True` prints a line such as `reqstorm: 3500/7000 (50%) ok 3493 failed 7 1.7 req/s ETA 35m 00s` to stderr.
122
+
123
+ ### Writing results to a database
124
+
125
+ `fetch_to_db` inserts one row per request through a database connection you already have:
126
+
127
+ ```python
128
+ import psycopg # or sqlite3, psycopg2, pymysql, MySQLdb, mysql.connector
129
+
130
+ with psycopg.connect("dbname=crawl") as connection:
131
+ summary = reqstorm.fetch_to_db_sync(urls, connection, table="api_results", resume=True)
132
+ ```
133
+
134
+ reqstorm creates the table if it does not exist. The fields you filter on are real columns, and the variable parts are JSON:
135
+
136
+ | Column | PostgreSQL | MySQL | SQLite |
137
+ |---|---|---|---|
138
+ | `id` | `BIGSERIAL` | `BIGINT AUTO_INCREMENT` | `INTEGER` |
139
+ | `request_index`, `status`, `attempts` | `INTEGER` | `INT` | `INTEGER` |
140
+ | `method`, `url`, `error`, `final_url` | `TEXT` | `VARCHAR` / `LONGTEXT` | `TEXT` |
141
+ | `ok` | `BOOLEAN` | `BOOLEAN` | `INTEGER` |
142
+ | `elapsed` | `DOUBLE PRECISION` | `DOUBLE` | `REAL` |
143
+ | `history`, `headers` | `JSONB` | `JSON` | `TEXT` (JSON) |
144
+ | `body` | `TEXT`, or `BYTEA` with `body="bytes"` | `LONGTEXT` / `LONGBLOB` | `TEXT` / `BLOB` |
145
+ | `created_at` | `TIMESTAMPTZ` | `TIMESTAMP` | `TEXT` |
146
+
147
+ So `SELECT url, error FROM api_results WHERE NOT ok` works directly. Rows are inserted in batches on a background thread, so a remote database does not slow down the requests. Existing rows are never deleted; with `resume=True`, requests that already have a successful row are skipped.
148
+
149
+ ### Methods, headers and bodies
150
+
151
+ Options given to `fetch_all` apply to every request. Use `reqstorm.Request` to vary them per request:
152
+
153
+ ```python
154
+ results = reqstorm.fetch_all_sync(
155
+ [
156
+ "https://api.example.com/items/1",
157
+ reqstorm.Request("https://api.example.com/items", method="POST", json={"name": "new"}),
158
+ ],
159
+ headers={"Authorization": "Bearer ..."}, # merged with each Request's own headers
160
+ )
161
+ ```
162
+
163
+ `fetch_all_sync(urls, "POST", json=...)` sends every plain URL as a POST. `params`, `json` and `data` work the same way.
164
+
165
+ ### Streaming results
166
+
167
+ `stream` (or `stream_sync`) yields each result as soon as it completes. `urls` can be a generator, which is read lazily, so millions of URLs never need to be in memory at once:
168
+
169
+ ```python
170
+ for result in reqstorm.stream_sync(read_urls_from_file(), concurrency=100):
171
+ save(result.url, result.status, result.body)
172
+ ```
173
+
174
+ `fetch_all` also accepts `callback=`, a function or coroutine function called with each result as it completes.
175
+
176
+ ### TLS
177
+
178
+ Certificates are verified by default. To trust a private certificate authority, pass an `ssl.SSLContext`:
179
+
180
+ ```python
181
+ import ssl
182
+
183
+ context = ssl.create_default_context(cafile="internal-ca.pem")
184
+ results = reqstorm.fetch_all_sync(urls, ssl=context)
185
+ ```
186
+
187
+ `verify_ssl=False` turns verification off. Only use it for hosts you control.
188
+
189
+ ### Using your own session
190
+
191
+ In async code, pass an existing `aiohttp.ClientSession` with `session=` to share cookies, connection pools or proxy settings. reqstorm will not close it.
192
+
193
+ ## Coming from reqt
194
+
195
+ reqstorm 2.0 is the next version of reqt, renamed because another project already uses that name. Install `reqstorm` and replace `import reqt` with `import reqstorm`. The `reqt` package on PyPI now just installs reqstorm and shows a deprecation warning, so existing code keeps working in the meantime.
196
+
197
+ reqt 1.x called a function with each raw response and returned nothing. That style still works, with a `DeprecationWarning`:
198
+
199
+ ```python
200
+ async def handle(response): # reqt 1.x
201
+ print(response.status)
202
+
203
+
204
+ await reqstorm.fetch_all(urls=urls, method=handle)
205
+ ```
206
+
207
+ Two things changed even in the 1.x style:
208
+
209
+ - **TLS certificates are now verified.** 1.x silently skipped verification, which let anyone in the network path impersonate the server. Pass `verify_ssl=False` only if you really need the old behaviour for hosts you control.
210
+ - **Every failed request is logged** to the `reqstorm` logger instead of only connection errors; other errors no longer abort the whole batch.
211
+
212
+ The 2.0 equivalent of the example above is:
213
+
214
+ ```python
215
+ results = await reqstorm.fetch_all(urls)
216
+ for result in results:
217
+ print(result.status)
218
+ ```
219
+
220
+ The reqt 1.x style will be removed in reqstorm 3.0.
221
+
222
+ ## Development
223
+
224
+ ```console
225
+ $ python -m pip install -e ".[test,lint]"
226
+ $ pytest
227
+ $ ruff check . && ruff format --check . && mypy reqstorm
228
+ ```
229
+
230
+ The tests run against a local server and need no network access. The PostgreSQL and MySQL tests run when `REQSTORM_TEST_POSTGRES` (a libpq connection string) and `REQSTORM_TEST_MYSQL` (`host:port:user:password:database`) are set.
231
+
232
+ ## License
233
+
234
+ MIT, see [LICENSE](LICENSE).
@@ -0,0 +1,74 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "reqstorm"
7
+ dynamic = ["version"]
8
+ description = "Send large numbers of HTTP requests concurrently with asyncio, with retries, timeouts and per-request results."
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ license-files = ["LICENSE"]
12
+ authors = [{ name = "Melih Colpan", email = "melihcolpan1@gmail.com" }]
13
+ requires-python = ">=3.9"
14
+ dependencies = ["aiohttp>=3.9,<4"]
15
+ keywords = ["asyncio", "aiohttp", "http", "requests", "concurrent", "bulk", "rate-limit", "retry", "crawler", "scraping"]
16
+ classifiers = [
17
+ "Development Status :: 5 - Production/Stable",
18
+ "Framework :: AsyncIO",
19
+ "Intended Audience :: Developers",
20
+ "Operating System :: OS Independent",
21
+ "Programming Language :: Python :: 3",
22
+ "Programming Language :: Python :: 3 :: Only",
23
+ "Programming Language :: Python :: 3.9",
24
+ "Programming Language :: Python :: 3.10",
25
+ "Programming Language :: Python :: 3.11",
26
+ "Programming Language :: Python :: 3.12",
27
+ "Programming Language :: Python :: 3.13",
28
+ "Topic :: Internet :: WWW/HTTP",
29
+ "Typing :: Typed",
30
+ ]
31
+
32
+ [project.optional-dependencies]
33
+ test = ["pytest>=8", "pytest-asyncio>=0.23", "trustme>=1.1"]
34
+ lint = ["ruff>=0.6", "mypy>=1.10"]
35
+
36
+ [project.urls]
37
+ Homepage = "https://reqstorm.github.io"
38
+ Source = "https://github.com/melihcolpan/reqstorm"
39
+ Changelog = "https://github.com/melihcolpan/reqstorm/blob/master/CHANGELOG.md"
40
+ Issues = "https://github.com/melihcolpan/reqstorm/issues"
41
+
42
+ [tool.setuptools]
43
+ packages = ["reqstorm"]
44
+
45
+ [tool.setuptools.dynamic]
46
+ version = { attr = "reqstorm.__version__" }
47
+
48
+ [tool.setuptools.package-data]
49
+ reqstorm = ["py.typed"]
50
+
51
+ [tool.pytest.ini_options]
52
+ asyncio_mode = "auto"
53
+ filterwarnings = [
54
+ "error",
55
+ # On Python 3.9, pytest-asyncio creates an event loop through asyncio.get_event_loop()
56
+ # after a test has used asyncio.run() (the blocking API does) and never closes it.
57
+ "ignore:unclosed event loop:ResourceWarning",
58
+ ]
59
+
60
+ [tool.ruff]
61
+ line-length = 110
62
+ target-version = "py39"
63
+ # Code samples in the docs and README are aligned by hand for reading
64
+ extend-exclude = ["*.md"]
65
+
66
+ [tool.ruff.lint]
67
+ select = ["E", "F", "W", "I", "B", "UP", "ASYNC"]
68
+ # Optional[...] / List[...] are kept for Python 3.9 support
69
+ ignore = ["UP006", "UP007", "UP035", "UP045"]
70
+
71
+ [tool.mypy]
72
+ python_version = "3.10"
73
+ strict = false
74
+ warn_unused_ignores = true
@@ -0,0 +1,53 @@
1
+ """reqstorm: send large numbers of HTTP requests concurrently with asyncio.
2
+
3
+ import asyncio
4
+ import reqstorm
5
+
6
+ async def main():
7
+ results = await reqstorm.fetch_all(["https://example.com", "https://example.org"])
8
+ for result in results:
9
+ print(result.url, result.status if result.ok else result.error)
10
+
11
+ asyncio.run(main())
12
+ """
13
+
14
+ from ._client import (
15
+ DEFAULT_RETRY_STATUSES,
16
+ Attempt,
17
+ HTTPStatusError,
18
+ Request,
19
+ Result,
20
+ Results,
21
+ fetch_all,
22
+ stream,
23
+ )
24
+ from ._files import Summary, fetch_to_db, fetch_to_file
25
+ from ._legacy import Reqt
26
+ from ._limits import parse_rate
27
+ from ._plan import Estimate, estimate
28
+ from ._sync import fetch_all_sync, fetch_to_db_sync, fetch_to_file_sync, stream_sync
29
+
30
+ __version__ = "2.0.1"
31
+
32
+ __all__ = [
33
+ "DEFAULT_RETRY_STATUSES",
34
+ "Attempt",
35
+ "Estimate",
36
+ "HTTPStatusError",
37
+ "Request",
38
+ "Reqt",
39
+ "Result",
40
+ "Results",
41
+ "Summary",
42
+ "estimate",
43
+ "fetch_all",
44
+ "fetch_all_sync",
45
+ "fetch_to_db",
46
+ "fetch_to_db_sync",
47
+ "fetch_to_file",
48
+ "fetch_to_file_sync",
49
+ "stream",
50
+ "parse_rate",
51
+ "stream_sync",
52
+ "__version__",
53
+ ]