reqstorm 2.2.0__tar.gz → 2.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {reqstorm-2.2.0/reqstorm.egg-info → reqstorm-2.4.0}/PKG-INFO +54 -8
- {reqstorm-2.2.0 → reqstorm-2.4.0}/README.md +49 -7
- {reqstorm-2.2.0 → reqstorm-2.4.0}/pyproject.toml +3 -2
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/__init__.py +3 -1
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_adaptive.py +21 -7
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_cli.py +127 -67
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_client.py +149 -24
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_files.py +105 -37
- reqstorm-2.4.0/reqstorm/_observe.py +408 -0
- reqstorm-2.4.0/reqstorm/_progress.py +15 -0
- reqstorm-2.4.0/reqstorm/_socks.py +84 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0/reqstorm.egg-info}/PKG-INFO +54 -8
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm.egg-info/SOURCES.txt +4 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm.egg-info/requires.txt +5 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_cli.py +37 -0
- reqstorm-2.4.0/tests/test_observe.py +267 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_report.py +7 -8
- reqstorm-2.4.0/tests/test_socks_backoff.py +144 -0
- reqstorm-2.2.0/reqstorm/_progress.py +0 -82
- {reqstorm-2.2.0 → reqstorm-2.4.0}/LICENSE +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/__main__.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_auth.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_cache.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_legacy.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_limits.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_paginate.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_plan.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_pydantic.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_report.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_schema.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_schema_sinks.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_sync.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_template.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/py.typed +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm.egg-info/dependency_links.txt +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm.egg-info/entry_points.txt +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm.egg-info/top_level.txt +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/setup.cfg +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_adaptive.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_auth.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_cache_proxy.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_fetch_all.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_files.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_legacy.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_limits.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_ordered_and_db.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_paginate.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_pydantic.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_run_report.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_schema.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_schema_db.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_stream.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_sync.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_template.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_tls.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: reqstorm
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.4.0
|
|
4
4
|
Summary: Send large numbers of HTTP requests concurrently with asyncio, with retries, timeouts and per-request results.
|
|
5
5
|
Author-email: Melih Colpan <colpanmelih@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -28,15 +28,19 @@ License-File: LICENSE
|
|
|
28
28
|
Requires-Dist: aiohttp<4,>=3.9
|
|
29
29
|
Provides-Extra: pydantic
|
|
30
30
|
Requires-Dist: pydantic>=2; extra == "pydantic"
|
|
31
|
+
Provides-Extra: socks
|
|
32
|
+
Requires-Dist: aiohttp-socks>=0.10; extra == "socks"
|
|
31
33
|
Provides-Extra: test
|
|
32
34
|
Requires-Dist: pytest>=8; extra == "test"
|
|
33
35
|
Requires-Dist: pytest-asyncio>=0.23; extra == "test"
|
|
34
36
|
Requires-Dist: trustme>=1.1; extra == "test"
|
|
35
37
|
Requires-Dist: pydantic>=2; extra == "test"
|
|
38
|
+
Requires-Dist: aiohttp-socks>=0.10; extra == "test"
|
|
36
39
|
Provides-Extra: lint
|
|
37
40
|
Requires-Dist: ruff>=0.6; extra == "lint"
|
|
38
41
|
Requires-Dist: mypy>=1.10; extra == "lint"
|
|
39
42
|
Requires-Dist: pydantic>=2; extra == "lint"
|
|
43
|
+
Requires-Dist: aiohttp-socks>=0.10; extra == "lint"
|
|
40
44
|
Dynamic: license-file
|
|
41
45
|
|
|
42
46
|
# reqstorm
|
|
@@ -90,6 +94,7 @@ for error in results.errors():
|
|
|
90
94
|
- [Quick start](#quick-start)
|
|
91
95
|
- [Results and reports](#results-and-reports)
|
|
92
96
|
- [Rate limits and concurrency](#rate-limits-and-concurrency)
|
|
97
|
+
- [Progress and logging](#progress-and-logging)
|
|
93
98
|
- [Timeouts and retries](#timeouts-and-retries)
|
|
94
99
|
- [Writing results to a file](#writing-results-to-a-file)
|
|
95
100
|
- [Writing results to a database](#writing-results-to-a-database)
|
|
@@ -123,16 +128,17 @@ reqstorm does all of that for you, with one call.
|
|
|
123
128
|
| One result per request | Failures are recorded, never raised; every attempt is kept |
|
|
124
129
|
| Rate limits | Per host, in any unit (`"100/min"`), or `"auto"` from the server's 429s and headers |
|
|
125
130
|
| Concurrency | Overall and per host |
|
|
126
|
-
| Retries |
|
|
131
|
+
| Retries | Exponential backoff with a cap and jitter, `Retry-After`; plus end-of-run rounds |
|
|
127
132
|
| Reports | Failures by reason, p50/p95/p99 response times, per-host figures |
|
|
128
133
|
| Pagination | Next links, `Link` headers, cursors and page numbers |
|
|
129
134
|
| Requests from data | URL templates over CSV rows or SQL query results |
|
|
130
|
-
| Tokens and proxies | Refresh an expired token on 401; rotate through proxies |
|
|
135
|
+
| Tokens and proxies | Refresh an expired token on 401; rotate through HTTP and SOCKS proxies |
|
|
131
136
|
| Caching | ETag / `If-None-Match`: unchanged resources cost a 304 |
|
|
132
137
|
| Output | JSONL, CSV, SQLite, PostgreSQL, MySQL; ordered or as completed |
|
|
133
138
|
| Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
|
|
134
139
|
| Resume | Skip what already succeeded after an interruption |
|
|
135
140
|
| Planning | `estimate()` before you start, progress with ETA while running |
|
|
141
|
+
| Logging | Retries, pauses and failures; per-run level, file and JSON, independent of the app |
|
|
136
142
|
| API | Blocking (scripts, Jupyter), asyncio and a `reqstorm` command |
|
|
137
143
|
| Safety | TLS verified, timeouts on, bounded concurrency, fully typed |
|
|
138
144
|
|
|
@@ -142,7 +148,7 @@ reqstorm does all of that for you, with one call.
|
|
|
142
148
|
$ python -m pip install reqstorm
|
|
143
149
|
```
|
|
144
150
|
|
|
145
|
-
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`.
|
|
151
|
+
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`; for SOCKS proxies, `reqstorm[socks]`.
|
|
146
152
|
|
|
147
153
|
## Quick start
|
|
148
154
|
|
|
@@ -298,6 +304,41 @@ reqstorm: 3500/7000 (50%) ok 3493 failed 7
|
|
|
298
304
|
1.7 req/s ETA 35m 00s
|
|
299
305
|
```
|
|
300
306
|
|
|
307
|
+
## Progress and logging
|
|
308
|
+
|
|
309
|
+
`progress=True` shows a line on stderr that keeps moving even while every request is waiting, so you can tell a long run is alive:
|
|
310
|
+
|
|
311
|
+
```text
|
|
312
|
+
reqstorm: 3500/7000 (50%) ok 3493
|
|
313
|
+
failed 7 active 12 retries 41
|
|
314
|
+
1.7 req/s ETA 34m 10s
|
|
315
|
+
```
|
|
316
|
+
|
|
317
|
+
It shows requests in flight, retries, the rate over the last minute, the retry round, hosts paused after a 429, and the time left. `progress=` also takes a function, which gets a `ProgressInfo` every two seconds. For a generator, pass `total=` to get a percentage.
|
|
318
|
+
|
|
319
|
+
**Logs.** reqstorm logs each retry and its reason, pauses, token refreshes and final failures. Turn them on for one run, independent of your application's logging setup:
|
|
320
|
+
|
|
321
|
+
```python
|
|
322
|
+
reqstorm.fetch_to_file_sync(
|
|
323
|
+
urls, "out.jsonl",
|
|
324
|
+
log_level="INFO", # or DEBUG: every attempt
|
|
325
|
+
log_file="logs/", # a new file per run
|
|
326
|
+
log_format="json", # optional
|
|
327
|
+
)
|
|
328
|
+
```
|
|
329
|
+
|
|
330
|
+
```text
|
|
331
|
+
14:32:41 WARNING reqstorm: GET .../items/412
|
|
332
|
+
failed (HTTP 503) on attempt 1 of 3;
|
|
333
|
+
retrying in 0.4s
|
|
334
|
+
15:42:18 ERROR reqstorm: GET .../items/913
|
|
335
|
+
failed after 3 attempts: HTTP 404
|
|
336
|
+
```
|
|
337
|
+
|
|
338
|
+
Log files are never overwritten: a directory or `{time}` in the name gives a time-stamped file per run, and an existing `run.log` becomes `run-2.log`. Without `log_level`, reqstorm logs to the standard `"reqstorm"` logger and follows your application's configuration.
|
|
339
|
+
|
|
340
|
+
**Read results while they are written.** Output files are written at least once a second (`flush_interval`), even while nothing finishes, and never with half a line; SQLite files are opened in WAL mode, so other programs can read them during the run.
|
|
341
|
+
|
|
301
342
|
## Timeouts and retries
|
|
302
343
|
|
|
303
344
|
```python
|
|
@@ -306,13 +347,16 @@ results = reqstorm.fetch_all_sync(
|
|
|
306
347
|
timeout=10, # seconds per attempt
|
|
307
348
|
retries=3, # retry right away...
|
|
308
349
|
backoff=0.5, # ...0.5 s, 1 s, 2 s apart
|
|
350
|
+
max_backoff=30, # never wait longer
|
|
309
351
|
retry_rounds=2, # then resend what still
|
|
310
352
|
retry_round_delay=30, # failed, 30 s later
|
|
311
353
|
)
|
|
312
354
|
```
|
|
313
355
|
|
|
314
356
|
- **`timeout`** is the time allowed for one attempt, including reading the body (default 30 s; `None` disables it). A slow server only fails its own requests.
|
|
315
|
-
- **`retries`** retries a request right away. The
|
|
357
|
+
- **`retries`** retries a request right away, also when no response came back at all (timeouts, dropped connections). The wait starts at `backoff` and doubles each time (0.5, 1, 2, 4, ... s), up to **`max_backoff`** (30 s by default), so many retries never wait minutes.
|
|
358
|
+
- **Jitter** (on by default) waits a random time between half and all of that delay, so thousands of requests that failed together do not retry at the same moment. `jitter=False` waits exactly.
|
|
359
|
+
- A **`Retry-After`** header from the server (up to 60 s) is followed as given instead.
|
|
316
360
|
- **`retry_rounds`** holds back requests that still failed for a retryable reason and sends them again after the rest of the batch, for temporary outages. Each request still produces exactly one result, with all its attempts in `history`.
|
|
317
361
|
|
|
318
362
|
| Retried | Not retried |
|
|
@@ -484,15 +528,17 @@ auth = reqstorm.BearerAuth(refresh=get_token)
|
|
|
484
528
|
reqstorm.fetch_all_sync(urls, auth=auth)
|
|
485
529
|
```
|
|
486
530
|
|
|
487
|
-
**Proxies.** One proxy, a pool used in turn, or one per `Request
|
|
531
|
+
**Proxies.** One proxy, a pool used in turn, or one per `Request`. HTTP and SOCKS proxies can be mixed:
|
|
488
532
|
|
|
489
533
|
```python
|
|
490
534
|
reqstorm.fetch_all_sync(urls, proxy=[
|
|
491
535
|
"http://proxy-1.example.com:8080",
|
|
492
|
-
"
|
|
536
|
+
"socks5h://user:pass@proxy-2.example.com:1080",
|
|
493
537
|
])
|
|
494
538
|
```
|
|
495
539
|
|
|
540
|
+
SOCKS4 and SOCKS5 (`socks5h://` and `socks4a://` let the proxy resolve host names, as with Tor) need `pip install "reqstorm[socks]"`.
|
|
541
|
+
|
|
496
542
|
**Caching.** `Cache` keeps responses in an SQLite file. The next run asks the server with `If-None-Match`; an unchanged resource comes back as a `304` with no body, and the stored response is used. With `ttl`, recent responses skip the network entirely:
|
|
497
543
|
|
|
498
544
|
```python
|
|
@@ -571,7 +617,7 @@ $ reqstorm urls.txt -o shop.db --report \
|
|
|
571
617
|
--schema products.json --explode items
|
|
572
618
|
```
|
|
573
619
|
|
|
574
|
-
`reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
|
|
620
|
+
`-v` / `-vv` log more, `-q` logs only errors, and `--log-file logs/` writes a new log file per run. `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
|
|
575
621
|
|
|
576
622
|
## When to use something else
|
|
577
623
|
|
|
@@ -49,6 +49,7 @@ for error in results.errors():
|
|
|
49
49
|
- [Quick start](#quick-start)
|
|
50
50
|
- [Results and reports](#results-and-reports)
|
|
51
51
|
- [Rate limits and concurrency](#rate-limits-and-concurrency)
|
|
52
|
+
- [Progress and logging](#progress-and-logging)
|
|
52
53
|
- [Timeouts and retries](#timeouts-and-retries)
|
|
53
54
|
- [Writing results to a file](#writing-results-to-a-file)
|
|
54
55
|
- [Writing results to a database](#writing-results-to-a-database)
|
|
@@ -82,16 +83,17 @@ reqstorm does all of that for you, with one call.
|
|
|
82
83
|
| One result per request | Failures are recorded, never raised; every attempt is kept |
|
|
83
84
|
| Rate limits | Per host, in any unit (`"100/min"`), or `"auto"` from the server's 429s and headers |
|
|
84
85
|
| Concurrency | Overall and per host |
|
|
85
|
-
| Retries |
|
|
86
|
+
| Retries | Exponential backoff with a cap and jitter, `Retry-After`; plus end-of-run rounds |
|
|
86
87
|
| Reports | Failures by reason, p50/p95/p99 response times, per-host figures |
|
|
87
88
|
| Pagination | Next links, `Link` headers, cursors and page numbers |
|
|
88
89
|
| Requests from data | URL templates over CSV rows or SQL query results |
|
|
89
|
-
| Tokens and proxies | Refresh an expired token on 401; rotate through proxies |
|
|
90
|
+
| Tokens and proxies | Refresh an expired token on 401; rotate through HTTP and SOCKS proxies |
|
|
90
91
|
| Caching | ETag / `If-None-Match`: unchanged resources cost a 304 |
|
|
91
92
|
| Output | JSONL, CSV, SQLite, PostgreSQL, MySQL; ordered or as completed |
|
|
92
93
|
| Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
|
|
93
94
|
| Resume | Skip what already succeeded after an interruption |
|
|
94
95
|
| Planning | `estimate()` before you start, progress with ETA while running |
|
|
96
|
+
| Logging | Retries, pauses and failures; per-run level, file and JSON, independent of the app |
|
|
95
97
|
| API | Blocking (scripts, Jupyter), asyncio and a `reqstorm` command |
|
|
96
98
|
| Safety | TLS verified, timeouts on, bounded concurrency, fully typed |
|
|
97
99
|
|
|
@@ -101,7 +103,7 @@ reqstorm does all of that for you, with one call.
|
|
|
101
103
|
$ python -m pip install reqstorm
|
|
102
104
|
```
|
|
103
105
|
|
|
104
|
-
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`.
|
|
106
|
+
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`; for SOCKS proxies, `reqstorm[socks]`.
|
|
105
107
|
|
|
106
108
|
## Quick start
|
|
107
109
|
|
|
@@ -257,6 +259,41 @@ reqstorm: 3500/7000 (50%) ok 3493 failed 7
|
|
|
257
259
|
1.7 req/s ETA 35m 00s
|
|
258
260
|
```
|
|
259
261
|
|
|
262
|
+
## Progress and logging
|
|
263
|
+
|
|
264
|
+
`progress=True` shows a line on stderr that keeps moving even while every request is waiting, so you can tell a long run is alive:
|
|
265
|
+
|
|
266
|
+
```text
|
|
267
|
+
reqstorm: 3500/7000 (50%) ok 3493
|
|
268
|
+
failed 7 active 12 retries 41
|
|
269
|
+
1.7 req/s ETA 34m 10s
|
|
270
|
+
```
|
|
271
|
+
|
|
272
|
+
It shows requests in flight, retries, the rate over the last minute, the retry round, hosts paused after a 429, and the time left. `progress=` also takes a function, which gets a `ProgressInfo` every two seconds. For a generator, pass `total=` to get a percentage.
|
|
273
|
+
|
|
274
|
+
**Logs.** reqstorm logs each retry and its reason, pauses, token refreshes and final failures. Turn them on for one run, independent of your application's logging setup:
|
|
275
|
+
|
|
276
|
+
```python
|
|
277
|
+
reqstorm.fetch_to_file_sync(
|
|
278
|
+
urls, "out.jsonl",
|
|
279
|
+
log_level="INFO", # or DEBUG: every attempt
|
|
280
|
+
log_file="logs/", # a new file per run
|
|
281
|
+
log_format="json", # optional
|
|
282
|
+
)
|
|
283
|
+
```
|
|
284
|
+
|
|
285
|
+
```text
|
|
286
|
+
14:32:41 WARNING reqstorm: GET .../items/412
|
|
287
|
+
failed (HTTP 503) on attempt 1 of 3;
|
|
288
|
+
retrying in 0.4s
|
|
289
|
+
15:42:18 ERROR reqstorm: GET .../items/913
|
|
290
|
+
failed after 3 attempts: HTTP 404
|
|
291
|
+
```
|
|
292
|
+
|
|
293
|
+
Log files are never overwritten: a directory or `{time}` in the name gives a time-stamped file per run, and an existing `run.log` becomes `run-2.log`. Without `log_level`, reqstorm logs to the standard `"reqstorm"` logger and follows your application's configuration.
|
|
294
|
+
|
|
295
|
+
**Read results while they are written.** Output files are written at least once a second (`flush_interval`), even while nothing finishes, and never with half a line; SQLite files are opened in WAL mode, so other programs can read them during the run.
|
|
296
|
+
|
|
260
297
|
## Timeouts and retries
|
|
261
298
|
|
|
262
299
|
```python
|
|
@@ -265,13 +302,16 @@ results = reqstorm.fetch_all_sync(
|
|
|
265
302
|
timeout=10, # seconds per attempt
|
|
266
303
|
retries=3, # retry right away...
|
|
267
304
|
backoff=0.5, # ...0.5 s, 1 s, 2 s apart
|
|
305
|
+
max_backoff=30, # never wait longer
|
|
268
306
|
retry_rounds=2, # then resend what still
|
|
269
307
|
retry_round_delay=30, # failed, 30 s later
|
|
270
308
|
)
|
|
271
309
|
```
|
|
272
310
|
|
|
273
311
|
- **`timeout`** is the time allowed for one attempt, including reading the body (default 30 s; `None` disables it). A slow server only fails its own requests.
|
|
274
|
-
- **`retries`** retries a request right away. The
|
|
312
|
+
- **`retries`** retries a request right away, also when no response came back at all (timeouts, dropped connections). The wait starts at `backoff` and doubles each time (0.5, 1, 2, 4, ... s), up to **`max_backoff`** (30 s by default), so many retries never wait minutes.
|
|
313
|
+
- **Jitter** (on by default) waits a random time between half and all of that delay, so thousands of requests that failed together do not retry at the same moment. `jitter=False` waits exactly.
|
|
314
|
+
- A **`Retry-After`** header from the server (up to 60 s) is followed as given instead.
|
|
275
315
|
- **`retry_rounds`** holds back requests that still failed for a retryable reason and sends them again after the rest of the batch, for temporary outages. Each request still produces exactly one result, with all its attempts in `history`.
|
|
276
316
|
|
|
277
317
|
| Retried | Not retried |
|
|
@@ -443,15 +483,17 @@ auth = reqstorm.BearerAuth(refresh=get_token)
|
|
|
443
483
|
reqstorm.fetch_all_sync(urls, auth=auth)
|
|
444
484
|
```
|
|
445
485
|
|
|
446
|
-
**Proxies.** One proxy, a pool used in turn, or one per `Request
|
|
486
|
+
**Proxies.** One proxy, a pool used in turn, or one per `Request`. HTTP and SOCKS proxies can be mixed:
|
|
447
487
|
|
|
448
488
|
```python
|
|
449
489
|
reqstorm.fetch_all_sync(urls, proxy=[
|
|
450
490
|
"http://proxy-1.example.com:8080",
|
|
451
|
-
"
|
|
491
|
+
"socks5h://user:pass@proxy-2.example.com:1080",
|
|
452
492
|
])
|
|
453
493
|
```
|
|
454
494
|
|
|
495
|
+
SOCKS4 and SOCKS5 (`socks5h://` and `socks4a://` let the proxy resolve host names, as with Tor) need `pip install "reqstorm[socks]"`.
|
|
496
|
+
|
|
455
497
|
**Caching.** `Cache` keeps responses in an SQLite file. The next run asks the server with `If-None-Match`; an unchanged resource comes back as a `304` with no body, and the stored response is used. With `ttl`, recent responses skip the network entirely:
|
|
456
498
|
|
|
457
499
|
```python
|
|
@@ -530,7 +572,7 @@ $ reqstorm urls.txt -o shop.db --report \
|
|
|
530
572
|
--schema products.json --explode items
|
|
531
573
|
```
|
|
532
574
|
|
|
533
|
-
`reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
|
|
575
|
+
`-v` / `-vv` log more, `-q` logs only errors, and `--log-file logs/` writes a new log file per run. `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
|
|
534
576
|
|
|
535
577
|
## When to use something else
|
|
536
578
|
|
|
@@ -31,8 +31,9 @@ classifiers = [
|
|
|
31
31
|
|
|
32
32
|
[project.optional-dependencies]
|
|
33
33
|
pydantic = ["pydantic>=2"]
|
|
34
|
-
|
|
35
|
-
|
|
34
|
+
socks = ["aiohttp-socks>=0.10"]
|
|
35
|
+
test = ["pytest>=8", "pytest-asyncio>=0.23", "trustme>=1.1", "pydantic>=2", "aiohttp-socks>=0.10"]
|
|
36
|
+
lint = ["ruff>=0.6", "mypy>=1.10", "pydantic>=2", "aiohttp-socks>=0.10"]
|
|
36
37
|
|
|
37
38
|
[project.scripts]
|
|
38
39
|
reqstorm = "reqstorm._cli:main"
|
|
@@ -26,13 +26,14 @@ from ._client import (
|
|
|
26
26
|
from ._files import Summary, fetch_to_db, fetch_to_file
|
|
27
27
|
from ._legacy import Reqt
|
|
28
28
|
from ._limits import parse_rate
|
|
29
|
+
from ._observe import ProgressInfo
|
|
29
30
|
from ._paginate import Cursor, LinkHeader, NextLink, PageNumber, Paginator
|
|
30
31
|
from ._plan import Estimate, estimate
|
|
31
32
|
from ._schema import Field, Schema, SchemaError, extract, infer_schema
|
|
32
33
|
from ._sync import fetch_all_sync, fetch_to_db_sync, fetch_to_file_sync, stream_sync
|
|
33
34
|
from ._template import from_template, read_csv, read_sql
|
|
34
35
|
|
|
35
|
-
__version__ = "2.
|
|
36
|
+
__version__ = "2.4.0"
|
|
36
37
|
|
|
37
38
|
__all__ = [
|
|
38
39
|
"Attempt",
|
|
@@ -46,6 +47,7 @@ __all__ = [
|
|
|
46
47
|
"LinkHeader",
|
|
47
48
|
"NextLink",
|
|
48
49
|
"PageNumber",
|
|
50
|
+
"ProgressInfo",
|
|
49
51
|
"Paginator",
|
|
50
52
|
"Reqt",
|
|
51
53
|
"Request",
|
|
@@ -6,10 +6,12 @@ import asyncio
|
|
|
6
6
|
import time
|
|
7
7
|
from dataclasses import dataclass
|
|
8
8
|
from email.utils import parsedate_to_datetime
|
|
9
|
-
from typing import Dict, Mapping, Optional
|
|
9
|
+
from typing import Dict, Mapping, Optional, Tuple
|
|
10
10
|
|
|
11
11
|
from ._limits import host_key
|
|
12
12
|
|
|
13
|
+
__all__ = ["AdaptiveRateLimiter", "seconds_until_reset"]
|
|
14
|
+
|
|
13
15
|
# Header names, most specific first. "RateLimit-*" is the IETF draft; "X-RateLimit-*" is the
|
|
14
16
|
# common convention (GitHub, Twitter/X, many others).
|
|
15
17
|
_REMAINING = ("RateLimit-Remaining", "X-RateLimit-Remaining", "X-Rate-Limit-Remaining")
|
|
@@ -91,17 +93,28 @@ class AdaptiveRateLimiter:
|
|
|
91
93
|
if slot > now:
|
|
92
94
|
await asyncio.sleep(slot - now)
|
|
93
95
|
|
|
94
|
-
def
|
|
96
|
+
def paused(self) -> Dict[str, float]:
|
|
97
|
+
"""Hosts that are paused right now, with the seconds left."""
|
|
98
|
+
now = time.monotonic()
|
|
99
|
+
return {
|
|
100
|
+
host: state.paused_until - now for host, state in self._hosts.items() if state.paused_until > now
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
def record(
|
|
104
|
+
self, url: str, status: Optional[int], headers: Mapping[str, str]
|
|
105
|
+
) -> Optional[Tuple[float, str]]:
|
|
106
|
+
"""Learn from a response. Returns ``(seconds, reason)`` when the host is paused."""
|
|
95
107
|
state = self._state(url)
|
|
96
108
|
if state is None or status is None:
|
|
97
|
-
return
|
|
109
|
+
return None
|
|
98
110
|
now = time.monotonic()
|
|
99
111
|
reset = seconds_until_reset(headers)
|
|
100
112
|
if status == 429:
|
|
101
113
|
state.interval = min(max(state.interval * 2, 0.1), 60.0)
|
|
102
114
|
pause = reset if reset is not None else max(state.interval, 1.0)
|
|
103
115
|
state.paused_until = max(state.paused_until, now + pause)
|
|
104
|
-
|
|
116
|
+
source = "Retry-After" if reset is not None else "no Retry-After"
|
|
117
|
+
return pause, f"429 Too Many Requests ({source}), spacing requests {state.interval:.2f}s apart"
|
|
105
118
|
remaining_text = _header(headers, _REMAINING)
|
|
106
119
|
if remaining_text is not None and reset is not None:
|
|
107
120
|
try:
|
|
@@ -111,8 +124,9 @@ class AdaptiveRateLimiter:
|
|
|
111
124
|
if remaining is not None:
|
|
112
125
|
if remaining <= 0:
|
|
113
126
|
state.paused_until = max(state.paused_until, now + reset)
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
return
|
|
127
|
+
return (reset, "rate limit used up until the window resets") if reset > 0 else None
|
|
128
|
+
state.interval = min(reset / remaining, 60.0)
|
|
129
|
+
return None
|
|
117
130
|
if 200 <= status < 400 and state.interval:
|
|
118
131
|
state.interval = state.interval * 0.9 if state.interval > 0.01 else 0.0
|
|
132
|
+
return None
|
|
@@ -19,90 +19,142 @@ from ._schema import Field, Schema, SchemaError, infer_schema
|
|
|
19
19
|
from ._sync import fetch_all_sync, fetch_to_file_sync, stream_sync
|
|
20
20
|
from ._template import from_template, read_csv
|
|
21
21
|
|
|
22
|
-
|
|
22
|
+
EPILOG = """\
|
|
23
23
|
examples:
|
|
24
24
|
reqstorm urls.txt -o results.jsonl --rate 100/min --retries 2
|
|
25
25
|
reqstorm urls.txt --estimate --rate 100/min
|
|
26
26
|
cat urls.txt | reqstorm --rate auto > results.jsonl
|
|
27
|
-
reqstorm
|
|
27
|
+
reqstorm urls.txt -o results.db -v --log-file logs/
|
|
28
|
+
reqstorm users.csv --template "https://api.example.com/users/{id}" -o users.jsonl
|
|
28
29
|
reqstorm urls.txt --paginate next:links.next --schema products.json --explode items -o shop.db
|
|
29
30
|
reqstorm urls.txt --infer-schema 5 --explode items > products.json
|
|
31
|
+
reqstorm urls.txt --proxy socks5h://127.0.0.1:9050 -o out.jsonl
|
|
30
32
|
|
|
31
|
-
|
|
32
|
-
2
|
|
33
|
+
rate limits:
|
|
34
|
+
5 or 0.5 (per second), 10/s, 100/min, 30/5min, 1000/h, 2/day, per host; or auto to
|
|
35
|
+
follow the server's 429 responses and X-RateLimit headers.
|
|
36
|
+
|
|
37
|
+
environment:
|
|
38
|
+
REQSTORM_TOKEN bearer token, instead of --bearer (keeps it out of shell history)
|
|
39
|
+
|
|
40
|
+
exit status:
|
|
41
|
+
0 every request succeeded 1 some failed or records were rejected
|
|
42
|
+
2 invalid arguments 130 interrupted
|
|
43
|
+
|
|
44
|
+
documentation: https://reqstorm.github.io/guide/cli/
|
|
33
45
|
"""
|
|
34
46
|
|
|
35
47
|
|
|
48
|
+
# fmt: off
|
|
36
49
|
def _parser() -> argparse.ArgumentParser:
|
|
37
50
|
parser = argparse.ArgumentParser(
|
|
38
51
|
prog="reqstorm",
|
|
39
52
|
description="Send many HTTP requests with rate limits, retries and progress, and save the results.",
|
|
40
|
-
epilog=
|
|
41
|
-
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
53
|
+
epilog=EPILOG,
|
|
54
|
+
formatter_class=lambda prog: argparse.RawDescriptionHelpFormatter(prog, max_help_position=32),
|
|
42
55
|
)
|
|
43
|
-
parser.add_argument
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
56
|
+
add = parser.add_argument
|
|
57
|
+
add("input", nargs="?", default="-",
|
|
58
|
+
help="file with one URL per line (blank lines and # comments are skipped), or a CSV file "
|
|
59
|
+
"with --template; default: standard input")
|
|
60
|
+
add("-o", "--output", metavar="FILE",
|
|
61
|
+
help="write results to FILE: .jsonl, .csv, or .db/.sqlite for SQLite; "
|
|
62
|
+
"default: JSON lines on standard output")
|
|
63
|
+
add("--version", action="version", version=f"reqstorm {__version__}")
|
|
49
64
|
|
|
50
65
|
request = parser.add_argument_group("requests")
|
|
51
|
-
request.add_argument
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
pace
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
)
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
66
|
+
add = request.add_argument
|
|
67
|
+
add("-X", "--method", default="GET", help="HTTP method for every request (default: GET)")
|
|
68
|
+
add("-H", "--header", action="append", default=[], metavar="'NAME: VALUE'",
|
|
69
|
+
help="header sent with every request; repeat for more")
|
|
70
|
+
add("--json", metavar="JSON", help="JSON body sent with every request")
|
|
71
|
+
add("--data", metavar="TEXT", help="raw body sent with every request")
|
|
72
|
+
add("--template", metavar="URL",
|
|
73
|
+
help="treat the input as CSV and build one URL per row from {column} placeholders, "
|
|
74
|
+
"e.g. 'https://api.example.com/users/{id}'")
|
|
75
|
+
add("--bearer", metavar="TOKEN", help="send 'Authorization: Bearer TOKEN' (or set REQSTORM_TOKEN)")
|
|
76
|
+
add("--proxy", action="append", default=[], metavar="URL",
|
|
77
|
+
help="send through a proxy: http://, socks5://, socks5h://, socks4:// or socks4a://; "
|
|
78
|
+
"repeat to rotate through several")
|
|
79
|
+
add("--no-verify", action="store_true",
|
|
80
|
+
help="do not verify TLS certificates (only for hosts you control)")
|
|
81
|
+
|
|
82
|
+
pace = parser.add_argument_group("pace")
|
|
83
|
+
add = pace.add_argument
|
|
84
|
+
add("--rate", metavar="LIMIT",
|
|
85
|
+
help="per-host rate limit: 5, 10/s, 100/min, 1000/h, or auto (default: none)")
|
|
86
|
+
add("-c", "--concurrency", type=int, default=100, metavar="N",
|
|
87
|
+
help="requests in flight at once (default: 100)")
|
|
88
|
+
add("--per-host", type=int, default=0, metavar="N",
|
|
89
|
+
help="requests in flight per host (default: no limit)")
|
|
90
|
+
add("--timeout", type=float, default=30.0, metavar="SECONDS",
|
|
91
|
+
help="time allowed for one attempt, including the body (default: 30)")
|
|
92
|
+
|
|
93
|
+
retry = parser.add_argument_group("retries")
|
|
94
|
+
add = retry.add_argument
|
|
95
|
+
add("--retries", type=int, default=0, metavar="N",
|
|
96
|
+
help="retry timeouts, connection errors and 429/5xx up to N times (default: 0)")
|
|
97
|
+
add("--backoff", type=float, default=0.5, metavar="SECONDS",
|
|
98
|
+
help="wait before the first retry, doubled after each (default: 0.5)")
|
|
99
|
+
add("--max-backoff", type=float, default=30.0, metavar="SECONDS",
|
|
100
|
+
help="longest wait between retries (default: 30)")
|
|
101
|
+
add("--no-jitter", action="store_true", help="wait exactly, instead of a random half to all of the delay")
|
|
102
|
+
add("--retry-rounds", type=int, default=0, metavar="N",
|
|
103
|
+
help="send requests that still failed again after all others, up to N rounds (needs -o)")
|
|
104
|
+
add("--retry-round-delay", type=float, default=5.0, metavar="SECONDS",
|
|
105
|
+
help="wait before each retry round (default: 5)")
|
|
106
|
+
|
|
107
|
+
more = parser.add_argument_group("pagination and caching")
|
|
108
|
+
add = more.add_argument
|
|
109
|
+
add("--paginate", metavar="STRATEGY",
|
|
110
|
+
help="follow pages: next:PATH (URL in the JSON), link (Link header), cursor:PATH[:PARAM], "
|
|
111
|
+
"or page[:PARAM[:ITEMS_PATH]]")
|
|
112
|
+
add("--max-pages", type=int, default=1000, metavar="N",
|
|
113
|
+
help="pages per starting URL at most (default: 1000)")
|
|
114
|
+
add("--cache", metavar="FILE", help="keep responses in this SQLite file and revalidate them with ETag")
|
|
115
|
+
add("--cache-ttl", type=float, metavar="SECONDS",
|
|
116
|
+
help="use a cached response without asking the server for this long (default: always ask)")
|
|
82
117
|
|
|
83
118
|
output = parser.add_argument_group("output")
|
|
84
|
-
output.add_argument
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
119
|
+
add = output.add_argument
|
|
120
|
+
add("--body", choices=["text", "base64", "none"], default="text",
|
|
121
|
+
help="how to store response bodies (default: text)")
|
|
122
|
+
add("--include-headers", action="store_true", help="store response headers too")
|
|
123
|
+
add("--resume", action="store_true",
|
|
124
|
+
help="skip requests already saved successfully in -o, append the rest")
|
|
125
|
+
add("--ordered", action="store_true", help="write results in input order instead of as they finish")
|
|
126
|
+
add("--table", default="reqstorm_results", metavar="NAME",
|
|
127
|
+
help="table for .db output (default: reqstorm_results)")
|
|
128
|
+
add("--flush-interval", type=float, default=1.0, metavar="SECONDS",
|
|
129
|
+
help="write waiting results to -o at least this often (default: 1)")
|
|
130
|
+
|
|
131
|
+
data = parser.add_argument_group("typed columns")
|
|
132
|
+
add = data.add_argument
|
|
133
|
+
add("--schema", metavar="FILE",
|
|
134
|
+
help="JSON file mapping columns to fields, e.g. {\"price\": {\"path\": \"pricing.amount\", "
|
|
135
|
+
"\"type\": \"float\"}}; writes typed rows instead of raw responses (needs -o)")
|
|
136
|
+
add("--explode", metavar="PATH", help="with --schema: one row per element of the array at PATH")
|
|
137
|
+
add("--rejects-table", metavar="NAME",
|
|
138
|
+
help="with --schema and .db output: also store rejected records here")
|
|
139
|
+
add("--infer-schema", type=int, metavar="N",
|
|
140
|
+
help="fetch the first N inputs, print a draft --schema file, exit")
|
|
141
|
+
|
|
142
|
+
info = parser.add_argument_group("progress and logging")
|
|
143
|
+
add = info.add_argument
|
|
144
|
+
add("-v", "--verbose", action="count", default=0,
|
|
145
|
+
help="log more: -v run events (INFO), -vv every attempt (DEBUG); "
|
|
146
|
+
"warnings and errors are always logged")
|
|
147
|
+
add("-q", "--quiet", action="store_true", help="no progress line, log errors only")
|
|
148
|
+
add("--log-file", metavar="PATH",
|
|
149
|
+
help="log to a file instead of stderr; a directory or a name with {time} gives a time-stamped "
|
|
150
|
+
"file per run, and an existing file is never overwritten")
|
|
151
|
+
add("--log-json", action="store_true", help="log one JSON object per line")
|
|
152
|
+
add("--report", action="store_true", help="print response times, statuses and hosts at the end")
|
|
153
|
+
add("--estimate", action="store_true", help="print how long the batch would take, send nothing, exit")
|
|
154
|
+
add("--latency", type=float, default=0.5, metavar="SECONDS",
|
|
155
|
+
help="typical response time assumed by --estimate (default: 0.5)")
|
|
105
156
|
return parser
|
|
157
|
+
# fmt: on
|
|
106
158
|
|
|
107
159
|
|
|
108
160
|
def _paginator(text: Optional[str], max_pages: int) -> Optional[Paginator]:
|
|
@@ -204,7 +256,7 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
|
204
256
|
args = parser.parse_args(argv)
|
|
205
257
|
try:
|
|
206
258
|
return _run(args)
|
|
207
|
-
except (ValueError, KeyError, SchemaError, OSError, json.JSONDecodeError) as error:
|
|
259
|
+
except (ValueError, KeyError, SchemaError, OSError, ImportError, json.JSONDecodeError) as error:
|
|
208
260
|
message = error.args[0] if isinstance(error, KeyError) and error.args else error
|
|
209
261
|
print(f"reqstorm: error: {message}", file=sys.stderr)
|
|
210
262
|
return 2
|
|
@@ -260,6 +312,8 @@ def _batch(args: argparse.Namespace, cache: Optional[Cache]) -> int:
|
|
|
260
312
|
timeout=args.timeout,
|
|
261
313
|
retries=args.retries,
|
|
262
314
|
backoff=args.backoff,
|
|
315
|
+
max_backoff=args.max_backoff,
|
|
316
|
+
jitter=not args.no_jitter,
|
|
263
317
|
verify_ssl=not args.no_verify,
|
|
264
318
|
rate_limit=rate,
|
|
265
319
|
auth=BearerAuth(token) if token else None,
|
|
@@ -281,17 +335,23 @@ def _batch(args: argparse.Namespace, cache: Optional[Cache]) -> int:
|
|
|
281
335
|
return 0
|
|
282
336
|
|
|
283
337
|
progress: Any = False if args.quiet else True
|
|
338
|
+
options["total"] = _count(args) if not args.paginate else None
|
|
339
|
+
options["log_level"] = "ERROR" if args.quiet else ("WARNING", "INFO", "DEBUG")[min(args.verbose, 2)]
|
|
340
|
+
options["log_file"] = args.log_file
|
|
341
|
+
options["log_format"] = "json" if args.log_json else "text"
|
|
284
342
|
if args.output:
|
|
285
343
|
summary = fetch_to_file_sync(
|
|
286
344
|
_inputs(args), args.output, body=args.body, include_headers=args.include_headers,
|
|
287
345
|
resume=args.resume, ordered=args.ordered, progress=progress, retry_rounds=args.retry_rounds,
|
|
288
346
|
retry_round_delay=args.retry_round_delay, table=args.table, schema=schema,
|
|
289
|
-
rejects_table=args.rejects_table, **options,
|
|
347
|
+
rejects_table=args.rejects_table, flush_interval=args.flush_interval, **options,
|
|
290
348
|
) # fmt: skip
|
|
291
349
|
message = f"reqstorm: {summary.ok} ok, {summary.failed} failed, {summary.skipped} skipped"
|
|
292
350
|
if schema is not None:
|
|
293
351
|
message += f", {summary.rows} rows, {summary.rejected} rejected"
|
|
294
352
|
print(message + f" -> {args.output}", file=sys.stderr)
|
|
353
|
+
if summary.log_file:
|
|
354
|
+
print(f"reqstorm: log written to {summary.log_file}", file=sys.stderr)
|
|
295
355
|
if args.report:
|
|
296
356
|
print(json.dumps(summary.report, indent=2), file=sys.stderr)
|
|
297
357
|
return 0 if summary.failed == 0 and summary.rejected == 0 else 1
|
|
@@ -302,7 +362,7 @@ def _batch(args: argparse.Namespace, cache: Optional[Cache]) -> int:
|
|
|
302
362
|
|
|
303
363
|
report = Report()
|
|
304
364
|
failed = 0
|
|
305
|
-
for result in stream_sync(_inputs(args), **options):
|
|
365
|
+
for result in stream_sync(_inputs(args), progress=progress, **options):
|
|
306
366
|
report.add(result)
|
|
307
367
|
failed += 0 if result.ok else 1
|
|
308
368
|
sys.stdout.write(json.dumps(result.to_dict(body=args.body, include_headers=args.include_headers),
|