reqstorm 2.2.0__tar.gz → 2.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. {reqstorm-2.2.0/reqstorm.egg-info → reqstorm-2.4.0}/PKG-INFO +54 -8
  2. {reqstorm-2.2.0 → reqstorm-2.4.0}/README.md +49 -7
  3. {reqstorm-2.2.0 → reqstorm-2.4.0}/pyproject.toml +3 -2
  4. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/__init__.py +3 -1
  5. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_adaptive.py +21 -7
  6. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_cli.py +127 -67
  7. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_client.py +149 -24
  8. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_files.py +105 -37
  9. reqstorm-2.4.0/reqstorm/_observe.py +408 -0
  10. reqstorm-2.4.0/reqstorm/_progress.py +15 -0
  11. reqstorm-2.4.0/reqstorm/_socks.py +84 -0
  12. {reqstorm-2.2.0 → reqstorm-2.4.0/reqstorm.egg-info}/PKG-INFO +54 -8
  13. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm.egg-info/SOURCES.txt +4 -0
  14. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm.egg-info/requires.txt +5 -0
  15. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_cli.py +37 -0
  16. reqstorm-2.4.0/tests/test_observe.py +267 -0
  17. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_report.py +7 -8
  18. reqstorm-2.4.0/tests/test_socks_backoff.py +144 -0
  19. reqstorm-2.2.0/reqstorm/_progress.py +0 -82
  20. {reqstorm-2.2.0 → reqstorm-2.4.0}/LICENSE +0 -0
  21. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/__main__.py +0 -0
  22. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_auth.py +0 -0
  23. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_cache.py +0 -0
  24. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_legacy.py +0 -0
  25. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_limits.py +0 -0
  26. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_paginate.py +0 -0
  27. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_plan.py +0 -0
  28. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_pydantic.py +0 -0
  29. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_report.py +0 -0
  30. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_schema.py +0 -0
  31. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_schema_sinks.py +0 -0
  32. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_sync.py +0 -0
  33. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/_template.py +0 -0
  34. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm/py.typed +0 -0
  35. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm.egg-info/dependency_links.txt +0 -0
  36. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm.egg-info/entry_points.txt +0 -0
  37. {reqstorm-2.2.0 → reqstorm-2.4.0}/reqstorm.egg-info/top_level.txt +0 -0
  38. {reqstorm-2.2.0 → reqstorm-2.4.0}/setup.cfg +0 -0
  39. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_adaptive.py +0 -0
  40. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_auth.py +0 -0
  41. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_cache_proxy.py +0 -0
  42. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_fetch_all.py +0 -0
  43. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_files.py +0 -0
  44. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_legacy.py +0 -0
  45. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_limits.py +0 -0
  46. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_ordered_and_db.py +0 -0
  47. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_paginate.py +0 -0
  48. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_pydantic.py +0 -0
  49. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_run_report.py +0 -0
  50. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_schema.py +0 -0
  51. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_schema_db.py +0 -0
  52. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_stream.py +0 -0
  53. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_sync.py +0 -0
  54. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_template.py +0 -0
  55. {reqstorm-2.2.0 → reqstorm-2.4.0}/tests/test_tls.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: reqstorm
3
- Version: 2.2.0
3
+ Version: 2.4.0
4
4
  Summary: Send large numbers of HTTP requests concurrently with asyncio, with retries, timeouts and per-request results.
5
5
  Author-email: Melih Colpan <colpanmelih@gmail.com>
6
6
  License-Expression: MIT
@@ -28,15 +28,19 @@ License-File: LICENSE
28
28
  Requires-Dist: aiohttp<4,>=3.9
29
29
  Provides-Extra: pydantic
30
30
  Requires-Dist: pydantic>=2; extra == "pydantic"
31
+ Provides-Extra: socks
32
+ Requires-Dist: aiohttp-socks>=0.10; extra == "socks"
31
33
  Provides-Extra: test
32
34
  Requires-Dist: pytest>=8; extra == "test"
33
35
  Requires-Dist: pytest-asyncio>=0.23; extra == "test"
34
36
  Requires-Dist: trustme>=1.1; extra == "test"
35
37
  Requires-Dist: pydantic>=2; extra == "test"
38
+ Requires-Dist: aiohttp-socks>=0.10; extra == "test"
36
39
  Provides-Extra: lint
37
40
  Requires-Dist: ruff>=0.6; extra == "lint"
38
41
  Requires-Dist: mypy>=1.10; extra == "lint"
39
42
  Requires-Dist: pydantic>=2; extra == "lint"
43
+ Requires-Dist: aiohttp-socks>=0.10; extra == "lint"
40
44
  Dynamic: license-file
41
45
 
42
46
  # reqstorm
@@ -90,6 +94,7 @@ for error in results.errors():
90
94
  - [Quick start](#quick-start)
91
95
  - [Results and reports](#results-and-reports)
92
96
  - [Rate limits and concurrency](#rate-limits-and-concurrency)
97
+ - [Progress and logging](#progress-and-logging)
93
98
  - [Timeouts and retries](#timeouts-and-retries)
94
99
  - [Writing results to a file](#writing-results-to-a-file)
95
100
  - [Writing results to a database](#writing-results-to-a-database)
@@ -123,16 +128,17 @@ reqstorm does all of that for you, with one call.
123
128
  | One result per request | Failures are recorded, never raised; every attempt is kept |
124
129
  | Rate limits | Per host, in any unit (`"100/min"`), or `"auto"` from the server's 429s and headers |
125
130
  | Concurrency | Overall and per host |
126
- | Retries | Immediate, with backoff and `Retry-After`; plus end-of-run rounds |
131
+ | Retries | Exponential backoff with a cap and jitter, `Retry-After`; plus end-of-run rounds |
127
132
  | Reports | Failures by reason, p50/p95/p99 response times, per-host figures |
128
133
  | Pagination | Next links, `Link` headers, cursors and page numbers |
129
134
  | Requests from data | URL templates over CSV rows or SQL query results |
130
- | Tokens and proxies | Refresh an expired token on 401; rotate through proxies |
135
+ | Tokens and proxies | Refresh an expired token on 401; rotate through HTTP and SOCKS proxies |
131
136
  | Caching | ETag / `If-None-Match`: unchanged resources cost a 304 |
132
137
  | Output | JSONL, CSV, SQLite, PostgreSQL, MySQL; ordered or as completed |
133
138
  | Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
134
139
  | Resume | Skip what already succeeded after an interruption |
135
140
  | Planning | `estimate()` before you start, progress with ETA while running |
141
+ | Logging | Retries, pauses and failures; per-run level, file and JSON, independent of the app |
136
142
  | API | Blocking (scripts, Jupyter), asyncio and a `reqstorm` command |
137
143
  | Safety | TLS verified, timeouts on, bounded concurrency, fully typed |
138
144
 
@@ -142,7 +148,7 @@ reqstorm does all of that for you, with one call.
142
148
  $ python -m pip install reqstorm
143
149
  ```
144
150
 
145
- reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`.
151
+ reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`; for SOCKS proxies, `reqstorm[socks]`.
146
152
 
147
153
  ## Quick start
148
154
 
@@ -298,6 +304,41 @@ reqstorm: 3500/7000 (50%) ok 3493 failed 7
298
304
  1.7 req/s ETA 35m 00s
299
305
  ```
300
306
 
307
+ ## Progress and logging
308
+
309
+ `progress=True` shows a line on stderr that keeps moving even while every request is waiting, so you can tell a long run is alive:
310
+
311
+ ```text
312
+ reqstorm: 3500/7000 (50%) ok 3493
313
+ failed 7 active 12 retries 41
314
+ 1.7 req/s ETA 34m 10s
315
+ ```
316
+
317
+ It shows requests in flight, retries, the rate over the last minute, the retry round, hosts paused after a 429, and the time left. `progress=` also takes a function, which gets a `ProgressInfo` every two seconds. For a generator, pass `total=` to get a percentage.
318
+
319
+ **Logs.** reqstorm logs each retry and its reason, pauses, token refreshes and final failures. Turn them on for one run, independent of your application's logging setup:
320
+
321
+ ```python
322
+ reqstorm.fetch_to_file_sync(
323
+ urls, "out.jsonl",
324
+ log_level="INFO", # or DEBUG: every attempt
325
+ log_file="logs/", # a new file per run
326
+ log_format="json", # optional
327
+ )
328
+ ```
329
+
330
+ ```text
331
+ 14:32:41 WARNING reqstorm: GET .../items/412
332
+ failed (HTTP 503) on attempt 1 of 3;
333
+ retrying in 0.4s
334
+ 15:42:18 ERROR reqstorm: GET .../items/913
335
+ failed after 3 attempts: HTTP 404
336
+ ```
337
+
338
+ Log files are never overwritten: a directory or `{time}` in the name gives a time-stamped file per run, and an existing `run.log` becomes `run-2.log`. Without `log_level`, reqstorm logs to the standard `"reqstorm"` logger and follows your application's configuration.
339
+
340
+ **Read results while they are written.** Output files are written at least once a second (`flush_interval`), even while nothing finishes, and never with half a line; SQLite files are opened in WAL mode, so other programs can read them during the run.
341
+
301
342
  ## Timeouts and retries
302
343
 
303
344
  ```python
@@ -306,13 +347,16 @@ results = reqstorm.fetch_all_sync(
306
347
  timeout=10, # seconds per attempt
307
348
  retries=3, # retry right away...
308
349
  backoff=0.5, # ...0.5 s, 1 s, 2 s apart
350
+ max_backoff=30, # never wait longer
309
351
  retry_rounds=2, # then resend what still
310
352
  retry_round_delay=30, # failed, 30 s later
311
353
  )
312
354
  ```
313
355
 
314
356
  - **`timeout`** is the time allowed for one attempt, including reading the body (default 30 s; `None` disables it). A slow server only fails its own requests.
315
- - **`retries`** retries a request right away. The delay starts at `backoff` and doubles; a `Retry-After` header (up to 60 s) takes precedence.
357
+ - **`retries`** retries a request right away, also when no response came back at all (timeouts, dropped connections). The wait starts at `backoff` and doubles each time (0.5, 1, 2, 4, ... s), up to **`max_backoff`** (30 s by default), so many retries never wait minutes.
358
+ - **Jitter** (on by default) waits a random time between half and all of that delay, so thousands of requests that failed together do not retry at the same moment. `jitter=False` waits exactly.
359
+ - A **`Retry-After`** header from the server (up to 60 s) is followed as given instead.
316
360
  - **`retry_rounds`** holds back requests that still failed for a retryable reason and sends them again after the rest of the batch, for temporary outages. Each request still produces exactly one result, with all its attempts in `history`.
317
361
 
318
362
  | Retried | Not retried |
@@ -484,15 +528,17 @@ auth = reqstorm.BearerAuth(refresh=get_token)
484
528
  reqstorm.fetch_all_sync(urls, auth=auth)
485
529
  ```
486
530
 
487
- **Proxies.** One proxy, a pool used in turn, or one per `Request`:
531
+ **Proxies.** One proxy, a pool used in turn, or one per `Request`. HTTP and SOCKS proxies can be mixed:
488
532
 
489
533
  ```python
490
534
  reqstorm.fetch_all_sync(urls, proxy=[
491
535
  "http://proxy-1.example.com:8080",
492
- "http://proxy-2.example.com:8080",
536
+ "socks5h://user:pass@proxy-2.example.com:1080",
493
537
  ])
494
538
  ```
495
539
 
540
+ SOCKS4 and SOCKS5 (`socks5h://` and `socks4a://` let the proxy resolve host names, as with Tor) need `pip install "reqstorm[socks]"`.
541
+
496
542
  **Caching.** `Cache` keeps responses in an SQLite file. The next run asks the server with `If-None-Match`; an unchanged resource comes back as a `304` with no body, and the stored response is used. With `ttl`, recent responses skip the network entirely:
497
543
 
498
544
  ```python
@@ -571,7 +617,7 @@ $ reqstorm urls.txt -o shop.db --report \
571
617
  --schema products.json --explode items
572
618
  ```
573
619
 
574
- `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
620
+ `-v` / `-vv` log more, `-q` logs only errors, and `--log-file logs/` writes a new log file per run. `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
575
621
 
576
622
  ## When to use something else
577
623
 
@@ -49,6 +49,7 @@ for error in results.errors():
49
49
  - [Quick start](#quick-start)
50
50
  - [Results and reports](#results-and-reports)
51
51
  - [Rate limits and concurrency](#rate-limits-and-concurrency)
52
+ - [Progress and logging](#progress-and-logging)
52
53
  - [Timeouts and retries](#timeouts-and-retries)
53
54
  - [Writing results to a file](#writing-results-to-a-file)
54
55
  - [Writing results to a database](#writing-results-to-a-database)
@@ -82,16 +83,17 @@ reqstorm does all of that for you, with one call.
82
83
  | One result per request | Failures are recorded, never raised; every attempt is kept |
83
84
  | Rate limits | Per host, in any unit (`"100/min"`), or `"auto"` from the server's 429s and headers |
84
85
  | Concurrency | Overall and per host |
85
- | Retries | Immediate, with backoff and `Retry-After`; plus end-of-run rounds |
86
+ | Retries | Exponential backoff with a cap and jitter, `Retry-After`; plus end-of-run rounds |
86
87
  | Reports | Failures by reason, p50/p95/p99 response times, per-host figures |
87
88
  | Pagination | Next links, `Link` headers, cursors and page numbers |
88
89
  | Requests from data | URL templates over CSV rows or SQL query results |
89
- | Tokens and proxies | Refresh an expired token on 401; rotate through proxies |
90
+ | Tokens and proxies | Refresh an expired token on 401; rotate through HTTP and SOCKS proxies |
90
91
  | Caching | ETag / `If-None-Match`: unchanged resources cost a 304 |
91
92
  | Output | JSONL, CSV, SQLite, PostgreSQL, MySQL; ordered or as completed |
92
93
  | Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
93
94
  | Resume | Skip what already succeeded after an interruption |
94
95
  | Planning | `estimate()` before you start, progress with ETA while running |
96
+ | Logging | Retries, pauses and failures; per-run level, file and JSON, independent of the app |
95
97
  | API | Blocking (scripts, Jupyter), asyncio and a `reqstorm` command |
96
98
  | Safety | TLS verified, timeouts on, bounded concurrency, fully typed |
97
99
 
@@ -101,7 +103,7 @@ reqstorm does all of that for you, with one call.
101
103
  $ python -m pip install reqstorm
102
104
  ```
103
105
 
104
- reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`.
106
+ reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`; for SOCKS proxies, `reqstorm[socks]`.
105
107
 
106
108
  ## Quick start
107
109
 
@@ -257,6 +259,41 @@ reqstorm: 3500/7000 (50%) ok 3493 failed 7
257
259
  1.7 req/s ETA 35m 00s
258
260
  ```
259
261
 
262
+ ## Progress and logging
263
+
264
+ `progress=True` shows a line on stderr that keeps moving even while every request is waiting, so you can tell a long run is alive:
265
+
266
+ ```text
267
+ reqstorm: 3500/7000 (50%) ok 3493
268
+ failed 7 active 12 retries 41
269
+ 1.7 req/s ETA 34m 10s
270
+ ```
271
+
272
+ It shows requests in flight, retries, the rate over the last minute, the retry round, hosts paused after a 429, and the time left. `progress=` also takes a function, which gets a `ProgressInfo` every two seconds. For a generator, pass `total=` to get a percentage.
273
+
274
+ **Logs.** reqstorm logs each retry and its reason, pauses, token refreshes and final failures. Turn them on for one run, independent of your application's logging setup:
275
+
276
+ ```python
277
+ reqstorm.fetch_to_file_sync(
278
+ urls, "out.jsonl",
279
+ log_level="INFO", # or DEBUG: every attempt
280
+ log_file="logs/", # a new file per run
281
+ log_format="json", # optional
282
+ )
283
+ ```
284
+
285
+ ```text
286
+ 14:32:41 WARNING reqstorm: GET .../items/412
287
+ failed (HTTP 503) on attempt 1 of 3;
288
+ retrying in 0.4s
289
+ 15:42:18 ERROR reqstorm: GET .../items/913
290
+ failed after 3 attempts: HTTP 404
291
+ ```
292
+
293
+ Log files are never overwritten: a directory or `{time}` in the name gives a time-stamped file per run, and an existing `run.log` becomes `run-2.log`. Without `log_level`, reqstorm logs to the standard `"reqstorm"` logger and follows your application's configuration.
294
+
295
+ **Read results while they are written.** Output files are written at least once a second (`flush_interval`), even while nothing finishes, and never with half a line; SQLite files are opened in WAL mode, so other programs can read them during the run.
296
+
260
297
  ## Timeouts and retries
261
298
 
262
299
  ```python
@@ -265,13 +302,16 @@ results = reqstorm.fetch_all_sync(
265
302
  timeout=10, # seconds per attempt
266
303
  retries=3, # retry right away...
267
304
  backoff=0.5, # ...0.5 s, 1 s, 2 s apart
305
+ max_backoff=30, # never wait longer
268
306
  retry_rounds=2, # then resend what still
269
307
  retry_round_delay=30, # failed, 30 s later
270
308
  )
271
309
  ```
272
310
 
273
311
  - **`timeout`** is the time allowed for one attempt, including reading the body (default 30 s; `None` disables it). A slow server only fails its own requests.
274
- - **`retries`** retries a request right away. The delay starts at `backoff` and doubles; a `Retry-After` header (up to 60 s) takes precedence.
312
+ - **`retries`** retries a request right away, also when no response came back at all (timeouts, dropped connections). The wait starts at `backoff` and doubles each time (0.5, 1, 2, 4, ... s), up to **`max_backoff`** (30 s by default), so many retries never wait minutes.
313
+ - **Jitter** (on by default) waits a random time between half and all of that delay, so thousands of requests that failed together do not retry at the same moment. `jitter=False` waits exactly.
314
+ - A **`Retry-After`** header from the server (up to 60 s) is followed as given instead.
275
315
  - **`retry_rounds`** holds back requests that still failed for a retryable reason and sends them again after the rest of the batch, for temporary outages. Each request still produces exactly one result, with all its attempts in `history`.
276
316
 
277
317
  | Retried | Not retried |
@@ -443,15 +483,17 @@ auth = reqstorm.BearerAuth(refresh=get_token)
443
483
  reqstorm.fetch_all_sync(urls, auth=auth)
444
484
  ```
445
485
 
446
- **Proxies.** One proxy, a pool used in turn, or one per `Request`:
486
+ **Proxies.** One proxy, a pool used in turn, or one per `Request`. HTTP and SOCKS proxies can be mixed:
447
487
 
448
488
  ```python
449
489
  reqstorm.fetch_all_sync(urls, proxy=[
450
490
  "http://proxy-1.example.com:8080",
451
- "http://proxy-2.example.com:8080",
491
+ "socks5h://user:pass@proxy-2.example.com:1080",
452
492
  ])
453
493
  ```
454
494
 
495
+ SOCKS4 and SOCKS5 (`socks5h://` and `socks4a://` let the proxy resolve host names, as with Tor) need `pip install "reqstorm[socks]"`.
496
+
455
497
  **Caching.** `Cache` keeps responses in an SQLite file. The next run asks the server with `If-None-Match`; an unchanged resource comes back as a `304` with no body, and the stored response is used. With `ttl`, recent responses skip the network entirely:
456
498
 
457
499
  ```python
@@ -530,7 +572,7 @@ $ reqstorm urls.txt -o shop.db --report \
530
572
  --schema products.json --explode items
531
573
  ```
532
574
 
533
- `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
575
+ `-v` / `-vv` log more, `-q` logs only errors, and `--log-file logs/` writes a new log file per run. `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
534
576
 
535
577
  ## When to use something else
536
578
 
@@ -31,8 +31,9 @@ classifiers = [
31
31
 
32
32
  [project.optional-dependencies]
33
33
  pydantic = ["pydantic>=2"]
34
- test = ["pytest>=8", "pytest-asyncio>=0.23", "trustme>=1.1", "pydantic>=2"]
35
- lint = ["ruff>=0.6", "mypy>=1.10", "pydantic>=2"]
34
+ socks = ["aiohttp-socks>=0.10"]
35
+ test = ["pytest>=8", "pytest-asyncio>=0.23", "trustme>=1.1", "pydantic>=2", "aiohttp-socks>=0.10"]
36
+ lint = ["ruff>=0.6", "mypy>=1.10", "pydantic>=2", "aiohttp-socks>=0.10"]
36
37
 
37
38
  [project.scripts]
38
39
  reqstorm = "reqstorm._cli:main"
@@ -26,13 +26,14 @@ from ._client import (
26
26
  from ._files import Summary, fetch_to_db, fetch_to_file
27
27
  from ._legacy import Reqt
28
28
  from ._limits import parse_rate
29
+ from ._observe import ProgressInfo
29
30
  from ._paginate import Cursor, LinkHeader, NextLink, PageNumber, Paginator
30
31
  from ._plan import Estimate, estimate
31
32
  from ._schema import Field, Schema, SchemaError, extract, infer_schema
32
33
  from ._sync import fetch_all_sync, fetch_to_db_sync, fetch_to_file_sync, stream_sync
33
34
  from ._template import from_template, read_csv, read_sql
34
35
 
35
- __version__ = "2.2.0"
36
+ __version__ = "2.4.0"
36
37
 
37
38
  __all__ = [
38
39
  "Attempt",
@@ -46,6 +47,7 @@ __all__ = [
46
47
  "LinkHeader",
47
48
  "NextLink",
48
49
  "PageNumber",
50
+ "ProgressInfo",
49
51
  "Paginator",
50
52
  "Reqt",
51
53
  "Request",
@@ -6,10 +6,12 @@ import asyncio
6
6
  import time
7
7
  from dataclasses import dataclass
8
8
  from email.utils import parsedate_to_datetime
9
- from typing import Dict, Mapping, Optional
9
+ from typing import Dict, Mapping, Optional, Tuple
10
10
 
11
11
  from ._limits import host_key
12
12
 
13
+ __all__ = ["AdaptiveRateLimiter", "seconds_until_reset"]
14
+
13
15
  # Header names, most specific first. "RateLimit-*" is the IETF draft; "X-RateLimit-*" is the
14
16
  # common convention (GitHub, Twitter/X, many others).
15
17
  _REMAINING = ("RateLimit-Remaining", "X-RateLimit-Remaining", "X-Rate-Limit-Remaining")
@@ -91,17 +93,28 @@ class AdaptiveRateLimiter:
91
93
  if slot > now:
92
94
  await asyncio.sleep(slot - now)
93
95
 
94
- def record(self, url: str, status: Optional[int], headers: Mapping[str, str]) -> None:
96
+ def paused(self) -> Dict[str, float]:
97
+ """Hosts that are paused right now, with the seconds left."""
98
+ now = time.monotonic()
99
+ return {
100
+ host: state.paused_until - now for host, state in self._hosts.items() if state.paused_until > now
101
+ }
102
+
103
+ def record(
104
+ self, url: str, status: Optional[int], headers: Mapping[str, str]
105
+ ) -> Optional[Tuple[float, str]]:
106
+ """Learn from a response. Returns ``(seconds, reason)`` when the host is paused."""
95
107
  state = self._state(url)
96
108
  if state is None or status is None:
97
- return
109
+ return None
98
110
  now = time.monotonic()
99
111
  reset = seconds_until_reset(headers)
100
112
  if status == 429:
101
113
  state.interval = min(max(state.interval * 2, 0.1), 60.0)
102
114
  pause = reset if reset is not None else max(state.interval, 1.0)
103
115
  state.paused_until = max(state.paused_until, now + pause)
104
- return
116
+ source = "Retry-After" if reset is not None else "no Retry-After"
117
+ return pause, f"429 Too Many Requests ({source}), spacing requests {state.interval:.2f}s apart"
105
118
  remaining_text = _header(headers, _REMAINING)
106
119
  if remaining_text is not None and reset is not None:
107
120
  try:
@@ -111,8 +124,9 @@ class AdaptiveRateLimiter:
111
124
  if remaining is not None:
112
125
  if remaining <= 0:
113
126
  state.paused_until = max(state.paused_until, now + reset)
114
- else:
115
- state.interval = min(reset / remaining, 60.0)
116
- return
127
+ return (reset, "rate limit used up until the window resets") if reset > 0 else None
128
+ state.interval = min(reset / remaining, 60.0)
129
+ return None
117
130
  if 200 <= status < 400 and state.interval:
118
131
  state.interval = state.interval * 0.9 if state.interval > 0.01 else 0.0
132
+ return None
@@ -19,90 +19,142 @@ from ._schema import Field, Schema, SchemaError, infer_schema
19
19
  from ._sync import fetch_all_sync, fetch_to_file_sync, stream_sync
20
20
  from ._template import from_template, read_csv
21
21
 
22
- EXAMPLES = """\
22
+ EPILOG = """\
23
23
  examples:
24
24
  reqstorm urls.txt -o results.jsonl --rate 100/min --retries 2
25
25
  reqstorm urls.txt --estimate --rate 100/min
26
26
  cat urls.txt | reqstorm --rate auto > results.jsonl
27
- reqstorm users.csv --template "https://api.example.com/users/{id}" -o users.db
27
+ reqstorm urls.txt -o results.db -v --log-file logs/
28
+ reqstorm users.csv --template "https://api.example.com/users/{id}" -o users.jsonl
28
29
  reqstorm urls.txt --paginate next:links.next --schema products.json --explode items -o shop.db
29
30
  reqstorm urls.txt --infer-schema 5 --explode items > products.json
31
+ reqstorm urls.txt --proxy socks5h://127.0.0.1:9050 -o out.jsonl
30
32
 
31
- exit status: 0 when every request succeeded, 1 when some failed or records were rejected,
32
- 2 for invalid arguments.
33
+ rate limits:
34
+ 5 or 0.5 (per second), 10/s, 100/min, 30/5min, 1000/h, 2/day, per host; or auto to
35
+ follow the server's 429 responses and X-RateLimit headers.
36
+
37
+ environment:
38
+ REQSTORM_TOKEN bearer token, instead of --bearer (keeps it out of shell history)
39
+
40
+ exit status:
41
+ 0 every request succeeded 1 some failed or records were rejected
42
+ 2 invalid arguments 130 interrupted
43
+
44
+ documentation: https://reqstorm.github.io/guide/cli/
33
45
  """
34
46
 
35
47
 
48
+ # fmt: off
36
49
  def _parser() -> argparse.ArgumentParser:
37
50
  parser = argparse.ArgumentParser(
38
51
  prog="reqstorm",
39
52
  description="Send many HTTP requests with rate limits, retries and progress, and save the results.",
40
- epilog=EXAMPLES,
41
- formatter_class=argparse.RawDescriptionHelpFormatter,
53
+ epilog=EPILOG,
54
+ formatter_class=lambda prog: argparse.RawDescriptionHelpFormatter(prog, max_help_position=32),
42
55
  )
43
- parser.add_argument("input", nargs="?", default="-",
44
- help="file with one URL per line, or a CSV with --template "
45
- "(default: standard input)") # fmt: skip
46
- parser.add_argument("-o", "--output", help="write results to a .jsonl, .csv or .db/.sqlite file "
47
- "(default: JSON lines on standard output)") # fmt: skip
48
- parser.add_argument("--version", action="version", version=f"reqstorm {__version__}")
56
+ add = parser.add_argument
57
+ add("input", nargs="?", default="-",
58
+ help="file with one URL per line (blank lines and # comments are skipped), or a CSV file "
59
+ "with --template; default: standard input")
60
+ add("-o", "--output", metavar="FILE",
61
+ help="write results to FILE: .jsonl, .csv, or .db/.sqlite for SQLite; "
62
+ "default: JSON lines on standard output")
63
+ add("--version", action="version", version=f"reqstorm {__version__}")
49
64
 
50
65
  request = parser.add_argument_group("requests")
51
- request.add_argument("-X", "--method", default="GET")
52
- request.add_argument("-H", "--header", action="append", default=[], metavar="'NAME: VALUE'")
53
- request.add_argument("--json", help="JSON body sent with every request")
54
- request.add_argument("--data", help="raw body sent with every request")
55
- request.add_argument("--template", metavar="URL", help="URL template filled from each CSV row, "
56
- "e.g. 'https://api.example.com/users/{id}'") # fmt: skip
57
- request.add_argument("--bearer", metavar="TOKEN", help="send 'Authorization: Bearer TOKEN' "
58
- "(or set REQSTORM_TOKEN)") # fmt: skip
59
- request.add_argument(
60
- "--proxy", action="append", default=[], metavar="URL", help="proxy URL; repeat to rotate"
61
- )
62
- request.add_argument("--no-verify", action="store_true", help="do not verify TLS certificates")
63
-
64
- pace = parser.add_argument_group("pace and retries")
65
- pace.add_argument("--rate", help="per-host rate limit: 5, 10/s, 100/min, 1000/h, or auto")
66
- pace.add_argument("-c", "--concurrency", type=int, default=100)
67
- pace.add_argument("--per-host", type=int, default=0, metavar="N", help="max requests in flight per host")
68
- pace.add_argument("--timeout", type=float, default=30.0, help="seconds per attempt (default 30)")
69
- pace.add_argument("--retries", type=int, default=0)
70
- pace.add_argument("--backoff", type=float, default=0.5)
71
- pace.add_argument("--retry-rounds", type=int, default=0)
72
- pace.add_argument("--retry-round-delay", type=float, default=5.0)
73
- pace.add_argument("--paginate", metavar="STRATEGY",
74
- help="next:PATH, link, cursor:PATH[:PARAM], or page[:PARAM[:ITEMS_PATH]]") # fmt: skip
75
- pace.add_argument("--max-pages", type=int, default=1000)
76
- pace.add_argument(
77
- "--cache", metavar="FILE", help="reuse responses from this cache file (revalidated with ETag)"
78
- )
79
- pace.add_argument(
80
- "--cache-ttl", type=float, help="seconds a cached response is used without asking the server"
81
- )
66
+ add = request.add_argument
67
+ add("-X", "--method", default="GET", help="HTTP method for every request (default: GET)")
68
+ add("-H", "--header", action="append", default=[], metavar="'NAME: VALUE'",
69
+ help="header sent with every request; repeat for more")
70
+ add("--json", metavar="JSON", help="JSON body sent with every request")
71
+ add("--data", metavar="TEXT", help="raw body sent with every request")
72
+ add("--template", metavar="URL",
73
+ help="treat the input as CSV and build one URL per row from {column} placeholders, "
74
+ "e.g. 'https://api.example.com/users/{id}'")
75
+ add("--bearer", metavar="TOKEN", help="send 'Authorization: Bearer TOKEN' (or set REQSTORM_TOKEN)")
76
+ add("--proxy", action="append", default=[], metavar="URL",
77
+ help="send through a proxy: http://, socks5://, socks5h://, socks4:// or socks4a://; "
78
+ "repeat to rotate through several")
79
+ add("--no-verify", action="store_true",
80
+ help="do not verify TLS certificates (only for hosts you control)")
81
+
82
+ pace = parser.add_argument_group("pace")
83
+ add = pace.add_argument
84
+ add("--rate", metavar="LIMIT",
85
+ help="per-host rate limit: 5, 10/s, 100/min, 1000/h, or auto (default: none)")
86
+ add("-c", "--concurrency", type=int, default=100, metavar="N",
87
+ help="requests in flight at once (default: 100)")
88
+ add("--per-host", type=int, default=0, metavar="N",
89
+ help="requests in flight per host (default: no limit)")
90
+ add("--timeout", type=float, default=30.0, metavar="SECONDS",
91
+ help="time allowed for one attempt, including the body (default: 30)")
92
+
93
+ retry = parser.add_argument_group("retries")
94
+ add = retry.add_argument
95
+ add("--retries", type=int, default=0, metavar="N",
96
+ help="retry timeouts, connection errors and 429/5xx up to N times (default: 0)")
97
+ add("--backoff", type=float, default=0.5, metavar="SECONDS",
98
+ help="wait before the first retry, doubled after each (default: 0.5)")
99
+ add("--max-backoff", type=float, default=30.0, metavar="SECONDS",
100
+ help="longest wait between retries (default: 30)")
101
+ add("--no-jitter", action="store_true", help="wait exactly, instead of a random half to all of the delay")
102
+ add("--retry-rounds", type=int, default=0, metavar="N",
103
+ help="send requests that still failed again after all others, up to N rounds (needs -o)")
104
+ add("--retry-round-delay", type=float, default=5.0, metavar="SECONDS",
105
+ help="wait before each retry round (default: 5)")
106
+
107
+ more = parser.add_argument_group("pagination and caching")
108
+ add = more.add_argument
109
+ add("--paginate", metavar="STRATEGY",
110
+ help="follow pages: next:PATH (URL in the JSON), link (Link header), cursor:PATH[:PARAM], "
111
+ "or page[:PARAM[:ITEMS_PATH]]")
112
+ add("--max-pages", type=int, default=1000, metavar="N",
113
+ help="pages per starting URL at most (default: 1000)")
114
+ add("--cache", metavar="FILE", help="keep responses in this SQLite file and revalidate them with ETag")
115
+ add("--cache-ttl", type=float, metavar="SECONDS",
116
+ help="use a cached response without asking the server for this long (default: always ask)")
82
117
 
83
118
  output = parser.add_argument_group("output")
84
- output.add_argument(
85
- "--resume", action="store_true", help="skip requests already saved successfully in --output"
86
- )
87
- output.add_argument("--ordered", action="store_true", help="write results in input order")
88
- output.add_argument("--body", choices=["text", "base64", "none"], default="text")
89
- output.add_argument("--include-headers", action="store_true")
90
- output.add_argument("--schema", metavar="FILE", help="JSON file mapping columns to fields: "
91
- '{"price": {"path": "pricing.amount", "type": "float"}}') # fmt: skip
92
- output.add_argument("--explode", metavar="PATH", help="with --schema: one row per element of this array")
93
- output.add_argument("--table", default="reqstorm_results", help="table name for .db output")
94
- output.add_argument("--rejects-table", help="with --schema and .db output: table for rejected records")
95
- output.add_argument("--infer-schema", type=int, metavar="N",
96
- help="fetch the first N inputs, print a draft --schema file and exit") # fmt: skip
97
-
98
- info = parser.add_argument_group("information")
99
- info.add_argument("--estimate", action="store_true", help="print how long the batch would take and exit")
100
- info.add_argument("--latency", type=float, default=0.5, help="typical response time for --estimate")
101
- info.add_argument("-q", "--quiet", action="store_true", help="no progress line")
102
- info.add_argument(
103
- "--report", action="store_true", help="print response times, statuses and hosts at the end"
104
- )
119
+ add = output.add_argument
120
+ add("--body", choices=["text", "base64", "none"], default="text",
121
+ help="how to store response bodies (default: text)")
122
+ add("--include-headers", action="store_true", help="store response headers too")
123
+ add("--resume", action="store_true",
124
+ help="skip requests already saved successfully in -o, append the rest")
125
+ add("--ordered", action="store_true", help="write results in input order instead of as they finish")
126
+ add("--table", default="reqstorm_results", metavar="NAME",
127
+ help="table for .db output (default: reqstorm_results)")
128
+ add("--flush-interval", type=float, default=1.0, metavar="SECONDS",
129
+ help="write waiting results to -o at least this often (default: 1)")
130
+
131
+ data = parser.add_argument_group("typed columns")
132
+ add = data.add_argument
133
+ add("--schema", metavar="FILE",
134
+ help="JSON file mapping columns to fields, e.g. {\"price\": {\"path\": \"pricing.amount\", "
135
+ "\"type\": \"float\"}}; writes typed rows instead of raw responses (needs -o)")
136
+ add("--explode", metavar="PATH", help="with --schema: one row per element of the array at PATH")
137
+ add("--rejects-table", metavar="NAME",
138
+ help="with --schema and .db output: also store rejected records here")
139
+ add("--infer-schema", type=int, metavar="N",
140
+ help="fetch the first N inputs, print a draft --schema file, exit")
141
+
142
+ info = parser.add_argument_group("progress and logging")
143
+ add = info.add_argument
144
+ add("-v", "--verbose", action="count", default=0,
145
+ help="log more: -v run events (INFO), -vv every attempt (DEBUG); "
146
+ "warnings and errors are always logged")
147
+ add("-q", "--quiet", action="store_true", help="no progress line, log errors only")
148
+ add("--log-file", metavar="PATH",
149
+ help="log to a file instead of stderr; a directory or a name with {time} gives a time-stamped "
150
+ "file per run, and an existing file is never overwritten")
151
+ add("--log-json", action="store_true", help="log one JSON object per line")
152
+ add("--report", action="store_true", help="print response times, statuses and hosts at the end")
153
+ add("--estimate", action="store_true", help="print how long the batch would take, send nothing, exit")
154
+ add("--latency", type=float, default=0.5, metavar="SECONDS",
155
+ help="typical response time assumed by --estimate (default: 0.5)")
105
156
  return parser
157
+ # fmt: on
106
158
 
107
159
 
108
160
  def _paginator(text: Optional[str], max_pages: int) -> Optional[Paginator]:
@@ -204,7 +256,7 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
204
256
  args = parser.parse_args(argv)
205
257
  try:
206
258
  return _run(args)
207
- except (ValueError, KeyError, SchemaError, OSError, json.JSONDecodeError) as error:
259
+ except (ValueError, KeyError, SchemaError, OSError, ImportError, json.JSONDecodeError) as error:
208
260
  message = error.args[0] if isinstance(error, KeyError) and error.args else error
209
261
  print(f"reqstorm: error: {message}", file=sys.stderr)
210
262
  return 2
@@ -260,6 +312,8 @@ def _batch(args: argparse.Namespace, cache: Optional[Cache]) -> int:
260
312
  timeout=args.timeout,
261
313
  retries=args.retries,
262
314
  backoff=args.backoff,
315
+ max_backoff=args.max_backoff,
316
+ jitter=not args.no_jitter,
263
317
  verify_ssl=not args.no_verify,
264
318
  rate_limit=rate,
265
319
  auth=BearerAuth(token) if token else None,
@@ -281,17 +335,23 @@ def _batch(args: argparse.Namespace, cache: Optional[Cache]) -> int:
281
335
  return 0
282
336
 
283
337
  progress: Any = False if args.quiet else True
338
+ options["total"] = _count(args) if not args.paginate else None
339
+ options["log_level"] = "ERROR" if args.quiet else ("WARNING", "INFO", "DEBUG")[min(args.verbose, 2)]
340
+ options["log_file"] = args.log_file
341
+ options["log_format"] = "json" if args.log_json else "text"
284
342
  if args.output:
285
343
  summary = fetch_to_file_sync(
286
344
  _inputs(args), args.output, body=args.body, include_headers=args.include_headers,
287
345
  resume=args.resume, ordered=args.ordered, progress=progress, retry_rounds=args.retry_rounds,
288
346
  retry_round_delay=args.retry_round_delay, table=args.table, schema=schema,
289
- rejects_table=args.rejects_table, **options,
347
+ rejects_table=args.rejects_table, flush_interval=args.flush_interval, **options,
290
348
  ) # fmt: skip
291
349
  message = f"reqstorm: {summary.ok} ok, {summary.failed} failed, {summary.skipped} skipped"
292
350
  if schema is not None:
293
351
  message += f", {summary.rows} rows, {summary.rejected} rejected"
294
352
  print(message + f" -> {args.output}", file=sys.stderr)
353
+ if summary.log_file:
354
+ print(f"reqstorm: log written to {summary.log_file}", file=sys.stderr)
295
355
  if args.report:
296
356
  print(json.dumps(summary.report, indent=2), file=sys.stderr)
297
357
  return 0 if summary.failed == 0 and summary.rejected == 0 else 1
@@ -302,7 +362,7 @@ def _batch(args: argparse.Namespace, cache: Optional[Cache]) -> int:
302
362
 
303
363
  report = Report()
304
364
  failed = 0
305
- for result in stream_sync(_inputs(args), **options):
365
+ for result in stream_sync(_inputs(args), progress=progress, **options):
306
366
  report.add(result)
307
367
  failed += 0 if result.ok else 1
308
368
  sys.stdout.write(json.dumps(result.to_dict(body=args.body, include_headers=args.include_headers),