reqstorm 2.3.0__tar.gz → 2.4.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {reqstorm-2.3.0/reqstorm.egg-info → reqstorm-2.4.1}/PKG-INFO +39 -2
- {reqstorm-2.3.0 → reqstorm-2.4.1}/README.md +38 -1
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/__init__.py +3 -1
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_adaptive.py +21 -7
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_cli.py +124 -76
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_client.py +98 -20
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_files.py +105 -37
- reqstorm-2.4.1/reqstorm/_observe.py +408 -0
- reqstorm-2.4.1/reqstorm/_progress.py +15 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_socks.py +19 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1/reqstorm.egg-info}/PKG-INFO +39 -2
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm.egg-info/SOURCES.txt +2 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_cli.py +37 -0
- reqstorm-2.4.1/tests/test_observe.py +267 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_report.py +7 -8
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_socks_backoff.py +20 -0
- reqstorm-2.3.0/reqstorm/_progress.py +0 -82
- {reqstorm-2.3.0 → reqstorm-2.4.1}/LICENSE +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/pyproject.toml +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/__main__.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_auth.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_cache.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_legacy.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_limits.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_paginate.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_plan.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_pydantic.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_report.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_schema.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_schema_sinks.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_sync.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_template.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/py.typed +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm.egg-info/dependency_links.txt +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm.egg-info/entry_points.txt +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm.egg-info/requires.txt +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm.egg-info/top_level.txt +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/setup.cfg +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_adaptive.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_auth.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_cache_proxy.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_fetch_all.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_files.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_legacy.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_limits.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_ordered_and_db.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_paginate.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_pydantic.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_run_report.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_schema.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_schema_db.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_stream.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_sync.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_template.py +0 -0
- {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_tls.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: reqstorm
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.4.1
|
|
4
4
|
Summary: Send large numbers of HTTP requests concurrently with asyncio, with retries, timeouts and per-request results.
|
|
5
5
|
Author-email: Melih Colpan <colpanmelih@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -94,6 +94,7 @@ for error in results.errors():
|
|
|
94
94
|
- [Quick start](#quick-start)
|
|
95
95
|
- [Results and reports](#results-and-reports)
|
|
96
96
|
- [Rate limits and concurrency](#rate-limits-and-concurrency)
|
|
97
|
+
- [Progress and logging](#progress-and-logging)
|
|
97
98
|
- [Timeouts and retries](#timeouts-and-retries)
|
|
98
99
|
- [Writing results to a file](#writing-results-to-a-file)
|
|
99
100
|
- [Writing results to a database](#writing-results-to-a-database)
|
|
@@ -137,6 +138,7 @@ reqstorm does all of that for you, with one call.
|
|
|
137
138
|
| Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
|
|
138
139
|
| Resume | Skip what already succeeded after an interruption |
|
|
139
140
|
| Planning | `estimate()` before you start, progress with ETA while running |
|
|
141
|
+
| Logging | Retries, pauses and failures; per-run level, file and JSON, independent of the app |
|
|
140
142
|
| API | Blocking (scripts, Jupyter), asyncio and a `reqstorm` command |
|
|
141
143
|
| Safety | TLS verified, timeouts on, bounded concurrency, fully typed |
|
|
142
144
|
|
|
@@ -302,6 +304,41 @@ reqstorm: 3500/7000 (50%) ok 3493 failed 7
|
|
|
302
304
|
1.7 req/s ETA 35m 00s
|
|
303
305
|
```
|
|
304
306
|
|
|
307
|
+
## Progress and logging
|
|
308
|
+
|
|
309
|
+
`progress=True` shows a line on stderr that keeps moving even while every request is waiting, so you can tell a long run is alive:
|
|
310
|
+
|
|
311
|
+
```text
|
|
312
|
+
reqstorm: 3500/7000 (50%) ok 3493
|
|
313
|
+
failed 7 active 12 retries 41
|
|
314
|
+
1.7 req/s ETA 34m 10s
|
|
315
|
+
```
|
|
316
|
+
|
|
317
|
+
It shows requests in flight, retries, the rate over the last minute, the retry round, hosts paused after a 429, and the time left. `progress=` also takes a function, which gets a `ProgressInfo` every two seconds. For a generator, pass `total=` to get a percentage.
|
|
318
|
+
|
|
319
|
+
**Logs.** reqstorm logs each retry and its reason, pauses, token refreshes and final failures. Turn them on for one run, independent of your application's logging setup:
|
|
320
|
+
|
|
321
|
+
```python
|
|
322
|
+
reqstorm.fetch_to_file_sync(
|
|
323
|
+
urls, "out.jsonl",
|
|
324
|
+
log_level="INFO", # or DEBUG: every attempt
|
|
325
|
+
log_file="logs/", # a new file per run
|
|
326
|
+
log_format="json", # optional
|
|
327
|
+
)
|
|
328
|
+
```
|
|
329
|
+
|
|
330
|
+
```text
|
|
331
|
+
14:32:41 WARNING reqstorm: GET .../items/412
|
|
332
|
+
failed (HTTP 503) on attempt 1 of 3;
|
|
333
|
+
retrying in 0.4s
|
|
334
|
+
15:42:18 ERROR reqstorm: GET .../items/913
|
|
335
|
+
failed after 3 attempts: HTTP 404
|
|
336
|
+
```
|
|
337
|
+
|
|
338
|
+
Log files are never overwritten: a directory or `{time}` in the name gives a time-stamped file per run, and an existing `run.log` becomes `run-2.log`. Without `log_level`, reqstorm logs to the standard `"reqstorm"` logger and follows your application's configuration.
|
|
339
|
+
|
|
340
|
+
**Read results while they are written.** Output files are written at least once a second (`flush_interval`), even while nothing finishes, and never with half a line; SQLite files are opened in WAL mode, so other programs can read them during the run.
|
|
341
|
+
|
|
305
342
|
## Timeouts and retries
|
|
306
343
|
|
|
307
344
|
```python
|
|
@@ -580,7 +617,7 @@ $ reqstorm urls.txt -o shop.db --report \
|
|
|
580
617
|
--schema products.json --explode items
|
|
581
618
|
```
|
|
582
619
|
|
|
583
|
-
`reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
|
|
620
|
+
`-v` / `-vv` log more, `-q` logs only errors, and `--log-file logs/` writes a new log file per run. `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
|
|
584
621
|
|
|
585
622
|
## When to use something else
|
|
586
623
|
|
|
@@ -49,6 +49,7 @@ for error in results.errors():
|
|
|
49
49
|
- [Quick start](#quick-start)
|
|
50
50
|
- [Results and reports](#results-and-reports)
|
|
51
51
|
- [Rate limits and concurrency](#rate-limits-and-concurrency)
|
|
52
|
+
- [Progress and logging](#progress-and-logging)
|
|
52
53
|
- [Timeouts and retries](#timeouts-and-retries)
|
|
53
54
|
- [Writing results to a file](#writing-results-to-a-file)
|
|
54
55
|
- [Writing results to a database](#writing-results-to-a-database)
|
|
@@ -92,6 +93,7 @@ reqstorm does all of that for you, with one call.
|
|
|
92
93
|
| Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
|
|
93
94
|
| Resume | Skip what already succeeded after an interruption |
|
|
94
95
|
| Planning | `estimate()` before you start, progress with ETA while running |
|
|
96
|
+
| Logging | Retries, pauses and failures; per-run level, file and JSON, independent of the app |
|
|
95
97
|
| API | Blocking (scripts, Jupyter), asyncio and a `reqstorm` command |
|
|
96
98
|
| Safety | TLS verified, timeouts on, bounded concurrency, fully typed |
|
|
97
99
|
|
|
@@ -257,6 +259,41 @@ reqstorm: 3500/7000 (50%) ok 3493 failed 7
|
|
|
257
259
|
1.7 req/s ETA 35m 00s
|
|
258
260
|
```
|
|
259
261
|
|
|
262
|
+
## Progress and logging
|
|
263
|
+
|
|
264
|
+
`progress=True` shows a line on stderr that keeps moving even while every request is waiting, so you can tell a long run is alive:
|
|
265
|
+
|
|
266
|
+
```text
|
|
267
|
+
reqstorm: 3500/7000 (50%) ok 3493
|
|
268
|
+
failed 7 active 12 retries 41
|
|
269
|
+
1.7 req/s ETA 34m 10s
|
|
270
|
+
```
|
|
271
|
+
|
|
272
|
+
It shows requests in flight, retries, the rate over the last minute, the retry round, hosts paused after a 429, and the time left. `progress=` also takes a function, which gets a `ProgressInfo` every two seconds. For a generator, pass `total=` to get a percentage.
|
|
273
|
+
|
|
274
|
+
**Logs.** reqstorm logs each retry and its reason, pauses, token refreshes and final failures. Turn them on for one run, independent of your application's logging setup:
|
|
275
|
+
|
|
276
|
+
```python
|
|
277
|
+
reqstorm.fetch_to_file_sync(
|
|
278
|
+
urls, "out.jsonl",
|
|
279
|
+
log_level="INFO", # or DEBUG: every attempt
|
|
280
|
+
log_file="logs/", # a new file per run
|
|
281
|
+
log_format="json", # optional
|
|
282
|
+
)
|
|
283
|
+
```
|
|
284
|
+
|
|
285
|
+
```text
|
|
286
|
+
14:32:41 WARNING reqstorm: GET .../items/412
|
|
287
|
+
failed (HTTP 503) on attempt 1 of 3;
|
|
288
|
+
retrying in 0.4s
|
|
289
|
+
15:42:18 ERROR reqstorm: GET .../items/913
|
|
290
|
+
failed after 3 attempts: HTTP 404
|
|
291
|
+
```
|
|
292
|
+
|
|
293
|
+
Log files are never overwritten: a directory or `{time}` in the name gives a time-stamped file per run, and an existing `run.log` becomes `run-2.log`. Without `log_level`, reqstorm logs to the standard `"reqstorm"` logger and follows your application's configuration.
|
|
294
|
+
|
|
295
|
+
**Read results while they are written.** Output files are written at least once a second (`flush_interval`), even while nothing finishes, and never with half a line; SQLite files are opened in WAL mode, so other programs can read them during the run.
|
|
296
|
+
|
|
260
297
|
## Timeouts and retries
|
|
261
298
|
|
|
262
299
|
```python
|
|
@@ -535,7 +572,7 @@ $ reqstorm urls.txt -o shop.db --report \
|
|
|
535
572
|
--schema products.json --explode items
|
|
536
573
|
```
|
|
537
574
|
|
|
538
|
-
`reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
|
|
575
|
+
`-v` / `-vv` log more, `-q` logs only errors, and `--log-file logs/` writes a new log file per run. `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
|
|
539
576
|
|
|
540
577
|
## When to use something else
|
|
541
578
|
|
|
@@ -26,13 +26,14 @@ from ._client import (
|
|
|
26
26
|
from ._files import Summary, fetch_to_db, fetch_to_file
|
|
27
27
|
from ._legacy import Reqt
|
|
28
28
|
from ._limits import parse_rate
|
|
29
|
+
from ._observe import ProgressInfo
|
|
29
30
|
from ._paginate import Cursor, LinkHeader, NextLink, PageNumber, Paginator
|
|
30
31
|
from ._plan import Estimate, estimate
|
|
31
32
|
from ._schema import Field, Schema, SchemaError, extract, infer_schema
|
|
32
33
|
from ._sync import fetch_all_sync, fetch_to_db_sync, fetch_to_file_sync, stream_sync
|
|
33
34
|
from ._template import from_template, read_csv, read_sql
|
|
34
35
|
|
|
35
|
-
__version__ = "2.
|
|
36
|
+
__version__ = "2.4.1"
|
|
36
37
|
|
|
37
38
|
__all__ = [
|
|
38
39
|
"Attempt",
|
|
@@ -46,6 +47,7 @@ __all__ = [
|
|
|
46
47
|
"LinkHeader",
|
|
47
48
|
"NextLink",
|
|
48
49
|
"PageNumber",
|
|
50
|
+
"ProgressInfo",
|
|
49
51
|
"Paginator",
|
|
50
52
|
"Reqt",
|
|
51
53
|
"Request",
|
|
@@ -6,10 +6,12 @@ import asyncio
|
|
|
6
6
|
import time
|
|
7
7
|
from dataclasses import dataclass
|
|
8
8
|
from email.utils import parsedate_to_datetime
|
|
9
|
-
from typing import Dict, Mapping, Optional
|
|
9
|
+
from typing import Dict, Mapping, Optional, Tuple
|
|
10
10
|
|
|
11
11
|
from ._limits import host_key
|
|
12
12
|
|
|
13
|
+
__all__ = ["AdaptiveRateLimiter", "seconds_until_reset"]
|
|
14
|
+
|
|
13
15
|
# Header names, most specific first. "RateLimit-*" is the IETF draft; "X-RateLimit-*" is the
|
|
14
16
|
# common convention (GitHub, Twitter/X, many others).
|
|
15
17
|
_REMAINING = ("RateLimit-Remaining", "X-RateLimit-Remaining", "X-Rate-Limit-Remaining")
|
|
@@ -91,17 +93,28 @@ class AdaptiveRateLimiter:
|
|
|
91
93
|
if slot > now:
|
|
92
94
|
await asyncio.sleep(slot - now)
|
|
93
95
|
|
|
94
|
-
def
|
|
96
|
+
def paused(self) -> Dict[str, float]:
|
|
97
|
+
"""Hosts that are paused right now, with the seconds left."""
|
|
98
|
+
now = time.monotonic()
|
|
99
|
+
return {
|
|
100
|
+
host: state.paused_until - now for host, state in self._hosts.items() if state.paused_until > now
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
def record(
|
|
104
|
+
self, url: str, status: Optional[int], headers: Mapping[str, str]
|
|
105
|
+
) -> Optional[Tuple[float, str]]:
|
|
106
|
+
"""Learn from a response. Returns ``(seconds, reason)`` when the host is paused."""
|
|
95
107
|
state = self._state(url)
|
|
96
108
|
if state is None or status is None:
|
|
97
|
-
return
|
|
109
|
+
return None
|
|
98
110
|
now = time.monotonic()
|
|
99
111
|
reset = seconds_until_reset(headers)
|
|
100
112
|
if status == 429:
|
|
101
113
|
state.interval = min(max(state.interval * 2, 0.1), 60.0)
|
|
102
114
|
pause = reset if reset is not None else max(state.interval, 1.0)
|
|
103
115
|
state.paused_until = max(state.paused_until, now + pause)
|
|
104
|
-
|
|
116
|
+
source = "Retry-After" if reset is not None else "no Retry-After"
|
|
117
|
+
return pause, f"429 Too Many Requests ({source}), spacing requests {state.interval:.2f}s apart"
|
|
105
118
|
remaining_text = _header(headers, _REMAINING)
|
|
106
119
|
if remaining_text is not None and reset is not None:
|
|
107
120
|
try:
|
|
@@ -111,8 +124,9 @@ class AdaptiveRateLimiter:
|
|
|
111
124
|
if remaining is not None:
|
|
112
125
|
if remaining <= 0:
|
|
113
126
|
state.paused_until = max(state.paused_until, now + reset)
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
return
|
|
127
|
+
return (reset, "rate limit used up until the window resets") if reset > 0 else None
|
|
128
|
+
state.interval = min(reset / remaining, 60.0)
|
|
129
|
+
return None
|
|
117
130
|
if 200 <= status < 400 and state.interval:
|
|
118
131
|
state.interval = state.interval * 0.9 if state.interval > 0.01 else 0.0
|
|
132
|
+
return None
|
|
@@ -19,100 +19,142 @@ from ._schema import Field, Schema, SchemaError, infer_schema
|
|
|
19
19
|
from ._sync import fetch_all_sync, fetch_to_file_sync, stream_sync
|
|
20
20
|
from ._template import from_template, read_csv
|
|
21
21
|
|
|
22
|
-
|
|
22
|
+
EPILOG = """\
|
|
23
23
|
examples:
|
|
24
24
|
reqstorm urls.txt -o results.jsonl --rate 100/min --retries 2
|
|
25
25
|
reqstorm urls.txt --estimate --rate 100/min
|
|
26
26
|
cat urls.txt | reqstorm --rate auto > results.jsonl
|
|
27
|
-
reqstorm
|
|
27
|
+
reqstorm urls.txt -o results.db -v --log-file logs/
|
|
28
|
+
reqstorm users.csv --template "https://api.example.com/users/{id}" -o users.jsonl
|
|
28
29
|
reqstorm urls.txt --paginate next:links.next --schema products.json --explode items -o shop.db
|
|
29
30
|
reqstorm urls.txt --infer-schema 5 --explode items > products.json
|
|
31
|
+
reqstorm urls.txt --proxy socks5h://127.0.0.1:9050 -o out.jsonl
|
|
30
32
|
|
|
31
|
-
|
|
32
|
-
2
|
|
33
|
+
rate limits:
|
|
34
|
+
5 or 0.5 (per second), 10/s, 100/min, 30/5min, 1000/h, 2/day, per host; or auto to
|
|
35
|
+
follow the server's 429 responses and X-RateLimit headers.
|
|
36
|
+
|
|
37
|
+
environment:
|
|
38
|
+
REQSTORM_TOKEN bearer token, instead of --bearer (keeps it out of shell history)
|
|
39
|
+
|
|
40
|
+
exit status:
|
|
41
|
+
0 every request succeeded 1 some failed or records were rejected
|
|
42
|
+
2 invalid arguments 130 interrupted
|
|
43
|
+
|
|
44
|
+
documentation: https://reqstorm.github.io/guide/cli/
|
|
33
45
|
"""
|
|
34
46
|
|
|
35
47
|
|
|
48
|
+
# fmt: off
|
|
36
49
|
def _parser() -> argparse.ArgumentParser:
|
|
37
50
|
parser = argparse.ArgumentParser(
|
|
38
51
|
prog="reqstorm",
|
|
39
52
|
description="Send many HTTP requests with rate limits, retries and progress, and save the results.",
|
|
40
|
-
epilog=
|
|
41
|
-
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
53
|
+
epilog=EPILOG,
|
|
54
|
+
formatter_class=lambda prog: argparse.RawDescriptionHelpFormatter(prog, max_help_position=32),
|
|
42
55
|
)
|
|
43
|
-
parser.add_argument
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
56
|
+
add = parser.add_argument
|
|
57
|
+
add("input", nargs="?", default="-",
|
|
58
|
+
help="file with one URL per line (blank lines and # comments are skipped), or a CSV file "
|
|
59
|
+
"with --template; default: standard input")
|
|
60
|
+
add("-o", "--output", metavar="FILE",
|
|
61
|
+
help="write results to FILE: .jsonl, .csv, or .db/.sqlite for SQLite; "
|
|
62
|
+
"default: JSON lines on standard output")
|
|
63
|
+
add("--version", action="version", version=f"reqstorm {__version__}")
|
|
49
64
|
|
|
50
65
|
request = parser.add_argument_group("requests")
|
|
51
|
-
request.add_argument
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
66
|
+
add = request.add_argument
|
|
67
|
+
add("-X", "--method", default="GET", help="HTTP method for every request (default: GET)")
|
|
68
|
+
add("-H", "--header", action="append", default=[], metavar="'NAME: VALUE'",
|
|
69
|
+
help="header sent with every request; repeat for more")
|
|
70
|
+
add("--json", metavar="JSON", help="JSON body sent with every request")
|
|
71
|
+
add("--data", metavar="TEXT", help="raw body sent with every request")
|
|
72
|
+
add("--template", metavar="URL",
|
|
73
|
+
help="treat the input as CSV and build one URL per row from {column} placeholders, "
|
|
74
|
+
"e.g. 'https://api.example.com/users/{id}'")
|
|
75
|
+
add("--bearer", metavar="TOKEN", help="send 'Authorization: Bearer TOKEN' (or set REQSTORM_TOKEN)")
|
|
76
|
+
add("--proxy", action="append", default=[], metavar="URL",
|
|
77
|
+
help="send through a proxy: http://, socks5://, socks5h://, socks4:// or socks4a://; "
|
|
78
|
+
"repeat to rotate through several")
|
|
79
|
+
add("--no-verify", action="store_true",
|
|
80
|
+
help="do not verify TLS certificates (only for hosts you control)")
|
|
81
|
+
|
|
82
|
+
pace = parser.add_argument_group("pace")
|
|
83
|
+
add = pace.add_argument
|
|
84
|
+
add("--rate", metavar="LIMIT",
|
|
85
|
+
help="per-host rate limit: 5, 10/s, 100/min, 1000/h, or auto (default: none)")
|
|
86
|
+
add("-c", "--concurrency", type=int, default=100, metavar="N",
|
|
87
|
+
help="requests in flight at once (default: 100)")
|
|
88
|
+
add("--per-host", type=int, default=0, metavar="N",
|
|
89
|
+
help="requests in flight per host (default: no limit)")
|
|
90
|
+
add("--timeout", type=float, default=30.0, metavar="SECONDS",
|
|
91
|
+
help="time allowed for one attempt, including the body (default: 30)")
|
|
92
|
+
|
|
93
|
+
retry = parser.add_argument_group("retries")
|
|
94
|
+
add = retry.add_argument
|
|
95
|
+
add("--retries", type=int, default=0, metavar="N",
|
|
96
|
+
help="retry timeouts, connection errors and 429/5xx up to N times (default: 0)")
|
|
97
|
+
add("--backoff", type=float, default=0.5, metavar="SECONDS",
|
|
98
|
+
help="wait before the first retry, doubled after each (default: 0.5)")
|
|
99
|
+
add("--max-backoff", type=float, default=30.0, metavar="SECONDS",
|
|
100
|
+
help="longest wait between retries (default: 30)")
|
|
101
|
+
add("--no-jitter", action="store_true", help="wait exactly, instead of a random half to all of the delay")
|
|
102
|
+
add("--retry-rounds", type=int, default=0, metavar="N",
|
|
103
|
+
help="send requests that still failed again after all others, up to N rounds (needs -o)")
|
|
104
|
+
add("--retry-round-delay", type=float, default=5.0, metavar="SECONDS",
|
|
105
|
+
help="wait before each retry round (default: 5)")
|
|
106
|
+
|
|
107
|
+
more = parser.add_argument_group("pagination and caching")
|
|
108
|
+
add = more.add_argument
|
|
109
|
+
add("--paginate", metavar="STRATEGY",
|
|
110
|
+
help="follow pages: next:PATH (URL in the JSON), link (Link header), cursor:PATH[:PARAM], "
|
|
111
|
+
"or page[:PARAM[:ITEMS_PATH]]")
|
|
112
|
+
add("--max-pages", type=int, default=1000, metavar="N",
|
|
113
|
+
help="pages per starting URL at most (default: 1000)")
|
|
114
|
+
add("--cache", metavar="FILE", help="keep responses in this SQLite file and revalidate them with ETag")
|
|
115
|
+
add("--cache-ttl", type=float, metavar="SECONDS",
|
|
116
|
+
help="use a cached response without asking the server for this long (default: always ask)")
|
|
92
117
|
|
|
93
118
|
output = parser.add_argument_group("output")
|
|
94
|
-
output.add_argument
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
119
|
+
add = output.add_argument
|
|
120
|
+
add("--body", choices=["text", "base64", "none"], default="text",
|
|
121
|
+
help="how to store response bodies (default: text)")
|
|
122
|
+
add("--include-headers", action="store_true", help="store response headers too")
|
|
123
|
+
add("--resume", action="store_true",
|
|
124
|
+
help="skip requests already saved successfully in -o, append the rest")
|
|
125
|
+
add("--ordered", action="store_true", help="write results in input order instead of as they finish")
|
|
126
|
+
add("--table", default="reqstorm_results", metavar="NAME",
|
|
127
|
+
help="table for .db output (default: reqstorm_results)")
|
|
128
|
+
add("--flush-interval", type=float, default=1.0, metavar="SECONDS",
|
|
129
|
+
help="write waiting results to -o at least this often (default: 1)")
|
|
130
|
+
|
|
131
|
+
data = parser.add_argument_group("typed columns")
|
|
132
|
+
add = data.add_argument
|
|
133
|
+
add("--schema", metavar="FILE",
|
|
134
|
+
help="JSON file mapping columns to fields, e.g. {\"price\": {\"path\": \"pricing.amount\", "
|
|
135
|
+
"\"type\": \"float\"}}; writes typed rows instead of raw responses (needs -o)")
|
|
136
|
+
add("--explode", metavar="PATH", help="with --schema: one row per element of the array at PATH")
|
|
137
|
+
add("--rejects-table", metavar="NAME",
|
|
138
|
+
help="with --schema and .db output: also store rejected records here")
|
|
139
|
+
add("--infer-schema", type=int, metavar="N",
|
|
140
|
+
help="fetch the first N inputs, print a draft --schema file, exit")
|
|
141
|
+
|
|
142
|
+
info = parser.add_argument_group("progress and logging")
|
|
143
|
+
add = info.add_argument
|
|
144
|
+
add("-v", "--verbose", action="count", default=0,
|
|
145
|
+
help="log more: -v run events (INFO), -vv every attempt (DEBUG); "
|
|
146
|
+
"warnings and errors are always logged")
|
|
147
|
+
add("-q", "--quiet", action="store_true", help="no progress line, log errors only")
|
|
148
|
+
add("--log-file", metavar="PATH",
|
|
149
|
+
help="log to a file instead of stderr; a directory or a name with {time} gives a time-stamped "
|
|
150
|
+
"file per run, and an existing file is never overwritten")
|
|
151
|
+
add("--log-json", action="store_true", help="log one JSON object per line")
|
|
152
|
+
add("--report", action="store_true", help="print response times, statuses and hosts at the end")
|
|
153
|
+
add("--estimate", action="store_true", help="print how long the batch would take, send nothing, exit")
|
|
154
|
+
add("--latency", type=float, default=0.5, metavar="SECONDS",
|
|
155
|
+
help="typical response time assumed by --estimate (default: 0.5)")
|
|
115
156
|
return parser
|
|
157
|
+
# fmt: on
|
|
116
158
|
|
|
117
159
|
|
|
118
160
|
def _paginator(text: Optional[str], max_pages: int) -> Optional[Paginator]:
|
|
@@ -293,17 +335,23 @@ def _batch(args: argparse.Namespace, cache: Optional[Cache]) -> int:
|
|
|
293
335
|
return 0
|
|
294
336
|
|
|
295
337
|
progress: Any = False if args.quiet else True
|
|
338
|
+
options["total"] = _count(args) if not args.paginate else None
|
|
339
|
+
options["log_level"] = "ERROR" if args.quiet else ("WARNING", "INFO", "DEBUG")[min(args.verbose, 2)]
|
|
340
|
+
options["log_file"] = args.log_file
|
|
341
|
+
options["log_format"] = "json" if args.log_json else "text"
|
|
296
342
|
if args.output:
|
|
297
343
|
summary = fetch_to_file_sync(
|
|
298
344
|
_inputs(args), args.output, body=args.body, include_headers=args.include_headers,
|
|
299
345
|
resume=args.resume, ordered=args.ordered, progress=progress, retry_rounds=args.retry_rounds,
|
|
300
346
|
retry_round_delay=args.retry_round_delay, table=args.table, schema=schema,
|
|
301
|
-
rejects_table=args.rejects_table, **options,
|
|
347
|
+
rejects_table=args.rejects_table, flush_interval=args.flush_interval, **options,
|
|
302
348
|
) # fmt: skip
|
|
303
349
|
message = f"reqstorm: {summary.ok} ok, {summary.failed} failed, {summary.skipped} skipped"
|
|
304
350
|
if schema is not None:
|
|
305
351
|
message += f", {summary.rows} rows, {summary.rejected} rejected"
|
|
306
352
|
print(message + f" -> {args.output}", file=sys.stderr)
|
|
353
|
+
if summary.log_file:
|
|
354
|
+
print(f"reqstorm: log written to {summary.log_file}", file=sys.stderr)
|
|
307
355
|
if args.report:
|
|
308
356
|
print(json.dumps(summary.report, indent=2), file=sys.stderr)
|
|
309
357
|
return 0 if summary.failed == 0 and summary.rejected == 0 else 1
|
|
@@ -314,7 +362,7 @@ def _batch(args: argparse.Namespace, cache: Optional[Cache]) -> int:
|
|
|
314
362
|
|
|
315
363
|
report = Report()
|
|
316
364
|
failed = 0
|
|
317
|
-
for result in stream_sync(_inputs(args), **options):
|
|
365
|
+
for result in stream_sync(_inputs(args), progress=progress, **options):
|
|
318
366
|
report.add(result)
|
|
319
367
|
failed += 0 if result.ok else 1
|
|
320
368
|
sys.stdout.write(json.dumps(result.to_dict(body=args.body, include_headers=args.include_headers),
|