reqstorm 2.3.0__tar.gz → 2.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. {reqstorm-2.3.0/reqstorm.egg-info → reqstorm-2.4.1}/PKG-INFO +39 -2
  2. {reqstorm-2.3.0 → reqstorm-2.4.1}/README.md +38 -1
  3. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/__init__.py +3 -1
  4. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_adaptive.py +21 -7
  5. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_cli.py +124 -76
  6. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_client.py +98 -20
  7. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_files.py +105 -37
  8. reqstorm-2.4.1/reqstorm/_observe.py +408 -0
  9. reqstorm-2.4.1/reqstorm/_progress.py +15 -0
  10. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_socks.py +19 -0
  11. {reqstorm-2.3.0 → reqstorm-2.4.1/reqstorm.egg-info}/PKG-INFO +39 -2
  12. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm.egg-info/SOURCES.txt +2 -0
  13. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_cli.py +37 -0
  14. reqstorm-2.4.1/tests/test_observe.py +267 -0
  15. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_report.py +7 -8
  16. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_socks_backoff.py +20 -0
  17. reqstorm-2.3.0/reqstorm/_progress.py +0 -82
  18. {reqstorm-2.3.0 → reqstorm-2.4.1}/LICENSE +0 -0
  19. {reqstorm-2.3.0 → reqstorm-2.4.1}/pyproject.toml +0 -0
  20. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/__main__.py +0 -0
  21. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_auth.py +0 -0
  22. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_cache.py +0 -0
  23. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_legacy.py +0 -0
  24. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_limits.py +0 -0
  25. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_paginate.py +0 -0
  26. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_plan.py +0 -0
  27. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_pydantic.py +0 -0
  28. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_report.py +0 -0
  29. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_schema.py +0 -0
  30. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_schema_sinks.py +0 -0
  31. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_sync.py +0 -0
  32. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/_template.py +0 -0
  33. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm/py.typed +0 -0
  34. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm.egg-info/dependency_links.txt +0 -0
  35. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm.egg-info/entry_points.txt +0 -0
  36. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm.egg-info/requires.txt +0 -0
  37. {reqstorm-2.3.0 → reqstorm-2.4.1}/reqstorm.egg-info/top_level.txt +0 -0
  38. {reqstorm-2.3.0 → reqstorm-2.4.1}/setup.cfg +0 -0
  39. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_adaptive.py +0 -0
  40. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_auth.py +0 -0
  41. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_cache_proxy.py +0 -0
  42. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_fetch_all.py +0 -0
  43. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_files.py +0 -0
  44. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_legacy.py +0 -0
  45. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_limits.py +0 -0
  46. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_ordered_and_db.py +0 -0
  47. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_paginate.py +0 -0
  48. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_pydantic.py +0 -0
  49. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_run_report.py +0 -0
  50. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_schema.py +0 -0
  51. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_schema_db.py +0 -0
  52. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_stream.py +0 -0
  53. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_sync.py +0 -0
  54. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_template.py +0 -0
  55. {reqstorm-2.3.0 → reqstorm-2.4.1}/tests/test_tls.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: reqstorm
3
- Version: 2.3.0
3
+ Version: 2.4.1
4
4
  Summary: Send large numbers of HTTP requests concurrently with asyncio, with retries, timeouts and per-request results.
5
5
  Author-email: Melih Colpan <colpanmelih@gmail.com>
6
6
  License-Expression: MIT
@@ -94,6 +94,7 @@ for error in results.errors():
94
94
  - [Quick start](#quick-start)
95
95
  - [Results and reports](#results-and-reports)
96
96
  - [Rate limits and concurrency](#rate-limits-and-concurrency)
97
+ - [Progress and logging](#progress-and-logging)
97
98
  - [Timeouts and retries](#timeouts-and-retries)
98
99
  - [Writing results to a file](#writing-results-to-a-file)
99
100
  - [Writing results to a database](#writing-results-to-a-database)
@@ -137,6 +138,7 @@ reqstorm does all of that for you, with one call.
137
138
  | Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
138
139
  | Resume | Skip what already succeeded after an interruption |
139
140
  | Planning | `estimate()` before you start, progress with ETA while running |
141
+ | Logging | Retries, pauses and failures; per-run level, file and JSON, independent of the app |
140
142
  | API | Blocking (scripts, Jupyter), asyncio and a `reqstorm` command |
141
143
  | Safety | TLS verified, timeouts on, bounded concurrency, fully typed |
142
144
 
@@ -302,6 +304,41 @@ reqstorm: 3500/7000 (50%) ok 3493 failed 7
302
304
  1.7 req/s ETA 35m 00s
303
305
  ```
304
306
 
307
+ ## Progress and logging
308
+
309
+ `progress=True` shows a line on stderr that keeps moving even while every request is waiting, so you can tell a long run is alive:
310
+
311
+ ```text
312
+ reqstorm: 3500/7000 (50%) ok 3493
313
+ failed 7 active 12 retries 41
314
+ 1.7 req/s ETA 34m 10s
315
+ ```
316
+
317
+ It shows requests in flight, retries, the rate over the last minute, the retry round, hosts paused after a 429, and the time left. `progress=` also takes a function, which gets a `ProgressInfo` every two seconds. For a generator, pass `total=` to get a percentage.
318
+
319
+ **Logs.** reqstorm logs each retry and its reason, pauses, token refreshes and final failures. Turn them on for one run, independent of your application's logging setup:
320
+
321
+ ```python
322
+ reqstorm.fetch_to_file_sync(
323
+ urls, "out.jsonl",
324
+ log_level="INFO", # or DEBUG: every attempt
325
+ log_file="logs/", # a new file per run
326
+ log_format="json", # optional
327
+ )
328
+ ```
329
+
330
+ ```text
331
+ 14:32:41 WARNING reqstorm: GET .../items/412
332
+ failed (HTTP 503) on attempt 1 of 3;
333
+ retrying in 0.4s
334
+ 15:42:18 ERROR reqstorm: GET .../items/913
335
+ failed after 3 attempts: HTTP 404
336
+ ```
337
+
338
+ Log files are never overwritten: a directory or `{time}` in the name gives a time-stamped file per run, and an existing `run.log` becomes `run-2.log`. Without `log_level`, reqstorm logs to the standard `"reqstorm"` logger and follows your application's configuration.
339
+
340
+ **Read results while they are written.** Output files are written at least once a second (`flush_interval`), even while nothing finishes, and never with half a line; SQLite files are opened in WAL mode, so other programs can read them during the run.
341
+
305
342
  ## Timeouts and retries
306
343
 
307
344
  ```python
@@ -580,7 +617,7 @@ $ reqstorm urls.txt -o shop.db --report \
580
617
  --schema products.json --explode items
581
618
  ```
582
619
 
583
- `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
620
+ `-v` / `-vv` log more, `-q` logs only errors, and `--log-file logs/` writes a new log file per run. `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
584
621
 
585
622
  ## When to use something else
586
623
 
@@ -49,6 +49,7 @@ for error in results.errors():
49
49
  - [Quick start](#quick-start)
50
50
  - [Results and reports](#results-and-reports)
51
51
  - [Rate limits and concurrency](#rate-limits-and-concurrency)
52
+ - [Progress and logging](#progress-and-logging)
52
53
  - [Timeouts and retries](#timeouts-and-retries)
53
54
  - [Writing results to a file](#writing-results-to-a-file)
54
55
  - [Writing results to a database](#writing-results-to-a-database)
@@ -92,6 +93,7 @@ reqstorm does all of that for you, with one call.
92
93
  | Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
93
94
  | Resume | Skip what already succeeded after an interruption |
94
95
  | Planning | `estimate()` before you start, progress with ETA while running |
96
+ | Logging | Retries, pauses and failures; per-run level, file and JSON, independent of the app |
95
97
  | API | Blocking (scripts, Jupyter), asyncio and a `reqstorm` command |
96
98
  | Safety | TLS verified, timeouts on, bounded concurrency, fully typed |
97
99
 
@@ -257,6 +259,41 @@ reqstorm: 3500/7000 (50%) ok 3493 failed 7
257
259
  1.7 req/s ETA 35m 00s
258
260
  ```
259
261
 
262
+ ## Progress and logging
263
+
264
+ `progress=True` shows a line on stderr that keeps moving even while every request is waiting, so you can tell a long run is alive:
265
+
266
+ ```text
267
+ reqstorm: 3500/7000 (50%) ok 3493
268
+ failed 7 active 12 retries 41
269
+ 1.7 req/s ETA 34m 10s
270
+ ```
271
+
272
+ It shows requests in flight, retries, the rate over the last minute, the retry round, hosts paused after a 429, and the time left. `progress=` also takes a function, which gets a `ProgressInfo` every two seconds. For a generator, pass `total=` to get a percentage.
273
+
274
+ **Logs.** reqstorm logs each retry and its reason, pauses, token refreshes and final failures. Turn them on for one run, independent of your application's logging setup:
275
+
276
+ ```python
277
+ reqstorm.fetch_to_file_sync(
278
+ urls, "out.jsonl",
279
+ log_level="INFO", # or DEBUG: every attempt
280
+ log_file="logs/", # a new file per run
281
+ log_format="json", # optional
282
+ )
283
+ ```
284
+
285
+ ```text
286
+ 14:32:41 WARNING reqstorm: GET .../items/412
287
+ failed (HTTP 503) on attempt 1 of 3;
288
+ retrying in 0.4s
289
+ 15:42:18 ERROR reqstorm: GET .../items/913
290
+ failed after 3 attempts: HTTP 404
291
+ ```
292
+
293
+ Log files are never overwritten: a directory or `{time}` in the name gives a time-stamped file per run, and an existing `run.log` becomes `run-2.log`. Without `log_level`, reqstorm logs to the standard `"reqstorm"` logger and follows your application's configuration.
294
+
295
+ **Read results while they are written.** Output files are written at least once a second (`flush_interval`), even while nothing finishes, and never with half a line; SQLite files are opened in WAL mode, so other programs can read them during the run.
296
+
260
297
  ## Timeouts and retries
261
298
 
262
299
  ```python
@@ -535,7 +572,7 @@ $ reqstorm urls.txt -o shop.db --report \
535
572
  --schema products.json --explode items
536
573
  ```
537
574
 
538
- `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
575
+ `-v` / `-vv` log more, `-q` logs only errors, and `--log-file logs/` writes a new log file per run. `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
539
576
 
540
577
  ## When to use something else
541
578
 
@@ -26,13 +26,14 @@ from ._client import (
26
26
  from ._files import Summary, fetch_to_db, fetch_to_file
27
27
  from ._legacy import Reqt
28
28
  from ._limits import parse_rate
29
+ from ._observe import ProgressInfo
29
30
  from ._paginate import Cursor, LinkHeader, NextLink, PageNumber, Paginator
30
31
  from ._plan import Estimate, estimate
31
32
  from ._schema import Field, Schema, SchemaError, extract, infer_schema
32
33
  from ._sync import fetch_all_sync, fetch_to_db_sync, fetch_to_file_sync, stream_sync
33
34
  from ._template import from_template, read_csv, read_sql
34
35
 
35
- __version__ = "2.3.0"
36
+ __version__ = "2.4.1"
36
37
 
37
38
  __all__ = [
38
39
  "Attempt",
@@ -46,6 +47,7 @@ __all__ = [
46
47
  "LinkHeader",
47
48
  "NextLink",
48
49
  "PageNumber",
50
+ "ProgressInfo",
49
51
  "Paginator",
50
52
  "Reqt",
51
53
  "Request",
@@ -6,10 +6,12 @@ import asyncio
6
6
  import time
7
7
  from dataclasses import dataclass
8
8
  from email.utils import parsedate_to_datetime
9
- from typing import Dict, Mapping, Optional
9
+ from typing import Dict, Mapping, Optional, Tuple
10
10
 
11
11
  from ._limits import host_key
12
12
 
13
+ __all__ = ["AdaptiveRateLimiter", "seconds_until_reset"]
14
+
13
15
  # Header names, most specific first. "RateLimit-*" is the IETF draft; "X-RateLimit-*" is the
14
16
  # common convention (GitHub, Twitter/X, many others).
15
17
  _REMAINING = ("RateLimit-Remaining", "X-RateLimit-Remaining", "X-Rate-Limit-Remaining")
@@ -91,17 +93,28 @@ class AdaptiveRateLimiter:
91
93
  if slot > now:
92
94
  await asyncio.sleep(slot - now)
93
95
 
94
- def record(self, url: str, status: Optional[int], headers: Mapping[str, str]) -> None:
96
+ def paused(self) -> Dict[str, float]:
97
+ """Hosts that are paused right now, with the seconds left."""
98
+ now = time.monotonic()
99
+ return {
100
+ host: state.paused_until - now for host, state in self._hosts.items() if state.paused_until > now
101
+ }
102
+
103
+ def record(
104
+ self, url: str, status: Optional[int], headers: Mapping[str, str]
105
+ ) -> Optional[Tuple[float, str]]:
106
+ """Learn from a response. Returns ``(seconds, reason)`` when the host is paused."""
95
107
  state = self._state(url)
96
108
  if state is None or status is None:
97
- return
109
+ return None
98
110
  now = time.monotonic()
99
111
  reset = seconds_until_reset(headers)
100
112
  if status == 429:
101
113
  state.interval = min(max(state.interval * 2, 0.1), 60.0)
102
114
  pause = reset if reset is not None else max(state.interval, 1.0)
103
115
  state.paused_until = max(state.paused_until, now + pause)
104
- return
116
+ source = "Retry-After" if reset is not None else "no Retry-After"
117
+ return pause, f"429 Too Many Requests ({source}), spacing requests {state.interval:.2f}s apart"
105
118
  remaining_text = _header(headers, _REMAINING)
106
119
  if remaining_text is not None and reset is not None:
107
120
  try:
@@ -111,8 +124,9 @@ class AdaptiveRateLimiter:
111
124
  if remaining is not None:
112
125
  if remaining <= 0:
113
126
  state.paused_until = max(state.paused_until, now + reset)
114
- else:
115
- state.interval = min(reset / remaining, 60.0)
116
- return
127
+ return (reset, "rate limit used up until the window resets") if reset > 0 else None
128
+ state.interval = min(reset / remaining, 60.0)
129
+ return None
117
130
  if 200 <= status < 400 and state.interval:
118
131
  state.interval = state.interval * 0.9 if state.interval > 0.01 else 0.0
132
+ return None
@@ -19,100 +19,142 @@ from ._schema import Field, Schema, SchemaError, infer_schema
19
19
  from ._sync import fetch_all_sync, fetch_to_file_sync, stream_sync
20
20
  from ._template import from_template, read_csv
21
21
 
22
- EXAMPLES = """\
22
+ EPILOG = """\
23
23
  examples:
24
24
  reqstorm urls.txt -o results.jsonl --rate 100/min --retries 2
25
25
  reqstorm urls.txt --estimate --rate 100/min
26
26
  cat urls.txt | reqstorm --rate auto > results.jsonl
27
- reqstorm users.csv --template "https://api.example.com/users/{id}" -o users.db
27
+ reqstorm urls.txt -o results.db -v --log-file logs/
28
+ reqstorm users.csv --template "https://api.example.com/users/{id}" -o users.jsonl
28
29
  reqstorm urls.txt --paginate next:links.next --schema products.json --explode items -o shop.db
29
30
  reqstorm urls.txt --infer-schema 5 --explode items > products.json
31
+ reqstorm urls.txt --proxy socks5h://127.0.0.1:9050 -o out.jsonl
30
32
 
31
- exit status: 0 when every request succeeded, 1 when some failed or records were rejected,
32
- 2 for invalid arguments.
33
+ rate limits:
34
+ 5 or 0.5 (per second), 10/s, 100/min, 30/5min, 1000/h, 2/day, per host; or auto to
35
+ follow the server's 429 responses and X-RateLimit headers.
36
+
37
+ environment:
38
+ REQSTORM_TOKEN bearer token, instead of --bearer (keeps it out of shell history)
39
+
40
+ exit status:
41
+ 0 every request succeeded 1 some failed or records were rejected
42
+ 2 invalid arguments 130 interrupted
43
+
44
+ documentation: https://reqstorm.github.io/guide/cli/
33
45
  """
34
46
 
35
47
 
48
+ # fmt: off
36
49
  def _parser() -> argparse.ArgumentParser:
37
50
  parser = argparse.ArgumentParser(
38
51
  prog="reqstorm",
39
52
  description="Send many HTTP requests with rate limits, retries and progress, and save the results.",
40
- epilog=EXAMPLES,
41
- formatter_class=argparse.RawDescriptionHelpFormatter,
53
+ epilog=EPILOG,
54
+ formatter_class=lambda prog: argparse.RawDescriptionHelpFormatter(prog, max_help_position=32),
42
55
  )
43
- parser.add_argument("input", nargs="?", default="-",
44
- help="file with one URL per line, or a CSV with --template "
45
- "(default: standard input)") # fmt: skip
46
- parser.add_argument("-o", "--output", help="write results to a .jsonl, .csv or .db/.sqlite file "
47
- "(default: JSON lines on standard output)") # fmt: skip
48
- parser.add_argument("--version", action="version", version=f"reqstorm {__version__}")
56
+ add = parser.add_argument
57
+ add("input", nargs="?", default="-",
58
+ help="file with one URL per line (blank lines and # comments are skipped), or a CSV file "
59
+ "with --template; default: standard input")
60
+ add("-o", "--output", metavar="FILE",
61
+ help="write results to FILE: .jsonl, .csv, or .db/.sqlite for SQLite; "
62
+ "default: JSON lines on standard output")
63
+ add("--version", action="version", version=f"reqstorm {__version__}")
49
64
 
50
65
  request = parser.add_argument_group("requests")
51
- request.add_argument("-X", "--method", default="GET")
52
- request.add_argument("-H", "--header", action="append", default=[], metavar="'NAME: VALUE'")
53
- request.add_argument("--json", help="JSON body sent with every request")
54
- request.add_argument("--data", help="raw body sent with every request")
55
- request.add_argument("--template", metavar="URL", help="URL template filled from each CSV row, "
56
- "e.g. 'https://api.example.com/users/{id}'") # fmt: skip
57
- request.add_argument("--bearer", metavar="TOKEN", help="send 'Authorization: Bearer TOKEN' "
58
- "(or set REQSTORM_TOKEN)") # fmt: skip
59
- request.add_argument(
60
- "--proxy",
61
- action="append",
62
- default=[],
63
- metavar="URL",
64
- help="http://, socks5://, socks5h:// or socks4:// proxy URL; repeat to rotate",
65
- )
66
- request.add_argument("--no-verify", action="store_true", help="do not verify TLS certificates")
67
-
68
- pace = parser.add_argument_group("pace and retries")
69
- pace.add_argument("--rate", help="per-host rate limit: 5, 10/s, 100/min, 1000/h, or auto")
70
- pace.add_argument("-c", "--concurrency", type=int, default=100)
71
- pace.add_argument("--per-host", type=int, default=0, metavar="N", help="max requests in flight per host")
72
- pace.add_argument("--timeout", type=float, default=30.0, help="seconds per attempt (default 30)")
73
- pace.add_argument("--retries", type=int, default=0)
74
- pace.add_argument(
75
- "--backoff", type=float, default=0.5, help="seconds before the first retry, doubled after"
76
- )
77
- pace.add_argument(
78
- "--max-backoff", type=float, default=30.0, help="longest wait between retries (default 30)"
79
- )
80
- pace.add_argument("--no-jitter", action="store_true", help="wait exactly, not a random part of the delay")
81
- pace.add_argument("--retry-rounds", type=int, default=0)
82
- pace.add_argument("--retry-round-delay", type=float, default=5.0)
83
- pace.add_argument("--paginate", metavar="STRATEGY",
84
- help="next:PATH, link, cursor:PATH[:PARAM], or page[:PARAM[:ITEMS_PATH]]") # fmt: skip
85
- pace.add_argument("--max-pages", type=int, default=1000)
86
- pace.add_argument(
87
- "--cache", metavar="FILE", help="reuse responses from this cache file (revalidated with ETag)"
88
- )
89
- pace.add_argument(
90
- "--cache-ttl", type=float, help="seconds a cached response is used without asking the server"
91
- )
66
+ add = request.add_argument
67
+ add("-X", "--method", default="GET", help="HTTP method for every request (default: GET)")
68
+ add("-H", "--header", action="append", default=[], metavar="'NAME: VALUE'",
69
+ help="header sent with every request; repeat for more")
70
+ add("--json", metavar="JSON", help="JSON body sent with every request")
71
+ add("--data", metavar="TEXT", help="raw body sent with every request")
72
+ add("--template", metavar="URL",
73
+ help="treat the input as CSV and build one URL per row from {column} placeholders, "
74
+ "e.g. 'https://api.example.com/users/{id}'")
75
+ add("--bearer", metavar="TOKEN", help="send 'Authorization: Bearer TOKEN' (or set REQSTORM_TOKEN)")
76
+ add("--proxy", action="append", default=[], metavar="URL",
77
+ help="send through a proxy: http://, socks5://, socks5h://, socks4:// or socks4a://; "
78
+ "repeat to rotate through several")
79
+ add("--no-verify", action="store_true",
80
+ help="do not verify TLS certificates (only for hosts you control)")
81
+
82
+ pace = parser.add_argument_group("pace")
83
+ add = pace.add_argument
84
+ add("--rate", metavar="LIMIT",
85
+ help="per-host rate limit: 5, 10/s, 100/min, 1000/h, or auto (default: none)")
86
+ add("-c", "--concurrency", type=int, default=100, metavar="N",
87
+ help="requests in flight at once (default: 100)")
88
+ add("--per-host", type=int, default=0, metavar="N",
89
+ help="requests in flight per host (default: no limit)")
90
+ add("--timeout", type=float, default=30.0, metavar="SECONDS",
91
+ help="time allowed for one attempt, including the body (default: 30)")
92
+
93
+ retry = parser.add_argument_group("retries")
94
+ add = retry.add_argument
95
+ add("--retries", type=int, default=0, metavar="N",
96
+ help="retry timeouts, connection errors and 429/5xx up to N times (default: 0)")
97
+ add("--backoff", type=float, default=0.5, metavar="SECONDS",
98
+ help="wait before the first retry, doubled after each (default: 0.5)")
99
+ add("--max-backoff", type=float, default=30.0, metavar="SECONDS",
100
+ help="longest wait between retries (default: 30)")
101
+ add("--no-jitter", action="store_true", help="wait exactly, instead of a random half to all of the delay")
102
+ add("--retry-rounds", type=int, default=0, metavar="N",
103
+ help="send requests that still failed again after all others, up to N rounds (needs -o)")
104
+ add("--retry-round-delay", type=float, default=5.0, metavar="SECONDS",
105
+ help="wait before each retry round (default: 5)")
106
+
107
+ more = parser.add_argument_group("pagination and caching")
108
+ add = more.add_argument
109
+ add("--paginate", metavar="STRATEGY",
110
+ help="follow pages: next:PATH (URL in the JSON), link (Link header), cursor:PATH[:PARAM], "
111
+ "or page[:PARAM[:ITEMS_PATH]]")
112
+ add("--max-pages", type=int, default=1000, metavar="N",
113
+ help="pages per starting URL at most (default: 1000)")
114
+ add("--cache", metavar="FILE", help="keep responses in this SQLite file and revalidate them with ETag")
115
+ add("--cache-ttl", type=float, metavar="SECONDS",
116
+ help="use a cached response without asking the server for this long (default: always ask)")
92
117
 
93
118
  output = parser.add_argument_group("output")
94
- output.add_argument(
95
- "--resume", action="store_true", help="skip requests already saved successfully in --output"
96
- )
97
- output.add_argument("--ordered", action="store_true", help="write results in input order")
98
- output.add_argument("--body", choices=["text", "base64", "none"], default="text")
99
- output.add_argument("--include-headers", action="store_true")
100
- output.add_argument("--schema", metavar="FILE", help="JSON file mapping columns to fields: "
101
- '{"price": {"path": "pricing.amount", "type": "float"}}') # fmt: skip
102
- output.add_argument("--explode", metavar="PATH", help="with --schema: one row per element of this array")
103
- output.add_argument("--table", default="reqstorm_results", help="table name for .db output")
104
- output.add_argument("--rejects-table", help="with --schema and .db output: table for rejected records")
105
- output.add_argument("--infer-schema", type=int, metavar="N",
106
- help="fetch the first N inputs, print a draft --schema file and exit") # fmt: skip
107
-
108
- info = parser.add_argument_group("information")
109
- info.add_argument("--estimate", action="store_true", help="print how long the batch would take and exit")
110
- info.add_argument("--latency", type=float, default=0.5, help="typical response time for --estimate")
111
- info.add_argument("-q", "--quiet", action="store_true", help="no progress line")
112
- info.add_argument(
113
- "--report", action="store_true", help="print response times, statuses and hosts at the end"
114
- )
119
+ add = output.add_argument
120
+ add("--body", choices=["text", "base64", "none"], default="text",
121
+ help="how to store response bodies (default: text)")
122
+ add("--include-headers", action="store_true", help="store response headers too")
123
+ add("--resume", action="store_true",
124
+ help="skip requests already saved successfully in -o, append the rest")
125
+ add("--ordered", action="store_true", help="write results in input order instead of as they finish")
126
+ add("--table", default="reqstorm_results", metavar="NAME",
127
+ help="table for .db output (default: reqstorm_results)")
128
+ add("--flush-interval", type=float, default=1.0, metavar="SECONDS",
129
+ help="write waiting results to -o at least this often (default: 1)")
130
+
131
+ data = parser.add_argument_group("typed columns")
132
+ add = data.add_argument
133
+ add("--schema", metavar="FILE",
134
+ help="JSON file mapping columns to fields, e.g. {\"price\": {\"path\": \"pricing.amount\", "
135
+ "\"type\": \"float\"}}; writes typed rows instead of raw responses (needs -o)")
136
+ add("--explode", metavar="PATH", help="with --schema: one row per element of the array at PATH")
137
+ add("--rejects-table", metavar="NAME",
138
+ help="with --schema and .db output: also store rejected records here")
139
+ add("--infer-schema", type=int, metavar="N",
140
+ help="fetch the first N inputs, print a draft --schema file, exit")
141
+
142
+ info = parser.add_argument_group("progress and logging")
143
+ add = info.add_argument
144
+ add("-v", "--verbose", action="count", default=0,
145
+ help="log more: -v run events (INFO), -vv every attempt (DEBUG); "
146
+ "warnings and errors are always logged")
147
+ add("-q", "--quiet", action="store_true", help="no progress line, log errors only")
148
+ add("--log-file", metavar="PATH",
149
+ help="log to a file instead of stderr; a directory or a name with {time} gives a time-stamped "
150
+ "file per run, and an existing file is never overwritten")
151
+ add("--log-json", action="store_true", help="log one JSON object per line")
152
+ add("--report", action="store_true", help="print response times, statuses and hosts at the end")
153
+ add("--estimate", action="store_true", help="print how long the batch would take, send nothing, exit")
154
+ add("--latency", type=float, default=0.5, metavar="SECONDS",
155
+ help="typical response time assumed by --estimate (default: 0.5)")
115
156
  return parser
157
+ # fmt: on
116
158
 
117
159
 
118
160
  def _paginator(text: Optional[str], max_pages: int) -> Optional[Paginator]:
@@ -293,17 +335,23 @@ def _batch(args: argparse.Namespace, cache: Optional[Cache]) -> int:
293
335
  return 0
294
336
 
295
337
  progress: Any = False if args.quiet else True
338
+ options["total"] = _count(args) if not args.paginate else None
339
+ options["log_level"] = "ERROR" if args.quiet else ("WARNING", "INFO", "DEBUG")[min(args.verbose, 2)]
340
+ options["log_file"] = args.log_file
341
+ options["log_format"] = "json" if args.log_json else "text"
296
342
  if args.output:
297
343
  summary = fetch_to_file_sync(
298
344
  _inputs(args), args.output, body=args.body, include_headers=args.include_headers,
299
345
  resume=args.resume, ordered=args.ordered, progress=progress, retry_rounds=args.retry_rounds,
300
346
  retry_round_delay=args.retry_round_delay, table=args.table, schema=schema,
301
- rejects_table=args.rejects_table, **options,
347
+ rejects_table=args.rejects_table, flush_interval=args.flush_interval, **options,
302
348
  ) # fmt: skip
303
349
  message = f"reqstorm: {summary.ok} ok, {summary.failed} failed, {summary.skipped} skipped"
304
350
  if schema is not None:
305
351
  message += f", {summary.rows} rows, {summary.rejected} rejected"
306
352
  print(message + f" -> {args.output}", file=sys.stderr)
353
+ if summary.log_file:
354
+ print(f"reqstorm: log written to {summary.log_file}", file=sys.stderr)
307
355
  if args.report:
308
356
  print(json.dumps(summary.report, indent=2), file=sys.stderr)
309
357
  return 0 if summary.failed == 0 and summary.rejected == 0 else 1
@@ -314,7 +362,7 @@ def _batch(args: argparse.Namespace, cache: Optional[Cache]) -> int:
314
362
 
315
363
  report = Report()
316
364
  failed = 0
317
- for result in stream_sync(_inputs(args), **options):
365
+ for result in stream_sync(_inputs(args), progress=progress, **options):
318
366
  report.add(result)
319
367
  failed += 0 if result.ok else 1
320
368
  sys.stdout.write(json.dumps(result.to_dict(body=args.body, include_headers=args.include_headers),