reqstorm 2.1.0__tar.gz → 2.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {reqstorm-2.1.0/reqstorm.egg-info → reqstorm-2.3.0}/PKG-INFO +138 -8
- {reqstorm-2.1.0 → reqstorm-2.3.0}/README.md +129 -7
- {reqstorm-2.1.0 → reqstorm-2.3.0}/pyproject.toml +7 -2
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm/__init__.py +18 -4
- reqstorm-2.3.0/reqstorm/__main__.py +5 -0
- reqstorm-2.3.0/reqstorm/_adaptive.py +118 -0
- reqstorm-2.3.0/reqstorm/_auth.py +79 -0
- reqstorm-2.3.0/reqstorm/_cache.py +141 -0
- reqstorm-2.3.0/reqstorm/_cli.py +329 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm/_client.py +256 -29
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm/_files.py +19 -4
- reqstorm-2.3.0/reqstorm/_paginate.py +144 -0
- reqstorm-2.3.0/reqstorm/_pydantic.py +103 -0
- reqstorm-2.3.0/reqstorm/_report.py +91 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm/_schema.py +9 -3
- reqstorm-2.3.0/reqstorm/_socks.py +84 -0
- reqstorm-2.3.0/reqstorm/_template.py +117 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0/reqstorm.egg-info}/PKG-INFO +138 -8
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm.egg-info/SOURCES.txt +20 -0
- reqstorm-2.3.0/reqstorm.egg-info/entry_points.txt +2 -0
- reqstorm-2.3.0/reqstorm.egg-info/requires.txt +20 -0
- reqstorm-2.3.0/tests/test_adaptive.py +86 -0
- reqstorm-2.3.0/tests/test_auth.py +74 -0
- reqstorm-2.3.0/tests/test_cache_proxy.py +119 -0
- reqstorm-2.3.0/tests/test_cli.py +121 -0
- reqstorm-2.3.0/tests/test_paginate.py +113 -0
- reqstorm-2.3.0/tests/test_pydantic.py +115 -0
- reqstorm-2.3.0/tests/test_run_report.py +52 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/tests/test_schema_db.py +22 -0
- reqstorm-2.3.0/tests/test_socks_backoff.py +144 -0
- reqstorm-2.3.0/tests/test_template.py +63 -0
- reqstorm-2.1.0/reqstorm.egg-info/requires.txt +0 -10
- {reqstorm-2.1.0 → reqstorm-2.3.0}/LICENSE +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm/_legacy.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm/_limits.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm/_plan.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm/_progress.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm/_schema_sinks.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm/_sync.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm/py.typed +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm.egg-info/dependency_links.txt +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/reqstorm.egg-info/top_level.txt +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/setup.cfg +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/tests/test_fetch_all.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/tests/test_files.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/tests/test_legacy.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/tests/test_limits.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/tests/test_ordered_and_db.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/tests/test_report.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/tests/test_schema.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/tests/test_stream.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/tests/test_sync.py +0 -0
- {reqstorm-2.1.0 → reqstorm-2.3.0}/tests/test_tls.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: reqstorm
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.3.0
|
|
4
4
|
Summary: Send large numbers of HTTP requests concurrently with asyncio, with retries, timeouts and per-request results.
|
|
5
5
|
Author-email: Melih Colpan <colpanmelih@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -26,13 +26,21 @@ Requires-Python: >=3.9
|
|
|
26
26
|
Description-Content-Type: text/markdown
|
|
27
27
|
License-File: LICENSE
|
|
28
28
|
Requires-Dist: aiohttp<4,>=3.9
|
|
29
|
+
Provides-Extra: pydantic
|
|
30
|
+
Requires-Dist: pydantic>=2; extra == "pydantic"
|
|
31
|
+
Provides-Extra: socks
|
|
32
|
+
Requires-Dist: aiohttp-socks>=0.10; extra == "socks"
|
|
29
33
|
Provides-Extra: test
|
|
30
34
|
Requires-Dist: pytest>=8; extra == "test"
|
|
31
35
|
Requires-Dist: pytest-asyncio>=0.23; extra == "test"
|
|
32
36
|
Requires-Dist: trustme>=1.1; extra == "test"
|
|
37
|
+
Requires-Dist: pydantic>=2; extra == "test"
|
|
38
|
+
Requires-Dist: aiohttp-socks>=0.10; extra == "test"
|
|
33
39
|
Provides-Extra: lint
|
|
34
40
|
Requires-Dist: ruff>=0.6; extra == "lint"
|
|
35
41
|
Requires-Dist: mypy>=1.10; extra == "lint"
|
|
42
|
+
Requires-Dist: pydantic>=2; extra == "lint"
|
|
43
|
+
Requires-Dist: aiohttp-socks>=0.10; extra == "lint"
|
|
36
44
|
Dynamic: license-file
|
|
37
45
|
|
|
38
46
|
# reqstorm
|
|
@@ -72,7 +80,8 @@ results = reqstorm.fetch_all_sync(
|
|
|
72
80
|
|
|
73
81
|
print(results.summary())
|
|
74
82
|
# {'total': 7000, 'ok': 6987, 'failed': 13,
|
|
75
|
-
# 'failures': {'HTTP 404': 9, 'TimeoutError': 4}
|
|
83
|
+
# 'failures': {'HTTP 404': 9, 'TimeoutError': 4},
|
|
84
|
+
# 'latency': {'p50': 0.21, 'p95': 0.73, ...}, ...}
|
|
76
85
|
|
|
77
86
|
for error in results.errors():
|
|
78
87
|
print(error["url"], error["error"] or error["status"])
|
|
@@ -89,9 +98,13 @@ for error in results.errors():
|
|
|
89
98
|
- [Writing results to a file](#writing-results-to-a-file)
|
|
90
99
|
- [Writing results to a database](#writing-results-to-a-database)
|
|
91
100
|
- [Structured data: JSON to typed columns](#structured-data-json-to-typed-columns)
|
|
101
|
+
- [Pagination](#pagination)
|
|
102
|
+
- [Requests from a CSV file or a table](#requests-from-a-csv-file-or-a-table)
|
|
103
|
+
- [Tokens, proxies and caching](#tokens-proxies-and-caching)
|
|
92
104
|
- [Requests, headers and bodies](#requests-headers-and-bodies)
|
|
93
105
|
- [Streaming results](#streaming-results)
|
|
94
106
|
- [TLS and sessions](#tls-and-sessions)
|
|
107
|
+
- [Command line](#command-line)
|
|
95
108
|
- [When to use something else](#when-to-use-something-else)
|
|
96
109
|
- [Coming from reqt](#coming-from-reqt)
|
|
97
110
|
- [Development](#development)
|
|
@@ -112,15 +125,19 @@ reqstorm does all of that for you, with one call.
|
|
|
112
125
|
| Feature | What you get |
|
|
113
126
|
|---|---|
|
|
114
127
|
| One result per request | Failures are recorded, never raised; every attempt is kept |
|
|
115
|
-
| Rate limits | Per host, in any unit
|
|
128
|
+
| Rate limits | Per host, in any unit (`"100/min"`), or `"auto"` from the server's 429s and headers |
|
|
116
129
|
| Concurrency | Overall and per host |
|
|
117
|
-
| Retries |
|
|
118
|
-
| Reports |
|
|
130
|
+
| Retries | Exponential backoff with a cap and jitter, `Retry-After`; plus end-of-run rounds |
|
|
131
|
+
| Reports | Failures by reason, p50/p95/p99 response times, per-host figures |
|
|
132
|
+
| Pagination | Next links, `Link` headers, cursors and page numbers |
|
|
133
|
+
| Requests from data | URL templates over CSV rows or SQL query results |
|
|
134
|
+
| Tokens and proxies | Refresh an expired token on 401; rotate through HTTP and SOCKS proxies |
|
|
135
|
+
| Caching | ETag / `If-None-Match`: unchanged resources cost a 304 |
|
|
119
136
|
| Output | JSONL, CSV, SQLite, PostgreSQL, MySQL; ordered or as completed |
|
|
120
137
|
| Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
|
|
121
138
|
| Resume | Skip what already succeeded after an interruption |
|
|
122
139
|
| Planning | `estimate()` before you start, progress with ETA while running |
|
|
123
|
-
| API | Blocking (scripts, Jupyter) and
|
|
140
|
+
| API | Blocking (scripts, Jupyter), asyncio and a `reqstorm` command |
|
|
124
141
|
| Safety | TLS verified, timeouts on, bounded concurrency, fully typed |
|
|
125
142
|
|
|
126
143
|
## Installation
|
|
@@ -129,7 +146,7 @@ reqstorm does all of that for you, with one call.
|
|
|
129
146
|
$ python -m pip install reqstorm
|
|
130
147
|
```
|
|
131
148
|
|
|
132
|
-
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`).
|
|
149
|
+
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`; for SOCKS proxies, `reqstorm[socks]`.
|
|
133
150
|
|
|
134
151
|
## Quick start
|
|
135
152
|
|
|
@@ -205,6 +222,17 @@ results.failed # failed results
|
|
|
205
222
|
results.summary() # counts, failures grouped by reason
|
|
206
223
|
results.errors() # failed requests as plain dicts
|
|
207
224
|
results.to_dicts() # every result as a dict
|
|
225
|
+
results.report() # response times, statuses, hosts
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
```python
|
|
229
|
+
>>> results.report()["latency"]
|
|
230
|
+
{'min': 0.081, 'p50': 0.214, 'p90': 0.502,
|
|
231
|
+
'p95': 0.733, 'p99': 1.902, 'max': 10.004,
|
|
232
|
+
'mean': 0.297}
|
|
233
|
+
>>> results.report()["hosts"]["api.example.com:443"]
|
|
234
|
+
{'requests': 7000, 'ok': 6987, 'failed': 13,
|
|
235
|
+
'latency': {...}}
|
|
208
236
|
```
|
|
209
237
|
|
|
210
238
|
`errors()` returns plain data, ready for a log file or a DataFrame:
|
|
@@ -255,6 +283,8 @@ results = reqstorm.fetch_all_sync(
|
|
|
255
283
|
|
|
256
284
|
The limit applies to each host (`host:port`) separately and counts retries too, so requests to different APIs never slow each other down. Requests to one host are spaced evenly.
|
|
257
285
|
|
|
286
|
+
**Don't know the limit?** `rate_limit="auto"` learns it from the server. A `429 Too Many Requests` pauses the host for `Retry-After` and slows it down; `X-RateLimit-Remaining` and `X-RateLimit-Reset` spread the remaining requests over the window; the rate recovers when the server stops pushing back. 429 responses are retried without using up `retries`.
|
|
287
|
+
|
|
258
288
|
**Plan before you send.** `estimate` predicts the duration and names the bottleneck, without sending anything:
|
|
259
289
|
|
|
260
290
|
```python
|
|
@@ -280,13 +310,16 @@ results = reqstorm.fetch_all_sync(
|
|
|
280
310
|
timeout=10, # seconds per attempt
|
|
281
311
|
retries=3, # retry right away...
|
|
282
312
|
backoff=0.5, # ...0.5 s, 1 s, 2 s apart
|
|
313
|
+
max_backoff=30, # never wait longer
|
|
283
314
|
retry_rounds=2, # then resend what still
|
|
284
315
|
retry_round_delay=30, # failed, 30 s later
|
|
285
316
|
)
|
|
286
317
|
```
|
|
287
318
|
|
|
288
319
|
- **`timeout`** is the time allowed for one attempt, including reading the body (default 30 s; `None` disables it). A slow server only fails its own requests.
|
|
289
|
-
- **`retries`** retries a request right away. The
|
|
320
|
+
- **`retries`** retries a request right away, also when no response came back at all (timeouts, dropped connections). The wait starts at `backoff` and doubles each time (0.5, 1, 2, 4, ... s), up to **`max_backoff`** (30 s by default), so many retries never wait minutes.
|
|
321
|
+
- **Jitter** (on by default) waits a random time between half and all of that delay, so thousands of requests that failed together do not retry at the same moment. `jitter=False` waits exactly.
|
|
322
|
+
- A **`Retry-After`** header from the server (up to 60 s) is followed as given instead.
|
|
290
323
|
- **`retry_rounds`** holds back requests that still failed for a retryable reason and sends them again after the rest of the batch, for temporary outages. Each request still produces exactly one result, with all its attempts in `history`.
|
|
291
324
|
|
|
292
325
|
| Retried | Not retried |
|
|
@@ -395,8 +428,87 @@ print(summary.rows, summary.rejected)
|
|
|
395
428
|
|
|
396
429
|
The same schema works for `.db`, `.jsonl` and `.csv` files, and `reqstorm.extract(results, schema)` returns the rows as Python lists. `reqstorm.infer_schema(samples)` drafts a schema from a few responses for you to review.
|
|
397
430
|
|
|
431
|
+
**Already have a Pydantic model?** Pass it as the schema; its fields become columns and Pydantic validates each record:
|
|
432
|
+
|
|
433
|
+
```python
|
|
434
|
+
class Product(BaseModel):
|
|
435
|
+
id: int = Field(
|
|
436
|
+
json_schema_extra={"key": True})
|
|
437
|
+
name: str
|
|
438
|
+
price: Optional[float] = Field(
|
|
439
|
+
None,
|
|
440
|
+
json_schema_extra={"path": "pricing.amount"})
|
|
441
|
+
|
|
442
|
+
reqstorm.fetch_to_file_sync(
|
|
443
|
+
urls, "shop.db", schema=Product, explode="items"
|
|
444
|
+
)
|
|
445
|
+
```
|
|
446
|
+
|
|
398
447
|
More in the [structured data guide](https://reqstorm.github.io/guide/structured-data/).
|
|
399
448
|
|
|
449
|
+
## Pagination
|
|
450
|
+
|
|
451
|
+
`paginate=` follows each starting URL through all of its pages, concurrently with the other URLs and under the same rate limits:
|
|
452
|
+
|
|
453
|
+
```python
|
|
454
|
+
results = reqstorm.fetch_all_sync(
|
|
455
|
+
["https://api.example.com/products"],
|
|
456
|
+
paginate=reqstorm.NextLink("links.next"),
|
|
457
|
+
)
|
|
458
|
+
```
|
|
459
|
+
|
|
460
|
+
| Strategy | Next page comes from |
|
|
461
|
+
|---|---|
|
|
462
|
+
| `NextLink("links.next")` | a URL in the JSON body |
|
|
463
|
+
| `LinkHeader()` | the `Link` header (GitHub style) |
|
|
464
|
+
| `Cursor("meta.next", param="cursor")` | a cursor in the body |
|
|
465
|
+
| `PageNumber("page", items="data")` | `?page=2, 3, ...` until empty |
|
|
466
|
+
|
|
467
|
+
Each result has `page` and `seed_index` (its starting URL). Combined with a schema and `explode`, every page of a catalogue becomes typed rows in one call. `max_pages` (1000 by default) stops an API that never ends.
|
|
468
|
+
|
|
469
|
+
## Requests from a CSV file or a table
|
|
470
|
+
|
|
471
|
+
`from_template` makes one request per row. Values are percent-encoded, and rows are read lazily:
|
|
472
|
+
|
|
473
|
+
```python
|
|
474
|
+
rows = reqstorm.read_csv("users.csv")
|
|
475
|
+
requests = reqstorm.from_template(
|
|
476
|
+
"https://api.example.com/users/{id}",
|
|
477
|
+
rows,
|
|
478
|
+
params={"country": "{country}"},
|
|
479
|
+
)
|
|
480
|
+
reqstorm.fetch_to_file_sync(requests, "users.jsonl")
|
|
481
|
+
```
|
|
482
|
+
|
|
483
|
+
`read_sql(connection, query)` reads rows from any database connection instead, and `json=` builds a request body per row.
|
|
484
|
+
|
|
485
|
+
## Tokens, proxies and caching
|
|
486
|
+
|
|
487
|
+
**Tokens that expire.** `BearerAuth` gets a new token when a response is 401 and sends the request again. Concurrent 401s share one refresh:
|
|
488
|
+
|
|
489
|
+
```python
|
|
490
|
+
auth = reqstorm.BearerAuth(refresh=get_token)
|
|
491
|
+
reqstorm.fetch_all_sync(urls, auth=auth)
|
|
492
|
+
```
|
|
493
|
+
|
|
494
|
+
**Proxies.** One proxy, a pool used in turn, or one per `Request`. HTTP and SOCKS proxies can be mixed:
|
|
495
|
+
|
|
496
|
+
```python
|
|
497
|
+
reqstorm.fetch_all_sync(urls, proxy=[
|
|
498
|
+
"http://proxy-1.example.com:8080",
|
|
499
|
+
"socks5h://user:pass@proxy-2.example.com:1080",
|
|
500
|
+
])
|
|
501
|
+
```
|
|
502
|
+
|
|
503
|
+
SOCKS4 and SOCKS5 (`socks5h://` and `socks4a://` let the proxy resolve host names, as with Tor) need `pip install "reqstorm[socks]"`.
|
|
504
|
+
|
|
505
|
+
**Caching.** `Cache` keeps responses in an SQLite file. The next run asks the server with `If-None-Match`; an unchanged resource comes back as a `304` with no body, and the stored response is used. With `ttl`, recent responses skip the network entirely:
|
|
506
|
+
|
|
507
|
+
```python
|
|
508
|
+
with reqstorm.Cache("responses.sqlite") as cache:
|
|
509
|
+
reqstorm.fetch_all_sync(urls, cache=cache)
|
|
510
|
+
```
|
|
511
|
+
|
|
400
512
|
## Requests, headers and bodies
|
|
401
513
|
|
|
402
514
|
Options given to `fetch_all` apply to every request; `reqstorm.Request` varies them per request:
|
|
@@ -452,6 +564,24 @@ results = reqstorm.fetch_all_sync(urls, ssl=context)
|
|
|
452
564
|
|
|
453
565
|
In async code, `session=` takes an existing `aiohttp.ClientSession` to share cookies, connection pools or proxy settings. reqstorm does not close it.
|
|
454
566
|
|
|
567
|
+
## Command line
|
|
568
|
+
|
|
569
|
+
The `reqstorm` command runs a batch without any Python code:
|
|
570
|
+
|
|
571
|
+
```console
|
|
572
|
+
$ reqstorm urls.txt -o results.jsonl \
|
|
573
|
+
--rate 100/min --retries 2
|
|
574
|
+
$ reqstorm urls.txt --estimate --rate 100/min
|
|
575
|
+
$ cat urls.txt | reqstorm --rate auto -q > out.jsonl
|
|
576
|
+
$ reqstorm users.csv -o users.db \
|
|
577
|
+
--template "https://api.example.com/users/{id}"
|
|
578
|
+
$ reqstorm urls.txt -o shop.db --report \
|
|
579
|
+
--paginate next:links.next \
|
|
580
|
+
--schema products.json --explode items
|
|
581
|
+
```
|
|
582
|
+
|
|
583
|
+
`reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
|
|
584
|
+
|
|
455
585
|
## When to use something else
|
|
456
586
|
|
|
457
587
|
reqstorm is built for batches. For other jobs, these are better fits:
|
|
@@ -35,7 +35,8 @@ results = reqstorm.fetch_all_sync(
|
|
|
35
35
|
|
|
36
36
|
print(results.summary())
|
|
37
37
|
# {'total': 7000, 'ok': 6987, 'failed': 13,
|
|
38
|
-
# 'failures': {'HTTP 404': 9, 'TimeoutError': 4}
|
|
38
|
+
# 'failures': {'HTTP 404': 9, 'TimeoutError': 4},
|
|
39
|
+
# 'latency': {'p50': 0.21, 'p95': 0.73, ...}, ...}
|
|
39
40
|
|
|
40
41
|
for error in results.errors():
|
|
41
42
|
print(error["url"], error["error"] or error["status"])
|
|
@@ -52,9 +53,13 @@ for error in results.errors():
|
|
|
52
53
|
- [Writing results to a file](#writing-results-to-a-file)
|
|
53
54
|
- [Writing results to a database](#writing-results-to-a-database)
|
|
54
55
|
- [Structured data: JSON to typed columns](#structured-data-json-to-typed-columns)
|
|
56
|
+
- [Pagination](#pagination)
|
|
57
|
+
- [Requests from a CSV file or a table](#requests-from-a-csv-file-or-a-table)
|
|
58
|
+
- [Tokens, proxies and caching](#tokens-proxies-and-caching)
|
|
55
59
|
- [Requests, headers and bodies](#requests-headers-and-bodies)
|
|
56
60
|
- [Streaming results](#streaming-results)
|
|
57
61
|
- [TLS and sessions](#tls-and-sessions)
|
|
62
|
+
- [Command line](#command-line)
|
|
58
63
|
- [When to use something else](#when-to-use-something-else)
|
|
59
64
|
- [Coming from reqt](#coming-from-reqt)
|
|
60
65
|
- [Development](#development)
|
|
@@ -75,15 +80,19 @@ reqstorm does all of that for you, with one call.
|
|
|
75
80
|
| Feature | What you get |
|
|
76
81
|
|---|---|
|
|
77
82
|
| One result per request | Failures are recorded, never raised; every attempt is kept |
|
|
78
|
-
| Rate limits | Per host, in any unit
|
|
83
|
+
| Rate limits | Per host, in any unit (`"100/min"`), or `"auto"` from the server's 429s and headers |
|
|
79
84
|
| Concurrency | Overall and per host |
|
|
80
|
-
| Retries |
|
|
81
|
-
| Reports |
|
|
85
|
+
| Retries | Exponential backoff with a cap and jitter, `Retry-After`; plus end-of-run rounds |
|
|
86
|
+
| Reports | Failures by reason, p50/p95/p99 response times, per-host figures |
|
|
87
|
+
| Pagination | Next links, `Link` headers, cursors and page numbers |
|
|
88
|
+
| Requests from data | URL templates over CSV rows or SQL query results |
|
|
89
|
+
| Tokens and proxies | Refresh an expired token on 401; rotate through HTTP and SOCKS proxies |
|
|
90
|
+
| Caching | ETag / `If-None-Match`: unchanged resources cost a 304 |
|
|
82
91
|
| Output | JSONL, CSV, SQLite, PostgreSQL, MySQL; ordered or as completed |
|
|
83
92
|
| Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
|
|
84
93
|
| Resume | Skip what already succeeded after an interruption |
|
|
85
94
|
| Planning | `estimate()` before you start, progress with ETA while running |
|
|
86
|
-
| API | Blocking (scripts, Jupyter) and
|
|
95
|
+
| API | Blocking (scripts, Jupyter), asyncio and a `reqstorm` command |
|
|
87
96
|
| Safety | TLS verified, timeouts on, bounded concurrency, fully typed |
|
|
88
97
|
|
|
89
98
|
## Installation
|
|
@@ -92,7 +101,7 @@ reqstorm does all of that for you, with one call.
|
|
|
92
101
|
$ python -m pip install reqstorm
|
|
93
102
|
```
|
|
94
103
|
|
|
95
|
-
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`).
|
|
104
|
+
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`; for SOCKS proxies, `reqstorm[socks]`.
|
|
96
105
|
|
|
97
106
|
## Quick start
|
|
98
107
|
|
|
@@ -168,6 +177,17 @@ results.failed # failed results
|
|
|
168
177
|
results.summary() # counts, failures grouped by reason
|
|
169
178
|
results.errors() # failed requests as plain dicts
|
|
170
179
|
results.to_dicts() # every result as a dict
|
|
180
|
+
results.report() # response times, statuses, hosts
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
```python
|
|
184
|
+
>>> results.report()["latency"]
|
|
185
|
+
{'min': 0.081, 'p50': 0.214, 'p90': 0.502,
|
|
186
|
+
'p95': 0.733, 'p99': 1.902, 'max': 10.004,
|
|
187
|
+
'mean': 0.297}
|
|
188
|
+
>>> results.report()["hosts"]["api.example.com:443"]
|
|
189
|
+
{'requests': 7000, 'ok': 6987, 'failed': 13,
|
|
190
|
+
'latency': {...}}
|
|
171
191
|
```
|
|
172
192
|
|
|
173
193
|
`errors()` returns plain data, ready for a log file or a DataFrame:
|
|
@@ -218,6 +238,8 @@ results = reqstorm.fetch_all_sync(
|
|
|
218
238
|
|
|
219
239
|
The limit applies to each host (`host:port`) separately and counts retries too, so requests to different APIs never slow each other down. Requests to one host are spaced evenly.
|
|
220
240
|
|
|
241
|
+
**Don't know the limit?** `rate_limit="auto"` learns it from the server. A `429 Too Many Requests` pauses the host for `Retry-After` and slows it down; `X-RateLimit-Remaining` and `X-RateLimit-Reset` spread the remaining requests over the window; the rate recovers when the server stops pushing back. 429 responses are retried without using up `retries`.
|
|
242
|
+
|
|
221
243
|
**Plan before you send.** `estimate` predicts the duration and names the bottleneck, without sending anything:
|
|
222
244
|
|
|
223
245
|
```python
|
|
@@ -243,13 +265,16 @@ results = reqstorm.fetch_all_sync(
|
|
|
243
265
|
timeout=10, # seconds per attempt
|
|
244
266
|
retries=3, # retry right away...
|
|
245
267
|
backoff=0.5, # ...0.5 s, 1 s, 2 s apart
|
|
268
|
+
max_backoff=30, # never wait longer
|
|
246
269
|
retry_rounds=2, # then resend what still
|
|
247
270
|
retry_round_delay=30, # failed, 30 s later
|
|
248
271
|
)
|
|
249
272
|
```
|
|
250
273
|
|
|
251
274
|
- **`timeout`** is the time allowed for one attempt, including reading the body (default 30 s; `None` disables it). A slow server only fails its own requests.
|
|
252
|
-
- **`retries`** retries a request right away. The
|
|
275
|
+
- **`retries`** retries a request right away, also when no response came back at all (timeouts, dropped connections). The wait starts at `backoff` and doubles each time (0.5, 1, 2, 4, ... s), up to **`max_backoff`** (30 s by default), so many retries never wait minutes.
|
|
276
|
+
- **Jitter** (on by default) waits a random time between half and all of that delay, so thousands of requests that failed together do not retry at the same moment. `jitter=False` waits exactly.
|
|
277
|
+
- A **`Retry-After`** header from the server (up to 60 s) is followed as given instead.
|
|
253
278
|
- **`retry_rounds`** holds back requests that still failed for a retryable reason and sends them again after the rest of the batch, for temporary outages. Each request still produces exactly one result, with all its attempts in `history`.
|
|
254
279
|
|
|
255
280
|
| Retried | Not retried |
|
|
@@ -358,8 +383,87 @@ print(summary.rows, summary.rejected)
|
|
|
358
383
|
|
|
359
384
|
The same schema works for `.db`, `.jsonl` and `.csv` files, and `reqstorm.extract(results, schema)` returns the rows as Python lists. `reqstorm.infer_schema(samples)` drafts a schema from a few responses for you to review.
|
|
360
385
|
|
|
386
|
+
**Already have a Pydantic model?** Pass it as the schema; its fields become columns and Pydantic validates each record:
|
|
387
|
+
|
|
388
|
+
```python
|
|
389
|
+
class Product(BaseModel):
|
|
390
|
+
id: int = Field(
|
|
391
|
+
json_schema_extra={"key": True})
|
|
392
|
+
name: str
|
|
393
|
+
price: Optional[float] = Field(
|
|
394
|
+
None,
|
|
395
|
+
json_schema_extra={"path": "pricing.amount"})
|
|
396
|
+
|
|
397
|
+
reqstorm.fetch_to_file_sync(
|
|
398
|
+
urls, "shop.db", schema=Product, explode="items"
|
|
399
|
+
)
|
|
400
|
+
```
|
|
401
|
+
|
|
361
402
|
More in the [structured data guide](https://reqstorm.github.io/guide/structured-data/).
|
|
362
403
|
|
|
404
|
+
## Pagination
|
|
405
|
+
|
|
406
|
+
`paginate=` follows each starting URL through all of its pages, concurrently with the other URLs and under the same rate limits:
|
|
407
|
+
|
|
408
|
+
```python
|
|
409
|
+
results = reqstorm.fetch_all_sync(
|
|
410
|
+
["https://api.example.com/products"],
|
|
411
|
+
paginate=reqstorm.NextLink("links.next"),
|
|
412
|
+
)
|
|
413
|
+
```
|
|
414
|
+
|
|
415
|
+
| Strategy | Next page comes from |
|
|
416
|
+
|---|---|
|
|
417
|
+
| `NextLink("links.next")` | a URL in the JSON body |
|
|
418
|
+
| `LinkHeader()` | the `Link` header (GitHub style) |
|
|
419
|
+
| `Cursor("meta.next", param="cursor")` | a cursor in the body |
|
|
420
|
+
| `PageNumber("page", items="data")` | `?page=2, 3, ...` until empty |
|
|
421
|
+
|
|
422
|
+
Each result has `page` and `seed_index` (its starting URL). Combined with a schema and `explode`, every page of a catalogue becomes typed rows in one call. `max_pages` (1000 by default) stops an API that never ends.
|
|
423
|
+
|
|
424
|
+
## Requests from a CSV file or a table
|
|
425
|
+
|
|
426
|
+
`from_template` makes one request per row. Values are percent-encoded, and rows are read lazily:
|
|
427
|
+
|
|
428
|
+
```python
|
|
429
|
+
rows = reqstorm.read_csv("users.csv")
|
|
430
|
+
requests = reqstorm.from_template(
|
|
431
|
+
"https://api.example.com/users/{id}",
|
|
432
|
+
rows,
|
|
433
|
+
params={"country": "{country}"},
|
|
434
|
+
)
|
|
435
|
+
reqstorm.fetch_to_file_sync(requests, "users.jsonl")
|
|
436
|
+
```
|
|
437
|
+
|
|
438
|
+
`read_sql(connection, query)` reads rows from any database connection instead, and `json=` builds a request body per row.
|
|
439
|
+
|
|
440
|
+
## Tokens, proxies and caching
|
|
441
|
+
|
|
442
|
+
**Tokens that expire.** `BearerAuth` gets a new token when a response is 401 and sends the request again. Concurrent 401s share one refresh:
|
|
443
|
+
|
|
444
|
+
```python
|
|
445
|
+
auth = reqstorm.BearerAuth(refresh=get_token)
|
|
446
|
+
reqstorm.fetch_all_sync(urls, auth=auth)
|
|
447
|
+
```
|
|
448
|
+
|
|
449
|
+
**Proxies.** One proxy, a pool used in turn, or one per `Request`. HTTP and SOCKS proxies can be mixed:
|
|
450
|
+
|
|
451
|
+
```python
|
|
452
|
+
reqstorm.fetch_all_sync(urls, proxy=[
|
|
453
|
+
"http://proxy-1.example.com:8080",
|
|
454
|
+
"socks5h://user:pass@proxy-2.example.com:1080",
|
|
455
|
+
])
|
|
456
|
+
```
|
|
457
|
+
|
|
458
|
+
SOCKS4 and SOCKS5 (`socks5h://` and `socks4a://` let the proxy resolve host names, as with Tor) need `pip install "reqstorm[socks]"`.
|
|
459
|
+
|
|
460
|
+
**Caching.** `Cache` keeps responses in an SQLite file. The next run asks the server with `If-None-Match`; an unchanged resource comes back as a `304` with no body, and the stored response is used. With `ttl`, recent responses skip the network entirely:
|
|
461
|
+
|
|
462
|
+
```python
|
|
463
|
+
with reqstorm.Cache("responses.sqlite") as cache:
|
|
464
|
+
reqstorm.fetch_all_sync(urls, cache=cache)
|
|
465
|
+
```
|
|
466
|
+
|
|
363
467
|
## Requests, headers and bodies
|
|
364
468
|
|
|
365
469
|
Options given to `fetch_all` apply to every request; `reqstorm.Request` varies them per request:
|
|
@@ -415,6 +519,24 @@ results = reqstorm.fetch_all_sync(urls, ssl=context)
|
|
|
415
519
|
|
|
416
520
|
In async code, `session=` takes an existing `aiohttp.ClientSession` to share cookies, connection pools or proxy settings. reqstorm does not close it.
|
|
417
521
|
|
|
522
|
+
## Command line
|
|
523
|
+
|
|
524
|
+
The `reqstorm` command runs a batch without any Python code:
|
|
525
|
+
|
|
526
|
+
```console
|
|
527
|
+
$ reqstorm urls.txt -o results.jsonl \
|
|
528
|
+
--rate 100/min --retries 2
|
|
529
|
+
$ reqstorm urls.txt --estimate --rate 100/min
|
|
530
|
+
$ cat urls.txt | reqstorm --rate auto -q > out.jsonl
|
|
531
|
+
$ reqstorm users.csv -o users.db \
|
|
532
|
+
--template "https://api.example.com/users/{id}"
|
|
533
|
+
$ reqstorm urls.txt -o shop.db --report \
|
|
534
|
+
--paginate next:links.next \
|
|
535
|
+
--schema products.json --explode items
|
|
536
|
+
```
|
|
537
|
+
|
|
538
|
+
`reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
|
|
539
|
+
|
|
418
540
|
## When to use something else
|
|
419
541
|
|
|
420
542
|
reqstorm is built for batches. For other jobs, these are better fits:
|
|
@@ -30,8 +30,13 @@ classifiers = [
|
|
|
30
30
|
]
|
|
31
31
|
|
|
32
32
|
[project.optional-dependencies]
|
|
33
|
-
|
|
34
|
-
|
|
33
|
+
pydantic = ["pydantic>=2"]
|
|
34
|
+
socks = ["aiohttp-socks>=0.10"]
|
|
35
|
+
test = ["pytest>=8", "pytest-asyncio>=0.23", "trustme>=1.1", "pydantic>=2", "aiohttp-socks>=0.10"]
|
|
36
|
+
lint = ["ruff>=0.6", "mypy>=1.10", "pydantic>=2", "aiohttp-socks>=0.10"]
|
|
37
|
+
|
|
38
|
+
[project.scripts]
|
|
39
|
+
reqstorm = "reqstorm._cli:main"
|
|
35
40
|
|
|
36
41
|
[project.urls]
|
|
37
42
|
Homepage = "https://reqstorm.github.io"
|
|
@@ -11,6 +11,8 @@ async def main():
|
|
|
11
11
|
asyncio.run(main())
|
|
12
12
|
"""
|
|
13
13
|
|
|
14
|
+
from ._auth import BearerAuth
|
|
15
|
+
from ._cache import Cache
|
|
14
16
|
from ._client import (
|
|
15
17
|
DEFAULT_RETRY_STATUSES,
|
|
16
18
|
Attempt,
|
|
@@ -24,20 +26,29 @@ from ._client import (
|
|
|
24
26
|
from ._files import Summary, fetch_to_db, fetch_to_file
|
|
25
27
|
from ._legacy import Reqt
|
|
26
28
|
from ._limits import parse_rate
|
|
29
|
+
from ._paginate import Cursor, LinkHeader, NextLink, PageNumber, Paginator
|
|
27
30
|
from ._plan import Estimate, estimate
|
|
28
31
|
from ._schema import Field, Schema, SchemaError, extract, infer_schema
|
|
29
32
|
from ._sync import fetch_all_sync, fetch_to_db_sync, fetch_to_file_sync, stream_sync
|
|
33
|
+
from ._template import from_template, read_csv, read_sql
|
|
30
34
|
|
|
31
|
-
__version__ = "2.
|
|
35
|
+
__version__ = "2.3.0"
|
|
32
36
|
|
|
33
37
|
__all__ = [
|
|
34
|
-
"DEFAULT_RETRY_STATUSES",
|
|
35
38
|
"Attempt",
|
|
39
|
+
"BearerAuth",
|
|
40
|
+
"Cache",
|
|
41
|
+
"Cursor",
|
|
42
|
+
"DEFAULT_RETRY_STATUSES",
|
|
36
43
|
"Estimate",
|
|
37
44
|
"Field",
|
|
38
45
|
"HTTPStatusError",
|
|
39
|
-
"
|
|
46
|
+
"LinkHeader",
|
|
47
|
+
"NextLink",
|
|
48
|
+
"PageNumber",
|
|
49
|
+
"Paginator",
|
|
40
50
|
"Reqt",
|
|
51
|
+
"Request",
|
|
41
52
|
"Result",
|
|
42
53
|
"Results",
|
|
43
54
|
"Schema",
|
|
@@ -51,9 +62,12 @@ __all__ = [
|
|
|
51
62
|
"fetch_to_db_sync",
|
|
52
63
|
"fetch_to_file",
|
|
53
64
|
"fetch_to_file_sync",
|
|
65
|
+
"from_template",
|
|
54
66
|
"infer_schema",
|
|
55
|
-
"stream",
|
|
56
67
|
"parse_rate",
|
|
68
|
+
"read_csv",
|
|
69
|
+
"read_sql",
|
|
70
|
+
"stream",
|
|
57
71
|
"stream_sync",
|
|
58
72
|
"__version__",
|
|
59
73
|
]
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""A per-host rate limiter that learns the rate from the server's responses."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import time
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from email.utils import parsedate_to_datetime
|
|
9
|
+
from typing import Dict, Mapping, Optional
|
|
10
|
+
|
|
11
|
+
from ._limits import host_key
|
|
12
|
+
|
|
13
|
+
# Header names, most specific first. "RateLimit-*" is the IETF draft; "X-RateLimit-*" is the
|
|
14
|
+
# common convention (GitHub, Twitter/X, many others).
|
|
15
|
+
_REMAINING = ("RateLimit-Remaining", "X-RateLimit-Remaining", "X-Rate-Limit-Remaining")
|
|
16
|
+
_RESET = ("RateLimit-Reset", "X-RateLimit-Reset", "X-Rate-Limit-Reset")
|
|
17
|
+
_MAX_PAUSE = 300.0
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _header(headers: Mapping[str, str], names: tuple) -> Optional[str]:
|
|
21
|
+
for name in names:
|
|
22
|
+
value = headers.get(name)
|
|
23
|
+
if value is not None:
|
|
24
|
+
return value
|
|
25
|
+
return None
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def seconds_until_reset(headers: Mapping[str, str], now: Optional[float] = None) -> Optional[float]:
|
|
29
|
+
"""Seconds until the server's rate-limit window resets, from ``*-Reset`` or ``Retry-After``.
|
|
30
|
+
|
|
31
|
+
Reset values above 10^9 are Unix timestamps (GitHub style); smaller ones are seconds
|
|
32
|
+
(IETF draft style). ``Retry-After`` may also be an HTTP date.
|
|
33
|
+
"""
|
|
34
|
+
now = time.time() if now is None else now
|
|
35
|
+
for value in (_header(headers, _RESET), headers.get("Retry-After")):
|
|
36
|
+
if value is None:
|
|
37
|
+
continue
|
|
38
|
+
value = value.strip()
|
|
39
|
+
try:
|
|
40
|
+
number = float(value)
|
|
41
|
+
except ValueError:
|
|
42
|
+
try:
|
|
43
|
+
return max(parsedate_to_datetime(value).timestamp() - now, 0.0)
|
|
44
|
+
except (TypeError, ValueError, IndexError, OverflowError):
|
|
45
|
+
continue
|
|
46
|
+
seconds = number - now if number > 1e9 else number
|
|
47
|
+
return min(max(seconds, 0.0), _MAX_PAUSE)
|
|
48
|
+
return None
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass
|
|
52
|
+
class _HostState:
|
|
53
|
+
interval: float = 0.0 # seconds between requests; 0 means no spacing
|
|
54
|
+
next_slot: float = 0.0
|
|
55
|
+
paused_until: float = 0.0
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class AdaptiveRateLimiter:
|
|
59
|
+
"""``rate_limit="auto"``: no limit until the server signals one, then follow it.
|
|
60
|
+
|
|
61
|
+
- A 429 response doubles the spacing between requests to that host and pauses it for
|
|
62
|
+
``Retry-After`` (or until the rate-limit window resets).
|
|
63
|
+
- ``*-RateLimit-Remaining`` / ``*-RateLimit-Reset`` headers spread the remaining requests
|
|
64
|
+
evenly over the rest of the window, and pause the host when none are left.
|
|
65
|
+
- Every successful response without such headers shortens the spacing by 10 %, so the
|
|
66
|
+
rate recovers once the server stops pushing back.
|
|
67
|
+
"""
|
|
68
|
+
|
|
69
|
+
adaptive = True
|
|
70
|
+
|
|
71
|
+
def __init__(self) -> None:
|
|
72
|
+
self._hosts: Dict[str, _HostState] = {}
|
|
73
|
+
|
|
74
|
+
def _state(self, url: str) -> Optional[_HostState]:
|
|
75
|
+
host = host_key(url)
|
|
76
|
+
if host is None:
|
|
77
|
+
return None
|
|
78
|
+
return self._hosts.setdefault(host, _HostState())
|
|
79
|
+
|
|
80
|
+
def interval(self, url: str) -> float:
|
|
81
|
+
state = self._state(url)
|
|
82
|
+
return state.interval if state else 0.0
|
|
83
|
+
|
|
84
|
+
async def wait(self, url: str) -> None:
|
|
85
|
+
state = self._state(url)
|
|
86
|
+
if state is None:
|
|
87
|
+
return
|
|
88
|
+
now = time.monotonic()
|
|
89
|
+
slot = max(now, state.next_slot, state.paused_until)
|
|
90
|
+
state.next_slot = slot + state.interval
|
|
91
|
+
if slot > now:
|
|
92
|
+
await asyncio.sleep(slot - now)
|
|
93
|
+
|
|
94
|
+
def record(self, url: str, status: Optional[int], headers: Mapping[str, str]) -> None:
|
|
95
|
+
state = self._state(url)
|
|
96
|
+
if state is None or status is None:
|
|
97
|
+
return
|
|
98
|
+
now = time.monotonic()
|
|
99
|
+
reset = seconds_until_reset(headers)
|
|
100
|
+
if status == 429:
|
|
101
|
+
state.interval = min(max(state.interval * 2, 0.1), 60.0)
|
|
102
|
+
pause = reset if reset is not None else max(state.interval, 1.0)
|
|
103
|
+
state.paused_until = max(state.paused_until, now + pause)
|
|
104
|
+
return
|
|
105
|
+
remaining_text = _header(headers, _REMAINING)
|
|
106
|
+
if remaining_text is not None and reset is not None:
|
|
107
|
+
try:
|
|
108
|
+
remaining = float(remaining_text)
|
|
109
|
+
except ValueError:
|
|
110
|
+
remaining = None
|
|
111
|
+
if remaining is not None:
|
|
112
|
+
if remaining <= 0:
|
|
113
|
+
state.paused_until = max(state.paused_until, now + reset)
|
|
114
|
+
else:
|
|
115
|
+
state.interval = min(reset / remaining, 60.0)
|
|
116
|
+
return
|
|
117
|
+
if 200 <= status < 400 and state.interval:
|
|
118
|
+
state.interval = state.interval * 0.9 if state.interval > 0.01 else 0.0
|