reqstorm 2.1.0__tar.gz → 2.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. {reqstorm-2.1.0/reqstorm.egg-info → reqstorm-2.2.0}/PKG-INFO +127 -6
  2. {reqstorm-2.1.0 → reqstorm-2.2.0}/README.md +122 -5
  3. {reqstorm-2.1.0 → reqstorm-2.2.0}/pyproject.toml +6 -2
  4. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm/__init__.py +18 -4
  5. reqstorm-2.2.0/reqstorm/__main__.py +5 -0
  6. reqstorm-2.2.0/reqstorm/_adaptive.py +118 -0
  7. reqstorm-2.2.0/reqstorm/_auth.py +79 -0
  8. reqstorm-2.2.0/reqstorm/_cache.py +141 -0
  9. reqstorm-2.2.0/reqstorm/_cli.py +317 -0
  10. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm/_client.py +207 -27
  11. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm/_files.py +19 -4
  12. reqstorm-2.2.0/reqstorm/_paginate.py +144 -0
  13. reqstorm-2.2.0/reqstorm/_pydantic.py +103 -0
  14. reqstorm-2.2.0/reqstorm/_report.py +91 -0
  15. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm/_schema.py +9 -3
  16. reqstorm-2.2.0/reqstorm/_template.py +117 -0
  17. {reqstorm-2.1.0 → reqstorm-2.2.0/reqstorm.egg-info}/PKG-INFO +127 -6
  18. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm.egg-info/SOURCES.txt +18 -0
  19. reqstorm-2.2.0/reqstorm.egg-info/entry_points.txt +2 -0
  20. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm.egg-info/requires.txt +5 -0
  21. reqstorm-2.2.0/tests/test_adaptive.py +86 -0
  22. reqstorm-2.2.0/tests/test_auth.py +74 -0
  23. reqstorm-2.2.0/tests/test_cache_proxy.py +119 -0
  24. reqstorm-2.2.0/tests/test_cli.py +121 -0
  25. reqstorm-2.2.0/tests/test_paginate.py +113 -0
  26. reqstorm-2.2.0/tests/test_pydantic.py +115 -0
  27. reqstorm-2.2.0/tests/test_run_report.py +52 -0
  28. {reqstorm-2.1.0 → reqstorm-2.2.0}/tests/test_schema_db.py +22 -0
  29. reqstorm-2.2.0/tests/test_template.py +63 -0
  30. {reqstorm-2.1.0 → reqstorm-2.2.0}/LICENSE +0 -0
  31. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm/_legacy.py +0 -0
  32. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm/_limits.py +0 -0
  33. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm/_plan.py +0 -0
  34. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm/_progress.py +0 -0
  35. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm/_schema_sinks.py +0 -0
  36. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm/_sync.py +0 -0
  37. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm/py.typed +0 -0
  38. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm.egg-info/dependency_links.txt +0 -0
  39. {reqstorm-2.1.0 → reqstorm-2.2.0}/reqstorm.egg-info/top_level.txt +0 -0
  40. {reqstorm-2.1.0 → reqstorm-2.2.0}/setup.cfg +0 -0
  41. {reqstorm-2.1.0 → reqstorm-2.2.0}/tests/test_fetch_all.py +0 -0
  42. {reqstorm-2.1.0 → reqstorm-2.2.0}/tests/test_files.py +0 -0
  43. {reqstorm-2.1.0 → reqstorm-2.2.0}/tests/test_legacy.py +0 -0
  44. {reqstorm-2.1.0 → reqstorm-2.2.0}/tests/test_limits.py +0 -0
  45. {reqstorm-2.1.0 → reqstorm-2.2.0}/tests/test_ordered_and_db.py +0 -0
  46. {reqstorm-2.1.0 → reqstorm-2.2.0}/tests/test_report.py +0 -0
  47. {reqstorm-2.1.0 → reqstorm-2.2.0}/tests/test_schema.py +0 -0
  48. {reqstorm-2.1.0 → reqstorm-2.2.0}/tests/test_stream.py +0 -0
  49. {reqstorm-2.1.0 → reqstorm-2.2.0}/tests/test_sync.py +0 -0
  50. {reqstorm-2.1.0 → reqstorm-2.2.0}/tests/test_tls.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: reqstorm
3
- Version: 2.1.0
3
+ Version: 2.2.0
4
4
  Summary: Send large numbers of HTTP requests concurrently with asyncio, with retries, timeouts and per-request results.
5
5
  Author-email: Melih Colpan <colpanmelih@gmail.com>
6
6
  License-Expression: MIT
@@ -26,13 +26,17 @@ Requires-Python: >=3.9
26
26
  Description-Content-Type: text/markdown
27
27
  License-File: LICENSE
28
28
  Requires-Dist: aiohttp<4,>=3.9
29
+ Provides-Extra: pydantic
30
+ Requires-Dist: pydantic>=2; extra == "pydantic"
29
31
  Provides-Extra: test
30
32
  Requires-Dist: pytest>=8; extra == "test"
31
33
  Requires-Dist: pytest-asyncio>=0.23; extra == "test"
32
34
  Requires-Dist: trustme>=1.1; extra == "test"
35
+ Requires-Dist: pydantic>=2; extra == "test"
33
36
  Provides-Extra: lint
34
37
  Requires-Dist: ruff>=0.6; extra == "lint"
35
38
  Requires-Dist: mypy>=1.10; extra == "lint"
39
+ Requires-Dist: pydantic>=2; extra == "lint"
36
40
  Dynamic: license-file
37
41
 
38
42
  # reqstorm
@@ -72,7 +76,8 @@ results = reqstorm.fetch_all_sync(
72
76
 
73
77
  print(results.summary())
74
78
  # {'total': 7000, 'ok': 6987, 'failed': 13,
75
- # 'failures': {'HTTP 404': 9, 'TimeoutError': 4}}
79
+ # 'failures': {'HTTP 404': 9, 'TimeoutError': 4},
80
+ # 'latency': {'p50': 0.21, 'p95': 0.73, ...}, ...}
76
81
 
77
82
  for error in results.errors():
78
83
  print(error["url"], error["error"] or error["status"])
@@ -89,9 +94,13 @@ for error in results.errors():
89
94
  - [Writing results to a file](#writing-results-to-a-file)
90
95
  - [Writing results to a database](#writing-results-to-a-database)
91
96
  - [Structured data: JSON to typed columns](#structured-data-json-to-typed-columns)
97
+ - [Pagination](#pagination)
98
+ - [Requests from a CSV file or a table](#requests-from-a-csv-file-or-a-table)
99
+ - [Tokens, proxies and caching](#tokens-proxies-and-caching)
92
100
  - [Requests, headers and bodies](#requests-headers-and-bodies)
93
101
  - [Streaming results](#streaming-results)
94
102
  - [TLS and sessions](#tls-and-sessions)
103
+ - [Command line](#command-line)
95
104
  - [When to use something else](#when-to-use-something-else)
96
105
  - [Coming from reqt](#coming-from-reqt)
97
106
  - [Development](#development)
@@ -112,15 +121,19 @@ reqstorm does all of that for you, with one call.
112
121
  | Feature | What you get |
113
122
  |---|---|
114
123
  | One result per request | Failures are recorded, never raised; every attempt is kept |
115
- | Rate limits | Per host, in any unit: `"10/s"`, `"100/min"`, `"1000/h"` |
124
+ | Rate limits | Per host, in any unit (`"100/min"`), or `"auto"` from the server's 429s and headers |
116
125
  | Concurrency | Overall and per host |
117
126
  | Retries | Immediate, with backoff and `Retry-After`; plus end-of-run rounds |
118
- | Reports | `summary()`, `errors()`, `to_dicts()` |
127
+ | Reports | Failures by reason, p50/p95/p99 response times, per-host figures |
128
+ | Pagination | Next links, `Link` headers, cursors and page numbers |
129
+ | Requests from data | URL templates over CSV rows or SQL query results |
130
+ | Tokens and proxies | Refresh an expired token on 401; rotate through proxies |
131
+ | Caching | ETag / `If-None-Match`: unchanged resources cost a 304 |
119
132
  | Output | JSONL, CSV, SQLite, PostgreSQL, MySQL; ordered or as completed |
120
133
  | Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
121
134
  | Resume | Skip what already succeeded after an interruption |
122
135
  | Planning | `estimate()` before you start, progress with ETA while running |
123
- | API | Blocking (scripts, Jupyter) and asyncio |
136
+ | API | Blocking (scripts, Jupyter), asyncio and a `reqstorm` command |
124
137
  | Safety | TLS verified, timeouts on, bounded concurrency, fully typed |
125
138
 
126
139
  ## Installation
@@ -129,7 +142,7 @@ reqstorm does all of that for you, with one call.
129
142
  $ python -m pip install reqstorm
130
143
  ```
131
144
 
132
- reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`).
145
+ reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`.
133
146
 
134
147
  ## Quick start
135
148
 
@@ -205,6 +218,17 @@ results.failed # failed results
205
218
  results.summary() # counts, failures grouped by reason
206
219
  results.errors() # failed requests as plain dicts
207
220
  results.to_dicts() # every result as a dict
221
+ results.report() # response times, statuses, hosts
222
+ ```
223
+
224
+ ```python
225
+ >>> results.report()["latency"]
226
+ {'min': 0.081, 'p50': 0.214, 'p90': 0.502,
227
+ 'p95': 0.733, 'p99': 1.902, 'max': 10.004,
228
+ 'mean': 0.297}
229
+ >>> results.report()["hosts"]["api.example.com:443"]
230
+ {'requests': 7000, 'ok': 6987, 'failed': 13,
231
+ 'latency': {...}}
208
232
  ```
209
233
 
210
234
  `errors()` returns plain data, ready for a log file or a DataFrame:
@@ -255,6 +279,8 @@ results = reqstorm.fetch_all_sync(
255
279
 
256
280
  The limit applies to each host (`host:port`) separately and counts retries too, so requests to different APIs never slow each other down. Requests to one host are spaced evenly.
257
281
 
282
+ **Don't know the limit?** `rate_limit="auto"` learns it from the server. A `429 Too Many Requests` pauses the host for `Retry-After` and slows it down; `X-RateLimit-Remaining` and `X-RateLimit-Reset` spread the remaining requests over the window; the rate recovers when the server stops pushing back. 429 responses are retried without using up `retries`.
283
+
258
284
  **Plan before you send.** `estimate` predicts the duration and names the bottleneck, without sending anything:
259
285
 
260
286
  ```python
@@ -395,8 +421,85 @@ print(summary.rows, summary.rejected)
395
421
 
396
422
  The same schema works for `.db`, `.jsonl` and `.csv` files, and `reqstorm.extract(results, schema)` returns the rows as Python lists. `reqstorm.infer_schema(samples)` drafts a schema from a few responses for you to review.
397
423
 
424
+ **Already have a Pydantic model?** Pass it as the schema; its fields become columns and Pydantic validates each record:
425
+
426
+ ```python
427
+ class Product(BaseModel):
428
+ id: int = Field(
429
+ json_schema_extra={"key": True})
430
+ name: str
431
+ price: Optional[float] = Field(
432
+ None,
433
+ json_schema_extra={"path": "pricing.amount"})
434
+
435
+ reqstorm.fetch_to_file_sync(
436
+ urls, "shop.db", schema=Product, explode="items"
437
+ )
438
+ ```
439
+
398
440
  More in the [structured data guide](https://reqstorm.github.io/guide/structured-data/).
399
441
 
442
+ ## Pagination
443
+
444
+ `paginate=` follows each starting URL through all of its pages, concurrently with the other URLs and under the same rate limits:
445
+
446
+ ```python
447
+ results = reqstorm.fetch_all_sync(
448
+ ["https://api.example.com/products"],
449
+ paginate=reqstorm.NextLink("links.next"),
450
+ )
451
+ ```
452
+
453
+ | Strategy | Next page comes from |
454
+ |---|---|
455
+ | `NextLink("links.next")` | a URL in the JSON body |
456
+ | `LinkHeader()` | the `Link` header (GitHub style) |
457
+ | `Cursor("meta.next", param="cursor")` | a cursor in the body |
458
+ | `PageNumber("page", items="data")` | `?page=2, 3, ...` until empty |
459
+
460
+ Each result has `page` and `seed_index` (its starting URL). Combined with a schema and `explode`, every page of a catalogue becomes typed rows in one call. `max_pages` (1000 by default) stops an API that never ends.
461
+
462
+ ## Requests from a CSV file or a table
463
+
464
+ `from_template` makes one request per row. Values are percent-encoded, and rows are read lazily:
465
+
466
+ ```python
467
+ rows = reqstorm.read_csv("users.csv")
468
+ requests = reqstorm.from_template(
469
+ "https://api.example.com/users/{id}",
470
+ rows,
471
+ params={"country": "{country}"},
472
+ )
473
+ reqstorm.fetch_to_file_sync(requests, "users.jsonl")
474
+ ```
475
+
476
+ `read_sql(connection, query)` reads rows from any database connection instead, and `json=` builds a request body per row.
477
+
478
+ ## Tokens, proxies and caching
479
+
480
+ **Tokens that expire.** `BearerAuth` gets a new token when a response is 401 and sends the request again. Concurrent 401s share one refresh:
481
+
482
+ ```python
483
+ auth = reqstorm.BearerAuth(refresh=get_token)
484
+ reqstorm.fetch_all_sync(urls, auth=auth)
485
+ ```
486
+
487
+ **Proxies.** One proxy, a pool used in turn, or one per `Request`:
488
+
489
+ ```python
490
+ reqstorm.fetch_all_sync(urls, proxy=[
491
+ "http://proxy-1.example.com:8080",
492
+ "http://proxy-2.example.com:8080",
493
+ ])
494
+ ```
495
+
496
+ **Caching.** `Cache` keeps responses in an SQLite file. The next run asks the server with `If-None-Match`; an unchanged resource comes back as a `304` with no body, and the stored response is used. With `ttl`, recent responses skip the network entirely:
497
+
498
+ ```python
499
+ with reqstorm.Cache("responses.sqlite") as cache:
500
+ reqstorm.fetch_all_sync(urls, cache=cache)
501
+ ```
502
+
400
503
  ## Requests, headers and bodies
401
504
 
402
505
  Options given to `fetch_all` apply to every request; `reqstorm.Request` varies them per request:
@@ -452,6 +555,24 @@ results = reqstorm.fetch_all_sync(urls, ssl=context)
452
555
 
453
556
  In async code, `session=` takes an existing `aiohttp.ClientSession` to share cookies, connection pools or proxy settings. reqstorm does not close it.
454
557
 
558
+ ## Command line
559
+
560
+ The `reqstorm` command runs a batch without any Python code:
561
+
562
+ ```console
563
+ $ reqstorm urls.txt -o results.jsonl \
564
+ --rate 100/min --retries 2
565
+ $ reqstorm urls.txt --estimate --rate 100/min
566
+ $ cat urls.txt | reqstorm --rate auto -q > out.jsonl
567
+ $ reqstorm users.csv -o users.db \
568
+ --template "https://api.example.com/users/{id}"
569
+ $ reqstorm urls.txt -o shop.db --report \
570
+ --paginate next:links.next \
571
+ --schema products.json --explode items
572
+ ```
573
+
574
+ `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
575
+
455
576
  ## When to use something else
456
577
 
457
578
  reqstorm is built for batches. For other jobs, these are better fits:
@@ -35,7 +35,8 @@ results = reqstorm.fetch_all_sync(
35
35
 
36
36
  print(results.summary())
37
37
  # {'total': 7000, 'ok': 6987, 'failed': 13,
38
- # 'failures': {'HTTP 404': 9, 'TimeoutError': 4}}
38
+ # 'failures': {'HTTP 404': 9, 'TimeoutError': 4},
39
+ # 'latency': {'p50': 0.21, 'p95': 0.73, ...}, ...}
39
40
 
40
41
  for error in results.errors():
41
42
  print(error["url"], error["error"] or error["status"])
@@ -52,9 +53,13 @@ for error in results.errors():
52
53
  - [Writing results to a file](#writing-results-to-a-file)
53
54
  - [Writing results to a database](#writing-results-to-a-database)
54
55
  - [Structured data: JSON to typed columns](#structured-data-json-to-typed-columns)
56
+ - [Pagination](#pagination)
57
+ - [Requests from a CSV file or a table](#requests-from-a-csv-file-or-a-table)
58
+ - [Tokens, proxies and caching](#tokens-proxies-and-caching)
55
59
  - [Requests, headers and bodies](#requests-headers-and-bodies)
56
60
  - [Streaming results](#streaming-results)
57
61
  - [TLS and sessions](#tls-and-sessions)
62
+ - [Command line](#command-line)
58
63
  - [When to use something else](#when-to-use-something-else)
59
64
  - [Coming from reqt](#coming-from-reqt)
60
65
  - [Development](#development)
@@ -75,15 +80,19 @@ reqstorm does all of that for you, with one call.
75
80
  | Feature | What you get |
76
81
  |---|---|
77
82
  | One result per request | Failures are recorded, never raised; every attempt is kept |
78
- | Rate limits | Per host, in any unit: `"10/s"`, `"100/min"`, `"1000/h"` |
83
+ | Rate limits | Per host, in any unit (`"100/min"`), or `"auto"` from the server's 429s and headers |
79
84
  | Concurrency | Overall and per host |
80
85
  | Retries | Immediate, with backoff and `Retry-After`; plus end-of-run rounds |
81
- | Reports | `summary()`, `errors()`, `to_dicts()` |
86
+ | Reports | Failures by reason, p50/p95/p99 response times, per-host figures |
87
+ | Pagination | Next links, `Link` headers, cursors and page numbers |
88
+ | Requests from data | URL templates over CSV rows or SQL query results |
89
+ | Tokens and proxies | Refresh an expired token on 401; rotate through proxies |
90
+ | Caching | ETag / `If-None-Match`: unchanged resources cost a 304 |
82
91
  | Output | JSONL, CSV, SQLite, PostgreSQL, MySQL; ordered or as completed |
83
92
  | Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
84
93
  | Resume | Skip what already succeeded after an interruption |
85
94
  | Planning | `estimate()` before you start, progress with ETA while running |
86
- | API | Blocking (scripts, Jupyter) and asyncio |
95
+ | API | Blocking (scripts, Jupyter), asyncio and a `reqstorm` command |
87
96
  | Safety | TLS verified, timeouts on, bounded concurrency, fully typed |
88
97
 
89
98
  ## Installation
@@ -92,7 +101,7 @@ reqstorm does all of that for you, with one call.
92
101
  $ python -m pip install reqstorm
93
102
  ```
94
103
 
95
- reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`).
104
+ reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`.
96
105
 
97
106
  ## Quick start
98
107
 
@@ -168,6 +177,17 @@ results.failed # failed results
168
177
  results.summary() # counts, failures grouped by reason
169
178
  results.errors() # failed requests as plain dicts
170
179
  results.to_dicts() # every result as a dict
180
+ results.report() # response times, statuses, hosts
181
+ ```
182
+
183
+ ```python
184
+ >>> results.report()["latency"]
185
+ {'min': 0.081, 'p50': 0.214, 'p90': 0.502,
186
+ 'p95': 0.733, 'p99': 1.902, 'max': 10.004,
187
+ 'mean': 0.297}
188
+ >>> results.report()["hosts"]["api.example.com:443"]
189
+ {'requests': 7000, 'ok': 6987, 'failed': 13,
190
+ 'latency': {...}}
171
191
  ```
172
192
 
173
193
  `errors()` returns plain data, ready for a log file or a DataFrame:
@@ -218,6 +238,8 @@ results = reqstorm.fetch_all_sync(
218
238
 
219
239
  The limit applies to each host (`host:port`) separately and counts retries too, so requests to different APIs never slow each other down. Requests to one host are spaced evenly.
220
240
 
241
+ **Don't know the limit?** `rate_limit="auto"` learns it from the server. A `429 Too Many Requests` pauses the host for `Retry-After` and slows it down; `X-RateLimit-Remaining` and `X-RateLimit-Reset` spread the remaining requests over the window; the rate recovers when the server stops pushing back. 429 responses are retried without using up `retries`.
242
+
221
243
  **Plan before you send.** `estimate` predicts the duration and names the bottleneck, without sending anything:
222
244
 
223
245
  ```python
@@ -358,8 +380,85 @@ print(summary.rows, summary.rejected)
358
380
 
359
381
  The same schema works for `.db`, `.jsonl` and `.csv` files, and `reqstorm.extract(results, schema)` returns the rows as Python lists. `reqstorm.infer_schema(samples)` drafts a schema from a few responses for you to review.
360
382
 
383
+ **Already have a Pydantic model?** Pass it as the schema; its fields become columns and Pydantic validates each record:
384
+
385
+ ```python
386
+ class Product(BaseModel):
387
+ id: int = Field(
388
+ json_schema_extra={"key": True})
389
+ name: str
390
+ price: Optional[float] = Field(
391
+ None,
392
+ json_schema_extra={"path": "pricing.amount"})
393
+
394
+ reqstorm.fetch_to_file_sync(
395
+ urls, "shop.db", schema=Product, explode="items"
396
+ )
397
+ ```
398
+
361
399
  More in the [structured data guide](https://reqstorm.github.io/guide/structured-data/).
362
400
 
401
+ ## Pagination
402
+
403
+ `paginate=` follows each starting URL through all of its pages, concurrently with the other URLs and under the same rate limits:
404
+
405
+ ```python
406
+ results = reqstorm.fetch_all_sync(
407
+ ["https://api.example.com/products"],
408
+ paginate=reqstorm.NextLink("links.next"),
409
+ )
410
+ ```
411
+
412
+ | Strategy | Next page comes from |
413
+ |---|---|
414
+ | `NextLink("links.next")` | a URL in the JSON body |
415
+ | `LinkHeader()` | the `Link` header (GitHub style) |
416
+ | `Cursor("meta.next", param="cursor")` | a cursor in the body |
417
+ | `PageNumber("page", items="data")` | `?page=2, 3, ...` until empty |
418
+
419
+ Each result has `page` and `seed_index` (its starting URL). Combined with a schema and `explode`, every page of a catalogue becomes typed rows in one call. `max_pages` (1000 by default) stops an API that never ends.
420
+
421
+ ## Requests from a CSV file or a table
422
+
423
+ `from_template` makes one request per row. Values are percent-encoded, and rows are read lazily:
424
+
425
+ ```python
426
+ rows = reqstorm.read_csv("users.csv")
427
+ requests = reqstorm.from_template(
428
+ "https://api.example.com/users/{id}",
429
+ rows,
430
+ params={"country": "{country}"},
431
+ )
432
+ reqstorm.fetch_to_file_sync(requests, "users.jsonl")
433
+ ```
434
+
435
+ `read_sql(connection, query)` reads rows from any database connection instead, and `json=` builds a request body per row.
436
+
437
+ ## Tokens, proxies and caching
438
+
439
+ **Tokens that expire.** `BearerAuth` gets a new token when a response is 401 and sends the request again. Concurrent 401s share one refresh:
440
+
441
+ ```python
442
+ auth = reqstorm.BearerAuth(refresh=get_token)
443
+ reqstorm.fetch_all_sync(urls, auth=auth)
444
+ ```
445
+
446
+ **Proxies.** One proxy, a pool used in turn, or one per `Request`:
447
+
448
+ ```python
449
+ reqstorm.fetch_all_sync(urls, proxy=[
450
+ "http://proxy-1.example.com:8080",
451
+ "http://proxy-2.example.com:8080",
452
+ ])
453
+ ```
454
+
455
+ **Caching.** `Cache` keeps responses in an SQLite file. The next run asks the server with `If-None-Match`; an unchanged resource comes back as a `304` with no body, and the stored response is used. With `ttl`, recent responses skip the network entirely:
456
+
457
+ ```python
458
+ with reqstorm.Cache("responses.sqlite") as cache:
459
+ reqstorm.fetch_all_sync(urls, cache=cache)
460
+ ```
461
+
363
462
  ## Requests, headers and bodies
364
463
 
365
464
  Options given to `fetch_all` apply to every request; `reqstorm.Request` varies them per request:
@@ -415,6 +514,24 @@ results = reqstorm.fetch_all_sync(urls, ssl=context)
415
514
 
416
515
  In async code, `session=` takes an existing `aiohttp.ClientSession` to share cookies, connection pools or proxy settings. reqstorm does not close it.
417
516
 
517
+ ## Command line
518
+
519
+ The `reqstorm` command runs a batch without any Python code:
520
+
521
+ ```console
522
+ $ reqstorm urls.txt -o results.jsonl \
523
+ --rate 100/min --retries 2
524
+ $ reqstorm urls.txt --estimate --rate 100/min
525
+ $ cat urls.txt | reqstorm --rate auto -q > out.jsonl
526
+ $ reqstorm users.csv -o users.db \
527
+ --template "https://api.example.com/users/{id}"
528
+ $ reqstorm urls.txt -o shop.db --report \
529
+ --paginate next:links.next \
530
+ --schema products.json --explode items
531
+ ```
532
+
533
+ `reqstorm --help` lists every option, and the [command line guide](https://reqstorm.github.io/guide/cli/) explains them.
534
+
418
535
  ## When to use something else
419
536
 
420
537
  reqstorm is built for batches. For other jobs, these are better fits:
@@ -30,8 +30,12 @@ classifiers = [
30
30
  ]
31
31
 
32
32
  [project.optional-dependencies]
33
- test = ["pytest>=8", "pytest-asyncio>=0.23", "trustme>=1.1"]
34
- lint = ["ruff>=0.6", "mypy>=1.10"]
33
+ pydantic = ["pydantic>=2"]
34
+ test = ["pytest>=8", "pytest-asyncio>=0.23", "trustme>=1.1", "pydantic>=2"]
35
+ lint = ["ruff>=0.6", "mypy>=1.10", "pydantic>=2"]
36
+
37
+ [project.scripts]
38
+ reqstorm = "reqstorm._cli:main"
35
39
 
36
40
  [project.urls]
37
41
  Homepage = "https://reqstorm.github.io"
@@ -11,6 +11,8 @@ async def main():
11
11
  asyncio.run(main())
12
12
  """
13
13
 
14
+ from ._auth import BearerAuth
15
+ from ._cache import Cache
14
16
  from ._client import (
15
17
  DEFAULT_RETRY_STATUSES,
16
18
  Attempt,
@@ -24,20 +26,29 @@ from ._client import (
24
26
  from ._files import Summary, fetch_to_db, fetch_to_file
25
27
  from ._legacy import Reqt
26
28
  from ._limits import parse_rate
29
+ from ._paginate import Cursor, LinkHeader, NextLink, PageNumber, Paginator
27
30
  from ._plan import Estimate, estimate
28
31
  from ._schema import Field, Schema, SchemaError, extract, infer_schema
29
32
  from ._sync import fetch_all_sync, fetch_to_db_sync, fetch_to_file_sync, stream_sync
33
+ from ._template import from_template, read_csv, read_sql
30
34
 
31
- __version__ = "2.1.0"
35
+ __version__ = "2.2.0"
32
36
 
33
37
  __all__ = [
34
- "DEFAULT_RETRY_STATUSES",
35
38
  "Attempt",
39
+ "BearerAuth",
40
+ "Cache",
41
+ "Cursor",
42
+ "DEFAULT_RETRY_STATUSES",
36
43
  "Estimate",
37
44
  "Field",
38
45
  "HTTPStatusError",
39
- "Request",
46
+ "LinkHeader",
47
+ "NextLink",
48
+ "PageNumber",
49
+ "Paginator",
40
50
  "Reqt",
51
+ "Request",
41
52
  "Result",
42
53
  "Results",
43
54
  "Schema",
@@ -51,9 +62,12 @@ __all__ = [
51
62
  "fetch_to_db_sync",
52
63
  "fetch_to_file",
53
64
  "fetch_to_file_sync",
65
+ "from_template",
54
66
  "infer_schema",
55
- "stream",
56
67
  "parse_rate",
68
+ "read_csv",
69
+ "read_sql",
70
+ "stream",
57
71
  "stream_sync",
58
72
  "__version__",
59
73
  ]
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from ._cli import main
4
+
5
+ sys.exit(main())
@@ -0,0 +1,118 @@
1
+ """A per-host rate limiter that learns the rate from the server's responses."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import time
7
+ from dataclasses import dataclass
8
+ from email.utils import parsedate_to_datetime
9
+ from typing import Dict, Mapping, Optional
10
+
11
+ from ._limits import host_key
12
+
13
+ # Header names, most specific first. "RateLimit-*" is the IETF draft; "X-RateLimit-*" is the
14
+ # common convention (GitHub, Twitter/X, many others).
15
+ _REMAINING = ("RateLimit-Remaining", "X-RateLimit-Remaining", "X-Rate-Limit-Remaining")
16
+ _RESET = ("RateLimit-Reset", "X-RateLimit-Reset", "X-Rate-Limit-Reset")
17
+ _MAX_PAUSE = 300.0
18
+
19
+
20
+ def _header(headers: Mapping[str, str], names: tuple) -> Optional[str]:
21
+ for name in names:
22
+ value = headers.get(name)
23
+ if value is not None:
24
+ return value
25
+ return None
26
+
27
+
28
+ def seconds_until_reset(headers: Mapping[str, str], now: Optional[float] = None) -> Optional[float]:
29
+ """Seconds until the server's rate-limit window resets, from ``*-Reset`` or ``Retry-After``.
30
+
31
+ Reset values above 10^9 are Unix timestamps (GitHub style); smaller ones are seconds
32
+ (IETF draft style). ``Retry-After`` may also be an HTTP date.
33
+ """
34
+ now = time.time() if now is None else now
35
+ for value in (_header(headers, _RESET), headers.get("Retry-After")):
36
+ if value is None:
37
+ continue
38
+ value = value.strip()
39
+ try:
40
+ number = float(value)
41
+ except ValueError:
42
+ try:
43
+ return max(parsedate_to_datetime(value).timestamp() - now, 0.0)
44
+ except (TypeError, ValueError, IndexError, OverflowError):
45
+ continue
46
+ seconds = number - now if number > 1e9 else number
47
+ return min(max(seconds, 0.0), _MAX_PAUSE)
48
+ return None
49
+
50
+
51
+ @dataclass
52
+ class _HostState:
53
+ interval: float = 0.0 # seconds between requests; 0 means no spacing
54
+ next_slot: float = 0.0
55
+ paused_until: float = 0.0
56
+
57
+
58
+ class AdaptiveRateLimiter:
59
+ """``rate_limit="auto"``: no limit until the server signals one, then follow it.
60
+
61
+ - A 429 response doubles the spacing between requests to that host and pauses it for
62
+ ``Retry-After`` (or until the rate-limit window resets).
63
+ - ``*-RateLimit-Remaining`` / ``*-RateLimit-Reset`` headers spread the remaining requests
64
+ evenly over the rest of the window, and pause the host when none are left.
65
+ - Every successful response without such headers shortens the spacing by 10 %, so the
66
+ rate recovers once the server stops pushing back.
67
+ """
68
+
69
+ adaptive = True
70
+
71
+ def __init__(self) -> None:
72
+ self._hosts: Dict[str, _HostState] = {}
73
+
74
+ def _state(self, url: str) -> Optional[_HostState]:
75
+ host = host_key(url)
76
+ if host is None:
77
+ return None
78
+ return self._hosts.setdefault(host, _HostState())
79
+
80
+ def interval(self, url: str) -> float:
81
+ state = self._state(url)
82
+ return state.interval if state else 0.0
83
+
84
+ async def wait(self, url: str) -> None:
85
+ state = self._state(url)
86
+ if state is None:
87
+ return
88
+ now = time.monotonic()
89
+ slot = max(now, state.next_slot, state.paused_until)
90
+ state.next_slot = slot + state.interval
91
+ if slot > now:
92
+ await asyncio.sleep(slot - now)
93
+
94
+ def record(self, url: str, status: Optional[int], headers: Mapping[str, str]) -> None:
95
+ state = self._state(url)
96
+ if state is None or status is None:
97
+ return
98
+ now = time.monotonic()
99
+ reset = seconds_until_reset(headers)
100
+ if status == 429:
101
+ state.interval = min(max(state.interval * 2, 0.1), 60.0)
102
+ pause = reset if reset is not None else max(state.interval, 1.0)
103
+ state.paused_until = max(state.paused_until, now + pause)
104
+ return
105
+ remaining_text = _header(headers, _REMAINING)
106
+ if remaining_text is not None and reset is not None:
107
+ try:
108
+ remaining = float(remaining_text)
109
+ except ValueError:
110
+ remaining = None
111
+ if remaining is not None:
112
+ if remaining <= 0:
113
+ state.paused_until = max(state.paused_until, now + reset)
114
+ else:
115
+ state.interval = min(reset / remaining, 60.0)
116
+ return
117
+ if 200 <= status < 400 and state.interval:
118
+ state.interval = state.interval * 0.9 if state.interval > 0.01 else 0.0
@@ -0,0 +1,79 @@
1
+ """Authentication that refreshes an expired token during a long run."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import inspect
7
+ from typing import Any, Awaitable, Callable, Dict, Optional, Union
8
+
9
+ __all__ = ["BearerAuth"]
10
+
11
+ TokenSource = Callable[[], Union[str, Awaitable[str]]]
12
+
13
+
14
+ class BearerAuth:
15
+ """Send ``Authorization: Bearer <token>``, and get a new token when the server says it expired.
16
+
17
+ auth = reqstorm.BearerAuth(refresh=get_token)
18
+ reqstorm.fetch_all_sync(urls, auth=auth)
19
+
20
+ Args:
21
+ token: The current token. Leave it out to fetch one with ``refresh`` before the
22
+ first request.
23
+ refresh: A function or coroutine function that returns a new token. It is called
24
+ when a response is 401 Unauthorized; the request is then sent again once with
25
+ the new token. Concurrent 401s share a single refresh.
26
+ header: Header to send the token in (default ``Authorization``).
27
+ scheme: Prefix before the token (default ``Bearer``); ``""`` sends the token alone.
28
+ """
29
+
30
+ def __init__(
31
+ self,
32
+ token: Optional[str] = None,
33
+ *,
34
+ refresh: Optional[TokenSource] = None,
35
+ header: str = "Authorization",
36
+ scheme: str = "Bearer",
37
+ ) -> None:
38
+ if token is None and refresh is None:
39
+ raise ValueError("BearerAuth needs a token, a refresh function, or both")
40
+ self.token = token
41
+ self._refresh = refresh
42
+ self.header = header
43
+ self.scheme = scheme
44
+ self.refreshes = 0
45
+ self._generation = 0
46
+ self._lock: Optional[asyncio.Lock] = None
47
+
48
+ @property
49
+ def can_refresh(self) -> bool:
50
+ return self._refresh is not None
51
+
52
+ async def headers(self) -> Dict[str, str]:
53
+ if self.token is None:
54
+ await self.refresh(self._generation)
55
+ value = f"{self.scheme} {self.token}" if self.scheme else str(self.token)
56
+ return {self.header: value}
57
+
58
+ @property
59
+ def generation(self) -> int:
60
+ """Changes each time the token is refreshed."""
61
+ return self._generation
62
+
63
+ async def refresh(self, seen_generation: int) -> None:
64
+ """Get a new token, unless another request already did since ``seen_generation``."""
65
+ if self._refresh is None:
66
+ return
67
+ if self._lock is None:
68
+ self._lock = asyncio.Lock()
69
+ async with self._lock:
70
+ if self._generation != seen_generation and self.token is not None:
71
+ return # refreshed by a concurrent request while we waited
72
+ token: Any = self._refresh()
73
+ if inspect.isawaitable(token):
74
+ token = await token
75
+ if not isinstance(token, str) or not token:
76
+ raise ValueError("the refresh function must return a non-empty token string")
77
+ self.token = token
78
+ self._generation += 1
79
+ self.refreshes += 1