reqstorm 2.2.0__tar.gz → 2.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {reqstorm-2.2.0/reqstorm.egg-info → reqstorm-2.3.0}/PKG-INFO +16 -7
- {reqstorm-2.2.0 → reqstorm-2.3.0}/README.md +11 -6
- {reqstorm-2.2.0 → reqstorm-2.3.0}/pyproject.toml +3 -2
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/__init__.py +1 -1
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_cli.py +15 -3
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_client.py +54 -7
- reqstorm-2.3.0/reqstorm/_socks.py +84 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0/reqstorm.egg-info}/PKG-INFO +16 -7
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm.egg-info/SOURCES.txt +2 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm.egg-info/requires.txt +5 -0
- reqstorm-2.3.0/tests/test_socks_backoff.py +144 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/LICENSE +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/__main__.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_adaptive.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_auth.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_cache.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_files.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_legacy.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_limits.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_paginate.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_plan.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_progress.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_pydantic.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_report.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_schema.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_schema_sinks.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_sync.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/_template.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm/py.typed +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm.egg-info/dependency_links.txt +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm.egg-info/entry_points.txt +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/reqstorm.egg-info/top_level.txt +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/setup.cfg +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_adaptive.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_auth.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_cache_proxy.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_cli.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_fetch_all.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_files.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_legacy.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_limits.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_ordered_and_db.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_paginate.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_pydantic.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_report.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_run_report.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_schema.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_schema_db.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_stream.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_sync.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_template.py +0 -0
- {reqstorm-2.2.0 → reqstorm-2.3.0}/tests/test_tls.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: reqstorm
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.3.0
|
|
4
4
|
Summary: Send large numbers of HTTP requests concurrently with asyncio, with retries, timeouts and per-request results.
|
|
5
5
|
Author-email: Melih Colpan <colpanmelih@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -28,15 +28,19 @@ License-File: LICENSE
|
|
|
28
28
|
Requires-Dist: aiohttp<4,>=3.9
|
|
29
29
|
Provides-Extra: pydantic
|
|
30
30
|
Requires-Dist: pydantic>=2; extra == "pydantic"
|
|
31
|
+
Provides-Extra: socks
|
|
32
|
+
Requires-Dist: aiohttp-socks>=0.10; extra == "socks"
|
|
31
33
|
Provides-Extra: test
|
|
32
34
|
Requires-Dist: pytest>=8; extra == "test"
|
|
33
35
|
Requires-Dist: pytest-asyncio>=0.23; extra == "test"
|
|
34
36
|
Requires-Dist: trustme>=1.1; extra == "test"
|
|
35
37
|
Requires-Dist: pydantic>=2; extra == "test"
|
|
38
|
+
Requires-Dist: aiohttp-socks>=0.10; extra == "test"
|
|
36
39
|
Provides-Extra: lint
|
|
37
40
|
Requires-Dist: ruff>=0.6; extra == "lint"
|
|
38
41
|
Requires-Dist: mypy>=1.10; extra == "lint"
|
|
39
42
|
Requires-Dist: pydantic>=2; extra == "lint"
|
|
43
|
+
Requires-Dist: aiohttp-socks>=0.10; extra == "lint"
|
|
40
44
|
Dynamic: license-file
|
|
41
45
|
|
|
42
46
|
# reqstorm
|
|
@@ -123,11 +127,11 @@ reqstorm does all of that for you, with one call.
|
|
|
123
127
|
| One result per request | Failures are recorded, never raised; every attempt is kept |
|
|
124
128
|
| Rate limits | Per host, in any unit (`"100/min"`), or `"auto"` from the server's 429s and headers |
|
|
125
129
|
| Concurrency | Overall and per host |
|
|
126
|
-
| Retries |
|
|
130
|
+
| Retries | Exponential backoff with a cap and jitter, `Retry-After`; plus end-of-run rounds |
|
|
127
131
|
| Reports | Failures by reason, p50/p95/p99 response times, per-host figures |
|
|
128
132
|
| Pagination | Next links, `Link` headers, cursors and page numbers |
|
|
129
133
|
| Requests from data | URL templates over CSV rows or SQL query results |
|
|
130
|
-
| Tokens and proxies | Refresh an expired token on 401; rotate through proxies |
|
|
134
|
+
| Tokens and proxies | Refresh an expired token on 401; rotate through HTTP and SOCKS proxies |
|
|
131
135
|
| Caching | ETag / `If-None-Match`: unchanged resources cost a 304 |
|
|
132
136
|
| Output | JSONL, CSV, SQLite, PostgreSQL, MySQL; ordered or as completed |
|
|
133
137
|
| Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
|
|
@@ -142,7 +146,7 @@ reqstorm does all of that for you, with one call.
|
|
|
142
146
|
$ python -m pip install reqstorm
|
|
143
147
|
```
|
|
144
148
|
|
|
145
|
-
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`.
|
|
149
|
+
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`; for SOCKS proxies, `reqstorm[socks]`.
|
|
146
150
|
|
|
147
151
|
## Quick start
|
|
148
152
|
|
|
@@ -306,13 +310,16 @@ results = reqstorm.fetch_all_sync(
|
|
|
306
310
|
timeout=10, # seconds per attempt
|
|
307
311
|
retries=3, # retry right away...
|
|
308
312
|
backoff=0.5, # ...0.5 s, 1 s, 2 s apart
|
|
313
|
+
max_backoff=30, # never wait longer
|
|
309
314
|
retry_rounds=2, # then resend what still
|
|
310
315
|
retry_round_delay=30, # failed, 30 s later
|
|
311
316
|
)
|
|
312
317
|
```
|
|
313
318
|
|
|
314
319
|
- **`timeout`** is the time allowed for one attempt, including reading the body (default 30 s; `None` disables it). A slow server only fails its own requests.
|
|
315
|
-
- **`retries`** retries a request right away. The
|
|
320
|
+
- **`retries`** retries a request right away, also when no response came back at all (timeouts, dropped connections). The wait starts at `backoff` and doubles each time (0.5, 1, 2, 4, ... s), up to **`max_backoff`** (30 s by default), so many retries never wait minutes.
|
|
321
|
+
- **Jitter** (on by default) waits a random time between half and all of that delay, so thousands of requests that failed together do not retry at the same moment. `jitter=False` waits exactly.
|
|
322
|
+
- A **`Retry-After`** header from the server (up to 60 s) is followed as given instead.
|
|
316
323
|
- **`retry_rounds`** holds back requests that still failed for a retryable reason and sends them again after the rest of the batch, for temporary outages. Each request still produces exactly one result, with all its attempts in `history`.
|
|
317
324
|
|
|
318
325
|
| Retried | Not retried |
|
|
@@ -484,15 +491,17 @@ auth = reqstorm.BearerAuth(refresh=get_token)
|
|
|
484
491
|
reqstorm.fetch_all_sync(urls, auth=auth)
|
|
485
492
|
```
|
|
486
493
|
|
|
487
|
-
**Proxies.** One proxy, a pool used in turn, or one per `Request
|
|
494
|
+
**Proxies.** One proxy, a pool used in turn, or one per `Request`. HTTP and SOCKS proxies can be mixed:
|
|
488
495
|
|
|
489
496
|
```python
|
|
490
497
|
reqstorm.fetch_all_sync(urls, proxy=[
|
|
491
498
|
"http://proxy-1.example.com:8080",
|
|
492
|
-
"
|
|
499
|
+
"socks5h://user:pass@proxy-2.example.com:1080",
|
|
493
500
|
])
|
|
494
501
|
```
|
|
495
502
|
|
|
503
|
+
SOCKS4 and SOCKS5 (`socks5h://` and `socks4a://` let the proxy resolve host names, as with Tor) need `pip install "reqstorm[socks]"`.
|
|
504
|
+
|
|
496
505
|
**Caching.** `Cache` keeps responses in an SQLite file. The next run asks the server with `If-None-Match`; an unchanged resource comes back as a `304` with no body, and the stored response is used. With `ttl`, recent responses skip the network entirely:
|
|
497
506
|
|
|
498
507
|
```python
|
|
@@ -82,11 +82,11 @@ reqstorm does all of that for you, with one call.
|
|
|
82
82
|
| One result per request | Failures are recorded, never raised; every attempt is kept |
|
|
83
83
|
| Rate limits | Per host, in any unit (`"100/min"`), or `"auto"` from the server's 429s and headers |
|
|
84
84
|
| Concurrency | Overall and per host |
|
|
85
|
-
| Retries |
|
|
85
|
+
| Retries | Exponential backoff with a cap and jitter, `Retry-After`; plus end-of-run rounds |
|
|
86
86
|
| Reports | Failures by reason, p50/p95/p99 response times, per-host figures |
|
|
87
87
|
| Pagination | Next links, `Link` headers, cursors and page numbers |
|
|
88
88
|
| Requests from data | URL templates over CSV rows or SQL query results |
|
|
89
|
-
| Tokens and proxies | Refresh an expired token on 401; rotate through proxies |
|
|
89
|
+
| Tokens and proxies | Refresh an expired token on 401; rotate through HTTP and SOCKS proxies |
|
|
90
90
|
| Caching | ETag / `If-None-Match`: unchanged resources cost a 304 |
|
|
91
91
|
| Output | JSONL, CSV, SQLite, PostgreSQL, MySQL; ordered or as completed |
|
|
92
92
|
| Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
|
|
@@ -101,7 +101,7 @@ reqstorm does all of that for you, with one call.
|
|
|
101
101
|
$ python -m pip install reqstorm
|
|
102
102
|
```
|
|
103
103
|
|
|
104
|
-
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`.
|
|
104
|
+
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`; for SOCKS proxies, `reqstorm[socks]`.
|
|
105
105
|
|
|
106
106
|
## Quick start
|
|
107
107
|
|
|
@@ -265,13 +265,16 @@ results = reqstorm.fetch_all_sync(
|
|
|
265
265
|
timeout=10, # seconds per attempt
|
|
266
266
|
retries=3, # retry right away...
|
|
267
267
|
backoff=0.5, # ...0.5 s, 1 s, 2 s apart
|
|
268
|
+
max_backoff=30, # never wait longer
|
|
268
269
|
retry_rounds=2, # then resend what still
|
|
269
270
|
retry_round_delay=30, # failed, 30 s later
|
|
270
271
|
)
|
|
271
272
|
```
|
|
272
273
|
|
|
273
274
|
- **`timeout`** is the time allowed for one attempt, including reading the body (default 30 s; `None` disables it). A slow server only fails its own requests.
|
|
274
|
-
- **`retries`** retries a request right away. The
|
|
275
|
+
- **`retries`** retries a request right away, also when no response came back at all (timeouts, dropped connections). The wait starts at `backoff` and doubles each time (0.5, 1, 2, 4, ... s), up to **`max_backoff`** (30 s by default), so many retries never wait minutes.
|
|
276
|
+
- **Jitter** (on by default) waits a random time between half and all of that delay, so thousands of requests that failed together do not retry at the same moment. `jitter=False` waits exactly.
|
|
277
|
+
- A **`Retry-After`** header from the server (up to 60 s) is followed as given instead.
|
|
275
278
|
- **`retry_rounds`** holds back requests that still failed for a retryable reason and sends them again after the rest of the batch, for temporary outages. Each request still produces exactly one result, with all its attempts in `history`.
|
|
276
279
|
|
|
277
280
|
| Retried | Not retried |
|
|
@@ -443,15 +446,17 @@ auth = reqstorm.BearerAuth(refresh=get_token)
|
|
|
443
446
|
reqstorm.fetch_all_sync(urls, auth=auth)
|
|
444
447
|
```
|
|
445
448
|
|
|
446
|
-
**Proxies.** One proxy, a pool used in turn, or one per `Request
|
|
449
|
+
**Proxies.** One proxy, a pool used in turn, or one per `Request`. HTTP and SOCKS proxies can be mixed:
|
|
447
450
|
|
|
448
451
|
```python
|
|
449
452
|
reqstorm.fetch_all_sync(urls, proxy=[
|
|
450
453
|
"http://proxy-1.example.com:8080",
|
|
451
|
-
"
|
|
454
|
+
"socks5h://user:pass@proxy-2.example.com:1080",
|
|
452
455
|
])
|
|
453
456
|
```
|
|
454
457
|
|
|
458
|
+
SOCKS4 and SOCKS5 (`socks5h://` and `socks4a://` let the proxy resolve host names, as with Tor) need `pip install "reqstorm[socks]"`.
|
|
459
|
+
|
|
455
460
|
**Caching.** `Cache` keeps responses in an SQLite file. The next run asks the server with `If-None-Match`; an unchanged resource comes back as a `304` with no body, and the stored response is used. With `ttl`, recent responses skip the network entirely:
|
|
456
461
|
|
|
457
462
|
```python
|
|
@@ -31,8 +31,9 @@ classifiers = [
|
|
|
31
31
|
|
|
32
32
|
[project.optional-dependencies]
|
|
33
33
|
pydantic = ["pydantic>=2"]
|
|
34
|
-
|
|
35
|
-
|
|
34
|
+
socks = ["aiohttp-socks>=0.10"]
|
|
35
|
+
test = ["pytest>=8", "pytest-asyncio>=0.23", "trustme>=1.1", "pydantic>=2", "aiohttp-socks>=0.10"]
|
|
36
|
+
lint = ["ruff>=0.6", "mypy>=1.10", "pydantic>=2", "aiohttp-socks>=0.10"]
|
|
36
37
|
|
|
37
38
|
[project.scripts]
|
|
38
39
|
reqstorm = "reqstorm._cli:main"
|
|
@@ -32,7 +32,7 @@ from ._schema import Field, Schema, SchemaError, extract, infer_schema
|
|
|
32
32
|
from ._sync import fetch_all_sync, fetch_to_db_sync, fetch_to_file_sync, stream_sync
|
|
33
33
|
from ._template import from_template, read_csv, read_sql
|
|
34
34
|
|
|
35
|
-
__version__ = "2.
|
|
35
|
+
__version__ = "2.3.0"
|
|
36
36
|
|
|
37
37
|
__all__ = [
|
|
38
38
|
"Attempt",
|
|
@@ -57,7 +57,11 @@ def _parser() -> argparse.ArgumentParser:
|
|
|
57
57
|
request.add_argument("--bearer", metavar="TOKEN", help="send 'Authorization: Bearer TOKEN' "
|
|
58
58
|
"(or set REQSTORM_TOKEN)") # fmt: skip
|
|
59
59
|
request.add_argument(
|
|
60
|
-
"--proxy",
|
|
60
|
+
"--proxy",
|
|
61
|
+
action="append",
|
|
62
|
+
default=[],
|
|
63
|
+
metavar="URL",
|
|
64
|
+
help="http://, socks5://, socks5h:// or socks4:// proxy URL; repeat to rotate",
|
|
61
65
|
)
|
|
62
66
|
request.add_argument("--no-verify", action="store_true", help="do not verify TLS certificates")
|
|
63
67
|
|
|
@@ -67,7 +71,13 @@ def _parser() -> argparse.ArgumentParser:
|
|
|
67
71
|
pace.add_argument("--per-host", type=int, default=0, metavar="N", help="max requests in flight per host")
|
|
68
72
|
pace.add_argument("--timeout", type=float, default=30.0, help="seconds per attempt (default 30)")
|
|
69
73
|
pace.add_argument("--retries", type=int, default=0)
|
|
70
|
-
pace.add_argument(
|
|
74
|
+
pace.add_argument(
|
|
75
|
+
"--backoff", type=float, default=0.5, help="seconds before the first retry, doubled after"
|
|
76
|
+
)
|
|
77
|
+
pace.add_argument(
|
|
78
|
+
"--max-backoff", type=float, default=30.0, help="longest wait between retries (default 30)"
|
|
79
|
+
)
|
|
80
|
+
pace.add_argument("--no-jitter", action="store_true", help="wait exactly, not a random part of the delay")
|
|
71
81
|
pace.add_argument("--retry-rounds", type=int, default=0)
|
|
72
82
|
pace.add_argument("--retry-round-delay", type=float, default=5.0)
|
|
73
83
|
pace.add_argument("--paginate", metavar="STRATEGY",
|
|
@@ -204,7 +214,7 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
|
204
214
|
args = parser.parse_args(argv)
|
|
205
215
|
try:
|
|
206
216
|
return _run(args)
|
|
207
|
-
except (ValueError, KeyError, SchemaError, OSError, json.JSONDecodeError) as error:
|
|
217
|
+
except (ValueError, KeyError, SchemaError, OSError, ImportError, json.JSONDecodeError) as error:
|
|
208
218
|
message = error.args[0] if isinstance(error, KeyError) and error.args else error
|
|
209
219
|
print(f"reqstorm: error: {message}", file=sys.stderr)
|
|
210
220
|
return 2
|
|
@@ -260,6 +270,8 @@ def _batch(args: argparse.Namespace, cache: Optional[Cache]) -> int:
|
|
|
260
270
|
timeout=args.timeout,
|
|
261
271
|
retries=args.retries,
|
|
262
272
|
backoff=args.backoff,
|
|
273
|
+
max_backoff=args.max_backoff,
|
|
274
|
+
jitter=not args.no_jitter,
|
|
263
275
|
verify_ssl=not args.no_verify,
|
|
264
276
|
rate_limit=rate,
|
|
265
277
|
auth=BearerAuth(token) if token else None,
|
|
@@ -7,6 +7,7 @@ import base64
|
|
|
7
7
|
import collections
|
|
8
8
|
import itertools
|
|
9
9
|
import json as jsonlib
|
|
10
|
+
import random
|
|
10
11
|
import ssl as ssllib
|
|
11
12
|
import time
|
|
12
13
|
from dataclasses import dataclass, field
|
|
@@ -34,6 +35,7 @@ from multidict import CIMultiDict, CIMultiDictProxy
|
|
|
34
35
|
from ._adaptive import AdaptiveRateLimiter
|
|
35
36
|
from ._limits import HostRateLimiter, RateLimit
|
|
36
37
|
from ._progress import Progress, ProgressTarget
|
|
38
|
+
from ._socks import SocksSessions, check_available, is_socks
|
|
37
39
|
|
|
38
40
|
__all__ = ["Attempt", "HTTPStatusError", "Request", "Result", "Results", "fetch_all", "stream"]
|
|
39
41
|
|
|
@@ -245,6 +247,18 @@ class _Options:
|
|
|
245
247
|
auth: Any = None
|
|
246
248
|
cache: Any = None
|
|
247
249
|
proxies: Optional[Iterator[str]] = None
|
|
250
|
+
socks: Optional[SocksSessions] = None
|
|
251
|
+
max_backoff: float = 30.0
|
|
252
|
+
jitter: bool = True
|
|
253
|
+
|
|
254
|
+
def backoff_delay(self, retry: int) -> float:
|
|
255
|
+
"""Seconds to wait before retry number ``retry`` (1 for the first)."""
|
|
256
|
+
delay = min(self.backoff * (2 ** (retry - 1)), self.max_backoff)
|
|
257
|
+
if self.jitter:
|
|
258
|
+
# "Equal jitter": somewhere between half the delay and the full delay, so
|
|
259
|
+
# requests that failed together do not all retry at the same moment
|
|
260
|
+
delay = delay / 2 + random.uniform(0, delay / 2)
|
|
261
|
+
return delay
|
|
248
262
|
|
|
249
263
|
def next_proxy(self) -> Optional[str]:
|
|
250
264
|
return next(self.proxies) if self.proxies is not None else None
|
|
@@ -327,7 +341,13 @@ async def _send(session: aiohttp.ClientSession, request: Request, index: int, op
|
|
|
327
341
|
if limiter is not None:
|
|
328
342
|
await limiter.wait(request.url)
|
|
329
343
|
attempt_started = time.monotonic()
|
|
330
|
-
|
|
344
|
+
proxy = request.proxy or options.next_proxy()
|
|
345
|
+
sender = session
|
|
346
|
+
if proxy is not None and is_socks(proxy):
|
|
347
|
+
if options.socks is None:
|
|
348
|
+
raise ValueError("a SOCKS proxy needs a session created by reqstorm, not session=")
|
|
349
|
+
sender, proxy = options.socks.session(proxy), None
|
|
350
|
+
async with sender.request(
|
|
331
351
|
request.method or "GET",
|
|
332
352
|
request.url,
|
|
333
353
|
headers=headers or None,
|
|
@@ -336,7 +356,7 @@ async def _send(session: aiohttp.ClientSession, request: Request, index: int, op
|
|
|
336
356
|
data=request.data,
|
|
337
357
|
timeout=options.timeout,
|
|
338
358
|
ssl=options.ssl,
|
|
339
|
-
proxy=
|
|
359
|
+
proxy=proxy,
|
|
340
360
|
) as response:
|
|
341
361
|
body = await response.read()
|
|
342
362
|
status, response_headers, final_url = response.status, response.headers, str(response.url)
|
|
@@ -361,7 +381,7 @@ async def _send(session: aiohttp.ClientSession, request: Request, index: int, op
|
|
|
361
381
|
if status not in options.retry_statuses or retries_used >= options.retries:
|
|
362
382
|
break
|
|
363
383
|
retries_used += 1
|
|
364
|
-
delay = _retry_after(response_headers) or options.
|
|
384
|
+
delay = _retry_after(response_headers) or options.backoff_delay(retries_used)
|
|
365
385
|
except asyncio.CancelledError:
|
|
366
386
|
raise
|
|
367
387
|
except Exception as error: # one failing request must not stop the others
|
|
@@ -371,7 +391,7 @@ async def _send(session: aiohttp.ClientSession, request: Request, index: int, op
|
|
|
371
391
|
if not _is_retryable_error(error) or retries_used >= options.retries:
|
|
372
392
|
break
|
|
373
393
|
retries_used += 1
|
|
374
|
-
delay = options.
|
|
394
|
+
delay = options.backoff_delay(retries_used)
|
|
375
395
|
if delay:
|
|
376
396
|
await asyncio.sleep(delay)
|
|
377
397
|
if cache_key and result.ok and result.status == 200 and not result.from_cache:
|
|
@@ -392,6 +412,8 @@ async def stream(
|
|
|
392
412
|
timeout: Optional[float] = 30.0,
|
|
393
413
|
retries: int = 0,
|
|
394
414
|
backoff: float = 0.5,
|
|
415
|
+
max_backoff: float = 30.0,
|
|
416
|
+
jitter: bool = True,
|
|
395
417
|
retry_statuses: Sequence[int] = DEFAULT_RETRY_STATUSES,
|
|
396
418
|
verify_ssl: bool = True,
|
|
397
419
|
ssl: Optional[ssllib.SSLContext] = None,
|
|
@@ -416,6 +438,8 @@ async def stream(
|
|
|
416
438
|
raise ValueError("retries must not be negative")
|
|
417
439
|
if concurrency_per_host < 0:
|
|
418
440
|
raise ValueError("concurrency_per_host must not be negative")
|
|
441
|
+
if backoff < 0 or max_backoff < 0:
|
|
442
|
+
raise ValueError("backoff and max_backoff must not be negative")
|
|
419
443
|
if rate_limit == "auto":
|
|
420
444
|
limiter: Any = AdaptiveRateLimiter()
|
|
421
445
|
elif rate_limit is not None:
|
|
@@ -423,6 +447,10 @@ async def stream(
|
|
|
423
447
|
else:
|
|
424
448
|
limiter = None
|
|
425
449
|
proxies = [proxy] if isinstance(proxy, str) else list(proxy or [])
|
|
450
|
+
if any(is_socks(item) for item in proxies):
|
|
451
|
+
check_available() # fail before sending anything, not on every request
|
|
452
|
+
if session is not None:
|
|
453
|
+
raise ValueError("SOCKS proxies cannot be combined with session=; reqstorm opens their sessions")
|
|
426
454
|
options = _Options(
|
|
427
455
|
method=method,
|
|
428
456
|
headers=headers,
|
|
@@ -432,12 +460,15 @@ async def stream(
|
|
|
432
460
|
timeout=aiohttp.ClientTimeout(total=timeout),
|
|
433
461
|
retries=retries,
|
|
434
462
|
backoff=backoff,
|
|
463
|
+
max_backoff=max_backoff,
|
|
464
|
+
jitter=jitter,
|
|
435
465
|
retry_statuses=frozenset(retry_statuses),
|
|
436
466
|
ssl=ssl if ssl is not None else verify_ssl,
|
|
437
467
|
rate_limiter=limiter,
|
|
438
468
|
auth=auth,
|
|
439
469
|
cache=cache,
|
|
440
470
|
proxies=itertools.cycle(proxies) if proxies else None,
|
|
471
|
+
socks=SocksSessions(concurrency, concurrency_per_host) if session is None else None,
|
|
441
472
|
)
|
|
442
473
|
|
|
443
474
|
owns_session = session is None
|
|
@@ -527,6 +558,8 @@ async def stream(
|
|
|
527
558
|
pass
|
|
528
559
|
if owns_session:
|
|
529
560
|
await session.close()
|
|
561
|
+
if options.socks is not None:
|
|
562
|
+
await options.socks.close()
|
|
530
563
|
|
|
531
564
|
|
|
532
565
|
async def _execute(
|
|
@@ -590,6 +623,8 @@ async def fetch_all(
|
|
|
590
623
|
timeout: Optional[float] = ...,
|
|
591
624
|
retries: int = ...,
|
|
592
625
|
backoff: float = ...,
|
|
626
|
+
max_backoff: float = ...,
|
|
627
|
+
jitter: bool = ...,
|
|
593
628
|
retry_statuses: Sequence[int] = ...,
|
|
594
629
|
retry_rounds: int = ...,
|
|
595
630
|
retry_round_delay: float = ...,
|
|
@@ -631,6 +666,8 @@ async def fetch_all(
|
|
|
631
666
|
timeout: Optional[float] = 30.0,
|
|
632
667
|
retries: int = 0,
|
|
633
668
|
backoff: float = 0.5,
|
|
669
|
+
max_backoff: float = 30.0,
|
|
670
|
+
jitter: bool = True,
|
|
634
671
|
retry_statuses: Sequence[int] = DEFAULT_RETRY_STATUSES,
|
|
635
672
|
retry_rounds: int = 0,
|
|
636
673
|
retry_round_delay: float = 5.0,
|
|
@@ -660,7 +697,13 @@ async def fetch_all(
|
|
|
660
697
|
timeout: Seconds allowed per attempt, including reading the body. ``None`` disables it.
|
|
661
698
|
retries: How many times to retry a request right away after a connection error,
|
|
662
699
|
a timeout or a status in ``retry_statuses``. Invalid URLs are not retried.
|
|
663
|
-
backoff: Delay before the first retry in seconds, doubled for each further retry
|
|
700
|
+
backoff: Delay before the first retry in seconds, doubled for each further retry
|
|
701
|
+
(0.5, 1, 2, 4, ... with the default).
|
|
702
|
+
max_backoff: Upper limit for that delay in seconds (default 30), so many retries
|
|
703
|
+
never wait minutes. A ``Retry-After`` from the server is followed instead, up
|
|
704
|
+
to 60 seconds.
|
|
705
|
+
jitter: Wait a random time between half and all of the delay, so requests that
|
|
706
|
+
failed together do not retry at the same moment. ``False`` waits exactly.
|
|
664
707
|
A ``Retry-After`` header (in seconds, up to 60) takes precedence.
|
|
665
708
|
retry_rounds: After all requests have finished, send the ones that still failed
|
|
666
709
|
for a retryable reason again, up to this many more rounds.
|
|
@@ -677,8 +720,10 @@ async def fetch_all(
|
|
|
677
720
|
concurrency_per_host: Maximum requests in flight to each host; 0 means no per-host limit.
|
|
678
721
|
auth: ``reqstorm.BearerAuth``: sends a token and refreshes it once on a 401.
|
|
679
722
|
cache: ``reqstorm.Cache``: reuses earlier GET responses, revalidating with ETag.
|
|
680
|
-
proxy: A proxy URL
|
|
681
|
-
|
|
723
|
+
proxy: A proxy URL, or a list of them used in turn, attempt by attempt:
|
|
724
|
+
``"http://user:pass@host:port"``, or ``socks5://``, ``socks5h://`` (the proxy
|
|
725
|
+
resolves host names), ``socks4://`` and ``socks4a://`` with
|
|
726
|
+
``pip install 'reqstorm[socks]'``. ``Request(proxy=...)`` sets one per request.
|
|
682
727
|
paginate: ``reqstorm.NextLink``, ``LinkHeader``, ``Cursor`` or ``PageNumber``: each
|
|
683
728
|
starting URL is followed through its pages. Results carry ``page`` and
|
|
684
729
|
``seed_index``. Cannot be combined with ``retry_rounds``.
|
|
@@ -708,6 +753,8 @@ async def fetch_all(
|
|
|
708
753
|
timeout=timeout,
|
|
709
754
|
retries=retries,
|
|
710
755
|
backoff=backoff,
|
|
756
|
+
max_backoff=max_backoff,
|
|
757
|
+
jitter=jitter,
|
|
711
758
|
retry_statuses=retry_statuses,
|
|
712
759
|
verify_ssl=verify_ssl,
|
|
713
760
|
ssl=ssl,
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""SOCKS4 and SOCKS5 proxies, through the optional aiohttp-socks package."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, Dict, Optional, Tuple
|
|
6
|
+
from urllib.parse import unquote, urlsplit
|
|
7
|
+
|
|
8
|
+
import aiohttp
|
|
9
|
+
|
|
10
|
+
SCHEMES = ("socks4", "socks4a", "socks5", "socks5h")
|
|
11
|
+
_MISSING = "SOCKS proxies need the aiohttp-socks package: pip install 'reqstorm[socks]'"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def is_socks(proxy: Optional[str]) -> bool:
|
|
15
|
+
return proxy is not None and urlsplit(proxy).scheme.lower() in SCHEMES
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def check_available() -> None:
|
|
19
|
+
try:
|
|
20
|
+
import aiohttp_socks # noqa: F401
|
|
21
|
+
except ImportError:
|
|
22
|
+
raise ImportError(_MISSING) from None
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def parse(proxy: str) -> Tuple[str, str, int, Optional[str], Optional[str], bool]:
|
|
26
|
+
"""(version, host, port, username, password, remote_dns) of a SOCKS proxy URL.
|
|
27
|
+
|
|
28
|
+
As with curl, ``socks5h://`` and ``socks4a://`` let the proxy resolve host names, while
|
|
29
|
+
``socks5://`` and ``socks4://`` resolve them locally.
|
|
30
|
+
"""
|
|
31
|
+
parts = urlsplit(proxy)
|
|
32
|
+
scheme = parts.scheme.lower()
|
|
33
|
+
if scheme not in SCHEMES:
|
|
34
|
+
raise ValueError(f"not a SOCKS proxy URL: {proxy!r}")
|
|
35
|
+
if not parts.hostname:
|
|
36
|
+
raise ValueError(f"SOCKS proxy URL without a host: {proxy!r}")
|
|
37
|
+
try:
|
|
38
|
+
port = parts.port or 1080
|
|
39
|
+
except ValueError:
|
|
40
|
+
raise ValueError(f"invalid port in SOCKS proxy URL {proxy!r}") from None
|
|
41
|
+
username = unquote(parts.username) if parts.username is not None else None
|
|
42
|
+
password = unquote(parts.password) if parts.password is not None else None
|
|
43
|
+
version = "socks5" if scheme.startswith("socks5") else "socks4"
|
|
44
|
+
return version, parts.hostname, port, username, password, scheme in ("socks5h", "socks4a")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class SocksSessions:
|
|
48
|
+
"""One aiohttp session per SOCKS proxy, created on first use and closed together.
|
|
49
|
+
|
|
50
|
+
A SOCKS proxy is a property of the connection, not of the request, so each proxy needs
|
|
51
|
+
its own connector. The connection limits match those of the main session.
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
def __init__(self, limit: int, limit_per_host: int) -> None:
|
|
55
|
+
self._limit = limit
|
|
56
|
+
self._limit_per_host = limit_per_host
|
|
57
|
+
self._sessions: Dict[str, aiohttp.ClientSession] = {}
|
|
58
|
+
|
|
59
|
+
def session(self, proxy: str) -> aiohttp.ClientSession:
|
|
60
|
+
existing = self._sessions.get(proxy)
|
|
61
|
+
if existing is not None:
|
|
62
|
+
return existing
|
|
63
|
+
check_available()
|
|
64
|
+
from aiohttp_socks import ProxyConnector, ProxyType
|
|
65
|
+
|
|
66
|
+
version, host, port, username, password, remote_dns = parse(proxy)
|
|
67
|
+
connector: Any = ProxyConnector(
|
|
68
|
+
proxy_type=ProxyType.SOCKS5 if version == "socks5" else ProxyType.SOCKS4,
|
|
69
|
+
host=host,
|
|
70
|
+
port=port,
|
|
71
|
+
username=username,
|
|
72
|
+
password=password,
|
|
73
|
+
rdns=remote_dns,
|
|
74
|
+
limit=self._limit,
|
|
75
|
+
limit_per_host=self._limit_per_host,
|
|
76
|
+
)
|
|
77
|
+
created = aiohttp.ClientSession(connector=connector)
|
|
78
|
+
self._sessions[proxy] = created
|
|
79
|
+
return created
|
|
80
|
+
|
|
81
|
+
async def close(self) -> None:
|
|
82
|
+
for session in self._sessions.values():
|
|
83
|
+
await session.close()
|
|
84
|
+
self._sessions.clear()
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: reqstorm
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.3.0
|
|
4
4
|
Summary: Send large numbers of HTTP requests concurrently with asyncio, with retries, timeouts and per-request results.
|
|
5
5
|
Author-email: Melih Colpan <colpanmelih@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -28,15 +28,19 @@ License-File: LICENSE
|
|
|
28
28
|
Requires-Dist: aiohttp<4,>=3.9
|
|
29
29
|
Provides-Extra: pydantic
|
|
30
30
|
Requires-Dist: pydantic>=2; extra == "pydantic"
|
|
31
|
+
Provides-Extra: socks
|
|
32
|
+
Requires-Dist: aiohttp-socks>=0.10; extra == "socks"
|
|
31
33
|
Provides-Extra: test
|
|
32
34
|
Requires-Dist: pytest>=8; extra == "test"
|
|
33
35
|
Requires-Dist: pytest-asyncio>=0.23; extra == "test"
|
|
34
36
|
Requires-Dist: trustme>=1.1; extra == "test"
|
|
35
37
|
Requires-Dist: pydantic>=2; extra == "test"
|
|
38
|
+
Requires-Dist: aiohttp-socks>=0.10; extra == "test"
|
|
36
39
|
Provides-Extra: lint
|
|
37
40
|
Requires-Dist: ruff>=0.6; extra == "lint"
|
|
38
41
|
Requires-Dist: mypy>=1.10; extra == "lint"
|
|
39
42
|
Requires-Dist: pydantic>=2; extra == "lint"
|
|
43
|
+
Requires-Dist: aiohttp-socks>=0.10; extra == "lint"
|
|
40
44
|
Dynamic: license-file
|
|
41
45
|
|
|
42
46
|
# reqstorm
|
|
@@ -123,11 +127,11 @@ reqstorm does all of that for you, with one call.
|
|
|
123
127
|
| One result per request | Failures are recorded, never raised; every attempt is kept |
|
|
124
128
|
| Rate limits | Per host, in any unit (`"100/min"`), or `"auto"` from the server's 429s and headers |
|
|
125
129
|
| Concurrency | Overall and per host |
|
|
126
|
-
| Retries |
|
|
130
|
+
| Retries | Exponential backoff with a cap and jitter, `Retry-After`; plus end-of-run rounds |
|
|
127
131
|
| Reports | Failures by reason, p50/p95/p99 response times, per-host figures |
|
|
128
132
|
| Pagination | Next links, `Link` headers, cursors and page numbers |
|
|
129
133
|
| Requests from data | URL templates over CSV rows or SQL query results |
|
|
130
|
-
| Tokens and proxies | Refresh an expired token on 401; rotate through proxies |
|
|
134
|
+
| Tokens and proxies | Refresh an expired token on 401; rotate through HTTP and SOCKS proxies |
|
|
131
135
|
| Caching | ETag / `If-None-Match`: unchanged resources cost a 304 |
|
|
132
136
|
| Output | JSONL, CSV, SQLite, PostgreSQL, MySQL; ordered or as completed |
|
|
133
137
|
| Typed columns | JSON fields to checked columns, nested paths, arrays to rows |
|
|
@@ -142,7 +146,7 @@ reqstorm does all of that for you, with one call.
|
|
|
142
146
|
$ python -m pip install reqstorm
|
|
143
147
|
```
|
|
144
148
|
|
|
145
|
-
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`.
|
|
149
|
+
reqstorm supports Python 3.9 to 3.13 on Linux, macOS and Windows. Its only dependency is [aiohttp](https://docs.aiohttp.org). For PostgreSQL or MySQL output, install the driver you already use (`psycopg`, `psycopg2`, `pymysql`, `mysqlclient` or `mysql-connector-python`). To use Pydantic models as schemas, install `reqstorm[pydantic]`; for SOCKS proxies, `reqstorm[socks]`.
|
|
146
150
|
|
|
147
151
|
## Quick start
|
|
148
152
|
|
|
@@ -306,13 +310,16 @@ results = reqstorm.fetch_all_sync(
|
|
|
306
310
|
timeout=10, # seconds per attempt
|
|
307
311
|
retries=3, # retry right away...
|
|
308
312
|
backoff=0.5, # ...0.5 s, 1 s, 2 s apart
|
|
313
|
+
max_backoff=30, # never wait longer
|
|
309
314
|
retry_rounds=2, # then resend what still
|
|
310
315
|
retry_round_delay=30, # failed, 30 s later
|
|
311
316
|
)
|
|
312
317
|
```
|
|
313
318
|
|
|
314
319
|
- **`timeout`** is the time allowed for one attempt, including reading the body (default 30 s; `None` disables it). A slow server only fails its own requests.
|
|
315
|
-
- **`retries`** retries a request right away. The
|
|
320
|
+
- **`retries`** retries a request right away, also when no response came back at all (timeouts, dropped connections). The wait starts at `backoff` and doubles each time (0.5, 1, 2, 4, ... s), up to **`max_backoff`** (30 s by default), so many retries never wait minutes.
|
|
321
|
+
- **Jitter** (on by default) waits a random time between half and all of that delay, so thousands of requests that failed together do not retry at the same moment. `jitter=False` waits exactly.
|
|
322
|
+
- A **`Retry-After`** header from the server (up to 60 s) is followed as given instead.
|
|
316
323
|
- **`retry_rounds`** holds back requests that still failed for a retryable reason and sends them again after the rest of the batch, for temporary outages. Each request still produces exactly one result, with all its attempts in `history`.
|
|
317
324
|
|
|
318
325
|
| Retried | Not retried |
|
|
@@ -484,15 +491,17 @@ auth = reqstorm.BearerAuth(refresh=get_token)
|
|
|
484
491
|
reqstorm.fetch_all_sync(urls, auth=auth)
|
|
485
492
|
```
|
|
486
493
|
|
|
487
|
-
**Proxies.** One proxy, a pool used in turn, or one per `Request
|
|
494
|
+
**Proxies.** One proxy, a pool used in turn, or one per `Request`. HTTP and SOCKS proxies can be mixed:
|
|
488
495
|
|
|
489
496
|
```python
|
|
490
497
|
reqstorm.fetch_all_sync(urls, proxy=[
|
|
491
498
|
"http://proxy-1.example.com:8080",
|
|
492
|
-
"
|
|
499
|
+
"socks5h://user:pass@proxy-2.example.com:1080",
|
|
493
500
|
])
|
|
494
501
|
```
|
|
495
502
|
|
|
503
|
+
SOCKS4 and SOCKS5 (`socks5h://` and `socks4a://` let the proxy resolve host names, as with Tor) need `pip install "reqstorm[socks]"`.
|
|
504
|
+
|
|
496
505
|
**Caching.** `Cache` keeps responses in an SQLite file. The next run asks the server with `If-None-Match`; an unchanged resource comes back as a `304` with no body, and the stored response is used. With `ttl`, recent responses skip the network entirely:
|
|
497
506
|
|
|
498
507
|
```python
|
|
@@ -18,6 +18,7 @@ reqstorm/_pydantic.py
|
|
|
18
18
|
reqstorm/_report.py
|
|
19
19
|
reqstorm/_schema.py
|
|
20
20
|
reqstorm/_schema_sinks.py
|
|
21
|
+
reqstorm/_socks.py
|
|
21
22
|
reqstorm/_sync.py
|
|
22
23
|
reqstorm/_template.py
|
|
23
24
|
reqstorm/py.typed
|
|
@@ -42,6 +43,7 @@ tests/test_report.py
|
|
|
42
43
|
tests/test_run_report.py
|
|
43
44
|
tests/test_schema.py
|
|
44
45
|
tests/test_schema_db.py
|
|
46
|
+
tests/test_socks_backoff.py
|
|
45
47
|
tests/test_stream.py
|
|
46
48
|
tests/test_sync.py
|
|
47
49
|
tests/test_template.py
|
|
@@ -4,12 +4,17 @@ aiohttp<4,>=3.9
|
|
|
4
4
|
ruff>=0.6
|
|
5
5
|
mypy>=1.10
|
|
6
6
|
pydantic>=2
|
|
7
|
+
aiohttp-socks>=0.10
|
|
7
8
|
|
|
8
9
|
[pydantic]
|
|
9
10
|
pydantic>=2
|
|
10
11
|
|
|
12
|
+
[socks]
|
|
13
|
+
aiohttp-socks>=0.10
|
|
14
|
+
|
|
11
15
|
[test]
|
|
12
16
|
pytest>=8
|
|
13
17
|
pytest-asyncio>=0.23
|
|
14
18
|
trustme>=1.1
|
|
15
19
|
pydantic>=2
|
|
20
|
+
aiohttp-socks>=0.10
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
import time
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
import reqstorm
|
|
6
|
+
from reqstorm._client import _Options
|
|
7
|
+
from reqstorm._socks import is_socks, parse
|
|
8
|
+
|
|
9
|
+
pytest.importorskip("aiohttp_socks")
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _port(server):
|
|
13
|
+
return int(server.rsplit(":", 1)[1])
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
async def test_socks5_connects_through_the_proxy(server, socks_proxy):
|
|
17
|
+
results = await reqstorm.fetch_all([server + "/ok"], proxy=socks_proxy.url("socks5"))
|
|
18
|
+
assert results[0].text() == "ok"
|
|
19
|
+
assert socks_proxy.requests == [(5, "127.0.0.1", _port(server))]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
async def test_socks5h_lets_the_proxy_resolve_names(server, socks_proxy):
|
|
23
|
+
url = f"http://api.example.test:{_port(server)}/ok"
|
|
24
|
+
results = await reqstorm.fetch_all([url], proxy=socks_proxy.url("socks5h"))
|
|
25
|
+
assert results[0].text() == "ok"
|
|
26
|
+
assert socks_proxy.requests == [(5, "api.example.test", _port(server))]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
async def test_socks5_resolves_names_locally(server, socks_proxy):
|
|
30
|
+
url = f"http://api.example.test:{_port(server)}/ok"
|
|
31
|
+
results = await reqstorm.fetch_all([url], proxy=socks_proxy.url("socks5"), timeout=5)
|
|
32
|
+
assert results[0].error is not None # the made-up name does not resolve here
|
|
33
|
+
assert socks_proxy.requests == []
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
async def test_socks5_with_username_and_password(server, socks_proxy_auth):
|
|
37
|
+
proxy = socks_proxy_auth.url("socks5", ("ayse", "p%40ss%3Aword")) # percent-encoded in the URL
|
|
38
|
+
results = await reqstorm.fetch_all([server + "/ok"], proxy=proxy)
|
|
39
|
+
assert results[0].text() == "ok"
|
|
40
|
+
wrong = await reqstorm.fetch_all([server + "/ok"], proxy=socks_proxy_auth.url("socks5", ("ayse", "x")))
|
|
41
|
+
assert wrong[0].error is not None and not wrong[0].ok
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@pytest.mark.parametrize("scheme, requested", [("socks4", "127.0.0.1"), ("socks4a", "api.example.test")])
|
|
45
|
+
async def test_socks4_and_4a(server, socks_proxy, scheme, requested):
|
|
46
|
+
host = "127.0.0.1" if scheme == "socks4" else "api.example.test"
|
|
47
|
+
results = await reqstorm.fetch_all([f"http://{host}:{_port(server)}/ok"], proxy=socks_proxy.url(scheme))
|
|
48
|
+
assert results[0].text() == "ok"
|
|
49
|
+
assert socks_proxy.requests == [(4, requested, _port(server))]
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
async def test_https_through_socks(tls_server, trusted_context, socks_proxy):
|
|
53
|
+
results = await reqstorm.fetch_all([tls_server + "/ok"], proxy=socks_proxy.url(), ssl=trusted_context)
|
|
54
|
+
assert results[0].text() == "ok" and len(socks_proxy.requests) == 1
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
async def test_pool_mixes_socks_and_http_proxies(server, second_server, socks_proxy):
|
|
58
|
+
# second_server stands in for an HTTP proxy and answers itself; the SOCKS proxy
|
|
59
|
+
# forwards to the real server. /proxy-check reports which port answered.
|
|
60
|
+
url = f"http://api.example.test:{_port(server)}/proxy-check"
|
|
61
|
+
proxies = [second_server, socks_proxy.url("socks5h")]
|
|
62
|
+
results = await reqstorm.fetch_all([url] * 4, proxy=proxies, concurrency=1)
|
|
63
|
+
ports = [result.json()["port"] for result in results]
|
|
64
|
+
assert ports == [_port(second_server), _port(server)] * 2
|
|
65
|
+
assert len(socks_proxy.requests) == 1 # the SOCKS connection is reused
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
async def test_per_request_socks_proxy_and_connection_reuse(server, socks_proxy):
|
|
69
|
+
requests = [reqstorm.Request(server + "/ok", proxy=socks_proxy.url()) for _ in range(5)]
|
|
70
|
+
results = await reqstorm.fetch_all(requests, concurrency=1)
|
|
71
|
+
assert all(result.ok for result in results)
|
|
72
|
+
assert len(socks_proxy.requests) == 1 # one connection, kept alive
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
async def test_socks_cannot_share_a_user_session(server, socks_proxy):
|
|
76
|
+
import aiohttp
|
|
77
|
+
|
|
78
|
+
async with aiohttp.ClientSession() as session:
|
|
79
|
+
with pytest.raises(ValueError, match="session"):
|
|
80
|
+
await reqstorm.fetch_all([server + "/ok"], proxy=socks_proxy.url(), session=session)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
async def test_missing_aiohttp_socks_is_explained(server, monkeypatch):
|
|
84
|
+
import builtins
|
|
85
|
+
|
|
86
|
+
real_import = builtins.__import__
|
|
87
|
+
|
|
88
|
+
def without_socks(name, *args, **kwargs):
|
|
89
|
+
if name.startswith("aiohttp_socks"):
|
|
90
|
+
raise ImportError(name)
|
|
91
|
+
return real_import(name, *args, **kwargs)
|
|
92
|
+
|
|
93
|
+
monkeypatch.setattr(builtins, "__import__", without_socks)
|
|
94
|
+
with pytest.raises(ImportError, match=r"reqstorm\[socks\]"):
|
|
95
|
+
await reqstorm.fetch_all([server + "/ok"], proxy="socks5://127.0.0.1:1080")
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def test_proxy_url_parsing():
|
|
99
|
+
assert parse("socks5h://user:p%40ss@proxy.example.com") == (
|
|
100
|
+
"socks5",
|
|
101
|
+
"proxy.example.com",
|
|
102
|
+
1080,
|
|
103
|
+
"user",
|
|
104
|
+
"p@ss",
|
|
105
|
+
True,
|
|
106
|
+
)
|
|
107
|
+
assert parse("SOCKS4://10.0.0.1:9050")[:3] == ("socks4", "10.0.0.1", 9050)
|
|
108
|
+
assert is_socks("socks5://x:1") and not is_socks("http://x:1") and not is_socks(None)
|
|
109
|
+
with pytest.raises(ValueError):
|
|
110
|
+
parse("socks5://")
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _options(**overrides):
|
|
114
|
+
import aiohttp
|
|
115
|
+
|
|
116
|
+
defaults = dict(method="GET", headers=None, params=None, json=None, data=None,
|
|
117
|
+
timeout=aiohttp.ClientTimeout(total=1), retries=10, backoff=0.5,
|
|
118
|
+
retry_statuses=frozenset(), ssl=True, rate_limiter=None) # fmt: skip
|
|
119
|
+
return _Options(**{**defaults, **overrides})
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def test_backoff_doubles_and_stops_at_max_backoff():
|
|
123
|
+
options = _options(jitter=False, max_backoff=4)
|
|
124
|
+
assert [options.backoff_delay(n) for n in range(1, 7)] == [0.5, 1, 2, 4, 4, 4]
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def test_jitter_stays_between_half_and_the_full_delay():
|
|
128
|
+
options = _options(max_backoff=30)
|
|
129
|
+
delays = [options.backoff_delay(4) for _ in range(500)] # 4 seconds without jitter
|
|
130
|
+
assert all(2 <= delay <= 4 for delay in delays)
|
|
131
|
+
assert max(delays) - min(delays) > 1 # actually spread out
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
async def test_max_backoff_limits_the_real_wait(server):
|
|
135
|
+
started = time.monotonic()
|
|
136
|
+
results = await reqstorm.fetch_all(
|
|
137
|
+
[server + "/flaky?key=cap&fail=3"], retries=3, backoff=10, max_backoff=0.1
|
|
138
|
+
)
|
|
139
|
+
assert results[0].ok and time.monotonic() - started < 1.5
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
async def test_negative_backoff_is_rejected(server):
|
|
143
|
+
with pytest.raises(ValueError, match="backoff"):
|
|
144
|
+
await reqstorm.fetch_all([server + "/ok"], max_backoff=-1)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|