proxy-scraper-cli 1.7.1__tar.gz → 1.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/PKG-INFO +33 -4
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/README.md +32 -3
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxy_scraper_cli.egg-info/PKG-INFO +33 -4
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxy_scraper_cli.egg-info/SOURCES.txt +6 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/__init__.py +3 -2
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/agent.py +21 -4
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/api.py +99 -3
- proxy_scraper_cli-1.8.0/proxyscraper/charts.py +135 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/mcp_server.py +12 -1
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/publish.py +99 -6
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/site/index.html +286 -67
- proxy_scraper_cli-1.8.0/proxyscraper/sites.py +155 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_agent.py +9 -1
- proxy_scraper_cli-1.8.0/tests/test_charts.py +57 -0
- proxy_scraper_cli-1.8.0/tests/test_live_api.py +79 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_server.py +4 -2
- proxy_scraper_cli-1.8.0/tests/test_sites.py +141 -0
- proxy_scraper_cli-1.8.0/tests/test_uptime.py +129 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/LICENSE +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxy_scraper_cli.egg-info/dependency_links.txt +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxy_scraper_cli.egg-info/entry_points.txt +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxy_scraper_cli.egg-info/requires.txt +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxy_scraper_cli.egg-info/top_level.txt +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/__main__.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/app.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/asndb.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/blocklist.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/checker.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/cli.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/compat.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/completion.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/exporters.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/fetchcache.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/geo.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/geodb.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/handshake.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/history.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/judges.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/mcp_entry.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/netio.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/options.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/output.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/pages.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/parsing.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/paths.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/pipeline.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/preferences.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/__init__.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/core.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/http.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/pool.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/socks.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/status.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/upstream.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/site/apple-touch-icon.png +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/site/googleaac1161b7853c5b5.html +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/site/logo.png +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/site/og.png +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/sources.json +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/sources.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/targets.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/__init__.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/dashboard.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/keys.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/report.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/serve.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/widgets.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/wizard.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/pyproject.toml +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/setup.cfg +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_api.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_app.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_asndb.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_auth.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_blocklist.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_checker.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_compat.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_completion.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_discovery_growth.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_exporters.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_fetchcache.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_geodb.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_judges.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_learning.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_mcp_server.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_options.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_output_stdout.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_packaging.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_pages.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_parsing.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_paths.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_pipeline.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_proxy_sources.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_publish.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_robustness.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_serve_password.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_serve_refill.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_serve_v2.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_server_fixes.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_tamper.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_targets.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_ui.py +0 -0
- {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_wizard.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: proxy-scraper-cli
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.8.0
|
|
4
4
|
Summary: Finds free HTTP, SOCKS4 and SOCKS5 proxies that actually work: 700+ sources, real checks, honeypot filtering and a rotating local proxy server.
|
|
5
5
|
Author: Maximilian Feix
|
|
6
6
|
License-Expression: MIT
|
|
@@ -175,7 +175,13 @@ Keys: <kbd>↑</kbd><kbd>↓</kbd> select · <kbd>Space</kbd> toggle · <kbd>1</
|
|
|
175
175
|
|
|
176
176
|
Don't want to scan yourself? Every hour **GitHub Actions** runs the tool and publishes the hits to the [`proxy-list`](../../tree/proxy-list) branch – every entry worked in the last run, fastest first.
|
|
177
177
|
|
|
178
|
-
|
|
178
|
+
<a href="https://maximilianfeix.github.io/proxy-scraper/"><picture>
|
|
179
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://maximilianfeix.github.io/proxy-scraper/chart-dark.svg">
|
|
180
|
+
<source media="(prefers-color-scheme: light)" srcset="https://maximilianfeix.github.io/proxy-scraper/chart-light.svg">
|
|
181
|
+
<img src="https://maximilianfeix.github.io/proxy-scraper/chart-dark.svg" alt="Working proxies over the last days, stacked by protocol – redrawn with every run" width="100%">
|
|
182
|
+
</picture></a>
|
|
183
|
+
|
|
184
|
+
**→ [Browse it on the website](https://maximilianfeix.github.io/proxy-scraper/)** – search, filter by type, country, HTTPS, provider and latency, see how long each proxy has been up and how often it was on the list this week, copy or download exactly the proxies you need.
|
|
179
185
|
|
|
180
186
|
Every protocol and country also has its own page with a plain download, e.g. [SOCKS5](https://maximilianfeix.github.io/proxy-scraper/socks5/) or [Germany](https://maximilianfeix.github.io/proxy-scraper/country/de/) (`country/de/proxies.txt`).
|
|
181
187
|
|
|
@@ -197,12 +203,22 @@ Every protocol and country also has its own page with a plain download, e.g. [SO
|
|
|
197
203
|
| HTTP · SOCKS4 · SOCKS5 | `1.2.3.4:8080` | [http.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/http.txt) · [socks4.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks4.txt) · [socks5.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks5.txt) |
|
|
198
204
|
| HTTPS-capable only | `type://ip:port` | [https.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/https.txt) |
|
|
199
205
|
| Elite only | `type://ip:port` | [elite.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/elite.txt) |
|
|
200
|
-
|
|
|
206
|
+
| Gets through to Google · Reddit · Amazon (no captcha, no 403 in the last run) | `type://ip:port` | [google.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/google.txt) · [reddit.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/reddit.txt) · [amazon.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/amazon.txt) |
|
|
207
|
+
| Stable, on the list in 90 %+ of this week's runs | `type://ip:port` | [stable.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/stable.txt) |
|
|
208
|
+
| With all details | latency, country, HTTPS, anonymity, exit IP, uptime, sites | [proxies.json](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.json) · [proxies.csv](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.csv) |
|
|
209
|
+
|
|
210
|
+
<a href="https://maximilianfeix.github.io/proxy-scraper/"><picture>
|
|
211
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://maximilianfeix.github.io/proxy-scraper/countries-dark.svg">
|
|
212
|
+
<source media="(prefers-color-scheme: light)" srcset="https://maximilianfeix.github.io/proxy-scraper/countries-light.svg">
|
|
213
|
+
<img src="https://maximilianfeix.github.io/proxy-scraper/countries-dark.svg" alt="Countries with the most working proxies right now" width="100%">
|
|
214
|
+
</picture></a>
|
|
201
215
|
|
|
202
216
|
```bash
|
|
203
217
|
curl -s https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks5.txt | head
|
|
204
218
|
```
|
|
205
219
|
|
|
220
|
+
Every file is also on GitHub Pages, which sits behind a CDN, isn't rate limited like raw.githubusercontent and sends CORS headers, so it works straight from the browser: `https://maximilianfeix.github.io/proxy-scraper/socks5.txt`. [jsDelivr](https://cdn.jsdelivr.net/gh/maximilianfeix/proxy-scraper@proxy-list/) works too, but can lag behind by a few hours.
|
|
221
|
+
|
|
206
222
|
Or let the tool start from it: `proxy-scraper --recheck live` downloads the list and checks it again from **your** network – about 30 seconds instead of a full scan (517 of 1,169 worked from here). With `--serve` you have a rotating proxy in under a minute.
|
|
207
223
|
|
|
208
224
|
<a id="mcp"></a>
|
|
@@ -215,7 +231,7 @@ Or let the tool start from it: `proxy-scraper --recheck live` downloads the list
|
|
|
215
231
|
|
|
216
232
|
| Tool | What it does |
|
|
217
233
|
|---|---|
|
|
218
|
-
| `get_proxies` | working proxies right now, from the hourly list – filter by protocol, country, HTTPS, elite, no datacenter, not blocklisted, stable, latency |
|
|
234
|
+
| `get_proxies` | working proxies right now, from the hourly list – filter by protocol, country, HTTPS, elite, no datacenter, not blocklisted, stable, uptime, latency, gets through to Google/Reddit/Amazon |
|
|
219
235
|
| `check_proxies` | checks proxies from your own network, so they work from where your code runs (30–90 s, reports progress) |
|
|
220
236
|
| `fetch_url` | loads a page through a verified proxy, switches proxies by itself when one fails, returns readable text – HTTPS only through proxies with verified TLS |
|
|
221
237
|
|
|
@@ -411,6 +427,19 @@ if __name__ == "__main__": # needed on macOS/Windows, the parser uses a process
|
|
|
411
427
|
|
|
412
428
|
Same run as the command line – sources, learning, every check, result files – just without terminal output. Each result has `url`, `latency`, `exit_ip`, `https`, `anonymity`, `country`, `asn`, `org` and `hosting`. There's an async version of both (`find_proxies_async`, `check_proxies_async`).
|
|
413
429
|
|
|
430
|
+
Don't need a fresh scan? `live_proxies` takes the [live list](#live-list) instead – no checks, one download, done in about a second:
|
|
431
|
+
|
|
432
|
+
```python
|
|
433
|
+
import itertools, requests
|
|
434
|
+
from proxyscraper import live_proxies
|
|
435
|
+
|
|
436
|
+
proxies = live_proxies(types=["socks5"], https=True, min_uptime=90) # the reliable ones this week
|
|
437
|
+
pool = itertools.cycle(p.url for p in proxies)
|
|
438
|
+
r = requests.get("https://api.ipify.org", proxies={"https": next(pool)}, timeout=15) # pip install "requests[socks]"
|
|
439
|
+
```
|
|
440
|
+
|
|
441
|
+
Same filters as `find_proxies`, plus `min_uptime`, `works_on` (`["google"]`, `"reddit"`, `"amazon"`) and `limit`. Every result also has `uptime_24h`, `uptime_7d`, `first_seen`, `up_for_hours` and `sites`.
|
|
442
|
+
|
|
414
443
|
<a id="proxy-server"></a>
|
|
415
444
|
|
|
416
445
|
## Rotating proxy server
|
|
@@ -135,7 +135,13 @@ Keys: <kbd>↑</kbd><kbd>↓</kbd> select · <kbd>Space</kbd> toggle · <kbd>1</
|
|
|
135
135
|
|
|
136
136
|
Don't want to scan yourself? Every hour **GitHub Actions** runs the tool and publishes the hits to the [`proxy-list`](../../tree/proxy-list) branch – every entry worked in the last run, fastest first.
|
|
137
137
|
|
|
138
|
-
|
|
138
|
+
<a href="https://maximilianfeix.github.io/proxy-scraper/"><picture>
|
|
139
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://maximilianfeix.github.io/proxy-scraper/chart-dark.svg">
|
|
140
|
+
<source media="(prefers-color-scheme: light)" srcset="https://maximilianfeix.github.io/proxy-scraper/chart-light.svg">
|
|
141
|
+
<img src="https://maximilianfeix.github.io/proxy-scraper/chart-dark.svg" alt="Working proxies over the last days, stacked by protocol – redrawn with every run" width="100%">
|
|
142
|
+
</picture></a>
|
|
143
|
+
|
|
144
|
+
**→ [Browse it on the website](https://maximilianfeix.github.io/proxy-scraper/)** – search, filter by type, country, HTTPS, provider and latency, see how long each proxy has been up and how often it was on the list this week, copy or download exactly the proxies you need.
|
|
139
145
|
|
|
140
146
|
Every protocol and country also has its own page with a plain download, e.g. [SOCKS5](https://maximilianfeix.github.io/proxy-scraper/socks5/) or [Germany](https://maximilianfeix.github.io/proxy-scraper/country/de/) (`country/de/proxies.txt`).
|
|
141
147
|
|
|
@@ -157,12 +163,22 @@ Every protocol and country also has its own page with a plain download, e.g. [SO
|
|
|
157
163
|
| HTTP · SOCKS4 · SOCKS5 | `1.2.3.4:8080` | [http.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/http.txt) · [socks4.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks4.txt) · [socks5.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks5.txt) |
|
|
158
164
|
| HTTPS-capable only | `type://ip:port` | [https.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/https.txt) |
|
|
159
165
|
| Elite only | `type://ip:port` | [elite.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/elite.txt) |
|
|
160
|
-
|
|
|
166
|
+
| Gets through to Google · Reddit · Amazon (no captcha, no 403 in the last run) | `type://ip:port` | [google.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/google.txt) · [reddit.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/reddit.txt) · [amazon.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/amazon.txt) |
|
|
167
|
+
| Stable, on the list in 90 %+ of this week's runs | `type://ip:port` | [stable.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/stable.txt) |
|
|
168
|
+
| With all details | latency, country, HTTPS, anonymity, exit IP, uptime, sites | [proxies.json](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.json) · [proxies.csv](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.csv) |
|
|
169
|
+
|
|
170
|
+
<a href="https://maximilianfeix.github.io/proxy-scraper/"><picture>
|
|
171
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://maximilianfeix.github.io/proxy-scraper/countries-dark.svg">
|
|
172
|
+
<source media="(prefers-color-scheme: light)" srcset="https://maximilianfeix.github.io/proxy-scraper/countries-light.svg">
|
|
173
|
+
<img src="https://maximilianfeix.github.io/proxy-scraper/countries-dark.svg" alt="Countries with the most working proxies right now" width="100%">
|
|
174
|
+
</picture></a>
|
|
161
175
|
|
|
162
176
|
```bash
|
|
163
177
|
curl -s https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks5.txt | head
|
|
164
178
|
```
|
|
165
179
|
|
|
180
|
+
Every file is also on GitHub Pages, which sits behind a CDN, isn't rate limited like raw.githubusercontent and sends CORS headers, so it works straight from the browser: `https://maximilianfeix.github.io/proxy-scraper/socks5.txt`. [jsDelivr](https://cdn.jsdelivr.net/gh/maximilianfeix/proxy-scraper@proxy-list/) works too, but can lag behind by a few hours.
|
|
181
|
+
|
|
166
182
|
Or let the tool start from it: `proxy-scraper --recheck live` downloads the list and checks it again from **your** network – about 30 seconds instead of a full scan (517 of 1,169 worked from here). With `--serve` you have a rotating proxy in under a minute.
|
|
167
183
|
|
|
168
184
|
<a id="mcp"></a>
|
|
@@ -175,7 +191,7 @@ Or let the tool start from it: `proxy-scraper --recheck live` downloads the list
|
|
|
175
191
|
|
|
176
192
|
| Tool | What it does |
|
|
177
193
|
|---|---|
|
|
178
|
-
| `get_proxies` | working proxies right now, from the hourly list – filter by protocol, country, HTTPS, elite, no datacenter, not blocklisted, stable, latency |
|
|
194
|
+
| `get_proxies` | working proxies right now, from the hourly list – filter by protocol, country, HTTPS, elite, no datacenter, not blocklisted, stable, uptime, latency, gets through to Google/Reddit/Amazon |
|
|
179
195
|
| `check_proxies` | checks proxies from your own network, so they work from where your code runs (30–90 s, reports progress) |
|
|
180
196
|
| `fetch_url` | loads a page through a verified proxy, switches proxies by itself when one fails, returns readable text – HTTPS only through proxies with verified TLS |
|
|
181
197
|
|
|
@@ -371,6 +387,19 @@ if __name__ == "__main__": # needed on macOS/Windows, the parser uses a process
|
|
|
371
387
|
|
|
372
388
|
Same run as the command line – sources, learning, every check, result files – just without terminal output. Each result has `url`, `latency`, `exit_ip`, `https`, `anonymity`, `country`, `asn`, `org` and `hosting`. There's an async version of both (`find_proxies_async`, `check_proxies_async`).
|
|
373
389
|
|
|
390
|
+
Don't need a fresh scan? `live_proxies` takes the [live list](#live-list) instead – no checks, one download, done in about a second:
|
|
391
|
+
|
|
392
|
+
```python
|
|
393
|
+
import itertools, requests
|
|
394
|
+
from proxyscraper import live_proxies
|
|
395
|
+
|
|
396
|
+
proxies = live_proxies(types=["socks5"], https=True, min_uptime=90) # the reliable ones this week
|
|
397
|
+
pool = itertools.cycle(p.url for p in proxies)
|
|
398
|
+
r = requests.get("https://api.ipify.org", proxies={"https": next(pool)}, timeout=15) # pip install "requests[socks]"
|
|
399
|
+
```
|
|
400
|
+
|
|
401
|
+
Same filters as `find_proxies`, plus `min_uptime`, `works_on` (`["google"]`, `"reddit"`, `"amazon"`) and `limit`. Every result also has `uptime_24h`, `uptime_7d`, `first_seen`, `up_for_hours` and `sites`.
|
|
402
|
+
|
|
374
403
|
<a id="proxy-server"></a>
|
|
375
404
|
|
|
376
405
|
## Rotating proxy server
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: proxy-scraper-cli
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.8.0
|
|
4
4
|
Summary: Finds free HTTP, SOCKS4 and SOCKS5 proxies that actually work: 700+ sources, real checks, honeypot filtering and a rotating local proxy server.
|
|
5
5
|
Author: Maximilian Feix
|
|
6
6
|
License-Expression: MIT
|
|
@@ -175,7 +175,13 @@ Keys: <kbd>↑</kbd><kbd>↓</kbd> select · <kbd>Space</kbd> toggle · <kbd>1</
|
|
|
175
175
|
|
|
176
176
|
Don't want to scan yourself? Every hour **GitHub Actions** runs the tool and publishes the hits to the [`proxy-list`](../../tree/proxy-list) branch – every entry worked in the last run, fastest first.
|
|
177
177
|
|
|
178
|
-
|
|
178
|
+
<a href="https://maximilianfeix.github.io/proxy-scraper/"><picture>
|
|
179
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://maximilianfeix.github.io/proxy-scraper/chart-dark.svg">
|
|
180
|
+
<source media="(prefers-color-scheme: light)" srcset="https://maximilianfeix.github.io/proxy-scraper/chart-light.svg">
|
|
181
|
+
<img src="https://maximilianfeix.github.io/proxy-scraper/chart-dark.svg" alt="Working proxies over the last days, stacked by protocol – redrawn with every run" width="100%">
|
|
182
|
+
</picture></a>
|
|
183
|
+
|
|
184
|
+
**→ [Browse it on the website](https://maximilianfeix.github.io/proxy-scraper/)** – search, filter by type, country, HTTPS, provider and latency, see how long each proxy has been up and how often it was on the list this week, copy or download exactly the proxies you need.
|
|
179
185
|
|
|
180
186
|
Every protocol and country also has its own page with a plain download, e.g. [SOCKS5](https://maximilianfeix.github.io/proxy-scraper/socks5/) or [Germany](https://maximilianfeix.github.io/proxy-scraper/country/de/) (`country/de/proxies.txt`).
|
|
181
187
|
|
|
@@ -197,12 +203,22 @@ Every protocol and country also has its own page with a plain download, e.g. [SO
|
|
|
197
203
|
| HTTP · SOCKS4 · SOCKS5 | `1.2.3.4:8080` | [http.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/http.txt) · [socks4.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks4.txt) · [socks5.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks5.txt) |
|
|
198
204
|
| HTTPS-capable only | `type://ip:port` | [https.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/https.txt) |
|
|
199
205
|
| Elite only | `type://ip:port` | [elite.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/elite.txt) |
|
|
200
|
-
|
|
|
206
|
+
| Gets through to Google · Reddit · Amazon (no captcha, no 403 in the last run) | `type://ip:port` | [google.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/google.txt) · [reddit.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/reddit.txt) · [amazon.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/amazon.txt) |
|
|
207
|
+
| Stable, on the list in 90 %+ of this week's runs | `type://ip:port` | [stable.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/stable.txt) |
|
|
208
|
+
| With all details | latency, country, HTTPS, anonymity, exit IP, uptime, sites | [proxies.json](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.json) · [proxies.csv](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.csv) |
|
|
209
|
+
|
|
210
|
+
<a href="https://maximilianfeix.github.io/proxy-scraper/"><picture>
|
|
211
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://maximilianfeix.github.io/proxy-scraper/countries-dark.svg">
|
|
212
|
+
<source media="(prefers-color-scheme: light)" srcset="https://maximilianfeix.github.io/proxy-scraper/countries-light.svg">
|
|
213
|
+
<img src="https://maximilianfeix.github.io/proxy-scraper/countries-dark.svg" alt="Countries with the most working proxies right now" width="100%">
|
|
214
|
+
</picture></a>
|
|
201
215
|
|
|
202
216
|
```bash
|
|
203
217
|
curl -s https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks5.txt | head
|
|
204
218
|
```
|
|
205
219
|
|
|
220
|
+
Every file is also on GitHub Pages, which sits behind a CDN, isn't rate limited like raw.githubusercontent and sends CORS headers, so it works straight from the browser: `https://maximilianfeix.github.io/proxy-scraper/socks5.txt`. [jsDelivr](https://cdn.jsdelivr.net/gh/maximilianfeix/proxy-scraper@proxy-list/) works too, but can lag behind by a few hours.
|
|
221
|
+
|
|
206
222
|
Or let the tool start from it: `proxy-scraper --recheck live` downloads the list and checks it again from **your** network – about 30 seconds instead of a full scan (517 of 1,169 worked from here). With `--serve` you have a rotating proxy in under a minute.
|
|
207
223
|
|
|
208
224
|
<a id="mcp"></a>
|
|
@@ -215,7 +231,7 @@ Or let the tool start from it: `proxy-scraper --recheck live` downloads the list
|
|
|
215
231
|
|
|
216
232
|
| Tool | What it does |
|
|
217
233
|
|---|---|
|
|
218
|
-
| `get_proxies` | working proxies right now, from the hourly list – filter by protocol, country, HTTPS, elite, no datacenter, not blocklisted, stable, latency |
|
|
234
|
+
| `get_proxies` | working proxies right now, from the hourly list – filter by protocol, country, HTTPS, elite, no datacenter, not blocklisted, stable, uptime, latency, gets through to Google/Reddit/Amazon |
|
|
219
235
|
| `check_proxies` | checks proxies from your own network, so they work from where your code runs (30–90 s, reports progress) |
|
|
220
236
|
| `fetch_url` | loads a page through a verified proxy, switches proxies by itself when one fails, returns readable text – HTTPS only through proxies with verified TLS |
|
|
221
237
|
|
|
@@ -411,6 +427,19 @@ if __name__ == "__main__": # needed on macOS/Windows, the parser uses a process
|
|
|
411
427
|
|
|
412
428
|
Same run as the command line – sources, learning, every check, result files – just without terminal output. Each result has `url`, `latency`, `exit_ip`, `https`, `anonymity`, `country`, `asn`, `org` and `hosting`. There's an async version of both (`find_proxies_async`, `check_proxies_async`).
|
|
413
429
|
|
|
430
|
+
Don't need a fresh scan? `live_proxies` takes the [live list](#live-list) instead – no checks, one download, done in about a second:
|
|
431
|
+
|
|
432
|
+
```python
|
|
433
|
+
import itertools, requests
|
|
434
|
+
from proxyscraper import live_proxies
|
|
435
|
+
|
|
436
|
+
proxies = live_proxies(types=["socks5"], https=True, min_uptime=90) # the reliable ones this week
|
|
437
|
+
pool = itertools.cycle(p.url for p in proxies)
|
|
438
|
+
r = requests.get("https://api.ipify.org", proxies={"https": next(pool)}, timeout=15) # pip install "requests[socks]"
|
|
439
|
+
```
|
|
440
|
+
|
|
441
|
+
Same filters as `find_proxies`, plus `min_uptime`, `works_on` (`["google"]`, `"reddit"`, `"amazon"`) and `limit`. Every result also has `uptime_24h`, `uptime_7d`, `first_seen`, `up_for_hours` and `sites`.
|
|
442
|
+
|
|
414
443
|
<a id="proxy-server"></a>
|
|
415
444
|
|
|
416
445
|
## Rotating proxy server
|
|
@@ -14,6 +14,7 @@ proxyscraper/api.py
|
|
|
14
14
|
proxyscraper/app.py
|
|
15
15
|
proxyscraper/asndb.py
|
|
16
16
|
proxyscraper/blocklist.py
|
|
17
|
+
proxyscraper/charts.py
|
|
17
18
|
proxyscraper/checker.py
|
|
18
19
|
proxyscraper/cli.py
|
|
19
20
|
proxyscraper/compat.py
|
|
@@ -36,6 +37,7 @@ proxyscraper/paths.py
|
|
|
36
37
|
proxyscraper/pipeline.py
|
|
37
38
|
proxyscraper/preferences.py
|
|
38
39
|
proxyscraper/publish.py
|
|
40
|
+
proxyscraper/sites.py
|
|
39
41
|
proxyscraper/sources.json
|
|
40
42
|
proxyscraper/sources.py
|
|
41
43
|
proxyscraper/targets.py
|
|
@@ -64,6 +66,7 @@ tests/test_app.py
|
|
|
64
66
|
tests/test_asndb.py
|
|
65
67
|
tests/test_auth.py
|
|
66
68
|
tests/test_blocklist.py
|
|
69
|
+
tests/test_charts.py
|
|
67
70
|
tests/test_checker.py
|
|
68
71
|
tests/test_compat.py
|
|
69
72
|
tests/test_completion.py
|
|
@@ -73,6 +76,7 @@ tests/test_fetchcache.py
|
|
|
73
76
|
tests/test_geodb.py
|
|
74
77
|
tests/test_judges.py
|
|
75
78
|
tests/test_learning.py
|
|
79
|
+
tests/test_live_api.py
|
|
76
80
|
tests/test_mcp_server.py
|
|
77
81
|
tests/test_options.py
|
|
78
82
|
tests/test_output_stdout.py
|
|
@@ -89,7 +93,9 @@ tests/test_serve_refill.py
|
|
|
89
93
|
tests/test_serve_v2.py
|
|
90
94
|
tests/test_server.py
|
|
91
95
|
tests/test_server_fixes.py
|
|
96
|
+
tests/test_sites.py
|
|
92
97
|
tests/test_tamper.py
|
|
93
98
|
tests/test_targets.py
|
|
94
99
|
tests/test_ui.py
|
|
100
|
+
tests/test_uptime.py
|
|
95
101
|
tests/test_wizard.py
|
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
"""proxy-scraper: collects public HTTP/SOCKS4/SOCKS5 proxies from hundreds of sources, checks them
|
|
2
2
|
in parallel with its own protocol handshakes and learns which sources deliver good proxies."""
|
|
3
3
|
|
|
4
|
-
__version__ = "1.
|
|
4
|
+
__version__ = "1.8.0"
|
|
5
5
|
|
|
6
6
|
|
|
7
7
|
def __getattr__(name):
|
|
8
8
|
# load the Python API only when needed – "import proxyscraper" should stay light (the CLI doesn't need it)
|
|
9
|
-
if name in ("find_proxies", "find_proxies_async", "check_proxies", "check_proxies_async"
|
|
9
|
+
if name in ("find_proxies", "find_proxies_async", "check_proxies", "check_proxies_async", "live_proxies",
|
|
10
|
+
"live_proxies_async", "LiveProxy"):
|
|
10
11
|
from . import api
|
|
11
12
|
return getattr(api, name)
|
|
12
13
|
raise AttributeError(f"module 'proxyscraper' has no attribute {name!r}")
|
|
@@ -35,6 +35,7 @@ from .pages import SITE_URL
|
|
|
35
35
|
from .parsing import PROXY_TYPES, make_key
|
|
36
36
|
from .publish import RAW_BASE
|
|
37
37
|
from .server import ProxyPool, RotatingServer
|
|
38
|
+
from .sites import SITES
|
|
38
39
|
|
|
39
40
|
LIVE_BASES = (SITE_URL.rstrip("/"), RAW_BASE) # GitHub Pages first, the raw branch as the mirror
|
|
40
41
|
CACHE_SECONDS = 300.0 # the list changes once an hour – no need to load 1 MB for every question
|
|
@@ -149,6 +150,15 @@ class LiveSource:
|
|
|
149
150
|
|
|
150
151
|
# --------------------------------------------------------------------------- filtering
|
|
151
152
|
|
|
153
|
+
def _check_sites(names: Iterable[str]) -> List[str]:
|
|
154
|
+
known = [s.name for s in SITES]
|
|
155
|
+
names = [str(n).lower() for n in names]
|
|
156
|
+
for n in names:
|
|
157
|
+
if n not in known:
|
|
158
|
+
raise AgentError(f"works_on takes {', '.join(known)}, not {n!r}")
|
|
159
|
+
return names
|
|
160
|
+
|
|
161
|
+
|
|
152
162
|
def _check_protocol(protocol: str) -> str:
|
|
153
163
|
protocol = (protocol or "any").lower()
|
|
154
164
|
if protocol not in PROTOCOLS:
|
|
@@ -168,20 +178,23 @@ def normalize_countries(countries: Iterable[str]) -> List[str]:
|
|
|
168
178
|
|
|
169
179
|
def select(rows: Iterable[dict], protocol: str = "any", countries: Iterable[str] = (), https_only: bool = False,
|
|
170
180
|
elite_only: bool = False, exclude_datacenter: bool = False, exclude_blocklisted: bool = False,
|
|
171
|
-
stable_only: bool = False, max_latency_ms: int = 0, run_hours: int = 1
|
|
181
|
+
stable_only: bool = False, max_latency_ms: int = 0, run_hours: int = 1, min_uptime: int = 0,
|
|
182
|
+
works_on: Iterable[str] = ()) -> List[dict]:
|
|
172
183
|
"""The rows that pass every filter, fastest first – the same filters as on the website."""
|
|
173
184
|
return sorted(matching(rows, protocol, countries, https_only, elite_only, exclude_datacenter,
|
|
174
|
-
exclude_blocklisted, stable_only, max_latency_ms, run_hours),
|
|
185
|
+
exclude_blocklisted, stable_only, max_latency_ms, run_hours, min_uptime, works_on),
|
|
175
186
|
key=lambda r: r.get("latency") or 0)
|
|
176
187
|
|
|
177
188
|
|
|
178
189
|
def matching(rows: Iterable[dict], protocol: str = "any", countries: Iterable[str] = (), https_only: bool = False,
|
|
179
190
|
elite_only: bool = False, exclude_datacenter: bool = False, exclude_blocklisted: bool = False,
|
|
180
|
-
stable_only: bool = False, max_latency_ms: int = 0, run_hours: int = 1
|
|
191
|
+
stable_only: bool = False, max_latency_ms: int = 0, run_hours: int = 1,
|
|
192
|
+
min_uptime: int = 0, works_on: Iterable[str] = ()) -> Iterator[dict]:
|
|
181
193
|
"""The rows that pass every filter, in list order – checks the filters before the first row is asked for."""
|
|
182
194
|
protocol = _check_protocol(protocol)
|
|
183
195
|
wanted = set(normalize_countries(countries))
|
|
184
196
|
stable_runs = -(-24 // max(run_hours, 1)) # runs in a row that make a day
|
|
197
|
+
sites = _check_sites(works_on)
|
|
185
198
|
return (r for r in rows
|
|
186
199
|
if (protocol == "any" or r.get("ptype") == protocol)
|
|
187
200
|
and (not wanted or r.get("country") in wanted)
|
|
@@ -190,7 +203,9 @@ def matching(rows: Iterable[dict], protocol: str = "any", countries: Iterable[st
|
|
|
190
203
|
and (not exclude_datacenter or not r.get("hosting"))
|
|
191
204
|
and (not exclude_blocklisted or not r.get("blocklisted"))
|
|
192
205
|
and (not stable_only or (r.get("streak") or 0) >= stable_runs)
|
|
193
|
-
and (not max_latency_ms or (r.get("latency") or 0) <= max_latency_ms)
|
|
206
|
+
and (not max_latency_ms or (r.get("latency") or 0) <= max_latency_ms)
|
|
207
|
+
and (not min_uptime or (r.get("uptime_7d") or 0) >= min_uptime)
|
|
208
|
+
and all((r.get("sites") or {}).get(s) for s in sites))
|
|
194
209
|
|
|
195
210
|
|
|
196
211
|
def describe(item: Union[dict, CheckResult], run_hours: Optional[int] = None) -> dict:
|
|
@@ -212,6 +227,8 @@ def describe(item: Union[dict, CheckResult], run_hours: Optional[int] = None) ->
|
|
|
212
227
|
"datacenter": item.get("hosting"),
|
|
213
228
|
"blocklisted": item.get("blocklisted"),
|
|
214
229
|
"up_for_hours": streak * run_hours if streak and run_hours else None,
|
|
230
|
+
"uptime_7d_percent": item.get("uptime_7d"),
|
|
231
|
+
"works_on": [name for name, ok in (item.get("sites") or {}).items() if ok],
|
|
215
232
|
}
|
|
216
233
|
|
|
217
234
|
|
|
@@ -6,7 +6,14 @@
|
|
|
6
6
|
for p in find_proxies(want=20, https=True, countries=["DE", "NL"]):
|
|
7
7
|
print(p.url, p.latency, p.country)
|
|
8
8
|
|
|
9
|
-
|
|
9
|
+
Or skip the checking and take the list that GitHub Actions checks every hour:
|
|
10
|
+
|
|
11
|
+
from proxyscraper import live_proxies
|
|
12
|
+
|
|
13
|
+
for p in live_proxies(types=["socks5"], countries="DE", https=True, min_uptime=90):
|
|
14
|
+
print(p.url, p.latency, p.uptime_7d)
|
|
15
|
+
|
|
16
|
+
Behind find_proxies runs exactly the same as on the command line (sources, learning, honeypot and
|
|
10
17
|
tampering checks, result files under results/), just without output in the terminal.
|
|
11
18
|
|
|
12
19
|
Large lists are parsed in a process pool. On macOS and Windows it starts the worker processes
|
|
@@ -20,9 +27,11 @@ import asyncio
|
|
|
20
27
|
import contextlib
|
|
21
28
|
import io
|
|
22
29
|
import os
|
|
30
|
+
import re
|
|
23
31
|
import tempfile
|
|
24
32
|
import threading
|
|
25
|
-
from
|
|
33
|
+
from dataclasses import dataclass, field
|
|
34
|
+
from typing import Dict, Iterable, List, Optional
|
|
26
35
|
|
|
27
36
|
from rich.console import Console
|
|
28
37
|
|
|
@@ -32,13 +41,22 @@ from .parsing import PROXY_TYPES
|
|
|
32
41
|
from .targets import parse_target
|
|
33
42
|
from .ui import widgets
|
|
34
43
|
|
|
35
|
-
__all__ = ["CheckResult", "check_proxies", "check_proxies_async", "find_proxies", "find_proxies_async"
|
|
44
|
+
__all__ = ["CheckResult", "LiveProxy", "check_proxies", "check_proxies_async", "find_proxies", "find_proxies_async",
|
|
45
|
+
"live_proxies", "live_proxies_async"]
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _types(types: Iterable[str]) -> List[str]:
|
|
49
|
+
"""["socks5"], "socks5" or "socks5,http" – a plain string shouldn't turn into its letters."""
|
|
50
|
+
if isinstance(types, str):
|
|
51
|
+
return [t.strip().lower() for t in types.split(",") if t.strip()]
|
|
52
|
+
return list(types)
|
|
36
53
|
|
|
37
54
|
|
|
38
55
|
def _options(types: Iterable[str], want: int, limit: int, https: bool, countries: Iterable[str], anonymity: str,
|
|
39
56
|
max_latency: int, targets: Iterable[str], no_datacenter: bool, no_blocklisted: bool, timeout: float,
|
|
40
57
|
concurrency: int,
|
|
41
58
|
recheck: Optional[str]) -> RunOptions:
|
|
59
|
+
types = _types(types)
|
|
42
60
|
if isinstance(countries, str):
|
|
43
61
|
countries = parse_countries(countries)
|
|
44
62
|
if anonymity not in ("", "anonymous", "elite"):
|
|
@@ -137,3 +155,81 @@ async def check_proxies_async(proxies: Iterable[str], **kwargs) -> List[CheckRes
|
|
|
137
155
|
|
|
138
156
|
def check_proxies(proxies: Iterable[str], **kwargs) -> List[CheckResult]:
|
|
139
157
|
return asyncio.run(check_proxies_async(proxies, **kwargs))
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
# --------------------------------------------------------------------------- the hourly list
|
|
161
|
+
|
|
162
|
+
@dataclass
|
|
163
|
+
class LiveProxy(CheckResult):
|
|
164
|
+
"""A proxy from the hourly list: the same fields as a CheckResult, plus how reliable it has been."""
|
|
165
|
+
uptime_24h: Optional[int] = None # share of today's hourly runs it was listed in, in percent
|
|
166
|
+
uptime_7d: Optional[int] = None # the same over the week
|
|
167
|
+
first_seen: str = "" # ISO time of the first run it was listed in
|
|
168
|
+
up_for_hours: int = 0 # listed without a gap for this long
|
|
169
|
+
sites: Dict[str, bool] = field(default_factory=dict) # "google", "reddit", "amazon" -> got through last run?
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _live_fetch(url: str, timeout: float = 20, headers=None):
|
|
173
|
+
from .netio import http_get
|
|
174
|
+
return http_get(url, timeout=timeout, headers=headers)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _live_proxy(row: dict, run_hours: int) -> LiveProxy:
|
|
178
|
+
streak = row.get("streak") if type(row.get("streak")) is int else 0
|
|
179
|
+
return LiveProxy(
|
|
180
|
+
key=f"{row['ptype']} {row['proxy']}", ptype=row["ptype"], proxy=row["proxy"],
|
|
181
|
+
latency=int(row.get("latency") or 0),
|
|
182
|
+
exit_ip=row.get("exit_ip") or "", https=row.get("https"), anonymity=row.get("anonymity") or "",
|
|
183
|
+
country=row.get("country") or "", targets=dict(row.get("targets") or {}), asn=row.get("asn") or 0,
|
|
184
|
+
org=row.get("org") or "", hosting=row.get("hosting"), blocklisted=row.get("blocklisted"),
|
|
185
|
+
uptime_24h=row.get("uptime_24h"), uptime_7d=row.get("uptime_7d"), first_seen=row.get("first_seen") or "",
|
|
186
|
+
up_for_hours=streak * run_hours,
|
|
187
|
+
sites={k: v for k, v in (row.get("sites") or {}).items() if isinstance(v, bool)})
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
async def live_proxies_async(*, types: Iterable[str] = PROXY_TYPES, countries: Iterable[str] = (), https: bool = False,
|
|
191
|
+
anonymity: str = "", max_latency: int = 0, no_datacenter: bool = False,
|
|
192
|
+
no_blocklisted: bool = False, min_uptime: int = 0, works_on: Iterable[str] = (),
|
|
193
|
+
limit: int = 0) -> List[LiveProxy]:
|
|
194
|
+
"""The proxies from the hourly list that pass the filters, fastest first. Nothing is checked here: they
|
|
195
|
+
worked from GitHub's servers in the last run (at most an hour ago). find_proxies checks from your network.
|
|
196
|
+
|
|
197
|
+
Same filters as find_proxies, plus
|
|
198
|
+
min_uptime only proxies listed in at least this share (percent) of the week's runs, e.g. 90
|
|
199
|
+
works_on only proxies that got through to these sites in the last run: "google", "reddit", "amazon"
|
|
200
|
+
limit at most this many (0 = all)
|
|
201
|
+
"""
|
|
202
|
+
from .agent import AgentError, LiveSource
|
|
203
|
+
from .sites import SITE
|
|
204
|
+
|
|
205
|
+
types = _types(types)
|
|
206
|
+
unknown = [t for t in types if t not in PROXY_TYPES]
|
|
207
|
+
if unknown:
|
|
208
|
+
raise ValueError(f"unknown proxy type {unknown[0]!r}, use {', '.join(PROXY_TYPES)}")
|
|
209
|
+
if isinstance(countries, str):
|
|
210
|
+
countries = parse_countries(countries)
|
|
211
|
+
countries = {c.strip().upper() for c in countries}
|
|
212
|
+
bad = [c for c in countries if not re.fullmatch(r"[A-Z]{2}", c)]
|
|
213
|
+
if bad:
|
|
214
|
+
raise ValueError(f"countries are two-letter codes like DE or US, not {bad[0]!r}")
|
|
215
|
+
works_on = [str(s).lower() for s in ([works_on] if isinstance(works_on, str) else works_on)]
|
|
216
|
+
unknown_sites = [s for s in works_on if s not in SITE]
|
|
217
|
+
if unknown_sites:
|
|
218
|
+
raise ValueError(f"works_on takes {', '.join(SITE)}, not {unknown_sites[0]!r}")
|
|
219
|
+
opts = _options(types, 0, 0, https, countries, anonymity, max_latency, (), no_datacenter, no_blocklisted, 8.0, 1,
|
|
220
|
+
None)
|
|
221
|
+
try:
|
|
222
|
+
data = await LiveSource(fetch=_live_fetch).get()
|
|
223
|
+
except AgentError as e:
|
|
224
|
+
raise ConnectionError(str(e)) from None
|
|
225
|
+
found = sorted((_live_proxy(r, data.run_hours) for r in data.rows
|
|
226
|
+
if r.get("ptype") in types and isinstance(r.get("proxy"), str)),
|
|
227
|
+
key=lambda p: p.latency)
|
|
228
|
+
found = [p for p in found if opts.filters.accepts(p) and (not min_uptime or (p.uptime_7d or 0) >= min_uptime)
|
|
229
|
+
and all(p.sites.get(s) for s in works_on)]
|
|
230
|
+
return found[:limit] if limit else found
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def live_proxies(**kwargs) -> List[LiveProxy]:
|
|
234
|
+
"""Like live_proxies_async, just synchronous (starts its own event loop)."""
|
|
235
|
+
return asyncio.run(live_proxies_async(**kwargs))
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""SVG charts of the live list for the README: the working proxies over the last week, stacked by protocol,
|
|
2
|
+
and where they are. Written with every run next to the list, so the README shows today's numbers.
|
|
3
|
+
|
|
4
|
+
Plain SVG without scripts or web fonts – GitHub shows README images through a proxy that allows neither.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from datetime import datetime, timedelta
|
|
10
|
+
from typing import Dict, List, Optional, Sequence, Tuple
|
|
11
|
+
from xml.sax.saxutils import escape
|
|
12
|
+
|
|
13
|
+
from .pages import COUNTRIES
|
|
14
|
+
from .parsing import PROXY_TYPES
|
|
15
|
+
|
|
16
|
+
THEMES = {
|
|
17
|
+
"dark": {"bg": "#121113", "panel": "#1A191C", "line": "#2B292F", "text": "#EDEBE6", "muted": "#9A97A0",
|
|
18
|
+
"http": "#8DB8FF", "socks4": "#C7A6FF", "socks5": "#D4F77A"},
|
|
19
|
+
"light": {"bg": "#FFFFFF", "panel": "#F6F6F3", "line": "#DAD9D4", "text": "#121113", "muted": "#5F5C66",
|
|
20
|
+
"http": "#1F5FD1", "socks4": "#7442D6", "socks5": "#3F5A00"},
|
|
21
|
+
}
|
|
22
|
+
FONT = "-apple-system, BlinkMacSystemFont, 'Segoe UI', Helvetica, Arial, sans-serif"
|
|
23
|
+
WIDTH, HEIGHT = 880, 300
|
|
24
|
+
DAYS = 7
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _when(run: dict) -> Optional[datetime]:
|
|
28
|
+
try:
|
|
29
|
+
return datetime.fromisoformat(run["updated"])
|
|
30
|
+
except (KeyError, TypeError, ValueError):
|
|
31
|
+
return None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def recent_runs(runs: Sequence[dict], now: datetime, days: int = DAYS) -> List[Tuple[datetime, dict]]:
|
|
35
|
+
"""The runs of the last `days`, oldest first. Runs less than 20 minutes apart (a manual run right after the
|
|
36
|
+
hourly one) count once, the later one wins – otherwise the chart gets spikes that mean nothing."""
|
|
37
|
+
since = now - timedelta(days=days)
|
|
38
|
+
timed = sorted(((t, r) for r in runs if (t := _when(r)) and t > since and isinstance(r.get("by_type"), dict)),
|
|
39
|
+
key=lambda x: x[0])
|
|
40
|
+
kept: List[Tuple[datetime, dict]] = []
|
|
41
|
+
for t, r in timed:
|
|
42
|
+
if kept and t - kept[-1][0] < timedelta(minutes=20):
|
|
43
|
+
kept[-1] = (t, r)
|
|
44
|
+
else:
|
|
45
|
+
kept.append((t, r))
|
|
46
|
+
return kept
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _num(n: int) -> str:
|
|
50
|
+
return f"{n:,}"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def trend_svg(runs: Sequence[dict], now: datetime, theme: str = "dark") -> str:
|
|
54
|
+
"""Working proxies per run over the last week, stacked by protocol, with today's total as the headline."""
|
|
55
|
+
c = THEMES[theme]
|
|
56
|
+
points = recent_runs(runs, now)
|
|
57
|
+
left, right, top, bottom = 250, WIDTH - 24, 36, HEIGHT - 44
|
|
58
|
+
parts = [f'<svg xmlns="http://www.w3.org/2000/svg" width="{WIDTH}" height="{HEIGHT}" '
|
|
59
|
+
f'viewBox="0 0 {WIDTH} {HEIGHT}" '
|
|
60
|
+
f'font-family="{FONT}" role="img" aria-label="Working proxies over the last {DAYS} days">',
|
|
61
|
+
f'<rect width="{WIDTH}" height="{HEIGHT}" rx="16" fill="{c["bg"]}"/>']
|
|
62
|
+
last = points[-1][1] if points else {"total": 0, "by_type": {}}
|
|
63
|
+
total = last.get("total", 0)
|
|
64
|
+
parts += [f'<text x="28" y="58" fill="{c["muted"]}" font-size="14">Working right now</text>',
|
|
65
|
+
f'<text x="26" y="112" fill="{c["text"]}" font-size="52" font-weight="600" letter-spacing="-1.5">'
|
|
66
|
+
f'{_num(total)}</text>']
|
|
67
|
+
for i, t in enumerate(PROXY_TYPES):
|
|
68
|
+
y = 158 + i * 28
|
|
69
|
+
n = last.get("by_type", {}).get(t, 0)
|
|
70
|
+
parts += [f'<rect x="28" y="{y - 10}" width="10" height="10" rx="2" fill="{c[t]}"/>',
|
|
71
|
+
f'<text x="48" y="{y}" fill="{c["muted"]}" font-size="14">{t.upper()}</text>',
|
|
72
|
+
f'<text x="200" y="{y}" fill="{c["text"]}" font-size="14" text-anchor="end">{_num(n)}</text>']
|
|
73
|
+
# the x axis covers the week, or less while the history is younger than that
|
|
74
|
+
start = max(now - timedelta(days=DAYS), points[0][0]) if len(points) >= 2 else now - timedelta(days=DAYS)
|
|
75
|
+
start = min(start, now - timedelta(hours=6))
|
|
76
|
+
span = (now - start).total_seconds()
|
|
77
|
+
|
|
78
|
+
def x_of(t: datetime) -> float:
|
|
79
|
+
return left + (right - left) * (t - start).total_seconds() / span
|
|
80
|
+
|
|
81
|
+
parts.append(f'<text x="28" y="{HEIGHT - 28}" fill="{c["muted"]}" font-size="12">checked every hour'
|
|
82
|
+
f'{f", last {DAYS} days" if span >= timedelta(days=DAYS).total_seconds() - 3600 else ""}</text>')
|
|
83
|
+
parts.append(f'<line x1="{left}" y1="{bottom}" x2="{right}" y2="{bottom}" stroke="{c["line"]}" stroke-width="1"/>')
|
|
84
|
+
midnight = start.replace(hour=0, minute=0, second=0, microsecond=0) + timedelta(days=1)
|
|
85
|
+
while midnight < now: # a line at the start of every day, labelled with the day
|
|
86
|
+
x = x_of(midnight)
|
|
87
|
+
parts.append(f'<line x1="{x:.1f}" y1="{top}" x2="{x:.1f}" y2="{bottom}" stroke="{c["line"]}"/>')
|
|
88
|
+
if right - x > 40:
|
|
89
|
+
parts.append(f'<text x="{x + 6:.1f}" y="{bottom + 22}" fill="{c["muted"]}" font-size="12">'
|
|
90
|
+
f'{escape(midnight.strftime("%a %d"))}</text>')
|
|
91
|
+
midnight += timedelta(days=1)
|
|
92
|
+
if len(points) >= 2:
|
|
93
|
+
peak = max(r.get("total", 0) for _, r in points) or 1
|
|
94
|
+
|
|
95
|
+
def y_of(v: float) -> float:
|
|
96
|
+
return bottom - (bottom - top) * v / (peak * 1.08)
|
|
97
|
+
|
|
98
|
+
below = [0.0] * len(points)
|
|
99
|
+
for t in PROXY_TYPES: # stacked: each protocol on top of the ones before it
|
|
100
|
+
above = [b + (r.get("by_type", {}).get(t, 0) or 0) for b, (_, r) in zip(below, points)]
|
|
101
|
+
upper = " ".join(f"{x_of(when):.1f},{y_of(v):.1f}" for (when, _), v in zip(points, above))
|
|
102
|
+
lower = " ".join(f"{x_of(when):.1f},{y_of(v):.1f}" for (when, _), v in reversed(list(zip(points, below))))
|
|
103
|
+
parts.append(f'<polygon points="{upper} {lower}" fill="{c[t]}" fill-opacity="0.85"/>')
|
|
104
|
+
below = above
|
|
105
|
+
parts.append(f'<text x="{right}" y="{top - 12}" fill="{c["muted"]}" font-size="12" text-anchor="end">'
|
|
106
|
+
f'peak {_num(peak)}</text>')
|
|
107
|
+
else:
|
|
108
|
+
parts.append(f'<text x="{(left + right) / 2:.0f}" y="{(top + bottom) / 2:.0f}" fill="{c["muted"]}" '
|
|
109
|
+
f'font-size="14" text-anchor="middle">The chart fills up with the next runs</text>')
|
|
110
|
+
parts.append("</svg>")
|
|
111
|
+
return "\n".join(parts) + "\n"
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def countries_svg(countries: Dict[str, int], theme: str = "dark", limit: int = 10) -> str:
|
|
115
|
+
"""Where the working proxies are: the top countries as bars."""
|
|
116
|
+
c = THEMES[theme]
|
|
117
|
+
top = sorted(countries.items(), key=lambda kv: -kv[1])[:limit]
|
|
118
|
+
row, pad = 24, 28
|
|
119
|
+
height = pad * 2 + 22 + row * max(len(top), 1)
|
|
120
|
+
most = max((n for _, n in top), default=1) or 1
|
|
121
|
+
parts = [f'<svg xmlns="http://www.w3.org/2000/svg" width="{WIDTH}" height="{height}" '
|
|
122
|
+
f'viewBox="0 0 {WIDTH} {height}" '
|
|
123
|
+
f'font-family="{FONT}" role="img" aria-label="Countries with the most working proxies">',
|
|
124
|
+
f'<rect width="{WIDTH}" height="{height}" rx="16" fill="{c["bg"]}"/>',
|
|
125
|
+
f'<text x="28" y="{pad + 10}" fill="{c["muted"]}" font-size="14">Where they are</text>']
|
|
126
|
+
for i, (cc, n) in enumerate(top):
|
|
127
|
+
y = pad + 36 + i * row
|
|
128
|
+
width = (WIDTH - 250 - 90) * n / most
|
|
129
|
+
name = COUNTRIES.get(cc, cc)
|
|
130
|
+
name = name[4:] if name.startswith("the ") else name
|
|
131
|
+
parts += [f'<text x="28" y="{y + 11}" fill="{c["text"]}" font-size="13">{escape(name)}</text>',
|
|
132
|
+
f'<rect x="250" y="{y}" width="{max(width, 2):.1f}" height="14" rx="3" fill="{c["socks5"]}"/>',
|
|
133
|
+
f'<text x="{250 + width + 10:.1f}" y="{y + 11}" fill="{c["muted"]}" font-size="12">{_num(n)}</text>']
|
|
134
|
+
parts.append("</svg>")
|
|
135
|
+
return "\n".join(parts) + "\n"
|