proxy-scraper-cli 1.7.1__tar.gz → 1.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/PKG-INFO +33 -4
  2. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/README.md +32 -3
  3. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxy_scraper_cli.egg-info/PKG-INFO +33 -4
  4. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxy_scraper_cli.egg-info/SOURCES.txt +6 -0
  5. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/__init__.py +3 -2
  6. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/agent.py +21 -4
  7. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/api.py +99 -3
  8. proxy_scraper_cli-1.8.0/proxyscraper/charts.py +135 -0
  9. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/mcp_server.py +12 -1
  10. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/publish.py +99 -6
  11. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/site/index.html +286 -67
  12. proxy_scraper_cli-1.8.0/proxyscraper/sites.py +155 -0
  13. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_agent.py +9 -1
  14. proxy_scraper_cli-1.8.0/tests/test_charts.py +57 -0
  15. proxy_scraper_cli-1.8.0/tests/test_live_api.py +79 -0
  16. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_server.py +4 -2
  17. proxy_scraper_cli-1.8.0/tests/test_sites.py +141 -0
  18. proxy_scraper_cli-1.8.0/tests/test_uptime.py +129 -0
  19. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/LICENSE +0 -0
  20. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxy_scraper_cli.egg-info/dependency_links.txt +0 -0
  21. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxy_scraper_cli.egg-info/entry_points.txt +0 -0
  22. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxy_scraper_cli.egg-info/requires.txt +0 -0
  23. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxy_scraper_cli.egg-info/top_level.txt +0 -0
  24. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/__main__.py +0 -0
  25. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/app.py +0 -0
  26. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/asndb.py +0 -0
  27. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/blocklist.py +0 -0
  28. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/checker.py +0 -0
  29. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/cli.py +0 -0
  30. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/compat.py +0 -0
  31. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/completion.py +0 -0
  32. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/exporters.py +0 -0
  33. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/fetchcache.py +0 -0
  34. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/geo.py +0 -0
  35. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/geodb.py +0 -0
  36. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/handshake.py +0 -0
  37. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/history.py +0 -0
  38. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/judges.py +0 -0
  39. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/mcp_entry.py +0 -0
  40. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/netio.py +0 -0
  41. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/options.py +0 -0
  42. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/output.py +0 -0
  43. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/pages.py +0 -0
  44. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/parsing.py +0 -0
  45. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/paths.py +0 -0
  46. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/pipeline.py +0 -0
  47. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/preferences.py +0 -0
  48. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/__init__.py +0 -0
  49. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/core.py +0 -0
  50. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/http.py +0 -0
  51. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/pool.py +0 -0
  52. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/socks.py +0 -0
  53. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/status.py +0 -0
  54. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/server/upstream.py +0 -0
  55. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/site/apple-touch-icon.png +0 -0
  56. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/site/googleaac1161b7853c5b5.html +0 -0
  57. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/site/logo.png +0 -0
  58. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/site/og.png +0 -0
  59. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/sources.json +0 -0
  60. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/sources.py +0 -0
  61. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/targets.py +0 -0
  62. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/__init__.py +0 -0
  63. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/dashboard.py +0 -0
  64. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/keys.py +0 -0
  65. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/report.py +0 -0
  66. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/serve.py +0 -0
  67. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/widgets.py +0 -0
  68. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/proxyscraper/ui/wizard.py +0 -0
  69. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/pyproject.toml +0 -0
  70. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/setup.cfg +0 -0
  71. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_api.py +0 -0
  72. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_app.py +0 -0
  73. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_asndb.py +0 -0
  74. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_auth.py +0 -0
  75. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_blocklist.py +0 -0
  76. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_checker.py +0 -0
  77. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_compat.py +0 -0
  78. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_completion.py +0 -0
  79. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_discovery_growth.py +0 -0
  80. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_exporters.py +0 -0
  81. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_fetchcache.py +0 -0
  82. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_geodb.py +0 -0
  83. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_judges.py +0 -0
  84. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_learning.py +0 -0
  85. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_mcp_server.py +0 -0
  86. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_options.py +0 -0
  87. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_output_stdout.py +0 -0
  88. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_packaging.py +0 -0
  89. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_pages.py +0 -0
  90. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_parsing.py +0 -0
  91. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_paths.py +0 -0
  92. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_pipeline.py +0 -0
  93. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_proxy_sources.py +0 -0
  94. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_publish.py +0 -0
  95. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_robustness.py +0 -0
  96. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_serve_password.py +0 -0
  97. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_serve_refill.py +0 -0
  98. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_serve_v2.py +0 -0
  99. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_server_fixes.py +0 -0
  100. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_tamper.py +0 -0
  101. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_targets.py +0 -0
  102. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_ui.py +0 -0
  103. {proxy_scraper_cli-1.7.1 → proxy_scraper_cli-1.8.0}/tests/test_wizard.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: proxy-scraper-cli
3
- Version: 1.7.1
3
+ Version: 1.8.0
4
4
  Summary: Finds free HTTP, SOCKS4 and SOCKS5 proxies that actually work: 700+ sources, real checks, honeypot filtering and a rotating local proxy server.
5
5
  Author: Maximilian Feix
6
6
  License-Expression: MIT
@@ -175,7 +175,13 @@ Keys: <kbd>↑</kbd><kbd>↓</kbd> select · <kbd>Space</kbd> toggle · <kbd>1</
175
175
 
176
176
  Don't want to scan yourself? Every hour **GitHub Actions** runs the tool and publishes the hits to the [`proxy-list`](../../tree/proxy-list) branch – every entry worked in the last run, fastest first.
177
177
 
178
- **→ [Browse it on the website](https://maximilianfeix.github.io/proxy-scraper/)** – search, filter by type, country, HTTPS, provider and latency, see how long each proxy has been up, copy or download exactly the proxies you need. `streaks.json` on the branch has the number of runs in a row for every proxy.
178
+ <a href="https://maximilianfeix.github.io/proxy-scraper/"><picture>
179
+ <source media="(prefers-color-scheme: dark)" srcset="https://maximilianfeix.github.io/proxy-scraper/chart-dark.svg">
180
+ <source media="(prefers-color-scheme: light)" srcset="https://maximilianfeix.github.io/proxy-scraper/chart-light.svg">
181
+ <img src="https://maximilianfeix.github.io/proxy-scraper/chart-dark.svg" alt="Working proxies over the last days, stacked by protocol – redrawn with every run" width="100%">
182
+ </picture></a>
183
+
184
+ **→ [Browse it on the website](https://maximilianfeix.github.io/proxy-scraper/)** – search, filter by type, country, HTTPS, provider and latency, see how long each proxy has been up and how often it was on the list this week, copy or download exactly the proxies you need.
179
185
 
180
186
  Every protocol and country also has its own page with a plain download, e.g. [SOCKS5](https://maximilianfeix.github.io/proxy-scraper/socks5/) or [Germany](https://maximilianfeix.github.io/proxy-scraper/country/de/) (`country/de/proxies.txt`).
181
187
 
@@ -197,12 +203,22 @@ Every protocol and country also has its own page with a plain download, e.g. [SO
197
203
  | HTTP · SOCKS4 · SOCKS5 | `1.2.3.4:8080` | [http.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/http.txt) · [socks4.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks4.txt) · [socks5.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks5.txt) |
198
204
  | HTTPS-capable only | `type://ip:port` | [https.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/https.txt) |
199
205
  | Elite only | `type://ip:port` | [elite.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/elite.txt) |
200
- | With all details | latency, country, HTTPS, anonymity, exit IP | [proxies.json](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.json) · [proxies.csv](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.csv) |
206
+ | Gets through to Google · Reddit · Amazon (no captcha, no 403 in the last run) | `type://ip:port` | [google.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/google.txt) · [reddit.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/reddit.txt) · [amazon.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/amazon.txt) |
207
+ | Stable, on the list in 90 %+ of this week's runs | `type://ip:port` | [stable.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/stable.txt) |
208
+ | With all details | latency, country, HTTPS, anonymity, exit IP, uptime, sites | [proxies.json](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.json) · [proxies.csv](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.csv) |
209
+
210
+ <a href="https://maximilianfeix.github.io/proxy-scraper/"><picture>
211
+ <source media="(prefers-color-scheme: dark)" srcset="https://maximilianfeix.github.io/proxy-scraper/countries-dark.svg">
212
+ <source media="(prefers-color-scheme: light)" srcset="https://maximilianfeix.github.io/proxy-scraper/countries-light.svg">
213
+ <img src="https://maximilianfeix.github.io/proxy-scraper/countries-dark.svg" alt="Countries with the most working proxies right now" width="100%">
214
+ </picture></a>
201
215
 
202
216
  ```bash
203
217
  curl -s https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks5.txt | head
204
218
  ```
205
219
 
220
+ Every file is also on GitHub Pages, which sits behind a CDN, isn't rate limited like raw.githubusercontent and sends CORS headers, so it works straight from the browser: `https://maximilianfeix.github.io/proxy-scraper/socks5.txt`. [jsDelivr](https://cdn.jsdelivr.net/gh/maximilianfeix/proxy-scraper@proxy-list/) works too, but can lag behind by a few hours.
221
+
206
222
  Or let the tool start from it: `proxy-scraper --recheck live` downloads the list and checks it again from **your** network – about 30 seconds instead of a full scan (517 of 1,169 worked from here). With `--serve` you have a rotating proxy in under a minute.
207
223
 
208
224
  <a id="mcp"></a>
@@ -215,7 +231,7 @@ Or let the tool start from it: `proxy-scraper --recheck live` downloads the list
215
231
 
216
232
  | Tool | What it does |
217
233
  |---|---|
218
- | `get_proxies` | working proxies right now, from the hourly list – filter by protocol, country, HTTPS, elite, no datacenter, not blocklisted, stable, latency |
234
+ | `get_proxies` | working proxies right now, from the hourly list – filter by protocol, country, HTTPS, elite, no datacenter, not blocklisted, stable, uptime, latency, gets through to Google/Reddit/Amazon |
219
235
  | `check_proxies` | checks proxies from your own network, so they work from where your code runs (30–90 s, reports progress) |
220
236
  | `fetch_url` | loads a page through a verified proxy, switches proxies by itself when one fails, returns readable text – HTTPS only through proxies with verified TLS |
221
237
 
@@ -411,6 +427,19 @@ if __name__ == "__main__": # needed on macOS/Windows, the parser uses a process
411
427
 
412
428
  Same run as the command line – sources, learning, every check, result files – just without terminal output. Each result has `url`, `latency`, `exit_ip`, `https`, `anonymity`, `country`, `asn`, `org` and `hosting`. There's an async version of both (`find_proxies_async`, `check_proxies_async`).
413
429
 
430
+ Don't need a fresh scan? `live_proxies` takes the [live list](#live-list) instead – no checks, one download, done in about a second:
431
+
432
+ ```python
433
+ import itertools, requests
434
+ from proxyscraper import live_proxies
435
+
436
+ proxies = live_proxies(types=["socks5"], https=True, min_uptime=90) # the reliable ones this week
437
+ pool = itertools.cycle(p.url for p in proxies)
438
+ r = requests.get("https://api.ipify.org", proxies={"https": next(pool)}, timeout=15) # pip install "requests[socks]"
439
+ ```
440
+
441
+ Same filters as `find_proxies`, plus `min_uptime`, `works_on` (`["google"]`, `"reddit"`, `"amazon"`) and `limit`. Every result also has `uptime_24h`, `uptime_7d`, `first_seen`, `up_for_hours` and `sites`.
442
+
414
443
  <a id="proxy-server"></a>
415
444
 
416
445
  ## Rotating proxy server
@@ -135,7 +135,13 @@ Keys: <kbd>↑</kbd><kbd>↓</kbd> select · <kbd>Space</kbd> toggle · <kbd>1</
135
135
 
136
136
  Don't want to scan yourself? Every hour **GitHub Actions** runs the tool and publishes the hits to the [`proxy-list`](../../tree/proxy-list) branch – every entry worked in the last run, fastest first.
137
137
 
138
- **→ [Browse it on the website](https://maximilianfeix.github.io/proxy-scraper/)** – search, filter by type, country, HTTPS, provider and latency, see how long each proxy has been up, copy or download exactly the proxies you need. `streaks.json` on the branch has the number of runs in a row for every proxy.
138
+ <a href="https://maximilianfeix.github.io/proxy-scraper/"><picture>
139
+ <source media="(prefers-color-scheme: dark)" srcset="https://maximilianfeix.github.io/proxy-scraper/chart-dark.svg">
140
+ <source media="(prefers-color-scheme: light)" srcset="https://maximilianfeix.github.io/proxy-scraper/chart-light.svg">
141
+ <img src="https://maximilianfeix.github.io/proxy-scraper/chart-dark.svg" alt="Working proxies over the last days, stacked by protocol – redrawn with every run" width="100%">
142
+ </picture></a>
143
+
144
+ **→ [Browse it on the website](https://maximilianfeix.github.io/proxy-scraper/)** – search, filter by type, country, HTTPS, provider and latency, see how long each proxy has been up and how often it was on the list this week, copy or download exactly the proxies you need.
139
145
 
140
146
  Every protocol and country also has its own page with a plain download, e.g. [SOCKS5](https://maximilianfeix.github.io/proxy-scraper/socks5/) or [Germany](https://maximilianfeix.github.io/proxy-scraper/country/de/) (`country/de/proxies.txt`).
141
147
 
@@ -157,12 +163,22 @@ Every protocol and country also has its own page with a plain download, e.g. [SO
157
163
  | HTTP · SOCKS4 · SOCKS5 | `1.2.3.4:8080` | [http.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/http.txt) · [socks4.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks4.txt) · [socks5.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks5.txt) |
158
164
  | HTTPS-capable only | `type://ip:port` | [https.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/https.txt) |
159
165
  | Elite only | `type://ip:port` | [elite.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/elite.txt) |
160
- | With all details | latency, country, HTTPS, anonymity, exit IP | [proxies.json](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.json) · [proxies.csv](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.csv) |
166
+ | Gets through to Google · Reddit · Amazon (no captcha, no 403 in the last run) | `type://ip:port` | [google.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/google.txt) · [reddit.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/reddit.txt) · [amazon.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/amazon.txt) |
167
+ | Stable, on the list in 90 %+ of this week's runs | `type://ip:port` | [stable.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/stable.txt) |
168
+ | With all details | latency, country, HTTPS, anonymity, exit IP, uptime, sites | [proxies.json](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.json) · [proxies.csv](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.csv) |
169
+
170
+ <a href="https://maximilianfeix.github.io/proxy-scraper/"><picture>
171
+ <source media="(prefers-color-scheme: dark)" srcset="https://maximilianfeix.github.io/proxy-scraper/countries-dark.svg">
172
+ <source media="(prefers-color-scheme: light)" srcset="https://maximilianfeix.github.io/proxy-scraper/countries-light.svg">
173
+ <img src="https://maximilianfeix.github.io/proxy-scraper/countries-dark.svg" alt="Countries with the most working proxies right now" width="100%">
174
+ </picture></a>
161
175
 
162
176
  ```bash
163
177
  curl -s https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks5.txt | head
164
178
  ```
165
179
 
180
+ Every file is also on GitHub Pages, which sits behind a CDN, isn't rate limited like raw.githubusercontent and sends CORS headers, so it works straight from the browser: `https://maximilianfeix.github.io/proxy-scraper/socks5.txt`. [jsDelivr](https://cdn.jsdelivr.net/gh/maximilianfeix/proxy-scraper@proxy-list/) works too, but can lag behind by a few hours.
181
+
166
182
  Or let the tool start from it: `proxy-scraper --recheck live` downloads the list and checks it again from **your** network – about 30 seconds instead of a full scan (517 of 1,169 worked from here). With `--serve` you have a rotating proxy in under a minute.
167
183
 
168
184
  <a id="mcp"></a>
@@ -175,7 +191,7 @@ Or let the tool start from it: `proxy-scraper --recheck live` downloads the list
175
191
 
176
192
  | Tool | What it does |
177
193
  |---|---|
178
- | `get_proxies` | working proxies right now, from the hourly list – filter by protocol, country, HTTPS, elite, no datacenter, not blocklisted, stable, latency |
194
+ | `get_proxies` | working proxies right now, from the hourly list – filter by protocol, country, HTTPS, elite, no datacenter, not blocklisted, stable, uptime, latency, gets through to Google/Reddit/Amazon |
179
195
  | `check_proxies` | checks proxies from your own network, so they work from where your code runs (30–90 s, reports progress) |
180
196
  | `fetch_url` | loads a page through a verified proxy, switches proxies by itself when one fails, returns readable text – HTTPS only through proxies with verified TLS |
181
197
 
@@ -371,6 +387,19 @@ if __name__ == "__main__": # needed on macOS/Windows, the parser uses a process
371
387
 
372
388
  Same run as the command line – sources, learning, every check, result files – just without terminal output. Each result has `url`, `latency`, `exit_ip`, `https`, `anonymity`, `country`, `asn`, `org` and `hosting`. There's an async version of both (`find_proxies_async`, `check_proxies_async`).
373
389
 
390
+ Don't need a fresh scan? `live_proxies` takes the [live list](#live-list) instead – no checks, one download, done in about a second:
391
+
392
+ ```python
393
+ import itertools, requests
394
+ from proxyscraper import live_proxies
395
+
396
+ proxies = live_proxies(types=["socks5"], https=True, min_uptime=90) # the reliable ones this week
397
+ pool = itertools.cycle(p.url for p in proxies)
398
+ r = requests.get("https://api.ipify.org", proxies={"https": next(pool)}, timeout=15) # pip install "requests[socks]"
399
+ ```
400
+
401
+ Same filters as `find_proxies`, plus `min_uptime`, `works_on` (`["google"]`, `"reddit"`, `"amazon"`) and `limit`. Every result also has `uptime_24h`, `uptime_7d`, `first_seen`, `up_for_hours` and `sites`.
402
+
374
403
  <a id="proxy-server"></a>
375
404
 
376
405
  ## Rotating proxy server
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: proxy-scraper-cli
3
- Version: 1.7.1
3
+ Version: 1.8.0
4
4
  Summary: Finds free HTTP, SOCKS4 and SOCKS5 proxies that actually work: 700+ sources, real checks, honeypot filtering and a rotating local proxy server.
5
5
  Author: Maximilian Feix
6
6
  License-Expression: MIT
@@ -175,7 +175,13 @@ Keys: <kbd>↑</kbd><kbd>↓</kbd> select · <kbd>Space</kbd> toggle · <kbd>1</
175
175
 
176
176
  Don't want to scan yourself? Every hour **GitHub Actions** runs the tool and publishes the hits to the [`proxy-list`](../../tree/proxy-list) branch – every entry worked in the last run, fastest first.
177
177
 
178
- **→ [Browse it on the website](https://maximilianfeix.github.io/proxy-scraper/)** – search, filter by type, country, HTTPS, provider and latency, see how long each proxy has been up, copy or download exactly the proxies you need. `streaks.json` on the branch has the number of runs in a row for every proxy.
178
+ <a href="https://maximilianfeix.github.io/proxy-scraper/"><picture>
179
+ <source media="(prefers-color-scheme: dark)" srcset="https://maximilianfeix.github.io/proxy-scraper/chart-dark.svg">
180
+ <source media="(prefers-color-scheme: light)" srcset="https://maximilianfeix.github.io/proxy-scraper/chart-light.svg">
181
+ <img src="https://maximilianfeix.github.io/proxy-scraper/chart-dark.svg" alt="Working proxies over the last days, stacked by protocol – redrawn with every run" width="100%">
182
+ </picture></a>
183
+
184
+ **→ [Browse it on the website](https://maximilianfeix.github.io/proxy-scraper/)** – search, filter by type, country, HTTPS, provider and latency, see how long each proxy has been up and how often it was on the list this week, copy or download exactly the proxies you need.
179
185
 
180
186
  Every protocol and country also has its own page with a plain download, e.g. [SOCKS5](https://maximilianfeix.github.io/proxy-scraper/socks5/) or [Germany](https://maximilianfeix.github.io/proxy-scraper/country/de/) (`country/de/proxies.txt`).
181
187
 
@@ -197,12 +203,22 @@ Every protocol and country also has its own page with a plain download, e.g. [SO
197
203
  | HTTP · SOCKS4 · SOCKS5 | `1.2.3.4:8080` | [http.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/http.txt) · [socks4.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks4.txt) · [socks5.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks5.txt) |
198
204
  | HTTPS-capable only | `type://ip:port` | [https.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/https.txt) |
199
205
  | Elite only | `type://ip:port` | [elite.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/elite.txt) |
200
- | With all details | latency, country, HTTPS, anonymity, exit IP | [proxies.json](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.json) · [proxies.csv](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.csv) |
206
+ | Gets through to Google · Reddit · Amazon (no captcha, no 403 in the last run) | `type://ip:port` | [google.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/google.txt) · [reddit.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/reddit.txt) · [amazon.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/works-with/amazon.txt) |
207
+ | Stable, on the list in 90 %+ of this week's runs | `type://ip:port` | [stable.txt](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/stable.txt) |
208
+ | With all details | latency, country, HTTPS, anonymity, exit IP, uptime, sites | [proxies.json](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.json) · [proxies.csv](https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/proxies.csv) |
209
+
210
+ <a href="https://maximilianfeix.github.io/proxy-scraper/"><picture>
211
+ <source media="(prefers-color-scheme: dark)" srcset="https://maximilianfeix.github.io/proxy-scraper/countries-dark.svg">
212
+ <source media="(prefers-color-scheme: light)" srcset="https://maximilianfeix.github.io/proxy-scraper/countries-light.svg">
213
+ <img src="https://maximilianfeix.github.io/proxy-scraper/countries-dark.svg" alt="Countries with the most working proxies right now" width="100%">
214
+ </picture></a>
201
215
 
202
216
  ```bash
203
217
  curl -s https://raw.githubusercontent.com/maximilianfeix/proxy-scraper/proxy-list/socks5.txt | head
204
218
  ```
205
219
 
220
+ Every file is also on GitHub Pages, which sits behind a CDN, isn't rate limited like raw.githubusercontent and sends CORS headers, so it works straight from the browser: `https://maximilianfeix.github.io/proxy-scraper/socks5.txt`. [jsDelivr](https://cdn.jsdelivr.net/gh/maximilianfeix/proxy-scraper@proxy-list/) works too, but can lag behind by a few hours.
221
+
206
222
  Or let the tool start from it: `proxy-scraper --recheck live` downloads the list and checks it again from **your** network – about 30 seconds instead of a full scan (517 of 1,169 worked from here). With `--serve` you have a rotating proxy in under a minute.
207
223
 
208
224
  <a id="mcp"></a>
@@ -215,7 +231,7 @@ Or let the tool start from it: `proxy-scraper --recheck live` downloads the list
215
231
 
216
232
  | Tool | What it does |
217
233
  |---|---|
218
- | `get_proxies` | working proxies right now, from the hourly list – filter by protocol, country, HTTPS, elite, no datacenter, not blocklisted, stable, latency |
234
+ | `get_proxies` | working proxies right now, from the hourly list – filter by protocol, country, HTTPS, elite, no datacenter, not blocklisted, stable, uptime, latency, gets through to Google/Reddit/Amazon |
219
235
  | `check_proxies` | checks proxies from your own network, so they work from where your code runs (30–90 s, reports progress) |
220
236
  | `fetch_url` | loads a page through a verified proxy, switches proxies by itself when one fails, returns readable text – HTTPS only through proxies with verified TLS |
221
237
 
@@ -411,6 +427,19 @@ if __name__ == "__main__": # needed on macOS/Windows, the parser uses a process
411
427
 
412
428
  Same run as the command line – sources, learning, every check, result files – just without terminal output. Each result has `url`, `latency`, `exit_ip`, `https`, `anonymity`, `country`, `asn`, `org` and `hosting`. There's an async version of both (`find_proxies_async`, `check_proxies_async`).
413
429
 
430
+ Don't need a fresh scan? `live_proxies` takes the [live list](#live-list) instead – no checks, one download, done in about a second:
431
+
432
+ ```python
433
+ import itertools, requests
434
+ from proxyscraper import live_proxies
435
+
436
+ proxies = live_proxies(types=["socks5"], https=True, min_uptime=90) # the reliable ones this week
437
+ pool = itertools.cycle(p.url for p in proxies)
438
+ r = requests.get("https://api.ipify.org", proxies={"https": next(pool)}, timeout=15) # pip install "requests[socks]"
439
+ ```
440
+
441
+ Same filters as `find_proxies`, plus `min_uptime`, `works_on` (`["google"]`, `"reddit"`, `"amazon"`) and `limit`. Every result also has `uptime_24h`, `uptime_7d`, `first_seen`, `up_for_hours` and `sites`.
442
+
414
443
  <a id="proxy-server"></a>
415
444
 
416
445
  ## Rotating proxy server
@@ -14,6 +14,7 @@ proxyscraper/api.py
14
14
  proxyscraper/app.py
15
15
  proxyscraper/asndb.py
16
16
  proxyscraper/blocklist.py
17
+ proxyscraper/charts.py
17
18
  proxyscraper/checker.py
18
19
  proxyscraper/cli.py
19
20
  proxyscraper/compat.py
@@ -36,6 +37,7 @@ proxyscraper/paths.py
36
37
  proxyscraper/pipeline.py
37
38
  proxyscraper/preferences.py
38
39
  proxyscraper/publish.py
40
+ proxyscraper/sites.py
39
41
  proxyscraper/sources.json
40
42
  proxyscraper/sources.py
41
43
  proxyscraper/targets.py
@@ -64,6 +66,7 @@ tests/test_app.py
64
66
  tests/test_asndb.py
65
67
  tests/test_auth.py
66
68
  tests/test_blocklist.py
69
+ tests/test_charts.py
67
70
  tests/test_checker.py
68
71
  tests/test_compat.py
69
72
  tests/test_completion.py
@@ -73,6 +76,7 @@ tests/test_fetchcache.py
73
76
  tests/test_geodb.py
74
77
  tests/test_judges.py
75
78
  tests/test_learning.py
79
+ tests/test_live_api.py
76
80
  tests/test_mcp_server.py
77
81
  tests/test_options.py
78
82
  tests/test_output_stdout.py
@@ -89,7 +93,9 @@ tests/test_serve_refill.py
89
93
  tests/test_serve_v2.py
90
94
  tests/test_server.py
91
95
  tests/test_server_fixes.py
96
+ tests/test_sites.py
92
97
  tests/test_tamper.py
93
98
  tests/test_targets.py
94
99
  tests/test_ui.py
100
+ tests/test_uptime.py
95
101
  tests/test_wizard.py
@@ -1,12 +1,13 @@
1
1
  """proxy-scraper: collects public HTTP/SOCKS4/SOCKS5 proxies from hundreds of sources, checks them
2
2
  in parallel with its own protocol handshakes and learns which sources deliver good proxies."""
3
3
 
4
- __version__ = "1.7.1"
4
+ __version__ = "1.8.0"
5
5
 
6
6
 
7
7
  def __getattr__(name):
8
8
  # load the Python API only when needed – "import proxyscraper" should stay light (the CLI doesn't need it)
9
- if name in ("find_proxies", "find_proxies_async", "check_proxies", "check_proxies_async"):
9
+ if name in ("find_proxies", "find_proxies_async", "check_proxies", "check_proxies_async", "live_proxies",
10
+ "live_proxies_async", "LiveProxy"):
10
11
  from . import api
11
12
  return getattr(api, name)
12
13
  raise AttributeError(f"module 'proxyscraper' has no attribute {name!r}")
@@ -35,6 +35,7 @@ from .pages import SITE_URL
35
35
  from .parsing import PROXY_TYPES, make_key
36
36
  from .publish import RAW_BASE
37
37
  from .server import ProxyPool, RotatingServer
38
+ from .sites import SITES
38
39
 
39
40
  LIVE_BASES = (SITE_URL.rstrip("/"), RAW_BASE) # GitHub Pages first, the raw branch as the mirror
40
41
  CACHE_SECONDS = 300.0 # the list changes once an hour – no need to load 1 MB for every question
@@ -149,6 +150,15 @@ class LiveSource:
149
150
 
150
151
  # --------------------------------------------------------------------------- filtering
151
152
 
153
+ def _check_sites(names: Iterable[str]) -> List[str]:
154
+ known = [s.name for s in SITES]
155
+ names = [str(n).lower() for n in names]
156
+ for n in names:
157
+ if n not in known:
158
+ raise AgentError(f"works_on takes {', '.join(known)}, not {n!r}")
159
+ return names
160
+
161
+
152
162
  def _check_protocol(protocol: str) -> str:
153
163
  protocol = (protocol or "any").lower()
154
164
  if protocol not in PROTOCOLS:
@@ -168,20 +178,23 @@ def normalize_countries(countries: Iterable[str]) -> List[str]:
168
178
 
169
179
  def select(rows: Iterable[dict], protocol: str = "any", countries: Iterable[str] = (), https_only: bool = False,
170
180
  elite_only: bool = False, exclude_datacenter: bool = False, exclude_blocklisted: bool = False,
171
- stable_only: bool = False, max_latency_ms: int = 0, run_hours: int = 1) -> List[dict]:
181
+ stable_only: bool = False, max_latency_ms: int = 0, run_hours: int = 1, min_uptime: int = 0,
182
+ works_on: Iterable[str] = ()) -> List[dict]:
172
183
  """The rows that pass every filter, fastest first – the same filters as on the website."""
173
184
  return sorted(matching(rows, protocol, countries, https_only, elite_only, exclude_datacenter,
174
- exclude_blocklisted, stable_only, max_latency_ms, run_hours),
185
+ exclude_blocklisted, stable_only, max_latency_ms, run_hours, min_uptime, works_on),
175
186
  key=lambda r: r.get("latency") or 0)
176
187
 
177
188
 
178
189
  def matching(rows: Iterable[dict], protocol: str = "any", countries: Iterable[str] = (), https_only: bool = False,
179
190
  elite_only: bool = False, exclude_datacenter: bool = False, exclude_blocklisted: bool = False,
180
- stable_only: bool = False, max_latency_ms: int = 0, run_hours: int = 1) -> Iterator[dict]:
191
+ stable_only: bool = False, max_latency_ms: int = 0, run_hours: int = 1,
192
+ min_uptime: int = 0, works_on: Iterable[str] = ()) -> Iterator[dict]:
181
193
  """The rows that pass every filter, in list order – checks the filters before the first row is asked for."""
182
194
  protocol = _check_protocol(protocol)
183
195
  wanted = set(normalize_countries(countries))
184
196
  stable_runs = -(-24 // max(run_hours, 1)) # runs in a row that make a day
197
+ sites = _check_sites(works_on)
185
198
  return (r for r in rows
186
199
  if (protocol == "any" or r.get("ptype") == protocol)
187
200
  and (not wanted or r.get("country") in wanted)
@@ -190,7 +203,9 @@ def matching(rows: Iterable[dict], protocol: str = "any", countries: Iterable[st
190
203
  and (not exclude_datacenter or not r.get("hosting"))
191
204
  and (not exclude_blocklisted or not r.get("blocklisted"))
192
205
  and (not stable_only or (r.get("streak") or 0) >= stable_runs)
193
- and (not max_latency_ms or (r.get("latency") or 0) <= max_latency_ms))
206
+ and (not max_latency_ms or (r.get("latency") or 0) <= max_latency_ms)
207
+ and (not min_uptime or (r.get("uptime_7d") or 0) >= min_uptime)
208
+ and all((r.get("sites") or {}).get(s) for s in sites))
194
209
 
195
210
 
196
211
  def describe(item: Union[dict, CheckResult], run_hours: Optional[int] = None) -> dict:
@@ -212,6 +227,8 @@ def describe(item: Union[dict, CheckResult], run_hours: Optional[int] = None) ->
212
227
  "datacenter": item.get("hosting"),
213
228
  "blocklisted": item.get("blocklisted"),
214
229
  "up_for_hours": streak * run_hours if streak and run_hours else None,
230
+ "uptime_7d_percent": item.get("uptime_7d"),
231
+ "works_on": [name for name, ok in (item.get("sites") or {}).items() if ok],
215
232
  }
216
233
 
217
234
 
@@ -6,7 +6,14 @@
6
6
  for p in find_proxies(want=20, https=True, countries=["DE", "NL"]):
7
7
  print(p.url, p.latency, p.country)
8
8
 
9
- Behind it runs exactly the same as on the command line (sources, learning, honeypot and
9
+ Or skip the checking and take the list that GitHub Actions checks every hour:
10
+
11
+ from proxyscraper import live_proxies
12
+
13
+ for p in live_proxies(types=["socks5"], countries="DE", https=True, min_uptime=90):
14
+ print(p.url, p.latency, p.uptime_7d)
15
+
16
+ Behind find_proxies runs exactly the same as on the command line (sources, learning, honeypot and
10
17
  tampering checks, result files under results/), just without output in the terminal.
11
18
 
12
19
  Large lists are parsed in a process pool. On macOS and Windows it starts the worker processes
@@ -20,9 +27,11 @@ import asyncio
20
27
  import contextlib
21
28
  import io
22
29
  import os
30
+ import re
23
31
  import tempfile
24
32
  import threading
25
- from typing import Iterable, List, Optional
33
+ from dataclasses import dataclass, field
34
+ from typing import Dict, Iterable, List, Optional
26
35
 
27
36
  from rich.console import Console
28
37
 
@@ -32,13 +41,22 @@ from .parsing import PROXY_TYPES
32
41
  from .targets import parse_target
33
42
  from .ui import widgets
34
43
 
35
- __all__ = ["CheckResult", "check_proxies", "check_proxies_async", "find_proxies", "find_proxies_async"]
44
+ __all__ = ["CheckResult", "LiveProxy", "check_proxies", "check_proxies_async", "find_proxies", "find_proxies_async",
45
+ "live_proxies", "live_proxies_async"]
46
+
47
+
48
+ def _types(types: Iterable[str]) -> List[str]:
49
+ """["socks5"], "socks5" or "socks5,http" – a plain string shouldn't turn into its letters."""
50
+ if isinstance(types, str):
51
+ return [t.strip().lower() for t in types.split(",") if t.strip()]
52
+ return list(types)
36
53
 
37
54
 
38
55
  def _options(types: Iterable[str], want: int, limit: int, https: bool, countries: Iterable[str], anonymity: str,
39
56
  max_latency: int, targets: Iterable[str], no_datacenter: bool, no_blocklisted: bool, timeout: float,
40
57
  concurrency: int,
41
58
  recheck: Optional[str]) -> RunOptions:
59
+ types = _types(types)
42
60
  if isinstance(countries, str):
43
61
  countries = parse_countries(countries)
44
62
  if anonymity not in ("", "anonymous", "elite"):
@@ -137,3 +155,81 @@ async def check_proxies_async(proxies: Iterable[str], **kwargs) -> List[CheckRes
137
155
 
138
156
  def check_proxies(proxies: Iterable[str], **kwargs) -> List[CheckResult]:
139
157
  return asyncio.run(check_proxies_async(proxies, **kwargs))
158
+
159
+
160
+ # --------------------------------------------------------------------------- the hourly list
161
+
162
+ @dataclass
163
+ class LiveProxy(CheckResult):
164
+ """A proxy from the hourly list: the same fields as a CheckResult, plus how reliable it has been."""
165
+ uptime_24h: Optional[int] = None # share of today's hourly runs it was listed in, in percent
166
+ uptime_7d: Optional[int] = None # the same over the week
167
+ first_seen: str = "" # ISO time of the first run it was listed in
168
+ up_for_hours: int = 0 # listed without a gap for this long
169
+ sites: Dict[str, bool] = field(default_factory=dict) # "google", "reddit", "amazon" -> got through last run?
170
+
171
+
172
+ def _live_fetch(url: str, timeout: float = 20, headers=None):
173
+ from .netio import http_get
174
+ return http_get(url, timeout=timeout, headers=headers)
175
+
176
+
177
+ def _live_proxy(row: dict, run_hours: int) -> LiveProxy:
178
+ streak = row.get("streak") if type(row.get("streak")) is int else 0
179
+ return LiveProxy(
180
+ key=f"{row['ptype']} {row['proxy']}", ptype=row["ptype"], proxy=row["proxy"],
181
+ latency=int(row.get("latency") or 0),
182
+ exit_ip=row.get("exit_ip") or "", https=row.get("https"), anonymity=row.get("anonymity") or "",
183
+ country=row.get("country") or "", targets=dict(row.get("targets") or {}), asn=row.get("asn") or 0,
184
+ org=row.get("org") or "", hosting=row.get("hosting"), blocklisted=row.get("blocklisted"),
185
+ uptime_24h=row.get("uptime_24h"), uptime_7d=row.get("uptime_7d"), first_seen=row.get("first_seen") or "",
186
+ up_for_hours=streak * run_hours,
187
+ sites={k: v for k, v in (row.get("sites") or {}).items() if isinstance(v, bool)})
188
+
189
+
190
+ async def live_proxies_async(*, types: Iterable[str] = PROXY_TYPES, countries: Iterable[str] = (), https: bool = False,
191
+ anonymity: str = "", max_latency: int = 0, no_datacenter: bool = False,
192
+ no_blocklisted: bool = False, min_uptime: int = 0, works_on: Iterable[str] = (),
193
+ limit: int = 0) -> List[LiveProxy]:
194
+ """The proxies from the hourly list that pass the filters, fastest first. Nothing is checked here: they
195
+ worked from GitHub's servers in the last run (at most an hour ago). find_proxies checks from your network.
196
+
197
+ Same filters as find_proxies, plus
198
+ min_uptime only proxies listed in at least this share (percent) of the week's runs, e.g. 90
199
+ works_on only proxies that got through to these sites in the last run: "google", "reddit", "amazon"
200
+ limit at most this many (0 = all)
201
+ """
202
+ from .agent import AgentError, LiveSource
203
+ from .sites import SITE
204
+
205
+ types = _types(types)
206
+ unknown = [t for t in types if t not in PROXY_TYPES]
207
+ if unknown:
208
+ raise ValueError(f"unknown proxy type {unknown[0]!r}, use {', '.join(PROXY_TYPES)}")
209
+ if isinstance(countries, str):
210
+ countries = parse_countries(countries)
211
+ countries = {c.strip().upper() for c in countries}
212
+ bad = [c for c in countries if not re.fullmatch(r"[A-Z]{2}", c)]
213
+ if bad:
214
+ raise ValueError(f"countries are two-letter codes like DE or US, not {bad[0]!r}")
215
+ works_on = [str(s).lower() for s in ([works_on] if isinstance(works_on, str) else works_on)]
216
+ unknown_sites = [s for s in works_on if s not in SITE]
217
+ if unknown_sites:
218
+ raise ValueError(f"works_on takes {', '.join(SITE)}, not {unknown_sites[0]!r}")
219
+ opts = _options(types, 0, 0, https, countries, anonymity, max_latency, (), no_datacenter, no_blocklisted, 8.0, 1,
220
+ None)
221
+ try:
222
+ data = await LiveSource(fetch=_live_fetch).get()
223
+ except AgentError as e:
224
+ raise ConnectionError(str(e)) from None
225
+ found = sorted((_live_proxy(r, data.run_hours) for r in data.rows
226
+ if r.get("ptype") in types and isinstance(r.get("proxy"), str)),
227
+ key=lambda p: p.latency)
228
+ found = [p for p in found if opts.filters.accepts(p) and (not min_uptime or (p.uptime_7d or 0) >= min_uptime)
229
+ and all(p.sites.get(s) for s in works_on)]
230
+ return found[:limit] if limit else found
231
+
232
+
233
+ def live_proxies(**kwargs) -> List[LiveProxy]:
234
+ """Like live_proxies_async, just synchronous (starts its own event loop)."""
235
+ return asyncio.run(live_proxies_async(**kwargs))
@@ -0,0 +1,135 @@
1
+ """SVG charts of the live list for the README: the working proxies over the last week, stacked by protocol,
2
+ and where they are. Written with every run next to the list, so the README shows today's numbers.
3
+
4
+ Plain SVG without scripts or web fonts – GitHub shows README images through a proxy that allows neither.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from datetime import datetime, timedelta
10
+ from typing import Dict, List, Optional, Sequence, Tuple
11
+ from xml.sax.saxutils import escape
12
+
13
+ from .pages import COUNTRIES
14
+ from .parsing import PROXY_TYPES
15
+
16
+ THEMES = {
17
+ "dark": {"bg": "#121113", "panel": "#1A191C", "line": "#2B292F", "text": "#EDEBE6", "muted": "#9A97A0",
18
+ "http": "#8DB8FF", "socks4": "#C7A6FF", "socks5": "#D4F77A"},
19
+ "light": {"bg": "#FFFFFF", "panel": "#F6F6F3", "line": "#DAD9D4", "text": "#121113", "muted": "#5F5C66",
20
+ "http": "#1F5FD1", "socks4": "#7442D6", "socks5": "#3F5A00"},
21
+ }
22
+ FONT = "-apple-system, BlinkMacSystemFont, 'Segoe UI', Helvetica, Arial, sans-serif"
23
+ WIDTH, HEIGHT = 880, 300
24
+ DAYS = 7
25
+
26
+
27
+ def _when(run: dict) -> Optional[datetime]:
28
+ try:
29
+ return datetime.fromisoformat(run["updated"])
30
+ except (KeyError, TypeError, ValueError):
31
+ return None
32
+
33
+
34
+ def recent_runs(runs: Sequence[dict], now: datetime, days: int = DAYS) -> List[Tuple[datetime, dict]]:
35
+ """The runs of the last `days`, oldest first. Runs less than 20 minutes apart (a manual run right after the
36
+ hourly one) count once, the later one wins – otherwise the chart gets spikes that mean nothing."""
37
+ since = now - timedelta(days=days)
38
+ timed = sorted(((t, r) for r in runs if (t := _when(r)) and t > since and isinstance(r.get("by_type"), dict)),
39
+ key=lambda x: x[0])
40
+ kept: List[Tuple[datetime, dict]] = []
41
+ for t, r in timed:
42
+ if kept and t - kept[-1][0] < timedelta(minutes=20):
43
+ kept[-1] = (t, r)
44
+ else:
45
+ kept.append((t, r))
46
+ return kept
47
+
48
+
49
+ def _num(n: int) -> str:
50
+ return f"{n:,}"
51
+
52
+
53
+ def trend_svg(runs: Sequence[dict], now: datetime, theme: str = "dark") -> str:
54
+ """Working proxies per run over the last week, stacked by protocol, with today's total as the headline."""
55
+ c = THEMES[theme]
56
+ points = recent_runs(runs, now)
57
+ left, right, top, bottom = 250, WIDTH - 24, 36, HEIGHT - 44
58
+ parts = [f'<svg xmlns="http://www.w3.org/2000/svg" width="{WIDTH}" height="{HEIGHT}" '
59
+ f'viewBox="0 0 {WIDTH} {HEIGHT}" '
60
+ f'font-family="{FONT}" role="img" aria-label="Working proxies over the last {DAYS} days">',
61
+ f'<rect width="{WIDTH}" height="{HEIGHT}" rx="16" fill="{c["bg"]}"/>']
62
+ last = points[-1][1] if points else {"total": 0, "by_type": {}}
63
+ total = last.get("total", 0)
64
+ parts += [f'<text x="28" y="58" fill="{c["muted"]}" font-size="14">Working right now</text>',
65
+ f'<text x="26" y="112" fill="{c["text"]}" font-size="52" font-weight="600" letter-spacing="-1.5">'
66
+ f'{_num(total)}</text>']
67
+ for i, t in enumerate(PROXY_TYPES):
68
+ y = 158 + i * 28
69
+ n = last.get("by_type", {}).get(t, 0)
70
+ parts += [f'<rect x="28" y="{y - 10}" width="10" height="10" rx="2" fill="{c[t]}"/>',
71
+ f'<text x="48" y="{y}" fill="{c["muted"]}" font-size="14">{t.upper()}</text>',
72
+ f'<text x="200" y="{y}" fill="{c["text"]}" font-size="14" text-anchor="end">{_num(n)}</text>']
73
+ # the x axis covers the week, or less while the history is younger than that
74
+ start = max(now - timedelta(days=DAYS), points[0][0]) if len(points) >= 2 else now - timedelta(days=DAYS)
75
+ start = min(start, now - timedelta(hours=6))
76
+ span = (now - start).total_seconds()
77
+
78
+ def x_of(t: datetime) -> float:
79
+ return left + (right - left) * (t - start).total_seconds() / span
80
+
81
+ parts.append(f'<text x="28" y="{HEIGHT - 28}" fill="{c["muted"]}" font-size="12">checked every hour'
82
+ f'{f", last {DAYS} days" if span >= timedelta(days=DAYS).total_seconds() - 3600 else ""}</text>')
83
+ parts.append(f'<line x1="{left}" y1="{bottom}" x2="{right}" y2="{bottom}" stroke="{c["line"]}" stroke-width="1"/>')
84
+ midnight = start.replace(hour=0, minute=0, second=0, microsecond=0) + timedelta(days=1)
85
+ while midnight < now: # a line at the start of every day, labelled with the day
86
+ x = x_of(midnight)
87
+ parts.append(f'<line x1="{x:.1f}" y1="{top}" x2="{x:.1f}" y2="{bottom}" stroke="{c["line"]}"/>')
88
+ if right - x > 40:
89
+ parts.append(f'<text x="{x + 6:.1f}" y="{bottom + 22}" fill="{c["muted"]}" font-size="12">'
90
+ f'{escape(midnight.strftime("%a %d"))}</text>')
91
+ midnight += timedelta(days=1)
92
+ if len(points) >= 2:
93
+ peak = max(r.get("total", 0) for _, r in points) or 1
94
+
95
+ def y_of(v: float) -> float:
96
+ return bottom - (bottom - top) * v / (peak * 1.08)
97
+
98
+ below = [0.0] * len(points)
99
+ for t in PROXY_TYPES: # stacked: each protocol on top of the ones before it
100
+ above = [b + (r.get("by_type", {}).get(t, 0) or 0) for b, (_, r) in zip(below, points)]
101
+ upper = " ".join(f"{x_of(when):.1f},{y_of(v):.1f}" for (when, _), v in zip(points, above))
102
+ lower = " ".join(f"{x_of(when):.1f},{y_of(v):.1f}" for (when, _), v in reversed(list(zip(points, below))))
103
+ parts.append(f'<polygon points="{upper} {lower}" fill="{c[t]}" fill-opacity="0.85"/>')
104
+ below = above
105
+ parts.append(f'<text x="{right}" y="{top - 12}" fill="{c["muted"]}" font-size="12" text-anchor="end">'
106
+ f'peak {_num(peak)}</text>')
107
+ else:
108
+ parts.append(f'<text x="{(left + right) / 2:.0f}" y="{(top + bottom) / 2:.0f}" fill="{c["muted"]}" '
109
+ f'font-size="14" text-anchor="middle">The chart fills up with the next runs</text>')
110
+ parts.append("</svg>")
111
+ return "\n".join(parts) + "\n"
112
+
113
+
114
+ def countries_svg(countries: Dict[str, int], theme: str = "dark", limit: int = 10) -> str:
115
+ """Where the working proxies are: the top countries as bars."""
116
+ c = THEMES[theme]
117
+ top = sorted(countries.items(), key=lambda kv: -kv[1])[:limit]
118
+ row, pad = 24, 28
119
+ height = pad * 2 + 22 + row * max(len(top), 1)
120
+ most = max((n for _, n in top), default=1) or 1
121
+ parts = [f'<svg xmlns="http://www.w3.org/2000/svg" width="{WIDTH}" height="{height}" '
122
+ f'viewBox="0 0 {WIDTH} {height}" '
123
+ f'font-family="{FONT}" role="img" aria-label="Countries with the most working proxies">',
124
+ f'<rect width="{WIDTH}" height="{height}" rx="16" fill="{c["bg"]}"/>',
125
+ f'<text x="28" y="{pad + 10}" fill="{c["muted"]}" font-size="14">Where they are</text>']
126
+ for i, (cc, n) in enumerate(top):
127
+ y = pad + 36 + i * row
128
+ width = (WIDTH - 250 - 90) * n / most
129
+ name = COUNTRIES.get(cc, cc)
130
+ name = name[4:] if name.startswith("the ") else name
131
+ parts += [f'<text x="28" y="{y + 11}" fill="{c["text"]}" font-size="13">{escape(name)}</text>',
132
+ f'<rect x="250" y="{y}" width="{max(width, 2):.1f}" height="14" rx="3" fill="{c["socks5"]}"/>',
133
+ f'<text x="{250 + width + 10:.1f}" y="{y + 11}" fill="{c["muted"]}" font-size="12">{_num(n)}</text>']
134
+ parts.append("</svg>")
135
+ return "\n".join(parts) + "\n"