proxy-scraper-cli 1.7.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- proxy_scraper_cli-1.7.1.dist-info/METADATA +794 -0
- proxy_scraper_cli-1.7.1.dist-info/RECORD +57 -0
- proxy_scraper_cli-1.7.1.dist-info/WHEEL +5 -0
- proxy_scraper_cli-1.7.1.dist-info/entry_points.txt +4 -0
- proxy_scraper_cli-1.7.1.dist-info/licenses/LICENSE +21 -0
- proxy_scraper_cli-1.7.1.dist-info/top_level.txt +1 -0
- proxyscraper/__init__.py +12 -0
- proxyscraper/__main__.py +7 -0
- proxyscraper/agent.py +635 -0
- proxyscraper/api.py +139 -0
- proxyscraper/app.py +595 -0
- proxyscraper/asndb.py +181 -0
- proxyscraper/blocklist.py +92 -0
- proxyscraper/checker.py +514 -0
- proxyscraper/cli.py +269 -0
- proxyscraper/compat.py +83 -0
- proxyscraper/completion.py +196 -0
- proxyscraper/exporters.py +118 -0
- proxyscraper/fetchcache.py +103 -0
- proxyscraper/geo.py +150 -0
- proxyscraper/geodb.py +143 -0
- proxyscraper/handshake.py +153 -0
- proxyscraper/history.py +106 -0
- proxyscraper/judges.py +159 -0
- proxyscraper/mcp_entry.py +34 -0
- proxyscraper/mcp_server.py +220 -0
- proxyscraper/netio.py +167 -0
- proxyscraper/options.py +274 -0
- proxyscraper/output.py +181 -0
- proxyscraper/pages.py +212 -0
- proxyscraper/parsing.py +192 -0
- proxyscraper/paths.py +55 -0
- proxyscraper/pipeline.py +434 -0
- proxyscraper/preferences.py +24 -0
- proxyscraper/publish.py +236 -0
- proxyscraper/server/__init__.py +41 -0
- proxyscraper/server/core.py +518 -0
- proxyscraper/server/http.py +164 -0
- proxyscraper/server/pool.py +195 -0
- proxyscraper/server/socks.py +65 -0
- proxyscraper/server/status.py +108 -0
- proxyscraper/server/upstream.py +119 -0
- proxyscraper/site/apple-touch-icon.png +0 -0
- proxyscraper/site/googleaac1161b7853c5b5.html +1 -0
- proxyscraper/site/index.html +934 -0
- proxyscraper/site/logo.png +0 -0
- proxyscraper/site/og.png +0 -0
- proxyscraper/sources.json +395 -0
- proxyscraper/sources.py +491 -0
- proxyscraper/targets.py +76 -0
- proxyscraper/ui/__init__.py +54 -0
- proxyscraper/ui/dashboard.py +344 -0
- proxyscraper/ui/keys.py +82 -0
- proxyscraper/ui/report.py +209 -0
- proxyscraper/ui/serve.py +142 -0
- proxyscraper/ui/widgets.py +250 -0
- proxyscraper/ui/wizard.py +595 -0
proxyscraper/options.py
ADDED
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
"""All settings of a run in one place – whether they come from the command line or
|
|
2
|
+
from the setup wizard."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import shlex
|
|
7
|
+
from dataclasses import dataclass, field
|
|
8
|
+
from typing import List, Optional, Set
|
|
9
|
+
|
|
10
|
+
from .checker import ANONYMITY_RANK, CheckResult
|
|
11
|
+
from .exporters import EXPORTERS
|
|
12
|
+
from .parsing import PROXY_TYPES
|
|
13
|
+
from .paths import is_checkout
|
|
14
|
+
from .server.pool import STRATEGIES
|
|
15
|
+
from .targets import target_label
|
|
16
|
+
|
|
17
|
+
DEFAULT_CONCURRENCY = 2000
|
|
18
|
+
DEFAULT_TIMEOUT = 8.0
|
|
19
|
+
DEFAULT_CONNECT_TIMEOUT = 4.0
|
|
20
|
+
DEFAULT_DISCOVER_REPOS = 400
|
|
21
|
+
DEFAULT_SERVE_PORT = 8899
|
|
22
|
+
STDOUT = "-" # -o -: hits to stdout
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def parse_countries(value: Optional[str]) -> Set[str]:
|
|
26
|
+
"""'de, at,CH' -> {'DE', 'AT', 'CH'}"""
|
|
27
|
+
return {c.strip().upper() for c in (value or "").split(",") if c.strip()}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class Filters:
|
|
32
|
+
countries: Set[str] = field(default_factory=set)
|
|
33
|
+
https_only: bool = False
|
|
34
|
+
min_anonymity: str = ""
|
|
35
|
+
max_latency: int = 0
|
|
36
|
+
targets: List[str] = field(default_factory=list) # target sites every proxy has to reach
|
|
37
|
+
no_datacenter: bool = False # only exits that are not (recognizably) in a datacenter
|
|
38
|
+
no_blocklisted: bool = False # only exits that are not on the SpamCop blocklist
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def needs_details(self) -> bool:
|
|
42
|
+
# anonymity comes from the confirmation (always runs) – HTTPS and target sites need the detail test
|
|
43
|
+
return self.https_only or bool(self.targets)
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def active(self) -> bool:
|
|
47
|
+
return bool(self.countries or self.https_only or self.min_anonymity or self.max_latency or self.targets
|
|
48
|
+
or self.no_datacenter or self.no_blocklisted)
|
|
49
|
+
|
|
50
|
+
def accepts(self, r: CheckResult) -> bool:
|
|
51
|
+
if self.max_latency and r.latency > self.max_latency:
|
|
52
|
+
return False
|
|
53
|
+
if self.https_only and r.https is not True:
|
|
54
|
+
return False
|
|
55
|
+
if self.min_anonymity and ANONYMITY_RANK.get(r.anonymity, -1) < ANONYMITY_RANK[self.min_anonymity]:
|
|
56
|
+
return False
|
|
57
|
+
if self.countries and r.country not in self.countries:
|
|
58
|
+
return False
|
|
59
|
+
if self.no_datacenter and r.hosting:
|
|
60
|
+
return False
|
|
61
|
+
if self.no_blocklisted and r.blocklisted:
|
|
62
|
+
return False
|
|
63
|
+
return all(r.targets.get(url) for url in self.targets) # reached every requested target site
|
|
64
|
+
|
|
65
|
+
def may_pass(self, r: CheckResult) -> bool:
|
|
66
|
+
"""Can `r` still pass the filters? HTTPS is still unknown at this point, possibly the country too.
|
|
67
|
+
|
|
68
|
+
Whatever is sure to fail already doesn't need an expensive HTTPS test any more.
|
|
69
|
+
"""
|
|
70
|
+
if self.max_latency and r.latency > self.max_latency:
|
|
71
|
+
return False
|
|
72
|
+
if self.min_anonymity and ANONYMITY_RANK.get(r.anonymity, -1) < ANONYMITY_RANK[self.min_anonymity]:
|
|
73
|
+
return False
|
|
74
|
+
if self.no_datacenter and r.hosting:
|
|
75
|
+
return False
|
|
76
|
+
if self.no_blocklisted and r.blocklisted:
|
|
77
|
+
return False
|
|
78
|
+
# country still unknown -> may still match
|
|
79
|
+
return not self.countries or not r.country or r.country in self.countries
|
|
80
|
+
|
|
81
|
+
def describe(self) -> str:
|
|
82
|
+
parts = []
|
|
83
|
+
if self.countries:
|
|
84
|
+
parts.append("country " + ",".join(sorted(self.countries)))
|
|
85
|
+
if self.https_only:
|
|
86
|
+
parts.append("HTTPS only")
|
|
87
|
+
if self.min_anonymity:
|
|
88
|
+
parts.append(f"min. {self.min_anonymity}")
|
|
89
|
+
if self.max_latency:
|
|
90
|
+
parts.append(f"≤ {self.max_latency} ms")
|
|
91
|
+
if self.targets:
|
|
92
|
+
parts.append("target " + ", ".join(target_label(u, self.targets) for u in self.targets))
|
|
93
|
+
if self.no_datacenter:
|
|
94
|
+
parts.append("no datacenters")
|
|
95
|
+
if self.no_blocklisted:
|
|
96
|
+
parts.append("not blocklisted")
|
|
97
|
+
return " · ".join(parts)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@dataclass
|
|
101
|
+
class RunOptions:
|
|
102
|
+
types: List[str] = field(default_factory=lambda: list(PROXY_TYPES))
|
|
103
|
+
filters: Filters = field(default_factory=Filters)
|
|
104
|
+
want: int = 0
|
|
105
|
+
limit: int = 0
|
|
106
|
+
fast: bool = False
|
|
107
|
+
no_geo: bool = False
|
|
108
|
+
recheck: Optional[str] = None
|
|
109
|
+
output: Optional[str] = None
|
|
110
|
+
exports: List[str] = field(default_factory=list) # extra formats, see exporters.py
|
|
111
|
+
concurrency: int = DEFAULT_CONCURRENCY
|
|
112
|
+
timeout: float = DEFAULT_TIMEOUT
|
|
113
|
+
connect_timeout: float = DEFAULT_CONNECT_TIMEOUT
|
|
114
|
+
discover: bool = False
|
|
115
|
+
no_discover: bool = False
|
|
116
|
+
discover_repos: int = DEFAULT_DISCOVER_REPOS
|
|
117
|
+
all_sources: bool = False
|
|
118
|
+
no_cache: bool = False # reload every list instead of taking unchanged ones from the cache
|
|
119
|
+
no_dnsbl: bool = False # skip the blocklist lookup of the exit IPs
|
|
120
|
+
serve: int = 0 # port of the rotating proxy server after the run, 0 = off
|
|
121
|
+
rotate: str = "weighted" # strategy of the proxy server, see server/pool.py
|
|
122
|
+
sticky: int = 0 # seconds a target site keeps the same proxy (0 = new one for every connection)
|
|
123
|
+
serve_host: str = "127.0.0.1" # address of the proxy server; anything else is reachable from outside
|
|
124
|
+
serve_password: str = field(default="", repr=False) # required from clients if set; never written to argv
|
|
125
|
+
serve_refill: float = 0 # hours between background refills of the proxy server's pool, 0 = off
|
|
126
|
+
|
|
127
|
+
def __post_init__(self) -> None:
|
|
128
|
+
unknown = set(self.types) - set(PROXY_TYPES)
|
|
129
|
+
if unknown:
|
|
130
|
+
raise ValueError(f"unknown proxy types: {', '.join(sorted(unknown))}")
|
|
131
|
+
# the types are really a set: fixed order, no duplicates.
|
|
132
|
+
# Otherwise "--types socks5 http" and "--types http socks5" would give different settings.
|
|
133
|
+
self.types = [t for t in PROXY_TYPES if t in self.types]
|
|
134
|
+
if not self.types:
|
|
135
|
+
raise ValueError("at least one proxy type needed") # else to_argv() gives "--types" without a value
|
|
136
|
+
if self.concurrency < 1:
|
|
137
|
+
raise ValueError("concurrency must be at least 1")
|
|
138
|
+
if self.timeout <= 0 or self.connect_timeout <= 0:
|
|
139
|
+
raise ValueError("timeouts must be greater than 0")
|
|
140
|
+
if min(self.want, self.limit, self.discover_repos, self.filters.max_latency) < 0:
|
|
141
|
+
raise ValueError("amounts and latency must not be negative")
|
|
142
|
+
unknown_exports = set(self.exports) - set(EXPORTERS)
|
|
143
|
+
if unknown_exports:
|
|
144
|
+
raise ValueError(f"unknown export formats: {', '.join(sorted(unknown_exports))}")
|
|
145
|
+
self.exports = [e for e in EXPORTERS if e in self.exports]
|
|
146
|
+
if self.rotate not in STRATEGIES:
|
|
147
|
+
raise ValueError(f"unknown strategy: {self.rotate}")
|
|
148
|
+
if self.sticky < 0:
|
|
149
|
+
raise ValueError("--sticky must not be negative")
|
|
150
|
+
if self.serve_refill < 0:
|
|
151
|
+
raise ValueError("--serve-refill must not be negative")
|
|
152
|
+
if not 0 <= self.serve <= 65535:
|
|
153
|
+
raise ValueError("port must be between 1 and 65535")
|
|
154
|
+
|
|
155
|
+
@property
|
|
156
|
+
def details(self) -> bool:
|
|
157
|
+
"""HTTPS test – can be switched off (--fast), unless the HTTPS filter needs it."""
|
|
158
|
+
return not self.fast or self.filters.needs_details
|
|
159
|
+
|
|
160
|
+
@property
|
|
161
|
+
def check_timeout(self) -> float:
|
|
162
|
+
"""Timeout of the basic check: whatever exceeds the latency limit drops out anyway –
|
|
163
|
+
nobody needs to wait that long. With 2000 parallel slots that's a lot more throughput."""
|
|
164
|
+
if self.filters.max_latency:
|
|
165
|
+
return min(self.timeout, self.filters.max_latency / 1000)
|
|
166
|
+
return self.timeout
|
|
167
|
+
|
|
168
|
+
@property
|
|
169
|
+
def check_connect_timeout(self) -> float:
|
|
170
|
+
return min(self.connect_timeout, self.check_timeout)
|
|
171
|
+
|
|
172
|
+
@property
|
|
173
|
+
def geo(self) -> bool:
|
|
174
|
+
return not self.no_geo or bool(self.filters.countries)
|
|
175
|
+
|
|
176
|
+
@classmethod
|
|
177
|
+
def from_args(cls, args) -> "RunOptions":
|
|
178
|
+
return cls(
|
|
179
|
+
types=list(args.types),
|
|
180
|
+
filters=Filters(
|
|
181
|
+
countries=parse_countries(args.country),
|
|
182
|
+
https_only=args.https_only,
|
|
183
|
+
min_anonymity=args.anonymity or "",
|
|
184
|
+
max_latency=args.max_latency,
|
|
185
|
+
targets=list(dict.fromkeys(args.target or [])),
|
|
186
|
+
no_datacenter=args.no_datacenter,
|
|
187
|
+
no_blocklisted=args.no_blocklisted,
|
|
188
|
+
),
|
|
189
|
+
want=args.want,
|
|
190
|
+
limit=args.limit,
|
|
191
|
+
fast=args.fast,
|
|
192
|
+
no_geo=args.no_geo,
|
|
193
|
+
recheck=args.recheck,
|
|
194
|
+
output=args.output,
|
|
195
|
+
exports=list(args.export or []),
|
|
196
|
+
concurrency=args.concurrency,
|
|
197
|
+
timeout=args.timeout,
|
|
198
|
+
connect_timeout=args.connect_timeout,
|
|
199
|
+
discover=args.discover,
|
|
200
|
+
no_discover=args.no_discover,
|
|
201
|
+
discover_repos=args.discover_repos,
|
|
202
|
+
all_sources=args.all_sources,
|
|
203
|
+
no_cache=args.no_cache,
|
|
204
|
+
no_dnsbl=args.no_dnsbl,
|
|
205
|
+
serve=args.serve,
|
|
206
|
+
rotate=args.rotate,
|
|
207
|
+
sticky=args.sticky,
|
|
208
|
+
serve_host=args.serve_host,
|
|
209
|
+
serve_password=args.serve_password,
|
|
210
|
+
serve_refill=args.serve_refill,
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
def to_argv(self) -> List[str]:
|
|
214
|
+
"""Command line arguments that produce exactly these settings (only deviations from the defaults)."""
|
|
215
|
+
argv: List[str] = []
|
|
216
|
+
if self.types != list(PROXY_TYPES):
|
|
217
|
+
argv += ["--types", *self.types]
|
|
218
|
+
f = self.filters
|
|
219
|
+
if f.countries:
|
|
220
|
+
argv += ["--country", ",".join(sorted(f.countries))]
|
|
221
|
+
if f.https_only:
|
|
222
|
+
argv.append("--https-only")
|
|
223
|
+
if f.min_anonymity:
|
|
224
|
+
argv += ["--anonymity", f.min_anonymity]
|
|
225
|
+
if f.max_latency:
|
|
226
|
+
argv += ["--max-latency", str(f.max_latency)]
|
|
227
|
+
for url in f.targets:
|
|
228
|
+
argv += ["--target", url]
|
|
229
|
+
_flag(argv, "--no-datacenter", f.no_datacenter)
|
|
230
|
+
_flag(argv, "--no-blocklisted", f.no_blocklisted)
|
|
231
|
+
_opt(argv, "--want", self.want, 0)
|
|
232
|
+
_opt(argv, "--limit", self.limit, 0)
|
|
233
|
+
_flag(argv, "--fast", self.fast)
|
|
234
|
+
_flag(argv, "--no-geo", self.no_geo)
|
|
235
|
+
if self.recheck is not None:
|
|
236
|
+
argv += ["--recheck", self.recheck] if self.recheck else ["--recheck"]
|
|
237
|
+
if self.output:
|
|
238
|
+
argv += ["--output", self.output]
|
|
239
|
+
if self.exports:
|
|
240
|
+
argv += ["--export", ",".join(self.exports)]
|
|
241
|
+
_opt(argv, "--concurrency", self.concurrency, DEFAULT_CONCURRENCY)
|
|
242
|
+
_opt(argv, "--timeout", self.timeout, DEFAULT_TIMEOUT)
|
|
243
|
+
_opt(argv, "--connect-timeout", self.connect_timeout, DEFAULT_CONNECT_TIMEOUT)
|
|
244
|
+
_flag(argv, "--discover", self.discover)
|
|
245
|
+
_flag(argv, "--no-discover", self.no_discover)
|
|
246
|
+
_opt(argv, "--discover-repos", self.discover_repos, DEFAULT_DISCOVER_REPOS)
|
|
247
|
+
_flag(argv, "--all-sources", self.all_sources)
|
|
248
|
+
_flag(argv, "--no-cache", self.no_cache)
|
|
249
|
+
_flag(argv, "--no-dnsbl", self.no_dnsbl)
|
|
250
|
+
if self.serve:
|
|
251
|
+
argv += ["--serve"] if self.serve == DEFAULT_SERVE_PORT else ["--serve", str(self.serve)]
|
|
252
|
+
if self.rotate != "weighted":
|
|
253
|
+
argv += ["--rotate", self.rotate]
|
|
254
|
+
if self.serve_host != "127.0.0.1":
|
|
255
|
+
argv += ["--serve-host", self.serve_host]
|
|
256
|
+
_opt(argv, "--sticky", self.sticky, 0)
|
|
257
|
+
_opt(argv, "--serve-refill", float(self.serve_refill), 0.0)
|
|
258
|
+
return argv
|
|
259
|
+
|
|
260
|
+
def to_command(self, program: Optional[str] = None) -> str:
|
|
261
|
+
if program is None: # started from the repo or installed?
|
|
262
|
+
program = "python3 proxy_scraper.py" if is_checkout() else "proxy-scraper"
|
|
263
|
+
return " ".join([program, *map(shlex.quote, self.to_argv())])
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _opt(argv: List[str], name: str, value, default) -> None:
|
|
267
|
+
if value != default:
|
|
268
|
+
argv += [name, f"{value:g}" if isinstance(value, float) else str(value)]
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def _flag(argv: List[str], name: str, value: bool) -> None:
|
|
272
|
+
if value:
|
|
273
|
+
argv.append(name)
|
|
274
|
+
|
proxyscraper/output.py
ADDED
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
"""Result files.
|
|
2
|
+
|
|
3
|
+
Every run gets its own folder results/<date>/ with
|
|
4
|
+
all.txt type://ip:port, fastest first (also live during the run)
|
|
5
|
+
http.txt … ip:port per protocol – ready for tools that just want a list
|
|
6
|
+
proxies.json every detail (latency, country, HTTPS, anonymity, exit IP)
|
|
7
|
+
proxies.csv the same as a table
|
|
8
|
+
and results/latest.txt (or the symlink results/latest) always points to the newest run.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import contextlib
|
|
14
|
+
import csv
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
from dataclasses import asdict
|
|
18
|
+
from datetime import datetime
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Dict, Iterable, List, Optional, Sequence, TextIO
|
|
21
|
+
|
|
22
|
+
from . import paths
|
|
23
|
+
from .checker import CheckResult
|
|
24
|
+
from .exporters import EXPORTERS
|
|
25
|
+
from .parsing import PROXY_TYPES
|
|
26
|
+
from .paths import atomic_write
|
|
27
|
+
from .targets import target_label
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def new_run_dir(base: Path) -> Path:
|
|
31
|
+
"""Folder with a timestamp; two runs in the same second get -2, -3 … instead of overwriting each other."""
|
|
32
|
+
stamp = f"{datetime.now():%Y-%m-%d_%H-%M-%S}"
|
|
33
|
+
base.mkdir(parents=True, exist_ok=True)
|
|
34
|
+
for n in range(1, 1000):
|
|
35
|
+
path = base / (stamp if n == 1 else f"{stamp}-{n}")
|
|
36
|
+
try:
|
|
37
|
+
path.mkdir()
|
|
38
|
+
except FileExistsError:
|
|
39
|
+
continue
|
|
40
|
+
return path
|
|
41
|
+
raise FileExistsError(f"too many runs in one second: {stamp}")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class ResultWriter:
|
|
45
|
+
def __init__(self, run_dir: Optional[Path] = None, extra_file: Optional[Path] = None,
|
|
46
|
+
exports: Sequence[str] = (), stdout: Optional[TextIO] = None):
|
|
47
|
+
self.run_dir = run_dir or new_run_dir(paths.RESULTS_DIR) # read now: the MCP server moves it
|
|
48
|
+
self.run_dir.mkdir(parents=True, exist_ok=True)
|
|
49
|
+
self.extra_file = extra_file
|
|
50
|
+
self.stdout = stdout # -o -: the hits also go here, one type://ip:port per line
|
|
51
|
+
self.exports = list(exports)
|
|
52
|
+
self.live_path = self.run_dir / "all.txt"
|
|
53
|
+
self._live = self.live_path.open("w", encoding="utf-8")
|
|
54
|
+
|
|
55
|
+
def add_live(self, r: CheckResult) -> None:
|
|
56
|
+
self._live.write(f"{r.ptype}://{r.proxy}\n")
|
|
57
|
+
self._live.flush()
|
|
58
|
+
|
|
59
|
+
def close(self) -> None:
|
|
60
|
+
"""Done without writing the result files (a refill in a scratch folder)."""
|
|
61
|
+
self._live.close()
|
|
62
|
+
|
|
63
|
+
def finalize(self, results: Iterable[CheckResult]) -> Dict[str, Path]:
|
|
64
|
+
self._live.close()
|
|
65
|
+
rows = sorted(results, key=lambda r: r.latency)
|
|
66
|
+
files: Dict[str, Path] = {}
|
|
67
|
+
|
|
68
|
+
lines = "".join(f"{r.ptype}://{r.proxy}\n" for r in rows)
|
|
69
|
+
self.live_path.write_text(lines, encoding="utf-8")
|
|
70
|
+
files["All (type://ip:port)"] = self.live_path
|
|
71
|
+
for t in PROXY_TYPES:
|
|
72
|
+
of_type = [r.proxy for r in rows if r.ptype == t]
|
|
73
|
+
if of_type:
|
|
74
|
+
path = self.run_dir / f"{t}.txt"
|
|
75
|
+
path.write_text("\n".join(of_type) + "\n", encoding="utf-8")
|
|
76
|
+
files[f"{t} (ip:port)"] = path
|
|
77
|
+
|
|
78
|
+
json_path = self.run_dir / "proxies.json"
|
|
79
|
+
json_path.write_text(json.dumps([_row(r) for r in rows], indent=1, ensure_ascii=False), encoding="utf-8")
|
|
80
|
+
files["Details (JSON)"] = json_path
|
|
81
|
+
|
|
82
|
+
csv_path = self.run_dir / "proxies.csv"
|
|
83
|
+
with csv_path.open("w", newline="", encoding="utf-8") as fh:
|
|
84
|
+
writer = csv.DictWriter(fh, fieldnames=list(_csv_row(rows[0]).keys()) if rows else ["proxy"])
|
|
85
|
+
writer.writeheader()
|
|
86
|
+
writer.writerows(_csv_row(r) for r in rows)
|
|
87
|
+
files["Details (CSV)"] = csv_path
|
|
88
|
+
|
|
89
|
+
now = datetime.now()
|
|
90
|
+
for name in self.exports:
|
|
91
|
+
filename, render = EXPORTERS[name]
|
|
92
|
+
path = self.run_dir / filename
|
|
93
|
+
path.write_text(render(rows, now), encoding="utf-8")
|
|
94
|
+
files[f"{name} ({filename})"] = path
|
|
95
|
+
|
|
96
|
+
if self.extra_file:
|
|
97
|
+
self.extra_file.parent.mkdir(parents=True, exist_ok=True)
|
|
98
|
+
self.extra_file.write_text(lines, encoding="utf-8")
|
|
99
|
+
files["Extra file (-o)"] = self.extra_file
|
|
100
|
+
if self.stdout is not None:
|
|
101
|
+
try:
|
|
102
|
+
self.stdout.write(lines)
|
|
103
|
+
self.stdout.flush()
|
|
104
|
+
except OSError: # `| head` stopped reading early – BrokenPipeError, on Windows EINVAL
|
|
105
|
+
_silence(self.stdout)
|
|
106
|
+
|
|
107
|
+
_point_latest(self.run_dir)
|
|
108
|
+
return files
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _silence(stream: TextIO) -> None:
|
|
112
|
+
"""Point a closed pipe at devnull, so the interpreter doesn't fail flushing it again on exit."""
|
|
113
|
+
with contextlib.suppress(OSError, ValueError, AttributeError):
|
|
114
|
+
os.dup2(os.open(os.devnull, os.O_WRONLY), stream.fileno())
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _row(r: CheckResult) -> dict:
|
|
118
|
+
d = asdict(r)
|
|
119
|
+
d.pop("key")
|
|
120
|
+
d["url"] = f"{r.ptype}://{r.proxy}"
|
|
121
|
+
return d
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _csv_row(r: CheckResult) -> dict:
|
|
125
|
+
"""Like _row, but flat: target sites as "google.com:ok;discord.com:no"."""
|
|
126
|
+
d = _row(r)
|
|
127
|
+
d["targets"] = ";".join(f"{target_label(u, r.targets)}:{'ok' if ok else 'no'}" for u, ok in r.targets.items())
|
|
128
|
+
return d
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
LATEST_POINTER = "latest.txt"
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _point_latest(run_dir: Path) -> None:
|
|
135
|
+
"""results/latest.txt always names the newest run; results/latest is also a symlink.
|
|
136
|
+
|
|
137
|
+
The symlink is handy for looking around, but on Windows it needs admin or developer
|
|
138
|
+
rights – the pointer file works everywhere.
|
|
139
|
+
"""
|
|
140
|
+
atomic_write(run_dir.parent / LATEST_POINTER, run_dir.name + "\n")
|
|
141
|
+
latest = run_dir.parent / "latest"
|
|
142
|
+
try:
|
|
143
|
+
if latest.is_symlink():
|
|
144
|
+
latest.unlink()
|
|
145
|
+
if not latest.exists():
|
|
146
|
+
os.symlink(run_dir.name, latest, target_is_directory=True)
|
|
147
|
+
except OSError:
|
|
148
|
+
pass # no symlinks allowed – latest.txt is enough
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def latest_run_dir(results_dir: Optional[Path] = None) -> Optional[Path]:
|
|
152
|
+
results_dir = results_dir or paths.RESULTS_DIR
|
|
153
|
+
pointer = results_dir / LATEST_POINTER
|
|
154
|
+
try:
|
|
155
|
+
name = pointer.read_text(encoding="utf-8").strip()
|
|
156
|
+
except (OSError, UnicodeDecodeError): # missing, no permission or broken -> try the symlink
|
|
157
|
+
name = ""
|
|
158
|
+
# only a folder name directly under results/ – empty, "..", or paths like "a/../.." would
|
|
159
|
+
# otherwise point to results/ itself or outside of it
|
|
160
|
+
if name and name not in (".", "..") and Path(name).name == name:
|
|
161
|
+
run_dir = results_dir / name
|
|
162
|
+
if run_dir.is_dir():
|
|
163
|
+
return run_dir
|
|
164
|
+
link = results_dir / "latest"
|
|
165
|
+
return link if link.is_dir() else None # runs from older versions without latest.txt
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def has_latest_results(results_dir: Optional[Path] = None) -> bool:
|
|
169
|
+
"""Is there a last run with hits? Doesn't read the whole file for that."""
|
|
170
|
+
run_dir = latest_run_dir(results_dir)
|
|
171
|
+
try:
|
|
172
|
+
return run_dir is not None and (run_dir / "all.txt").stat().st_size > 0
|
|
173
|
+
except OSError:
|
|
174
|
+
return False
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def latest_results(results_dir: Optional[Path] = None) -> List[str]:
|
|
178
|
+
"""Proxies of the last run for --recheck without a file."""
|
|
179
|
+
run_dir = latest_run_dir(results_dir)
|
|
180
|
+
path = run_dir / "all.txt" if run_dir else None
|
|
181
|
+
return path.read_text(encoding="utf-8").splitlines() if path and path.exists() else []
|
proxyscraper/pages.py
ADDED
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
"""Static pages for search engines: one per protocol, per filter and per country, plus a sitemap.
|
|
2
|
+
|
|
3
|
+
The website itself loads everything with JavaScript, so a search engine mostly sees an empty shell. These
|
|
4
|
+
pages carry the proxies as plain HTML: "free socks5 proxy list" or "free proxies germany" land on a page
|
|
5
|
+
that answers exactly that, and every page links back to the full, filterable list.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from datetime import datetime
|
|
11
|
+
from html import escape
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Callable, Dict, List, Tuple
|
|
14
|
+
|
|
15
|
+
SITE_URL = "https://maximilianfeix.github.io/proxy-scraper/"
|
|
16
|
+
REPO_URL = "https://github.com/maximilianfeix/proxy-scraper"
|
|
17
|
+
ROWS_PER_PAGE = 200
|
|
18
|
+
SITEMAP_MIN = 3 # countries with fewer proxies still get their page, but aren't advertised in the sitemap
|
|
19
|
+
|
|
20
|
+
# every country here always gets a page, so its URL never disappears between runs (empty ones say so)
|
|
21
|
+
COUNTRIES = {
|
|
22
|
+
"AE": "the United Arab Emirates", "AR": "Argentina", "AT": "Austria", "AU": "Australia", "BD": "Bangladesh",
|
|
23
|
+
"BE": "Belgium", "BG": "Bulgaria", "BR": "Brazil", "CA": "Canada", "CH": "Switzerland", "CL": "Chile",
|
|
24
|
+
"CN": "China", "CO": "Colombia", "CZ": "Czechia", "DE": "Germany", "DK": "Denmark", "EC": "Ecuador",
|
|
25
|
+
"EE": "Estonia", "EG": "Egypt", "ES": "Spain", "FI": "Finland", "FR": "France", "GB": "the United Kingdom",
|
|
26
|
+
"GR": "Greece", "HK": "Hong Kong", "HR": "Croatia", "HU": "Hungary", "ID": "Indonesia", "IE": "Ireland",
|
|
27
|
+
"IL": "Israel", "IN": "India", "IQ": "Iraq", "IR": "Iran", "IT": "Italy", "JP": "Japan", "KE": "Kenya",
|
|
28
|
+
"KH": "Cambodia", "KR": "South Korea", "KZ": "Kazakhstan", "LT": "Lithuania", "LV": "Latvia", "MA": "Morocco",
|
|
29
|
+
"MX": "Mexico", "MY": "Malaysia", "NG": "Nigeria", "NL": "the Netherlands", "NO": "Norway", "NP": "Nepal",
|
|
30
|
+
"NZ": "New Zealand", "PE": "Peru", "PH": "the Philippines", "PK": "Pakistan", "PL": "Poland", "PT": "Portugal",
|
|
31
|
+
"RO": "Romania", "RS": "Serbia", "RU": "Russia", "SA": "Saudi Arabia", "SE": "Sweden", "SG": "Singapore",
|
|
32
|
+
"TH": "Thailand", "TR": "Turkey", "TW": "Taiwan", "UA": "Ukraine", "US": "the United States",
|
|
33
|
+
"VE": "Venezuela", "VN": "Vietnam", "ZA": "South Africa",
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
Filter = Callable[[dict], bool]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _short_name(country: str) -> str:
|
|
40
|
+
return country[4:] if country.startswith("the ") else country
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def page_specs() -> List[Tuple[str, str, str, Filter]]:
|
|
44
|
+
"""(path, title, what one proxy is called in the lede, filter) for every page."""
|
|
45
|
+
specs: List[Tuple[str, str, str, Filter]] = [
|
|
46
|
+
("http/", "Free HTTP proxy list", "HTTP proxies", lambda r: r["ptype"] == "http"),
|
|
47
|
+
("socks4/", "Free SOCKS4 proxy list", "SOCKS4 proxies", lambda r: r["ptype"] == "socks4"),
|
|
48
|
+
("socks5/", "Free SOCKS5 proxy list", "SOCKS5 proxies", lambda r: r["ptype"] == "socks5"),
|
|
49
|
+
("https/", "Free HTTPS proxy list", "proxies that tunnel HTTPS with verified TLS",
|
|
50
|
+
lambda r: bool(r.get("https"))),
|
|
51
|
+
("elite/", "Free elite proxy list", "elite proxies (no forwarded IP, no Via header)",
|
|
52
|
+
lambda r: r.get("anonymity") == "elite"),
|
|
53
|
+
]
|
|
54
|
+
for cc, name in sorted(COUNTRIES.items(), key=lambda kv: _short_name(kv[1])):
|
|
55
|
+
specs.append((f"country/{cc.lower()}/", f"Free proxies in {_short_name(name)}",
|
|
56
|
+
f"proxies in {name}", lambda r, cc=cc: r.get("country") == cc))
|
|
57
|
+
return specs
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
CSS = """
|
|
61
|
+
:root { --bg: #121113; --panel: #1A191C; --line: #2B292F; --text: #EDEBE6; --muted: #9A97A0; --signal: #D4F77A;
|
|
62
|
+
--signal-ink: #D4F77A; --on-signal: #121113; color-scheme: dark; }
|
|
63
|
+
@media (prefers-color-scheme: light) {
|
|
64
|
+
:root { --bg: #EFEFEC; --panel: #FFFFFF; --line: #DAD9D4; --text: #121113; --muted: #5F5C66;
|
|
65
|
+
--signal-ink: #3F5A00; color-scheme: light; }
|
|
66
|
+
}
|
|
67
|
+
* { box-sizing: border-box; }
|
|
68
|
+
body { margin: 0; background: var(--bg); color: var(--text); font: 400 16px/1.55 "Geist", ui-sans-serif, system-ui,
|
|
69
|
+
sans-serif; padding-inline: 16px; }
|
|
70
|
+
.wrap { max-width: 1080px; margin: 0 auto; padding-block: 28px 64px; display: grid; gap: 36px; }
|
|
71
|
+
a { color: var(--signal-ink); }
|
|
72
|
+
:focus-visible { outline: 2px solid var(--signal-ink); outline-offset: 3px; border-radius: 6px; }
|
|
73
|
+
.mono { font-family: "Geist Mono", ui-monospace, Menlo, monospace; }
|
|
74
|
+
header { display: flex; justify-content: space-between; align-items: center; gap: 16px; flex-wrap: wrap; }
|
|
75
|
+
.brand { display: flex; align-items: center; gap: 10px; color: var(--text); text-decoration: none; font-weight: 600;
|
|
76
|
+
letter-spacing: -.03em; font-size: 18px; }
|
|
77
|
+
.brand img { width: 30px; height: 30px; border-radius: 9px; }
|
|
78
|
+
h1 { margin: 0; font-size: clamp(36px, 6vw, 72px); line-height: 1; letter-spacing: -.05em; font-weight: 600;
|
|
79
|
+
text-wrap: balance; }
|
|
80
|
+
.lede { margin: 16px 0 0; color: var(--muted); font-size: 18px; max-width: 62ch; }
|
|
81
|
+
.actions { display: flex; gap: 10px; flex-wrap: wrap; margin-top: 22px; }
|
|
82
|
+
.btn { display: inline-flex; align-items: center; padding: 10px 18px; border-radius: 99px; text-decoration: none;
|
|
83
|
+
font-weight: 500; border: 1px solid var(--line); color: var(--text); }
|
|
84
|
+
.btn.primary { background: var(--signal); color: var(--on-signal); border-color: var(--signal); }
|
|
85
|
+
.table { overflow-x: auto; background: var(--panel); border: 1px solid var(--line); border-radius: 16px; }
|
|
86
|
+
table { border-collapse: collapse; width: 100%; min-width: 640px; font-size: 14px; }
|
|
87
|
+
caption { text-align: left; padding: 14px 16px 0; color: var(--muted); font-size: 13px; }
|
|
88
|
+
th { text-align: left; font-weight: 500; color: var(--muted); padding: 12px 16px;
|
|
89
|
+
border-bottom: 1px solid var(--line); }
|
|
90
|
+
td { padding: 10px 16px; border-bottom: 1px solid var(--line); font-variant-numeric: tabular-nums;
|
|
91
|
+
white-space: nowrap; }
|
|
92
|
+
tr:last-child td { border-bottom: 0; }
|
|
93
|
+
.empty { padding: 28px 16px; color: var(--muted); }
|
|
94
|
+
nav h2 { font-size: 15px; font-weight: 500; color: var(--muted); margin: 0 0 10px; }
|
|
95
|
+
nav ul { list-style: none; margin: 0; padding: 0; display: flex; flex-wrap: wrap; gap: 8px; }
|
|
96
|
+
nav a { display: inline-block; padding: 5px 12px; border: 1px solid var(--line); border-radius: 99px;
|
|
97
|
+
text-decoration: none; color: var(--text); font-size: 14px; }
|
|
98
|
+
nav a[aria-current] { background: var(--signal); color: var(--on-signal); border-color: var(--signal); }
|
|
99
|
+
footer { color: var(--muted); font-size: 14px; display: grid; gap: 6px; }
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _render(path: str, title: str, what: str, rows: List[dict], updated: datetime, nav: str) -> str:
|
|
104
|
+
depth = path.count("/")
|
|
105
|
+
root = "../" * depth
|
|
106
|
+
total = len(rows)
|
|
107
|
+
when = updated.strftime("%d %b %Y, %H:%M UTC")
|
|
108
|
+
if total:
|
|
109
|
+
lede = (f"{total:,} {what} that worked in the last check ({when}). Each one passed a real handshake, "
|
|
110
|
+
"a honeypot check on two sites and a content check, so none of them rewrite the pages you load.")
|
|
111
|
+
else:
|
|
112
|
+
lede = (f"No {what} passed every check in the last run ({when}). The list is checked again every hour, "
|
|
113
|
+
"so look again later or try the full list.")
|
|
114
|
+
body_rows = "\n".join(
|
|
115
|
+
f"<tr><td class=\"mono\">{escape(r['proxy'])}</td><td>{escape(r['ptype'].upper())}</td>"
|
|
116
|
+
f"<td>{escape(r.get('country') or '–')}</td><td>{r['latency']:,} ms</td>"
|
|
117
|
+
f"<td>{'yes' if r.get('https') else 'no'}</td><td>{escape(r.get('anonymity') or '–')}</td>"
|
|
118
|
+
f"<td>{escape((r.get('org') or '–')[:40])}</td></tr>"
|
|
119
|
+
for r in rows[:ROWS_PER_PAGE])
|
|
120
|
+
table = (f"<div class=\"table\"><table><caption>The {min(total, ROWS_PER_PAGE):,} fastest, as ip:port. "
|
|
121
|
+
f"The download has all {total:,} as type://ip:port.</caption>"
|
|
122
|
+
"<thead><tr><th scope=\"col\">Proxy</th><th scope=\"col\">Type</th><th scope=\"col\">Country</th>"
|
|
123
|
+
"<th scope=\"col\">Latency</th><th scope=\"col\">HTTPS</th><th scope=\"col\">Anonymity</th>"
|
|
124
|
+
f"<th scope=\"col\">Provider</th></tr></thead><tbody>{body_rows}</tbody></table></div>"
|
|
125
|
+
if total else "")
|
|
126
|
+
description = (f"{total:,} free {what}, checked every hour: real handshake, honeypot and content check. "
|
|
127
|
+
"Download as text or filter the full list.")
|
|
128
|
+
return f"""<!doctype html>
|
|
129
|
+
<html lang="en">
|
|
130
|
+
<head>
|
|
131
|
+
<meta charset="utf-8">
|
|
132
|
+
<meta name="viewport" content="width=device-width, initial-scale=1">
|
|
133
|
+
<title>{escape(title)} – checked every hour | proxy-scraper</title>
|
|
134
|
+
<meta name="description" content="{escape(description)}">
|
|
135
|
+
<link rel="canonical" href="{SITE_URL}{path}">
|
|
136
|
+
<meta property="og:title" content="{escape(title)}">
|
|
137
|
+
<meta property="og:description" content="{escape(description)}">
|
|
138
|
+
<meta property="og:image" content="{SITE_URL}og.png">
|
|
139
|
+
<meta property="og:url" content="{SITE_URL}{path}">
|
|
140
|
+
<meta name="twitter:card" content="summary_large_image">
|
|
141
|
+
<link rel="icon" href="{root}logo.png">
|
|
142
|
+
<link rel="apple-touch-icon" href="{root}apple-touch-icon.png">
|
|
143
|
+
<link rel="preconnect" href="https://fonts.googleapis.com">
|
|
144
|
+
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
|
145
|
+
<link href="https://fonts.googleapis.com/css2?family=Geist:wght@400..700&family=Geist+Mono:wght@400;500&display=swap"
|
|
146
|
+
rel="stylesheet">
|
|
147
|
+
<style>{CSS}</style>
|
|
148
|
+
</head>
|
|
149
|
+
<body>
|
|
150
|
+
<div class="wrap">
|
|
151
|
+
<header>
|
|
152
|
+
<a class="brand" href="{root}"><img src="{root}logo.png" alt="">proxy-scraper</a>
|
|
153
|
+
<a href="{root}">All proxies</a>
|
|
154
|
+
</header>
|
|
155
|
+
<main>
|
|
156
|
+
<h1>{escape(title)}</h1>
|
|
157
|
+
<p class="lede">{escape(lede)}</p>
|
|
158
|
+
<div class="actions">
|
|
159
|
+
<a class="btn primary" href="proxies.txt" download>Download {total:,} as .txt</a>
|
|
160
|
+
<a class="btn" href="{root}#list">Filter the full list</a>
|
|
161
|
+
<a class="btn" href="{REPO_URL}">Check them from your network</a>
|
|
162
|
+
</div>
|
|
163
|
+
</main>
|
|
164
|
+
{table}
|
|
165
|
+
{nav}
|
|
166
|
+
<footer>
|
|
167
|
+
<span>Free proxies are run by strangers. Never send passwords or personal data through them.</span>
|
|
168
|
+
<span>Collected and checked by <a href="{REPO_URL}">proxy-scraper</a>, open source, MIT licensed.</span>
|
|
169
|
+
</footer>
|
|
170
|
+
</div>
|
|
171
|
+
</body>
|
|
172
|
+
</html>
|
|
173
|
+
"""
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _nav(specs, counts: Dict[str, int], current: str, root: str) -> str:
|
|
177
|
+
def links(items):
|
|
178
|
+
here = ' aria-current="page"'
|
|
179
|
+
return "".join(f'<li><a href="{root}{p}"{here if p == current else ""}>{escape(label)} ({counts[p]:,})</a></li>'
|
|
180
|
+
for p, label in items)
|
|
181
|
+
|
|
182
|
+
kinds = [(p, t.replace("Free ", "").replace(" proxy list", ""))
|
|
183
|
+
for p, t, _, _ in specs if not p.startswith("country/")]
|
|
184
|
+
countries = sorted(((p, t.replace("Free proxies in ", "")) for p, t, _, _ in specs
|
|
185
|
+
if p.startswith("country/") and counts[p]), key=lambda pt: -counts[pt[0]])
|
|
186
|
+
return (f"<nav aria-label=\"More lists\"><h2>By protocol</h2><ul>{links(kinds)}</ul></nav>"
|
|
187
|
+
f"<nav aria-label=\"By country\"><h2>By country</h2><ul>{links(countries)}</ul></nav>")
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def write_pages(rows: List[dict], out: Path, updated: datetime) -> List[str]:
|
|
191
|
+
"""Writes every page with its proxies.txt and the sitemap. -> the paths listed in the sitemap."""
|
|
192
|
+
rows = sorted(rows, key=lambda r: r["latency"])
|
|
193
|
+
specs = page_specs()
|
|
194
|
+
selected = {p: [r for r in rows if keep(r)] for p, _, _, keep in specs}
|
|
195
|
+
counts = {p: len(v) for p, v in selected.items()}
|
|
196
|
+
listed = [""]
|
|
197
|
+
for path, title, what, _ in specs:
|
|
198
|
+
page_rows = selected[path]
|
|
199
|
+
folder = out / path
|
|
200
|
+
folder.mkdir(parents=True, exist_ok=True)
|
|
201
|
+
nav = _nav(specs, counts, path, "../" * path.count("/"))
|
|
202
|
+
(folder / "index.html").write_text(_render(path, title, what, page_rows, updated, nav), encoding="utf-8")
|
|
203
|
+
(folder / "proxies.txt").write_text("".join(f"{r['url']}\n" for r in page_rows), encoding="utf-8")
|
|
204
|
+
if not path.startswith("country/") or counts[path] >= SITEMAP_MIN:
|
|
205
|
+
listed.append(path)
|
|
206
|
+
lastmod = updated.strftime("%Y-%m-%dT%H:%M:%S+00:00")
|
|
207
|
+
urls = "".join(f"<url><loc>{SITE_URL}{p}</loc><lastmod>{lastmod}</lastmod><changefreq>hourly</changefreq></url>"
|
|
208
|
+
for p in listed)
|
|
209
|
+
(out / "sitemap.xml").write_text('<?xml version="1.0" encoding="UTF-8"?>\n'
|
|
210
|
+
f'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">{urls}</urlset>\n',
|
|
211
|
+
encoding="utf-8")
|
|
212
|
+
return listed
|