proxy-scraper-cli 1.7.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. proxy_scraper_cli-1.7.1.dist-info/METADATA +794 -0
  2. proxy_scraper_cli-1.7.1.dist-info/RECORD +57 -0
  3. proxy_scraper_cli-1.7.1.dist-info/WHEEL +5 -0
  4. proxy_scraper_cli-1.7.1.dist-info/entry_points.txt +4 -0
  5. proxy_scraper_cli-1.7.1.dist-info/licenses/LICENSE +21 -0
  6. proxy_scraper_cli-1.7.1.dist-info/top_level.txt +1 -0
  7. proxyscraper/__init__.py +12 -0
  8. proxyscraper/__main__.py +7 -0
  9. proxyscraper/agent.py +635 -0
  10. proxyscraper/api.py +139 -0
  11. proxyscraper/app.py +595 -0
  12. proxyscraper/asndb.py +181 -0
  13. proxyscraper/blocklist.py +92 -0
  14. proxyscraper/checker.py +514 -0
  15. proxyscraper/cli.py +269 -0
  16. proxyscraper/compat.py +83 -0
  17. proxyscraper/completion.py +196 -0
  18. proxyscraper/exporters.py +118 -0
  19. proxyscraper/fetchcache.py +103 -0
  20. proxyscraper/geo.py +150 -0
  21. proxyscraper/geodb.py +143 -0
  22. proxyscraper/handshake.py +153 -0
  23. proxyscraper/history.py +106 -0
  24. proxyscraper/judges.py +159 -0
  25. proxyscraper/mcp_entry.py +34 -0
  26. proxyscraper/mcp_server.py +220 -0
  27. proxyscraper/netio.py +167 -0
  28. proxyscraper/options.py +274 -0
  29. proxyscraper/output.py +181 -0
  30. proxyscraper/pages.py +212 -0
  31. proxyscraper/parsing.py +192 -0
  32. proxyscraper/paths.py +55 -0
  33. proxyscraper/pipeline.py +434 -0
  34. proxyscraper/preferences.py +24 -0
  35. proxyscraper/publish.py +236 -0
  36. proxyscraper/server/__init__.py +41 -0
  37. proxyscraper/server/core.py +518 -0
  38. proxyscraper/server/http.py +164 -0
  39. proxyscraper/server/pool.py +195 -0
  40. proxyscraper/server/socks.py +65 -0
  41. proxyscraper/server/status.py +108 -0
  42. proxyscraper/server/upstream.py +119 -0
  43. proxyscraper/site/apple-touch-icon.png +0 -0
  44. proxyscraper/site/googleaac1161b7853c5b5.html +1 -0
  45. proxyscraper/site/index.html +934 -0
  46. proxyscraper/site/logo.png +0 -0
  47. proxyscraper/site/og.png +0 -0
  48. proxyscraper/sources.json +395 -0
  49. proxyscraper/sources.py +491 -0
  50. proxyscraper/targets.py +76 -0
  51. proxyscraper/ui/__init__.py +54 -0
  52. proxyscraper/ui/dashboard.py +344 -0
  53. proxyscraper/ui/keys.py +82 -0
  54. proxyscraper/ui/report.py +209 -0
  55. proxyscraper/ui/serve.py +142 -0
  56. proxyscraper/ui/widgets.py +250 -0
  57. proxyscraper/ui/wizard.py +595 -0
@@ -0,0 +1,274 @@
1
+ """All settings of a run in one place – whether they come from the command line or
2
+ from the setup wizard."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import shlex
7
+ from dataclasses import dataclass, field
8
+ from typing import List, Optional, Set
9
+
10
+ from .checker import ANONYMITY_RANK, CheckResult
11
+ from .exporters import EXPORTERS
12
+ from .parsing import PROXY_TYPES
13
+ from .paths import is_checkout
14
+ from .server.pool import STRATEGIES
15
+ from .targets import target_label
16
+
17
+ DEFAULT_CONCURRENCY = 2000
18
+ DEFAULT_TIMEOUT = 8.0
19
+ DEFAULT_CONNECT_TIMEOUT = 4.0
20
+ DEFAULT_DISCOVER_REPOS = 400
21
+ DEFAULT_SERVE_PORT = 8899
22
+ STDOUT = "-" # -o -: hits to stdout
23
+
24
+
25
+ def parse_countries(value: Optional[str]) -> Set[str]:
26
+ """'de, at,CH' -> {'DE', 'AT', 'CH'}"""
27
+ return {c.strip().upper() for c in (value or "").split(",") if c.strip()}
28
+
29
+
30
+ @dataclass
31
+ class Filters:
32
+ countries: Set[str] = field(default_factory=set)
33
+ https_only: bool = False
34
+ min_anonymity: str = ""
35
+ max_latency: int = 0
36
+ targets: List[str] = field(default_factory=list) # target sites every proxy has to reach
37
+ no_datacenter: bool = False # only exits that are not (recognizably) in a datacenter
38
+ no_blocklisted: bool = False # only exits that are not on the SpamCop blocklist
39
+
40
+ @property
41
+ def needs_details(self) -> bool:
42
+ # anonymity comes from the confirmation (always runs) – HTTPS and target sites need the detail test
43
+ return self.https_only or bool(self.targets)
44
+
45
+ @property
46
+ def active(self) -> bool:
47
+ return bool(self.countries or self.https_only or self.min_anonymity or self.max_latency or self.targets
48
+ or self.no_datacenter or self.no_blocklisted)
49
+
50
+ def accepts(self, r: CheckResult) -> bool:
51
+ if self.max_latency and r.latency > self.max_latency:
52
+ return False
53
+ if self.https_only and r.https is not True:
54
+ return False
55
+ if self.min_anonymity and ANONYMITY_RANK.get(r.anonymity, -1) < ANONYMITY_RANK[self.min_anonymity]:
56
+ return False
57
+ if self.countries and r.country not in self.countries:
58
+ return False
59
+ if self.no_datacenter and r.hosting:
60
+ return False
61
+ if self.no_blocklisted and r.blocklisted:
62
+ return False
63
+ return all(r.targets.get(url) for url in self.targets) # reached every requested target site
64
+
65
+ def may_pass(self, r: CheckResult) -> bool:
66
+ """Can `r` still pass the filters? HTTPS is still unknown at this point, possibly the country too.
67
+
68
+ Whatever is sure to fail already doesn't need an expensive HTTPS test any more.
69
+ """
70
+ if self.max_latency and r.latency > self.max_latency:
71
+ return False
72
+ if self.min_anonymity and ANONYMITY_RANK.get(r.anonymity, -1) < ANONYMITY_RANK[self.min_anonymity]:
73
+ return False
74
+ if self.no_datacenter and r.hosting:
75
+ return False
76
+ if self.no_blocklisted and r.blocklisted:
77
+ return False
78
+ # country still unknown -> may still match
79
+ return not self.countries or not r.country or r.country in self.countries
80
+
81
+ def describe(self) -> str:
82
+ parts = []
83
+ if self.countries:
84
+ parts.append("country " + ",".join(sorted(self.countries)))
85
+ if self.https_only:
86
+ parts.append("HTTPS only")
87
+ if self.min_anonymity:
88
+ parts.append(f"min. {self.min_anonymity}")
89
+ if self.max_latency:
90
+ parts.append(f"≤ {self.max_latency} ms")
91
+ if self.targets:
92
+ parts.append("target " + ", ".join(target_label(u, self.targets) for u in self.targets))
93
+ if self.no_datacenter:
94
+ parts.append("no datacenters")
95
+ if self.no_blocklisted:
96
+ parts.append("not blocklisted")
97
+ return " · ".join(parts)
98
+
99
+
100
+ @dataclass
101
+ class RunOptions:
102
+ types: List[str] = field(default_factory=lambda: list(PROXY_TYPES))
103
+ filters: Filters = field(default_factory=Filters)
104
+ want: int = 0
105
+ limit: int = 0
106
+ fast: bool = False
107
+ no_geo: bool = False
108
+ recheck: Optional[str] = None
109
+ output: Optional[str] = None
110
+ exports: List[str] = field(default_factory=list) # extra formats, see exporters.py
111
+ concurrency: int = DEFAULT_CONCURRENCY
112
+ timeout: float = DEFAULT_TIMEOUT
113
+ connect_timeout: float = DEFAULT_CONNECT_TIMEOUT
114
+ discover: bool = False
115
+ no_discover: bool = False
116
+ discover_repos: int = DEFAULT_DISCOVER_REPOS
117
+ all_sources: bool = False
118
+ no_cache: bool = False # reload every list instead of taking unchanged ones from the cache
119
+ no_dnsbl: bool = False # skip the blocklist lookup of the exit IPs
120
+ serve: int = 0 # port of the rotating proxy server after the run, 0 = off
121
+ rotate: str = "weighted" # strategy of the proxy server, see server/pool.py
122
+ sticky: int = 0 # seconds a target site keeps the same proxy (0 = new one for every connection)
123
+ serve_host: str = "127.0.0.1" # address of the proxy server; anything else is reachable from outside
124
+ serve_password: str = field(default="", repr=False) # required from clients if set; never written to argv
125
+ serve_refill: float = 0 # hours between background refills of the proxy server's pool, 0 = off
126
+
127
+ def __post_init__(self) -> None:
128
+ unknown = set(self.types) - set(PROXY_TYPES)
129
+ if unknown:
130
+ raise ValueError(f"unknown proxy types: {', '.join(sorted(unknown))}")
131
+ # the types are really a set: fixed order, no duplicates.
132
+ # Otherwise "--types socks5 http" and "--types http socks5" would give different settings.
133
+ self.types = [t for t in PROXY_TYPES if t in self.types]
134
+ if not self.types:
135
+ raise ValueError("at least one proxy type needed") # else to_argv() gives "--types" without a value
136
+ if self.concurrency < 1:
137
+ raise ValueError("concurrency must be at least 1")
138
+ if self.timeout <= 0 or self.connect_timeout <= 0:
139
+ raise ValueError("timeouts must be greater than 0")
140
+ if min(self.want, self.limit, self.discover_repos, self.filters.max_latency) < 0:
141
+ raise ValueError("amounts and latency must not be negative")
142
+ unknown_exports = set(self.exports) - set(EXPORTERS)
143
+ if unknown_exports:
144
+ raise ValueError(f"unknown export formats: {', '.join(sorted(unknown_exports))}")
145
+ self.exports = [e for e in EXPORTERS if e in self.exports]
146
+ if self.rotate not in STRATEGIES:
147
+ raise ValueError(f"unknown strategy: {self.rotate}")
148
+ if self.sticky < 0:
149
+ raise ValueError("--sticky must not be negative")
150
+ if self.serve_refill < 0:
151
+ raise ValueError("--serve-refill must not be negative")
152
+ if not 0 <= self.serve <= 65535:
153
+ raise ValueError("port must be between 1 and 65535")
154
+
155
+ @property
156
+ def details(self) -> bool:
157
+ """HTTPS test – can be switched off (--fast), unless the HTTPS filter needs it."""
158
+ return not self.fast or self.filters.needs_details
159
+
160
+ @property
161
+ def check_timeout(self) -> float:
162
+ """Timeout of the basic check: whatever exceeds the latency limit drops out anyway –
163
+ nobody needs to wait that long. With 2000 parallel slots that's a lot more throughput."""
164
+ if self.filters.max_latency:
165
+ return min(self.timeout, self.filters.max_latency / 1000)
166
+ return self.timeout
167
+
168
+ @property
169
+ def check_connect_timeout(self) -> float:
170
+ return min(self.connect_timeout, self.check_timeout)
171
+
172
+ @property
173
+ def geo(self) -> bool:
174
+ return not self.no_geo or bool(self.filters.countries)
175
+
176
+ @classmethod
177
+ def from_args(cls, args) -> "RunOptions":
178
+ return cls(
179
+ types=list(args.types),
180
+ filters=Filters(
181
+ countries=parse_countries(args.country),
182
+ https_only=args.https_only,
183
+ min_anonymity=args.anonymity or "",
184
+ max_latency=args.max_latency,
185
+ targets=list(dict.fromkeys(args.target or [])),
186
+ no_datacenter=args.no_datacenter,
187
+ no_blocklisted=args.no_blocklisted,
188
+ ),
189
+ want=args.want,
190
+ limit=args.limit,
191
+ fast=args.fast,
192
+ no_geo=args.no_geo,
193
+ recheck=args.recheck,
194
+ output=args.output,
195
+ exports=list(args.export or []),
196
+ concurrency=args.concurrency,
197
+ timeout=args.timeout,
198
+ connect_timeout=args.connect_timeout,
199
+ discover=args.discover,
200
+ no_discover=args.no_discover,
201
+ discover_repos=args.discover_repos,
202
+ all_sources=args.all_sources,
203
+ no_cache=args.no_cache,
204
+ no_dnsbl=args.no_dnsbl,
205
+ serve=args.serve,
206
+ rotate=args.rotate,
207
+ sticky=args.sticky,
208
+ serve_host=args.serve_host,
209
+ serve_password=args.serve_password,
210
+ serve_refill=args.serve_refill,
211
+ )
212
+
213
+ def to_argv(self) -> List[str]:
214
+ """Command line arguments that produce exactly these settings (only deviations from the defaults)."""
215
+ argv: List[str] = []
216
+ if self.types != list(PROXY_TYPES):
217
+ argv += ["--types", *self.types]
218
+ f = self.filters
219
+ if f.countries:
220
+ argv += ["--country", ",".join(sorted(f.countries))]
221
+ if f.https_only:
222
+ argv.append("--https-only")
223
+ if f.min_anonymity:
224
+ argv += ["--anonymity", f.min_anonymity]
225
+ if f.max_latency:
226
+ argv += ["--max-latency", str(f.max_latency)]
227
+ for url in f.targets:
228
+ argv += ["--target", url]
229
+ _flag(argv, "--no-datacenter", f.no_datacenter)
230
+ _flag(argv, "--no-blocklisted", f.no_blocklisted)
231
+ _opt(argv, "--want", self.want, 0)
232
+ _opt(argv, "--limit", self.limit, 0)
233
+ _flag(argv, "--fast", self.fast)
234
+ _flag(argv, "--no-geo", self.no_geo)
235
+ if self.recheck is not None:
236
+ argv += ["--recheck", self.recheck] if self.recheck else ["--recheck"]
237
+ if self.output:
238
+ argv += ["--output", self.output]
239
+ if self.exports:
240
+ argv += ["--export", ",".join(self.exports)]
241
+ _opt(argv, "--concurrency", self.concurrency, DEFAULT_CONCURRENCY)
242
+ _opt(argv, "--timeout", self.timeout, DEFAULT_TIMEOUT)
243
+ _opt(argv, "--connect-timeout", self.connect_timeout, DEFAULT_CONNECT_TIMEOUT)
244
+ _flag(argv, "--discover", self.discover)
245
+ _flag(argv, "--no-discover", self.no_discover)
246
+ _opt(argv, "--discover-repos", self.discover_repos, DEFAULT_DISCOVER_REPOS)
247
+ _flag(argv, "--all-sources", self.all_sources)
248
+ _flag(argv, "--no-cache", self.no_cache)
249
+ _flag(argv, "--no-dnsbl", self.no_dnsbl)
250
+ if self.serve:
251
+ argv += ["--serve"] if self.serve == DEFAULT_SERVE_PORT else ["--serve", str(self.serve)]
252
+ if self.rotate != "weighted":
253
+ argv += ["--rotate", self.rotate]
254
+ if self.serve_host != "127.0.0.1":
255
+ argv += ["--serve-host", self.serve_host]
256
+ _opt(argv, "--sticky", self.sticky, 0)
257
+ _opt(argv, "--serve-refill", float(self.serve_refill), 0.0)
258
+ return argv
259
+
260
+ def to_command(self, program: Optional[str] = None) -> str:
261
+ if program is None: # started from the repo or installed?
262
+ program = "python3 proxy_scraper.py" if is_checkout() else "proxy-scraper"
263
+ return " ".join([program, *map(shlex.quote, self.to_argv())])
264
+
265
+
266
+ def _opt(argv: List[str], name: str, value, default) -> None:
267
+ if value != default:
268
+ argv += [name, f"{value:g}" if isinstance(value, float) else str(value)]
269
+
270
+
271
+ def _flag(argv: List[str], name: str, value: bool) -> None:
272
+ if value:
273
+ argv.append(name)
274
+
proxyscraper/output.py ADDED
@@ -0,0 +1,181 @@
1
+ """Result files.
2
+
3
+ Every run gets its own folder results/<date>/ with
4
+ all.txt type://ip:port, fastest first (also live during the run)
5
+ http.txt … ip:port per protocol – ready for tools that just want a list
6
+ proxies.json every detail (latency, country, HTTPS, anonymity, exit IP)
7
+ proxies.csv the same as a table
8
+ and results/latest.txt (or the symlink results/latest) always points to the newest run.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import contextlib
14
+ import csv
15
+ import json
16
+ import os
17
+ from dataclasses import asdict
18
+ from datetime import datetime
19
+ from pathlib import Path
20
+ from typing import Dict, Iterable, List, Optional, Sequence, TextIO
21
+
22
+ from . import paths
23
+ from .checker import CheckResult
24
+ from .exporters import EXPORTERS
25
+ from .parsing import PROXY_TYPES
26
+ from .paths import atomic_write
27
+ from .targets import target_label
28
+
29
+
30
+ def new_run_dir(base: Path) -> Path:
31
+ """Folder with a timestamp; two runs in the same second get -2, -3 … instead of overwriting each other."""
32
+ stamp = f"{datetime.now():%Y-%m-%d_%H-%M-%S}"
33
+ base.mkdir(parents=True, exist_ok=True)
34
+ for n in range(1, 1000):
35
+ path = base / (stamp if n == 1 else f"{stamp}-{n}")
36
+ try:
37
+ path.mkdir()
38
+ except FileExistsError:
39
+ continue
40
+ return path
41
+ raise FileExistsError(f"too many runs in one second: {stamp}")
42
+
43
+
44
+ class ResultWriter:
45
+ def __init__(self, run_dir: Optional[Path] = None, extra_file: Optional[Path] = None,
46
+ exports: Sequence[str] = (), stdout: Optional[TextIO] = None):
47
+ self.run_dir = run_dir or new_run_dir(paths.RESULTS_DIR) # read now: the MCP server moves it
48
+ self.run_dir.mkdir(parents=True, exist_ok=True)
49
+ self.extra_file = extra_file
50
+ self.stdout = stdout # -o -: the hits also go here, one type://ip:port per line
51
+ self.exports = list(exports)
52
+ self.live_path = self.run_dir / "all.txt"
53
+ self._live = self.live_path.open("w", encoding="utf-8")
54
+
55
+ def add_live(self, r: CheckResult) -> None:
56
+ self._live.write(f"{r.ptype}://{r.proxy}\n")
57
+ self._live.flush()
58
+
59
+ def close(self) -> None:
60
+ """Done without writing the result files (a refill in a scratch folder)."""
61
+ self._live.close()
62
+
63
+ def finalize(self, results: Iterable[CheckResult]) -> Dict[str, Path]:
64
+ self._live.close()
65
+ rows = sorted(results, key=lambda r: r.latency)
66
+ files: Dict[str, Path] = {}
67
+
68
+ lines = "".join(f"{r.ptype}://{r.proxy}\n" for r in rows)
69
+ self.live_path.write_text(lines, encoding="utf-8")
70
+ files["All (type://ip:port)"] = self.live_path
71
+ for t in PROXY_TYPES:
72
+ of_type = [r.proxy for r in rows if r.ptype == t]
73
+ if of_type:
74
+ path = self.run_dir / f"{t}.txt"
75
+ path.write_text("\n".join(of_type) + "\n", encoding="utf-8")
76
+ files[f"{t} (ip:port)"] = path
77
+
78
+ json_path = self.run_dir / "proxies.json"
79
+ json_path.write_text(json.dumps([_row(r) for r in rows], indent=1, ensure_ascii=False), encoding="utf-8")
80
+ files["Details (JSON)"] = json_path
81
+
82
+ csv_path = self.run_dir / "proxies.csv"
83
+ with csv_path.open("w", newline="", encoding="utf-8") as fh:
84
+ writer = csv.DictWriter(fh, fieldnames=list(_csv_row(rows[0]).keys()) if rows else ["proxy"])
85
+ writer.writeheader()
86
+ writer.writerows(_csv_row(r) for r in rows)
87
+ files["Details (CSV)"] = csv_path
88
+
89
+ now = datetime.now()
90
+ for name in self.exports:
91
+ filename, render = EXPORTERS[name]
92
+ path = self.run_dir / filename
93
+ path.write_text(render(rows, now), encoding="utf-8")
94
+ files[f"{name} ({filename})"] = path
95
+
96
+ if self.extra_file:
97
+ self.extra_file.parent.mkdir(parents=True, exist_ok=True)
98
+ self.extra_file.write_text(lines, encoding="utf-8")
99
+ files["Extra file (-o)"] = self.extra_file
100
+ if self.stdout is not None:
101
+ try:
102
+ self.stdout.write(lines)
103
+ self.stdout.flush()
104
+ except OSError: # `| head` stopped reading early – BrokenPipeError, on Windows EINVAL
105
+ _silence(self.stdout)
106
+
107
+ _point_latest(self.run_dir)
108
+ return files
109
+
110
+
111
+ def _silence(stream: TextIO) -> None:
112
+ """Point a closed pipe at devnull, so the interpreter doesn't fail flushing it again on exit."""
113
+ with contextlib.suppress(OSError, ValueError, AttributeError):
114
+ os.dup2(os.open(os.devnull, os.O_WRONLY), stream.fileno())
115
+
116
+
117
+ def _row(r: CheckResult) -> dict:
118
+ d = asdict(r)
119
+ d.pop("key")
120
+ d["url"] = f"{r.ptype}://{r.proxy}"
121
+ return d
122
+
123
+
124
+ def _csv_row(r: CheckResult) -> dict:
125
+ """Like _row, but flat: target sites as "google.com:ok;discord.com:no"."""
126
+ d = _row(r)
127
+ d["targets"] = ";".join(f"{target_label(u, r.targets)}:{'ok' if ok else 'no'}" for u, ok in r.targets.items())
128
+ return d
129
+
130
+
131
+ LATEST_POINTER = "latest.txt"
132
+
133
+
134
+ def _point_latest(run_dir: Path) -> None:
135
+ """results/latest.txt always names the newest run; results/latest is also a symlink.
136
+
137
+ The symlink is handy for looking around, but on Windows it needs admin or developer
138
+ rights – the pointer file works everywhere.
139
+ """
140
+ atomic_write(run_dir.parent / LATEST_POINTER, run_dir.name + "\n")
141
+ latest = run_dir.parent / "latest"
142
+ try:
143
+ if latest.is_symlink():
144
+ latest.unlink()
145
+ if not latest.exists():
146
+ os.symlink(run_dir.name, latest, target_is_directory=True)
147
+ except OSError:
148
+ pass # no symlinks allowed – latest.txt is enough
149
+
150
+
151
+ def latest_run_dir(results_dir: Optional[Path] = None) -> Optional[Path]:
152
+ results_dir = results_dir or paths.RESULTS_DIR
153
+ pointer = results_dir / LATEST_POINTER
154
+ try:
155
+ name = pointer.read_text(encoding="utf-8").strip()
156
+ except (OSError, UnicodeDecodeError): # missing, no permission or broken -> try the symlink
157
+ name = ""
158
+ # only a folder name directly under results/ – empty, "..", or paths like "a/../.." would
159
+ # otherwise point to results/ itself or outside of it
160
+ if name and name not in (".", "..") and Path(name).name == name:
161
+ run_dir = results_dir / name
162
+ if run_dir.is_dir():
163
+ return run_dir
164
+ link = results_dir / "latest"
165
+ return link if link.is_dir() else None # runs from older versions without latest.txt
166
+
167
+
168
+ def has_latest_results(results_dir: Optional[Path] = None) -> bool:
169
+ """Is there a last run with hits? Doesn't read the whole file for that."""
170
+ run_dir = latest_run_dir(results_dir)
171
+ try:
172
+ return run_dir is not None and (run_dir / "all.txt").stat().st_size > 0
173
+ except OSError:
174
+ return False
175
+
176
+
177
+ def latest_results(results_dir: Optional[Path] = None) -> List[str]:
178
+ """Proxies of the last run for --recheck without a file."""
179
+ run_dir = latest_run_dir(results_dir)
180
+ path = run_dir / "all.txt" if run_dir else None
181
+ return path.read_text(encoding="utf-8").splitlines() if path and path.exists() else []
proxyscraper/pages.py ADDED
@@ -0,0 +1,212 @@
1
+ """Static pages for search engines: one per protocol, per filter and per country, plus a sitemap.
2
+
3
+ The website itself loads everything with JavaScript, so a search engine mostly sees an empty shell. These
4
+ pages carry the proxies as plain HTML: "free socks5 proxy list" or "free proxies germany" land on a page
5
+ that answers exactly that, and every page links back to the full, filterable list.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from datetime import datetime
11
+ from html import escape
12
+ from pathlib import Path
13
+ from typing import Callable, Dict, List, Tuple
14
+
15
+ SITE_URL = "https://maximilianfeix.github.io/proxy-scraper/"
16
+ REPO_URL = "https://github.com/maximilianfeix/proxy-scraper"
17
+ ROWS_PER_PAGE = 200
18
+ SITEMAP_MIN = 3 # countries with fewer proxies still get their page, but aren't advertised in the sitemap
19
+
20
+ # every country here always gets a page, so its URL never disappears between runs (empty ones say so)
21
+ COUNTRIES = {
22
+ "AE": "the United Arab Emirates", "AR": "Argentina", "AT": "Austria", "AU": "Australia", "BD": "Bangladesh",
23
+ "BE": "Belgium", "BG": "Bulgaria", "BR": "Brazil", "CA": "Canada", "CH": "Switzerland", "CL": "Chile",
24
+ "CN": "China", "CO": "Colombia", "CZ": "Czechia", "DE": "Germany", "DK": "Denmark", "EC": "Ecuador",
25
+ "EE": "Estonia", "EG": "Egypt", "ES": "Spain", "FI": "Finland", "FR": "France", "GB": "the United Kingdom",
26
+ "GR": "Greece", "HK": "Hong Kong", "HR": "Croatia", "HU": "Hungary", "ID": "Indonesia", "IE": "Ireland",
27
+ "IL": "Israel", "IN": "India", "IQ": "Iraq", "IR": "Iran", "IT": "Italy", "JP": "Japan", "KE": "Kenya",
28
+ "KH": "Cambodia", "KR": "South Korea", "KZ": "Kazakhstan", "LT": "Lithuania", "LV": "Latvia", "MA": "Morocco",
29
+ "MX": "Mexico", "MY": "Malaysia", "NG": "Nigeria", "NL": "the Netherlands", "NO": "Norway", "NP": "Nepal",
30
+ "NZ": "New Zealand", "PE": "Peru", "PH": "the Philippines", "PK": "Pakistan", "PL": "Poland", "PT": "Portugal",
31
+ "RO": "Romania", "RS": "Serbia", "RU": "Russia", "SA": "Saudi Arabia", "SE": "Sweden", "SG": "Singapore",
32
+ "TH": "Thailand", "TR": "Turkey", "TW": "Taiwan", "UA": "Ukraine", "US": "the United States",
33
+ "VE": "Venezuela", "VN": "Vietnam", "ZA": "South Africa",
34
+ }
35
+
36
+ Filter = Callable[[dict], bool]
37
+
38
+
39
+ def _short_name(country: str) -> str:
40
+ return country[4:] if country.startswith("the ") else country
41
+
42
+
43
+ def page_specs() -> List[Tuple[str, str, str, Filter]]:
44
+ """(path, title, what one proxy is called in the lede, filter) for every page."""
45
+ specs: List[Tuple[str, str, str, Filter]] = [
46
+ ("http/", "Free HTTP proxy list", "HTTP proxies", lambda r: r["ptype"] == "http"),
47
+ ("socks4/", "Free SOCKS4 proxy list", "SOCKS4 proxies", lambda r: r["ptype"] == "socks4"),
48
+ ("socks5/", "Free SOCKS5 proxy list", "SOCKS5 proxies", lambda r: r["ptype"] == "socks5"),
49
+ ("https/", "Free HTTPS proxy list", "proxies that tunnel HTTPS with verified TLS",
50
+ lambda r: bool(r.get("https"))),
51
+ ("elite/", "Free elite proxy list", "elite proxies (no forwarded IP, no Via header)",
52
+ lambda r: r.get("anonymity") == "elite"),
53
+ ]
54
+ for cc, name in sorted(COUNTRIES.items(), key=lambda kv: _short_name(kv[1])):
55
+ specs.append((f"country/{cc.lower()}/", f"Free proxies in {_short_name(name)}",
56
+ f"proxies in {name}", lambda r, cc=cc: r.get("country") == cc))
57
+ return specs
58
+
59
+
60
+ CSS = """
61
+ :root { --bg: #121113; --panel: #1A191C; --line: #2B292F; --text: #EDEBE6; --muted: #9A97A0; --signal: #D4F77A;
62
+ --signal-ink: #D4F77A; --on-signal: #121113; color-scheme: dark; }
63
+ @media (prefers-color-scheme: light) {
64
+ :root { --bg: #EFEFEC; --panel: #FFFFFF; --line: #DAD9D4; --text: #121113; --muted: #5F5C66;
65
+ --signal-ink: #3F5A00; color-scheme: light; }
66
+ }
67
+ * { box-sizing: border-box; }
68
+ body { margin: 0; background: var(--bg); color: var(--text); font: 400 16px/1.55 "Geist", ui-sans-serif, system-ui,
69
+ sans-serif; padding-inline: 16px; }
70
+ .wrap { max-width: 1080px; margin: 0 auto; padding-block: 28px 64px; display: grid; gap: 36px; }
71
+ a { color: var(--signal-ink); }
72
+ :focus-visible { outline: 2px solid var(--signal-ink); outline-offset: 3px; border-radius: 6px; }
73
+ .mono { font-family: "Geist Mono", ui-monospace, Menlo, monospace; }
74
+ header { display: flex; justify-content: space-between; align-items: center; gap: 16px; flex-wrap: wrap; }
75
+ .brand { display: flex; align-items: center; gap: 10px; color: var(--text); text-decoration: none; font-weight: 600;
76
+ letter-spacing: -.03em; font-size: 18px; }
77
+ .brand img { width: 30px; height: 30px; border-radius: 9px; }
78
+ h1 { margin: 0; font-size: clamp(36px, 6vw, 72px); line-height: 1; letter-spacing: -.05em; font-weight: 600;
79
+ text-wrap: balance; }
80
+ .lede { margin: 16px 0 0; color: var(--muted); font-size: 18px; max-width: 62ch; }
81
+ .actions { display: flex; gap: 10px; flex-wrap: wrap; margin-top: 22px; }
82
+ .btn { display: inline-flex; align-items: center; padding: 10px 18px; border-radius: 99px; text-decoration: none;
83
+ font-weight: 500; border: 1px solid var(--line); color: var(--text); }
84
+ .btn.primary { background: var(--signal); color: var(--on-signal); border-color: var(--signal); }
85
+ .table { overflow-x: auto; background: var(--panel); border: 1px solid var(--line); border-radius: 16px; }
86
+ table { border-collapse: collapse; width: 100%; min-width: 640px; font-size: 14px; }
87
+ caption { text-align: left; padding: 14px 16px 0; color: var(--muted); font-size: 13px; }
88
+ th { text-align: left; font-weight: 500; color: var(--muted); padding: 12px 16px;
89
+ border-bottom: 1px solid var(--line); }
90
+ td { padding: 10px 16px; border-bottom: 1px solid var(--line); font-variant-numeric: tabular-nums;
91
+ white-space: nowrap; }
92
+ tr:last-child td { border-bottom: 0; }
93
+ .empty { padding: 28px 16px; color: var(--muted); }
94
+ nav h2 { font-size: 15px; font-weight: 500; color: var(--muted); margin: 0 0 10px; }
95
+ nav ul { list-style: none; margin: 0; padding: 0; display: flex; flex-wrap: wrap; gap: 8px; }
96
+ nav a { display: inline-block; padding: 5px 12px; border: 1px solid var(--line); border-radius: 99px;
97
+ text-decoration: none; color: var(--text); font-size: 14px; }
98
+ nav a[aria-current] { background: var(--signal); color: var(--on-signal); border-color: var(--signal); }
99
+ footer { color: var(--muted); font-size: 14px; display: grid; gap: 6px; }
100
+ """
101
+
102
+
103
+ def _render(path: str, title: str, what: str, rows: List[dict], updated: datetime, nav: str) -> str:
104
+ depth = path.count("/")
105
+ root = "../" * depth
106
+ total = len(rows)
107
+ when = updated.strftime("%d %b %Y, %H:%M UTC")
108
+ if total:
109
+ lede = (f"{total:,} {what} that worked in the last check ({when}). Each one passed a real handshake, "
110
+ "a honeypot check on two sites and a content check, so none of them rewrite the pages you load.")
111
+ else:
112
+ lede = (f"No {what} passed every check in the last run ({when}). The list is checked again every hour, "
113
+ "so look again later or try the full list.")
114
+ body_rows = "\n".join(
115
+ f"<tr><td class=\"mono\">{escape(r['proxy'])}</td><td>{escape(r['ptype'].upper())}</td>"
116
+ f"<td>{escape(r.get('country') or '–')}</td><td>{r['latency']:,} ms</td>"
117
+ f"<td>{'yes' if r.get('https') else 'no'}</td><td>{escape(r.get('anonymity') or '–')}</td>"
118
+ f"<td>{escape((r.get('org') or '–')[:40])}</td></tr>"
119
+ for r in rows[:ROWS_PER_PAGE])
120
+ table = (f"<div class=\"table\"><table><caption>The {min(total, ROWS_PER_PAGE):,} fastest, as ip:port. "
121
+ f"The download has all {total:,} as type://ip:port.</caption>"
122
+ "<thead><tr><th scope=\"col\">Proxy</th><th scope=\"col\">Type</th><th scope=\"col\">Country</th>"
123
+ "<th scope=\"col\">Latency</th><th scope=\"col\">HTTPS</th><th scope=\"col\">Anonymity</th>"
124
+ f"<th scope=\"col\">Provider</th></tr></thead><tbody>{body_rows}</tbody></table></div>"
125
+ if total else "")
126
+ description = (f"{total:,} free {what}, checked every hour: real handshake, honeypot and content check. "
127
+ "Download as text or filter the full list.")
128
+ return f"""<!doctype html>
129
+ <html lang="en">
130
+ <head>
131
+ <meta charset="utf-8">
132
+ <meta name="viewport" content="width=device-width, initial-scale=1">
133
+ <title>{escape(title)} – checked every hour | proxy-scraper</title>
134
+ <meta name="description" content="{escape(description)}">
135
+ <link rel="canonical" href="{SITE_URL}{path}">
136
+ <meta property="og:title" content="{escape(title)}">
137
+ <meta property="og:description" content="{escape(description)}">
138
+ <meta property="og:image" content="{SITE_URL}og.png">
139
+ <meta property="og:url" content="{SITE_URL}{path}">
140
+ <meta name="twitter:card" content="summary_large_image">
141
+ <link rel="icon" href="{root}logo.png">
142
+ <link rel="apple-touch-icon" href="{root}apple-touch-icon.png">
143
+ <link rel="preconnect" href="https://fonts.googleapis.com">
144
+ <link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
145
+ <link href="https://fonts.googleapis.com/css2?family=Geist:wght@400..700&family=Geist+Mono:wght@400;500&display=swap"
146
+ rel="stylesheet">
147
+ <style>{CSS}</style>
148
+ </head>
149
+ <body>
150
+ <div class="wrap">
151
+ <header>
152
+ <a class="brand" href="{root}"><img src="{root}logo.png" alt="">proxy-scraper</a>
153
+ <a href="{root}">All proxies</a>
154
+ </header>
155
+ <main>
156
+ <h1>{escape(title)}</h1>
157
+ <p class="lede">{escape(lede)}</p>
158
+ <div class="actions">
159
+ <a class="btn primary" href="proxies.txt" download>Download {total:,} as .txt</a>
160
+ <a class="btn" href="{root}#list">Filter the full list</a>
161
+ <a class="btn" href="{REPO_URL}">Check them from your network</a>
162
+ </div>
163
+ </main>
164
+ {table}
165
+ {nav}
166
+ <footer>
167
+ <span>Free proxies are run by strangers. Never send passwords or personal data through them.</span>
168
+ <span>Collected and checked by <a href="{REPO_URL}">proxy-scraper</a>, open source, MIT licensed.</span>
169
+ </footer>
170
+ </div>
171
+ </body>
172
+ </html>
173
+ """
174
+
175
+
176
+ def _nav(specs, counts: Dict[str, int], current: str, root: str) -> str:
177
+ def links(items):
178
+ here = ' aria-current="page"'
179
+ return "".join(f'<li><a href="{root}{p}"{here if p == current else ""}>{escape(label)} ({counts[p]:,})</a></li>'
180
+ for p, label in items)
181
+
182
+ kinds = [(p, t.replace("Free ", "").replace(" proxy list", ""))
183
+ for p, t, _, _ in specs if not p.startswith("country/")]
184
+ countries = sorted(((p, t.replace("Free proxies in ", "")) for p, t, _, _ in specs
185
+ if p.startswith("country/") and counts[p]), key=lambda pt: -counts[pt[0]])
186
+ return (f"<nav aria-label=\"More lists\"><h2>By protocol</h2><ul>{links(kinds)}</ul></nav>"
187
+ f"<nav aria-label=\"By country\"><h2>By country</h2><ul>{links(countries)}</ul></nav>")
188
+
189
+
190
+ def write_pages(rows: List[dict], out: Path, updated: datetime) -> List[str]:
191
+ """Writes every page with its proxies.txt and the sitemap. -> the paths listed in the sitemap."""
192
+ rows = sorted(rows, key=lambda r: r["latency"])
193
+ specs = page_specs()
194
+ selected = {p: [r for r in rows if keep(r)] for p, _, _, keep in specs}
195
+ counts = {p: len(v) for p, v in selected.items()}
196
+ listed = [""]
197
+ for path, title, what, _ in specs:
198
+ page_rows = selected[path]
199
+ folder = out / path
200
+ folder.mkdir(parents=True, exist_ok=True)
201
+ nav = _nav(specs, counts, path, "../" * path.count("/"))
202
+ (folder / "index.html").write_text(_render(path, title, what, page_rows, updated, nav), encoding="utf-8")
203
+ (folder / "proxies.txt").write_text("".join(f"{r['url']}\n" for r in page_rows), encoding="utf-8")
204
+ if not path.startswith("country/") or counts[path] >= SITEMAP_MIN:
205
+ listed.append(path)
206
+ lastmod = updated.strftime("%Y-%m-%dT%H:%M:%S+00:00")
207
+ urls = "".join(f"<url><loc>{SITE_URL}{p}</loc><lastmod>{lastmod}</lastmod><changefreq>hourly</changefreq></url>"
208
+ for p in listed)
209
+ (out / "sitemap.xml").write_text('<?xml version="1.0" encoding="UTF-8"?>\n'
210
+ f'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">{urls}</urlset>\n',
211
+ encoding="utf-8")
212
+ return listed