google-browser-scraper 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- google_browser_scraper/__init__.py +27 -0
- google_browser_scraper/__main__.py +5 -0
- google_browser_scraper/browser.py +203 -0
- google_browser_scraper/classify.py +105 -0
- google_browser_scraper/cli.py +281 -0
- google_browser_scraper/doctor.py +155 -0
- google_browser_scraper/engines.py +54 -0
- google_browser_scraper/links.py +125 -0
- google_browser_scraper/mcp.py +172 -0
- google_browser_scraper/output.py +144 -0
- google_browser_scraper/parse.py +270 -0
- google_browser_scraper/proxy.py +120 -0
- google_browser_scraper/relay.py +277 -0
- google_browser_scraper/scraper.py +449 -0
- google_browser_scraper/server.py +240 -0
- google_browser_scraper/warmup.py +40 -0
- google_browser_scraper-0.0.1.dist-info/METADATA +223 -0
- google_browser_scraper-0.0.1.dist-info/RECORD +22 -0
- google_browser_scraper-0.0.1.dist-info/WHEEL +5 -0
- google_browser_scraper-0.0.1.dist-info/entry_points.txt +2 -0
- google_browser_scraper-0.0.1.dist-info/licenses/LICENSE +21 -0
- google_browser_scraper-0.0.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Google search results from a real browser through your own proxy.
|
|
2
|
+
|
|
3
|
+
from google_browser_scraper import Scraper, Settings, ProxyTemplate
|
|
4
|
+
|
|
5
|
+
proxy = ProxyTemplate("http://user-session-{session}:pass@gate.example.com:7000")
|
|
6
|
+
for record in Scraper(proxy, Settings(pages=2)).run(["best running shoes"]):
|
|
7
|
+
print(record["organic_results"])
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
__version__ = "0.0.1"
|
|
11
|
+
|
|
12
|
+
from .classify import Verdict, classify # noqa: E402
|
|
13
|
+
from .parse import parse_serp # noqa: E402
|
|
14
|
+
from .proxy import NodeMavenSource, ProxyTemplate # noqa: E402
|
|
15
|
+
from .scraper import ExitsRefused, Scraper, Settings # noqa: E402
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"ExitsRefused",
|
|
19
|
+
"NodeMavenSource",
|
|
20
|
+
"ProxyTemplate",
|
|
21
|
+
"Scraper",
|
|
22
|
+
"Settings",
|
|
23
|
+
"Verdict",
|
|
24
|
+
"__version__",
|
|
25
|
+
"classify",
|
|
26
|
+
"parse_serp",
|
|
27
|
+
]
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
"""One browser identity: one profile, one proxy session, one cookie jar.
|
|
2
|
+
|
|
3
|
+
- Patchright, headful: headless builds announce `HeadlessChrome` in the UA.
|
|
4
|
+
- A persistent context with `no_viewport=True`, so the page reports the real
|
|
5
|
+
screen rather than an emulated one.
|
|
6
|
+
- `locale` is never set: setting it through the context makes the main thread
|
|
7
|
+
and Web Workers disagree. `timezone_id` goes through CDP emulation and agrees
|
|
8
|
+
in workers, and is off unless asked for.
|
|
9
|
+
- The query is typed into the box, not sent as `/search?q=`.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import random
|
|
15
|
+
import shutil
|
|
16
|
+
import tempfile
|
|
17
|
+
import time
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
|
|
20
|
+
HOME_URL = "https://www.google.com/?hl=en"
|
|
21
|
+
SEARCH_BOX = "textarea[name='q'], input[name='q']"
|
|
22
|
+
READY = "#rso, #search"
|
|
23
|
+
# Reject first: it sets the smaller cookie. The middle selector is the
|
|
24
|
+
# redirect form of the wall served to EU exits.
|
|
25
|
+
CONSENT_BUTTONS = (
|
|
26
|
+
"button#W0wltc",
|
|
27
|
+
'form[action="https://consent.google.com/save"]'
|
|
28
|
+
':has(input[name="set_eom"][value="true"]) button',
|
|
29
|
+
"button#L2AGLb",
|
|
30
|
+
)
|
|
31
|
+
NAV_TIMEOUT_MS = 60_000
|
|
32
|
+
WARM_TIMEOUT_MS = 30_000
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True)
|
|
36
|
+
class Fetched:
|
|
37
|
+
"""What one navigation left on screen."""
|
|
38
|
+
|
|
39
|
+
url: str
|
|
40
|
+
status: int | None
|
|
41
|
+
html: str
|
|
42
|
+
consent_dismissed: bool = False
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class BrowserSession:
|
|
46
|
+
"""A Patchright persistent context bound to one proxy session.
|
|
47
|
+
|
|
48
|
+
Use as a context manager. The profile directory is temporary and removed on
|
|
49
|
+
close, so an identity is never reused by accident after it is retired.
|
|
50
|
+
|
|
51
|
+
`browser_args` are extra Chromium switches. **Never pass a
|
|
52
|
+
`--disable-features` here**: Chromium honours only the last such switch, so
|
|
53
|
+
one from the caller silently discards Playwright's own list, which includes
|
|
54
|
+
`OptimizationHints`.
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
name = "patchright"
|
|
58
|
+
|
|
59
|
+
def __init__(
|
|
60
|
+
self,
|
|
61
|
+
proxy: dict[str, str] | None = None,
|
|
62
|
+
*,
|
|
63
|
+
headless: bool = False,
|
|
64
|
+
channel: str | None = None,
|
|
65
|
+
timezone_id: str | None = None,
|
|
66
|
+
browser_args: list[str] | tuple[str, ...] = (),
|
|
67
|
+
rng: random.Random | None = None,
|
|
68
|
+
) -> None:
|
|
69
|
+
if any(a.startswith(("--disable-features", "--enable-features")) for a in browser_args):
|
|
70
|
+
raise ValueError(
|
|
71
|
+
"--disable-features/--enable-features would replace the browser's "
|
|
72
|
+
"own list rather than add to it; not accepted"
|
|
73
|
+
)
|
|
74
|
+
self._proxy = proxy
|
|
75
|
+
self._headless = headless
|
|
76
|
+
self._channel = channel
|
|
77
|
+
self._timezone_id = timezone_id
|
|
78
|
+
self._args = list(browser_args)
|
|
79
|
+
self._rng = rng or random.Random()
|
|
80
|
+
self._pw = None
|
|
81
|
+
self._context = None
|
|
82
|
+
self._profile = None
|
|
83
|
+
self.page = None
|
|
84
|
+
|
|
85
|
+
def __enter__(self) -> BrowserSession:
|
|
86
|
+
self._profile = tempfile.mkdtemp(prefix="gbs-profile-")
|
|
87
|
+
try:
|
|
88
|
+
self._context = self._launch(self._profile)
|
|
89
|
+
except Exception:
|
|
90
|
+
self.close()
|
|
91
|
+
raise
|
|
92
|
+
pages = self._context.pages
|
|
93
|
+
self.page = pages[0] if pages else self._context.new_page()
|
|
94
|
+
return self
|
|
95
|
+
|
|
96
|
+
def _launch(self, profile: str):
|
|
97
|
+
from patchright.sync_api import sync_playwright
|
|
98
|
+
|
|
99
|
+
self._pw = sync_playwright().start()
|
|
100
|
+
return self._pw.chromium.launch_persistent_context(
|
|
101
|
+
user_data_dir=profile,
|
|
102
|
+
headless=self._headless,
|
|
103
|
+
channel=self._channel,
|
|
104
|
+
proxy=self._proxy,
|
|
105
|
+
no_viewport=True,
|
|
106
|
+
timezone_id=self._timezone_id,
|
|
107
|
+
args=self._args or None,
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
def __exit__(self, *exc) -> None:
|
|
111
|
+
self.close()
|
|
112
|
+
|
|
113
|
+
def close(self) -> None:
|
|
114
|
+
for step in (
|
|
115
|
+
lambda: self._context and self._context.close(),
|
|
116
|
+
lambda: self._pw and self._pw.stop(),
|
|
117
|
+
):
|
|
118
|
+
try:
|
|
119
|
+
step()
|
|
120
|
+
except Exception:
|
|
121
|
+
pass
|
|
122
|
+
if self._profile:
|
|
123
|
+
shutil.rmtree(self._profile, ignore_errors=True)
|
|
124
|
+
self._context = self._pw = self._profile = self.page = None
|
|
125
|
+
|
|
126
|
+
# -- warming -------------------------------------------------------------
|
|
127
|
+
|
|
128
|
+
def visit(self, url: str, dwell_seconds: float) -> bool:
|
|
129
|
+
"""Open a warm-up page and stay on it. Returns whether it arrived."""
|
|
130
|
+
try:
|
|
131
|
+
self.page.goto(url, wait_until="domcontentloaded", timeout=WARM_TIMEOUT_MS)
|
|
132
|
+
except Exception:
|
|
133
|
+
return False
|
|
134
|
+
time.sleep(dwell_seconds)
|
|
135
|
+
return True
|
|
136
|
+
|
|
137
|
+
# -- searching -----------------------------------------------------------
|
|
138
|
+
|
|
139
|
+
def search(self, query: str) -> Fetched:
|
|
140
|
+
"""Type `query` into Google's box and return the page it led to."""
|
|
141
|
+
page = self.page
|
|
142
|
+
if not any(h.is_visible() for h in page.query_selector_all(SEARCH_BOX)):
|
|
143
|
+
page.goto(HOME_URL, wait_until="domcontentloaded", timeout=NAV_TIMEOUT_MS)
|
|
144
|
+
dismissed = self._dismiss_consent()
|
|
145
|
+
# Act on the handle that was found, never on the selector again: the
|
|
146
|
+
# selector also matches a hidden field, and `page.click` would pick it.
|
|
147
|
+
box = page.wait_for_selector(SEARCH_BOX, state="visible", timeout=NAV_TIMEOUT_MS)
|
|
148
|
+
box.click()
|
|
149
|
+
# A results page keeps the previous query in the box. Typing after it
|
|
150
|
+
# appends, and the appended query returns a real page for the wrong
|
|
151
|
+
# question. Select it so the typing replaces it.
|
|
152
|
+
if box.input_value():
|
|
153
|
+
box.press("ControlOrMeta+a")
|
|
154
|
+
box.type(query, delay=self._rng.randint(45, 140))
|
|
155
|
+
typed = box.input_value()
|
|
156
|
+
if typed != query:
|
|
157
|
+
raise RuntimeError(f"search box holds {typed!r} after typing {query!r}; not submitting")
|
|
158
|
+
# `no_wait_after=True` keeps `press` from waiting for the navigation a
|
|
159
|
+
# second time with Playwright's own 30 s default under ours.
|
|
160
|
+
with page.expect_navigation(wait_until="domcontentloaded", timeout=NAV_TIMEOUT_MS) as nav:
|
|
161
|
+
box.press("Enter", no_wait_after=True, timeout=NAV_TIMEOUT_MS)
|
|
162
|
+
return self._snapshot(nav.value, dismissed)
|
|
163
|
+
|
|
164
|
+
def next_page(self) -> Fetched | None:
|
|
165
|
+
"""Click "Next" on the current results page, if there is one."""
|
|
166
|
+
link = self.page.query_selector("a#pnnext")
|
|
167
|
+
if link is None or not link.is_visible():
|
|
168
|
+
return None
|
|
169
|
+
with self.page.expect_navigation(
|
|
170
|
+
wait_until="domcontentloaded", timeout=NAV_TIMEOUT_MS
|
|
171
|
+
) as nav:
|
|
172
|
+
link.click()
|
|
173
|
+
return self._snapshot(nav.value, False)
|
|
174
|
+
|
|
175
|
+
def _snapshot(self, response, dismissed: bool) -> Fetched:
|
|
176
|
+
# Google builds results in the browser; reading at domcontentloaded
|
|
177
|
+
# catches the "enable JavaScript" scaffold. A refusal never grows the
|
|
178
|
+
# container, so the short timeout is expected to expire on those.
|
|
179
|
+
try:
|
|
180
|
+
self.page.wait_for_selector(READY, state="attached", timeout=8_000)
|
|
181
|
+
except Exception:
|
|
182
|
+
pass
|
|
183
|
+
return Fetched(
|
|
184
|
+
url=self.page.url,
|
|
185
|
+
status=response.status if response is not None else None,
|
|
186
|
+
html=self.page.content(),
|
|
187
|
+
consent_dismissed=dismissed,
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
def _dismiss_consent(self) -> bool:
|
|
191
|
+
"""Clear the consent overlay if one is up. It is intermittent, so
|
|
192
|
+
finding none is normal and not an error."""
|
|
193
|
+
for selector in CONSENT_BUTTONS:
|
|
194
|
+
for handle in self.page.query_selector_all(selector):
|
|
195
|
+
if handle.is_visible():
|
|
196
|
+
handle.click(timeout=10_000)
|
|
197
|
+
try:
|
|
198
|
+
self.page.wait_for_load_state("domcontentloaded", timeout=10_000)
|
|
199
|
+
except Exception:
|
|
200
|
+
pass
|
|
201
|
+
self.page.wait_for_timeout(500)
|
|
202
|
+
return True
|
|
203
|
+
return False
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""Decide what Google served before anything tries to read results out of it.
|
|
2
|
+
|
|
3
|
+
A refusal must never reach the parser, because a parser handed a refusal returns
|
|
4
|
+
zero results, and "Google found nothing" and "Google refused this exit" need
|
|
5
|
+
opposite responses: the first is an answer, the second means drop the exit and
|
|
6
|
+
ask again from a fresh one.
|
|
7
|
+
|
|
8
|
+
Only structural markers are used - ids, paths and endpoints - so a translated
|
|
9
|
+
page classifies the same as an English one.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from urllib.parse import urlsplit
|
|
16
|
+
|
|
17
|
+
OK = "ok"
|
|
18
|
+
NO_RESULTS = "no_results"
|
|
19
|
+
SORRY = "sorry"
|
|
20
|
+
WALL = "wall"
|
|
21
|
+
CONSENT = "consent"
|
|
22
|
+
NO_JS = "no_js"
|
|
23
|
+
BLOCK = "block"
|
|
24
|
+
EMPTY = "empty"
|
|
25
|
+
|
|
26
|
+
#: Google looked at this exit and refused it. Drop the exit.
|
|
27
|
+
REFUSED = frozenset({SORRY, WALL, BLOCK})
|
|
28
|
+
#: Google answered the query. Keep the exit.
|
|
29
|
+
SERVED = frozenset({OK, NO_RESULTS})
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass(frozen=True)
|
|
33
|
+
class Verdict:
|
|
34
|
+
kind: str
|
|
35
|
+
reason: str
|
|
36
|
+
|
|
37
|
+
@property
|
|
38
|
+
def served(self) -> bool:
|
|
39
|
+
return self.kind in SERVED
|
|
40
|
+
|
|
41
|
+
@property
|
|
42
|
+
def refused(self) -> bool:
|
|
43
|
+
return self.kind in REFUSED
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def classify(url: str | None, html: str | None, status: int | None = None) -> Verdict:
|
|
47
|
+
"""Classify one Google response from its final URL, body and status.
|
|
48
|
+
|
|
49
|
+
Order matters. A served page can carry a dormant consent panel, so the
|
|
50
|
+
results test runs before the inline consent test; reversing them scores
|
|
51
|
+
working pages as interstitials.
|
|
52
|
+
"""
|
|
53
|
+
if not html:
|
|
54
|
+
return Verdict(EMPTY, "no body")
|
|
55
|
+
|
|
56
|
+
parts = urlsplit(url or "")
|
|
57
|
+
host = parts.netloc.lower()
|
|
58
|
+
path = parts.path.lower()
|
|
59
|
+
low = html.lower()
|
|
60
|
+
|
|
61
|
+
# The refusal redirects to /sorry/ with a reCAPTCHA. It has arrived as 429
|
|
62
|
+
# and as 200 with the same body, so the status is a hint and not the rule.
|
|
63
|
+
if host.startswith("sorry.") or path.startswith("/sorry"):
|
|
64
|
+
return Verdict(SORRY, "redirected to /sorry/ with a captcha")
|
|
65
|
+
# A redirect stub that was not followed: a tiny body pointing at /sorry/
|
|
66
|
+
# is the refusal, not a page.
|
|
67
|
+
if len(html) < 2000 and "/sorry/" in low:
|
|
68
|
+
return Verdict(SORRY, "short body pointing at /sorry/")
|
|
69
|
+
if status == 429:
|
|
70
|
+
return Verdict(SORRY, "HTTP 429")
|
|
71
|
+
if "unusual traffic from your computer" in low:
|
|
72
|
+
return Verdict(SORRY, "unusual-traffic page")
|
|
73
|
+
if host.startswith("consent."):
|
|
74
|
+
return Verdict(CONSENT, "consent interstitial")
|
|
75
|
+
|
|
76
|
+
has_rso = 'id="rso"' in low
|
|
77
|
+
has_search = 'id="search"' in low
|
|
78
|
+
if has_rso:
|
|
79
|
+
return Verdict(OK, "results container present")
|
|
80
|
+
|
|
81
|
+
# A real empty SERP keeps `#search`. Retiring an exit on it would throw away
|
|
82
|
+
# a working identity for a query that simply has no results.
|
|
83
|
+
if has_search:
|
|
84
|
+
return Verdict(NO_RESULTS, "search container present, no results in it")
|
|
85
|
+
|
|
86
|
+
if "consent.google.com" in low:
|
|
87
|
+
return Verdict(CONSENT, "consent wall served inline, no results behind it")
|
|
88
|
+
|
|
89
|
+
# The "enable JavaScript" scaffold. It also carries `emsg=SG_REL`, so it is
|
|
90
|
+
# tested before the wall below. From a browser that ran scripts it means
|
|
91
|
+
# Google declined to render, so the scraper drops the exit.
|
|
92
|
+
if "enablejs" in low:
|
|
93
|
+
return Verdict(NO_JS, "served the no-JavaScript scaffold")
|
|
94
|
+
|
|
95
|
+
# A SERP-shaped page with an error marker and no results container: a
|
|
96
|
+
# refusal that does not redirect.
|
|
97
|
+
if "emsg=sg_rel" in low:
|
|
98
|
+
return Verdict(WALL, "SERP-shaped wall without results (emsg=SG_REL)")
|
|
99
|
+
|
|
100
|
+
if "<noscript" in low and "<h3" not in low:
|
|
101
|
+
return Verdict(NO_JS, "a scaffold with no results and a noscript block")
|
|
102
|
+
|
|
103
|
+
if "error 403" in low or status == 403:
|
|
104
|
+
return Verdict(BLOCK, "HTTP 403 page")
|
|
105
|
+
return Verdict(BLOCK, "no results and no known interstitial")
|
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
"""Command line: `search`, `serve`, `mcp`, `parse` and `doctor`."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import os
|
|
8
|
+
import shlex
|
|
9
|
+
import sys
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from . import __version__, engines, warmup
|
|
13
|
+
from .classify import classify
|
|
14
|
+
from .output import WRITERS, open_writer, use_utf8
|
|
15
|
+
from .parse import parse_serp
|
|
16
|
+
from .proxy import NodeMavenSource, ProxyTemplate
|
|
17
|
+
from .scraper import ExitsRefused, Scraper, Settings
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def main(argv: list[str] | None = None) -> int:
|
|
21
|
+
use_utf8(sys.stdout)
|
|
22
|
+
use_utf8(sys.stderr, errors="backslashreplace")
|
|
23
|
+
parser = _parser()
|
|
24
|
+
args = parser.parse_args(argv)
|
|
25
|
+
if not getattr(args, "command", None):
|
|
26
|
+
parser.print_help()
|
|
27
|
+
return 2
|
|
28
|
+
try:
|
|
29
|
+
return args.handler(args)
|
|
30
|
+
except (ValueError, RuntimeError) as exc:
|
|
31
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
32
|
+
return 1
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _parser() -> argparse.ArgumentParser:
|
|
36
|
+
parser = argparse.ArgumentParser(
|
|
37
|
+
prog="google-browser-scraper",
|
|
38
|
+
description="Google results from a real browser through your own proxy.",
|
|
39
|
+
)
|
|
40
|
+
parser.add_argument("--version", action="version", version=__version__)
|
|
41
|
+
sub = parser.add_subparsers(dest="command")
|
|
42
|
+
|
|
43
|
+
search = sub.add_parser("search", help="run queries and write results")
|
|
44
|
+
search.add_argument("queries", nargs="*", help="queries; or use --file")
|
|
45
|
+
search.add_argument("-f", "--file", type=Path, help="one query per line")
|
|
46
|
+
_scraper_options(search)
|
|
47
|
+
search.add_argument("--pages", type=int, default=1, help="result pages per query (default 1)")
|
|
48
|
+
search.add_argument("--format", choices=list(WRITERS), default="jsonl")
|
|
49
|
+
search.add_argument("-o", "--output", type=Path, help="output file (default stdout)")
|
|
50
|
+
search.add_argument(
|
|
51
|
+
"--price-per-gb",
|
|
52
|
+
type=float,
|
|
53
|
+
help="your proxy price, to print the cost per 1000 pages at the end",
|
|
54
|
+
)
|
|
55
|
+
search.set_defaults(handler=_search)
|
|
56
|
+
|
|
57
|
+
serve = sub.add_parser("serve", help="SerpApi-style HTTP API: /search.json?q=...")
|
|
58
|
+
_scraper_options(serve)
|
|
59
|
+
serve.add_argument("--host", default="127.0.0.1")
|
|
60
|
+
serve.add_argument("--port", type=int, default=8000)
|
|
61
|
+
serve.add_argument("--max-pages", type=int, default=3)
|
|
62
|
+
serve.add_argument("--prewarm", action="store_true", help="warm an exit at startup")
|
|
63
|
+
serve.set_defaults(handler=_serve)
|
|
64
|
+
|
|
65
|
+
mcp = sub.add_parser("mcp", help="MCP server over stdio with a google_search tool")
|
|
66
|
+
_scraper_options(mcp)
|
|
67
|
+
mcp.add_argument("--prewarm", action="store_true", help="warm an exit at startup")
|
|
68
|
+
mcp.set_defaults(handler=_mcp)
|
|
69
|
+
|
|
70
|
+
parse = sub.add_parser("parse", help="parse a saved results page, offline")
|
|
71
|
+
parse.add_argument("file", type=Path)
|
|
72
|
+
parse.set_defaults(handler=_parse)
|
|
73
|
+
|
|
74
|
+
doctor = sub.add_parser("doctor", help="check this machine's browser; sends nothing")
|
|
75
|
+
doctor.add_argument("--engine", choices=list(engines.ENGINES), default=engines.DEFAULT_ENGINE)
|
|
76
|
+
doctor.add_argument("--headless", action="store_true")
|
|
77
|
+
doctor.add_argument("--channel")
|
|
78
|
+
doctor.add_argument("--browser-arg", action="append", default=[], dest="browser_args")
|
|
79
|
+
doctor.set_defaults(handler=_doctor)
|
|
80
|
+
return parser
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _scraper_options(p: argparse.ArgumentParser) -> None:
|
|
84
|
+
source = p.add_mutually_exclusive_group()
|
|
85
|
+
source.add_argument(
|
|
86
|
+
"--proxy",
|
|
87
|
+
help="proxy URL with a {session} placeholder for sticky sessions; also read from GBS_PROXY",
|
|
88
|
+
)
|
|
89
|
+
source.add_argument(
|
|
90
|
+
"--nodemaven",
|
|
91
|
+
action="store_true",
|
|
92
|
+
help="NodeMaven gateway from NODEMAVEN_LOGIN / NODEMAVEN_PASSWORD",
|
|
93
|
+
)
|
|
94
|
+
source.add_argument("--no-proxy", action="store_true", help="use this machine's own connection")
|
|
95
|
+
p.add_argument("--country", help="exit country for --nodemaven, e.g. us")
|
|
96
|
+
p.add_argument(
|
|
97
|
+
"--engine",
|
|
98
|
+
choices=list(engines.ENGINES),
|
|
99
|
+
default=engines.DEFAULT_ENGINE,
|
|
100
|
+
help="browser engine: patchright (default) or the experimental cloak",
|
|
101
|
+
)
|
|
102
|
+
p.add_argument(
|
|
103
|
+
"--warmup",
|
|
104
|
+
choices=list(warmup.LADDERS),
|
|
105
|
+
default=warmup.DEFAULT,
|
|
106
|
+
help=f"pages opened before the first query on an exit (default {warmup.DEFAULT})",
|
|
107
|
+
)
|
|
108
|
+
p.add_argument(
|
|
109
|
+
"--max-per-exit",
|
|
110
|
+
type=int,
|
|
111
|
+
default=10,
|
|
112
|
+
help="queries before an exit is retired even if still served",
|
|
113
|
+
)
|
|
114
|
+
p.add_argument(
|
|
115
|
+
"--gap",
|
|
116
|
+
type=_pair,
|
|
117
|
+
default=(8.0, 20.0),
|
|
118
|
+
metavar="LOW,HIGH",
|
|
119
|
+
help="seconds between queries on one exit (default 8,20)",
|
|
120
|
+
)
|
|
121
|
+
p.add_argument(
|
|
122
|
+
"--headless",
|
|
123
|
+
action="store_true",
|
|
124
|
+
help="run without a window. Not recommended: see `doctor`",
|
|
125
|
+
)
|
|
126
|
+
p.add_argument("--channel", help="browser channel, e.g. chrome to use installed Chrome")
|
|
127
|
+
p.add_argument("--timezone", help="IANA timezone for the browser")
|
|
128
|
+
p.add_argument(
|
|
129
|
+
"--browser-arg",
|
|
130
|
+
action="append",
|
|
131
|
+
default=[],
|
|
132
|
+
dest="browser_args",
|
|
133
|
+
help="extra Chromium switch, repeatable; also GBS_BROWSER_ARGS",
|
|
134
|
+
)
|
|
135
|
+
p.add_argument(
|
|
136
|
+
"--no-relay",
|
|
137
|
+
action="store_true",
|
|
138
|
+
help="hand the proxy to the browser directly, without the traffic count",
|
|
139
|
+
)
|
|
140
|
+
p.add_argument(
|
|
141
|
+
"--save-html",
|
|
142
|
+
type=Path,
|
|
143
|
+
metavar="DIR",
|
|
144
|
+
help="keep every page's HTML here, gzipped (they carry the exit's location)",
|
|
145
|
+
)
|
|
146
|
+
p.add_argument(
|
|
147
|
+
"--no-resolve-links",
|
|
148
|
+
action="store_true",
|
|
149
|
+
help="leave result links as Google /goto redirects (saves one small request per result)",
|
|
150
|
+
)
|
|
151
|
+
p.add_argument("-v", "--verbose", action="store_true", help="log identities to stderr")
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _build(args, *, pages: int = 1) -> tuple[Scraper, str]:
|
|
155
|
+
if args.nodemaven:
|
|
156
|
+
proxy = NodeMavenSource(country=args.country)
|
|
157
|
+
elif args.no_proxy:
|
|
158
|
+
proxy = None
|
|
159
|
+
else:
|
|
160
|
+
url = args.proxy or os.environ.get("GBS_PROXY")
|
|
161
|
+
if not url:
|
|
162
|
+
raise ValueError("give --proxy, --nodemaven or --no-proxy")
|
|
163
|
+
proxy = ProxyTemplate(url)
|
|
164
|
+
if args.country and not args.nodemaven:
|
|
165
|
+
raise ValueError(
|
|
166
|
+
"--country only applies to --nodemaven; put the country in your "
|
|
167
|
+
"provider's proxy URL instead"
|
|
168
|
+
)
|
|
169
|
+
if proxy is not None and not proxy.sticky:
|
|
170
|
+
print(
|
|
171
|
+
"warning: the proxy URL has no {session} placeholder, so the exit may change "
|
|
172
|
+
"between requests and warming it does nothing",
|
|
173
|
+
file=sys.stderr,
|
|
174
|
+
)
|
|
175
|
+
browser_args = tuple(args.browser_args) + tuple(
|
|
176
|
+
shlex.split(os.environ.get("GBS_BROWSER_ARGS", ""))
|
|
177
|
+
)
|
|
178
|
+
settings = Settings(
|
|
179
|
+
engine=args.engine,
|
|
180
|
+
warmup=args.warmup,
|
|
181
|
+
pages=pages,
|
|
182
|
+
max_queries_per_exit=args.max_per_exit,
|
|
183
|
+
gap_seconds=args.gap,
|
|
184
|
+
headless=args.headless,
|
|
185
|
+
channel=args.channel,
|
|
186
|
+
timezone_id=args.timezone,
|
|
187
|
+
browser_args=browser_args,
|
|
188
|
+
use_relay=not args.no_relay,
|
|
189
|
+
save_html=str(args.save_html) if args.save_html else None,
|
|
190
|
+
resolve_links=not args.no_resolve_links,
|
|
191
|
+
)
|
|
192
|
+
log = (lambda e: print(json.dumps(e), file=sys.stderr)) if args.verbose else None
|
|
193
|
+
described = proxy.describe() if proxy else "none (this machine's own address)"
|
|
194
|
+
if args.verbose:
|
|
195
|
+
print(f"proxy: {described}; engine: {args.engine}", file=sys.stderr)
|
|
196
|
+
return Scraper(proxy, settings, on_event=log), described
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _search(args) -> int:
|
|
200
|
+
queries = list(args.queries)
|
|
201
|
+
if args.file:
|
|
202
|
+
queries += args.file.read_text(encoding="utf-8").splitlines()
|
|
203
|
+
if not any(q.strip() for q in queries):
|
|
204
|
+
raise ValueError("no queries given")
|
|
205
|
+
scraper, _ = _build(args, pages=args.pages)
|
|
206
|
+
stream = args.output.open("w", encoding="utf-8", newline="") if args.output else sys.stdout
|
|
207
|
+
writer = open_writer(args.format, stream)
|
|
208
|
+
failed, code = 0, 0
|
|
209
|
+
try:
|
|
210
|
+
for record in scraper.run(queries):
|
|
211
|
+
failed += record["search_metadata"]["status"] != "Success"
|
|
212
|
+
writer.write(record)
|
|
213
|
+
except ExitsRefused as exc:
|
|
214
|
+
print(f"stopped: {exc}", file=sys.stderr)
|
|
215
|
+
code = 3
|
|
216
|
+
finally:
|
|
217
|
+
writer.close()
|
|
218
|
+
if args.output:
|
|
219
|
+
stream.close()
|
|
220
|
+
print("summary: " + json.dumps(scraper.stats.summary(args.price_per_gb)), file=sys.stderr)
|
|
221
|
+
return code or (4 if failed else 0)
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _serve(args) -> int:
|
|
225
|
+
from .server import Worker, serve
|
|
226
|
+
|
|
227
|
+
worker = Worker(lambda: _build(args)[0], prewarm=args.prewarm).start()
|
|
228
|
+
serve(worker, host=args.host, port=args.port, max_pages=args.max_pages)
|
|
229
|
+
return 0
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _mcp(args) -> int:
|
|
233
|
+
from .mcp import McpServer
|
|
234
|
+
from .server import Worker
|
|
235
|
+
|
|
236
|
+
worker = Worker(lambda: _build(args)[0], prewarm=args.prewarm).start()
|
|
237
|
+
try:
|
|
238
|
+
McpServer(worker).serve()
|
|
239
|
+
finally:
|
|
240
|
+
worker.stop()
|
|
241
|
+
return 0
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _parse(args) -> int:
|
|
245
|
+
page = args.file.read_text(encoding="utf-8", errors="replace")
|
|
246
|
+
verdict = classify("https://www.google.com/search", page)
|
|
247
|
+
out = {"page_verdict": verdict.kind, "verdict_reason": verdict.reason}
|
|
248
|
+
if verdict.served:
|
|
249
|
+
out.update(parse_serp(page))
|
|
250
|
+
json.dump(out, sys.stdout, ensure_ascii=False, indent=2)
|
|
251
|
+
sys.stdout.write("\n")
|
|
252
|
+
return 0 if verdict.served else 4
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _doctor(args) -> int:
|
|
256
|
+
from .doctor import FAIL, assess, read_fingerprint
|
|
257
|
+
|
|
258
|
+
browser_args = tuple(args.browser_args) + tuple(
|
|
259
|
+
shlex.split(os.environ.get("GBS_BROWSER_ARGS", ""))
|
|
260
|
+
)
|
|
261
|
+
fp = read_fingerprint(
|
|
262
|
+
engine=args.engine, headless=args.headless, channel=args.channel, browser_args=browser_args
|
|
263
|
+
)
|
|
264
|
+
checks = assess(fp)
|
|
265
|
+
width = max(len(c.name) for c in checks)
|
|
266
|
+
print(f"engine: {args.engine}")
|
|
267
|
+
for c in checks:
|
|
268
|
+
print(f"[{c.status:4}] {c.name:<{width}} {c.detail}")
|
|
269
|
+
return 1 if any(c.status == FAIL for c in checks) else 0
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _pair(text: str) -> tuple[float, float]:
|
|
273
|
+
try:
|
|
274
|
+
low, high = (float(x) for x in text.split(","))
|
|
275
|
+
except ValueError:
|
|
276
|
+
raise argparse.ArgumentTypeError("expected LOW,HIGH, e.g. 8,20") from None
|
|
277
|
+
return low, high
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
if __name__ == "__main__": # pragma: no cover
|
|
281
|
+
sys.exit(main())
|