google-browser-scraper 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,27 @@
1
+ """Google search results from a real browser through your own proxy.
2
+
3
+ from google_browser_scraper import Scraper, Settings, ProxyTemplate
4
+
5
+ proxy = ProxyTemplate("http://user-session-{session}:pass@gate.example.com:7000")
6
+ for record in Scraper(proxy, Settings(pages=2)).run(["best running shoes"]):
7
+ print(record["organic_results"])
8
+ """
9
+
10
+ __version__ = "0.0.1"
11
+
12
+ from .classify import Verdict, classify # noqa: E402
13
+ from .parse import parse_serp # noqa: E402
14
+ from .proxy import NodeMavenSource, ProxyTemplate # noqa: E402
15
+ from .scraper import ExitsRefused, Scraper, Settings # noqa: E402
16
+
17
+ __all__ = [
18
+ "ExitsRefused",
19
+ "NodeMavenSource",
20
+ "ProxyTemplate",
21
+ "Scraper",
22
+ "Settings",
23
+ "Verdict",
24
+ "__version__",
25
+ "classify",
26
+ "parse_serp",
27
+ ]
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
@@ -0,0 +1,203 @@
1
+ """One browser identity: one profile, one proxy session, one cookie jar.
2
+
3
+ - Patchright, headful: headless builds announce `HeadlessChrome` in the UA.
4
+ - A persistent context with `no_viewport=True`, so the page reports the real
5
+ screen rather than an emulated one.
6
+ - `locale` is never set: setting it through the context makes the main thread
7
+ and Web Workers disagree. `timezone_id` goes through CDP emulation and agrees
8
+ in workers, and is off unless asked for.
9
+ - The query is typed into the box, not sent as `/search?q=`.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import random
15
+ import shutil
16
+ import tempfile
17
+ import time
18
+ from dataclasses import dataclass
19
+
20
+ HOME_URL = "https://www.google.com/?hl=en"
21
+ SEARCH_BOX = "textarea[name='q'], input[name='q']"
22
+ READY = "#rso, #search"
23
+ # Reject first: it sets the smaller cookie. The middle selector is the
24
+ # redirect form of the wall served to EU exits.
25
+ CONSENT_BUTTONS = (
26
+ "button#W0wltc",
27
+ 'form[action="https://consent.google.com/save"]'
28
+ ':has(input[name="set_eom"][value="true"]) button',
29
+ "button#L2AGLb",
30
+ )
31
+ NAV_TIMEOUT_MS = 60_000
32
+ WARM_TIMEOUT_MS = 30_000
33
+
34
+
35
+ @dataclass(frozen=True)
36
+ class Fetched:
37
+ """What one navigation left on screen."""
38
+
39
+ url: str
40
+ status: int | None
41
+ html: str
42
+ consent_dismissed: bool = False
43
+
44
+
45
+ class BrowserSession:
46
+ """A Patchright persistent context bound to one proxy session.
47
+
48
+ Use as a context manager. The profile directory is temporary and removed on
49
+ close, so an identity is never reused by accident after it is retired.
50
+
51
+ `browser_args` are extra Chromium switches. **Never pass a
52
+ `--disable-features` here**: Chromium honours only the last such switch, so
53
+ one from the caller silently discards Playwright's own list, which includes
54
+ `OptimizationHints`.
55
+ """
56
+
57
+ name = "patchright"
58
+
59
+ def __init__(
60
+ self,
61
+ proxy: dict[str, str] | None = None,
62
+ *,
63
+ headless: bool = False,
64
+ channel: str | None = None,
65
+ timezone_id: str | None = None,
66
+ browser_args: list[str] | tuple[str, ...] = (),
67
+ rng: random.Random | None = None,
68
+ ) -> None:
69
+ if any(a.startswith(("--disable-features", "--enable-features")) for a in browser_args):
70
+ raise ValueError(
71
+ "--disable-features/--enable-features would replace the browser's "
72
+ "own list rather than add to it; not accepted"
73
+ )
74
+ self._proxy = proxy
75
+ self._headless = headless
76
+ self._channel = channel
77
+ self._timezone_id = timezone_id
78
+ self._args = list(browser_args)
79
+ self._rng = rng or random.Random()
80
+ self._pw = None
81
+ self._context = None
82
+ self._profile = None
83
+ self.page = None
84
+
85
+ def __enter__(self) -> BrowserSession:
86
+ self._profile = tempfile.mkdtemp(prefix="gbs-profile-")
87
+ try:
88
+ self._context = self._launch(self._profile)
89
+ except Exception:
90
+ self.close()
91
+ raise
92
+ pages = self._context.pages
93
+ self.page = pages[0] if pages else self._context.new_page()
94
+ return self
95
+
96
+ def _launch(self, profile: str):
97
+ from patchright.sync_api import sync_playwright
98
+
99
+ self._pw = sync_playwright().start()
100
+ return self._pw.chromium.launch_persistent_context(
101
+ user_data_dir=profile,
102
+ headless=self._headless,
103
+ channel=self._channel,
104
+ proxy=self._proxy,
105
+ no_viewport=True,
106
+ timezone_id=self._timezone_id,
107
+ args=self._args or None,
108
+ )
109
+
110
+ def __exit__(self, *exc) -> None:
111
+ self.close()
112
+
113
+ def close(self) -> None:
114
+ for step in (
115
+ lambda: self._context and self._context.close(),
116
+ lambda: self._pw and self._pw.stop(),
117
+ ):
118
+ try:
119
+ step()
120
+ except Exception:
121
+ pass
122
+ if self._profile:
123
+ shutil.rmtree(self._profile, ignore_errors=True)
124
+ self._context = self._pw = self._profile = self.page = None
125
+
126
+ # -- warming -------------------------------------------------------------
127
+
128
+ def visit(self, url: str, dwell_seconds: float) -> bool:
129
+ """Open a warm-up page and stay on it. Returns whether it arrived."""
130
+ try:
131
+ self.page.goto(url, wait_until="domcontentloaded", timeout=WARM_TIMEOUT_MS)
132
+ except Exception:
133
+ return False
134
+ time.sleep(dwell_seconds)
135
+ return True
136
+
137
+ # -- searching -----------------------------------------------------------
138
+
139
+ def search(self, query: str) -> Fetched:
140
+ """Type `query` into Google's box and return the page it led to."""
141
+ page = self.page
142
+ if not any(h.is_visible() for h in page.query_selector_all(SEARCH_BOX)):
143
+ page.goto(HOME_URL, wait_until="domcontentloaded", timeout=NAV_TIMEOUT_MS)
144
+ dismissed = self._dismiss_consent()
145
+ # Act on the handle that was found, never on the selector again: the
146
+ # selector also matches a hidden field, and `page.click` would pick it.
147
+ box = page.wait_for_selector(SEARCH_BOX, state="visible", timeout=NAV_TIMEOUT_MS)
148
+ box.click()
149
+ # A results page keeps the previous query in the box. Typing after it
150
+ # appends, and the appended query returns a real page for the wrong
151
+ # question. Select it so the typing replaces it.
152
+ if box.input_value():
153
+ box.press("ControlOrMeta+a")
154
+ box.type(query, delay=self._rng.randint(45, 140))
155
+ typed = box.input_value()
156
+ if typed != query:
157
+ raise RuntimeError(f"search box holds {typed!r} after typing {query!r}; not submitting")
158
+ # `no_wait_after=True` keeps `press` from waiting for the navigation a
159
+ # second time with Playwright's own 30 s default under ours.
160
+ with page.expect_navigation(wait_until="domcontentloaded", timeout=NAV_TIMEOUT_MS) as nav:
161
+ box.press("Enter", no_wait_after=True, timeout=NAV_TIMEOUT_MS)
162
+ return self._snapshot(nav.value, dismissed)
163
+
164
+ def next_page(self) -> Fetched | None:
165
+ """Click "Next" on the current results page, if there is one."""
166
+ link = self.page.query_selector("a#pnnext")
167
+ if link is None or not link.is_visible():
168
+ return None
169
+ with self.page.expect_navigation(
170
+ wait_until="domcontentloaded", timeout=NAV_TIMEOUT_MS
171
+ ) as nav:
172
+ link.click()
173
+ return self._snapshot(nav.value, False)
174
+
175
+ def _snapshot(self, response, dismissed: bool) -> Fetched:
176
+ # Google builds results in the browser; reading at domcontentloaded
177
+ # catches the "enable JavaScript" scaffold. A refusal never grows the
178
+ # container, so the short timeout is expected to expire on those.
179
+ try:
180
+ self.page.wait_for_selector(READY, state="attached", timeout=8_000)
181
+ except Exception:
182
+ pass
183
+ return Fetched(
184
+ url=self.page.url,
185
+ status=response.status if response is not None else None,
186
+ html=self.page.content(),
187
+ consent_dismissed=dismissed,
188
+ )
189
+
190
+ def _dismiss_consent(self) -> bool:
191
+ """Clear the consent overlay if one is up. It is intermittent, so
192
+ finding none is normal and not an error."""
193
+ for selector in CONSENT_BUTTONS:
194
+ for handle in self.page.query_selector_all(selector):
195
+ if handle.is_visible():
196
+ handle.click(timeout=10_000)
197
+ try:
198
+ self.page.wait_for_load_state("domcontentloaded", timeout=10_000)
199
+ except Exception:
200
+ pass
201
+ self.page.wait_for_timeout(500)
202
+ return True
203
+ return False
@@ -0,0 +1,105 @@
1
+ """Decide what Google served before anything tries to read results out of it.
2
+
3
+ A refusal must never reach the parser, because a parser handed a refusal returns
4
+ zero results, and "Google found nothing" and "Google refused this exit" need
5
+ opposite responses: the first is an answer, the second means drop the exit and
6
+ ask again from a fresh one.
7
+
8
+ Only structural markers are used - ids, paths and endpoints - so a translated
9
+ page classifies the same as an English one.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from dataclasses import dataclass
15
+ from urllib.parse import urlsplit
16
+
17
+ OK = "ok"
18
+ NO_RESULTS = "no_results"
19
+ SORRY = "sorry"
20
+ WALL = "wall"
21
+ CONSENT = "consent"
22
+ NO_JS = "no_js"
23
+ BLOCK = "block"
24
+ EMPTY = "empty"
25
+
26
+ #: Google looked at this exit and refused it. Drop the exit.
27
+ REFUSED = frozenset({SORRY, WALL, BLOCK})
28
+ #: Google answered the query. Keep the exit.
29
+ SERVED = frozenset({OK, NO_RESULTS})
30
+
31
+
32
+ @dataclass(frozen=True)
33
+ class Verdict:
34
+ kind: str
35
+ reason: str
36
+
37
+ @property
38
+ def served(self) -> bool:
39
+ return self.kind in SERVED
40
+
41
+ @property
42
+ def refused(self) -> bool:
43
+ return self.kind in REFUSED
44
+
45
+
46
+ def classify(url: str | None, html: str | None, status: int | None = None) -> Verdict:
47
+ """Classify one Google response from its final URL, body and status.
48
+
49
+ Order matters. A served page can carry a dormant consent panel, so the
50
+ results test runs before the inline consent test; reversing them scores
51
+ working pages as interstitials.
52
+ """
53
+ if not html:
54
+ return Verdict(EMPTY, "no body")
55
+
56
+ parts = urlsplit(url or "")
57
+ host = parts.netloc.lower()
58
+ path = parts.path.lower()
59
+ low = html.lower()
60
+
61
+ # The refusal redirects to /sorry/ with a reCAPTCHA. It has arrived as 429
62
+ # and as 200 with the same body, so the status is a hint and not the rule.
63
+ if host.startswith("sorry.") or path.startswith("/sorry"):
64
+ return Verdict(SORRY, "redirected to /sorry/ with a captcha")
65
+ # A redirect stub that was not followed: a tiny body pointing at /sorry/
66
+ # is the refusal, not a page.
67
+ if len(html) < 2000 and "/sorry/" in low:
68
+ return Verdict(SORRY, "short body pointing at /sorry/")
69
+ if status == 429:
70
+ return Verdict(SORRY, "HTTP 429")
71
+ if "unusual traffic from your computer" in low:
72
+ return Verdict(SORRY, "unusual-traffic page")
73
+ if host.startswith("consent."):
74
+ return Verdict(CONSENT, "consent interstitial")
75
+
76
+ has_rso = 'id="rso"' in low
77
+ has_search = 'id="search"' in low
78
+ if has_rso:
79
+ return Verdict(OK, "results container present")
80
+
81
+ # A real empty SERP keeps `#search`. Retiring an exit on it would throw away
82
+ # a working identity for a query that simply has no results.
83
+ if has_search:
84
+ return Verdict(NO_RESULTS, "search container present, no results in it")
85
+
86
+ if "consent.google.com" in low:
87
+ return Verdict(CONSENT, "consent wall served inline, no results behind it")
88
+
89
+ # The "enable JavaScript" scaffold. It also carries `emsg=SG_REL`, so it is
90
+ # tested before the wall below. From a browser that ran scripts it means
91
+ # Google declined to render, so the scraper drops the exit.
92
+ if "enablejs" in low:
93
+ return Verdict(NO_JS, "served the no-JavaScript scaffold")
94
+
95
+ # A SERP-shaped page with an error marker and no results container: a
96
+ # refusal that does not redirect.
97
+ if "emsg=sg_rel" in low:
98
+ return Verdict(WALL, "SERP-shaped wall without results (emsg=SG_REL)")
99
+
100
+ if "<noscript" in low and "<h3" not in low:
101
+ return Verdict(NO_JS, "a scaffold with no results and a noscript block")
102
+
103
+ if "error 403" in low or status == 403:
104
+ return Verdict(BLOCK, "HTTP 403 page")
105
+ return Verdict(BLOCK, "no results and no known interstitial")
@@ -0,0 +1,281 @@
1
+ """Command line: `search`, `serve`, `mcp`, `parse` and `doctor`."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import os
8
+ import shlex
9
+ import sys
10
+ from pathlib import Path
11
+
12
+ from . import __version__, engines, warmup
13
+ from .classify import classify
14
+ from .output import WRITERS, open_writer, use_utf8
15
+ from .parse import parse_serp
16
+ from .proxy import NodeMavenSource, ProxyTemplate
17
+ from .scraper import ExitsRefused, Scraper, Settings
18
+
19
+
20
+ def main(argv: list[str] | None = None) -> int:
21
+ use_utf8(sys.stdout)
22
+ use_utf8(sys.stderr, errors="backslashreplace")
23
+ parser = _parser()
24
+ args = parser.parse_args(argv)
25
+ if not getattr(args, "command", None):
26
+ parser.print_help()
27
+ return 2
28
+ try:
29
+ return args.handler(args)
30
+ except (ValueError, RuntimeError) as exc:
31
+ print(f"error: {exc}", file=sys.stderr)
32
+ return 1
33
+
34
+
35
+ def _parser() -> argparse.ArgumentParser:
36
+ parser = argparse.ArgumentParser(
37
+ prog="google-browser-scraper",
38
+ description="Google results from a real browser through your own proxy.",
39
+ )
40
+ parser.add_argument("--version", action="version", version=__version__)
41
+ sub = parser.add_subparsers(dest="command")
42
+
43
+ search = sub.add_parser("search", help="run queries and write results")
44
+ search.add_argument("queries", nargs="*", help="queries; or use --file")
45
+ search.add_argument("-f", "--file", type=Path, help="one query per line")
46
+ _scraper_options(search)
47
+ search.add_argument("--pages", type=int, default=1, help="result pages per query (default 1)")
48
+ search.add_argument("--format", choices=list(WRITERS), default="jsonl")
49
+ search.add_argument("-o", "--output", type=Path, help="output file (default stdout)")
50
+ search.add_argument(
51
+ "--price-per-gb",
52
+ type=float,
53
+ help="your proxy price, to print the cost per 1000 pages at the end",
54
+ )
55
+ search.set_defaults(handler=_search)
56
+
57
+ serve = sub.add_parser("serve", help="SerpApi-style HTTP API: /search.json?q=...")
58
+ _scraper_options(serve)
59
+ serve.add_argument("--host", default="127.0.0.1")
60
+ serve.add_argument("--port", type=int, default=8000)
61
+ serve.add_argument("--max-pages", type=int, default=3)
62
+ serve.add_argument("--prewarm", action="store_true", help="warm an exit at startup")
63
+ serve.set_defaults(handler=_serve)
64
+
65
+ mcp = sub.add_parser("mcp", help="MCP server over stdio with a google_search tool")
66
+ _scraper_options(mcp)
67
+ mcp.add_argument("--prewarm", action="store_true", help="warm an exit at startup")
68
+ mcp.set_defaults(handler=_mcp)
69
+
70
+ parse = sub.add_parser("parse", help="parse a saved results page, offline")
71
+ parse.add_argument("file", type=Path)
72
+ parse.set_defaults(handler=_parse)
73
+
74
+ doctor = sub.add_parser("doctor", help="check this machine's browser; sends nothing")
75
+ doctor.add_argument("--engine", choices=list(engines.ENGINES), default=engines.DEFAULT_ENGINE)
76
+ doctor.add_argument("--headless", action="store_true")
77
+ doctor.add_argument("--channel")
78
+ doctor.add_argument("--browser-arg", action="append", default=[], dest="browser_args")
79
+ doctor.set_defaults(handler=_doctor)
80
+ return parser
81
+
82
+
83
+ def _scraper_options(p: argparse.ArgumentParser) -> None:
84
+ source = p.add_mutually_exclusive_group()
85
+ source.add_argument(
86
+ "--proxy",
87
+ help="proxy URL with a {session} placeholder for sticky sessions; also read from GBS_PROXY",
88
+ )
89
+ source.add_argument(
90
+ "--nodemaven",
91
+ action="store_true",
92
+ help="NodeMaven gateway from NODEMAVEN_LOGIN / NODEMAVEN_PASSWORD",
93
+ )
94
+ source.add_argument("--no-proxy", action="store_true", help="use this machine's own connection")
95
+ p.add_argument("--country", help="exit country for --nodemaven, e.g. us")
96
+ p.add_argument(
97
+ "--engine",
98
+ choices=list(engines.ENGINES),
99
+ default=engines.DEFAULT_ENGINE,
100
+ help="browser engine: patchright (default) or the experimental cloak",
101
+ )
102
+ p.add_argument(
103
+ "--warmup",
104
+ choices=list(warmup.LADDERS),
105
+ default=warmup.DEFAULT,
106
+ help=f"pages opened before the first query on an exit (default {warmup.DEFAULT})",
107
+ )
108
+ p.add_argument(
109
+ "--max-per-exit",
110
+ type=int,
111
+ default=10,
112
+ help="queries before an exit is retired even if still served",
113
+ )
114
+ p.add_argument(
115
+ "--gap",
116
+ type=_pair,
117
+ default=(8.0, 20.0),
118
+ metavar="LOW,HIGH",
119
+ help="seconds between queries on one exit (default 8,20)",
120
+ )
121
+ p.add_argument(
122
+ "--headless",
123
+ action="store_true",
124
+ help="run without a window. Not recommended: see `doctor`",
125
+ )
126
+ p.add_argument("--channel", help="browser channel, e.g. chrome to use installed Chrome")
127
+ p.add_argument("--timezone", help="IANA timezone for the browser")
128
+ p.add_argument(
129
+ "--browser-arg",
130
+ action="append",
131
+ default=[],
132
+ dest="browser_args",
133
+ help="extra Chromium switch, repeatable; also GBS_BROWSER_ARGS",
134
+ )
135
+ p.add_argument(
136
+ "--no-relay",
137
+ action="store_true",
138
+ help="hand the proxy to the browser directly, without the traffic count",
139
+ )
140
+ p.add_argument(
141
+ "--save-html",
142
+ type=Path,
143
+ metavar="DIR",
144
+ help="keep every page's HTML here, gzipped (they carry the exit's location)",
145
+ )
146
+ p.add_argument(
147
+ "--no-resolve-links",
148
+ action="store_true",
149
+ help="leave result links as Google /goto redirects (saves one small request per result)",
150
+ )
151
+ p.add_argument("-v", "--verbose", action="store_true", help="log identities to stderr")
152
+
153
+
154
+ def _build(args, *, pages: int = 1) -> tuple[Scraper, str]:
155
+ if args.nodemaven:
156
+ proxy = NodeMavenSource(country=args.country)
157
+ elif args.no_proxy:
158
+ proxy = None
159
+ else:
160
+ url = args.proxy or os.environ.get("GBS_PROXY")
161
+ if not url:
162
+ raise ValueError("give --proxy, --nodemaven or --no-proxy")
163
+ proxy = ProxyTemplate(url)
164
+ if args.country and not args.nodemaven:
165
+ raise ValueError(
166
+ "--country only applies to --nodemaven; put the country in your "
167
+ "provider's proxy URL instead"
168
+ )
169
+ if proxy is not None and not proxy.sticky:
170
+ print(
171
+ "warning: the proxy URL has no {session} placeholder, so the exit may change "
172
+ "between requests and warming it does nothing",
173
+ file=sys.stderr,
174
+ )
175
+ browser_args = tuple(args.browser_args) + tuple(
176
+ shlex.split(os.environ.get("GBS_BROWSER_ARGS", ""))
177
+ )
178
+ settings = Settings(
179
+ engine=args.engine,
180
+ warmup=args.warmup,
181
+ pages=pages,
182
+ max_queries_per_exit=args.max_per_exit,
183
+ gap_seconds=args.gap,
184
+ headless=args.headless,
185
+ channel=args.channel,
186
+ timezone_id=args.timezone,
187
+ browser_args=browser_args,
188
+ use_relay=not args.no_relay,
189
+ save_html=str(args.save_html) if args.save_html else None,
190
+ resolve_links=not args.no_resolve_links,
191
+ )
192
+ log = (lambda e: print(json.dumps(e), file=sys.stderr)) if args.verbose else None
193
+ described = proxy.describe() if proxy else "none (this machine's own address)"
194
+ if args.verbose:
195
+ print(f"proxy: {described}; engine: {args.engine}", file=sys.stderr)
196
+ return Scraper(proxy, settings, on_event=log), described
197
+
198
+
199
+ def _search(args) -> int:
200
+ queries = list(args.queries)
201
+ if args.file:
202
+ queries += args.file.read_text(encoding="utf-8").splitlines()
203
+ if not any(q.strip() for q in queries):
204
+ raise ValueError("no queries given")
205
+ scraper, _ = _build(args, pages=args.pages)
206
+ stream = args.output.open("w", encoding="utf-8", newline="") if args.output else sys.stdout
207
+ writer = open_writer(args.format, stream)
208
+ failed, code = 0, 0
209
+ try:
210
+ for record in scraper.run(queries):
211
+ failed += record["search_metadata"]["status"] != "Success"
212
+ writer.write(record)
213
+ except ExitsRefused as exc:
214
+ print(f"stopped: {exc}", file=sys.stderr)
215
+ code = 3
216
+ finally:
217
+ writer.close()
218
+ if args.output:
219
+ stream.close()
220
+ print("summary: " + json.dumps(scraper.stats.summary(args.price_per_gb)), file=sys.stderr)
221
+ return code or (4 if failed else 0)
222
+
223
+
224
+ def _serve(args) -> int:
225
+ from .server import Worker, serve
226
+
227
+ worker = Worker(lambda: _build(args)[0], prewarm=args.prewarm).start()
228
+ serve(worker, host=args.host, port=args.port, max_pages=args.max_pages)
229
+ return 0
230
+
231
+
232
+ def _mcp(args) -> int:
233
+ from .mcp import McpServer
234
+ from .server import Worker
235
+
236
+ worker = Worker(lambda: _build(args)[0], prewarm=args.prewarm).start()
237
+ try:
238
+ McpServer(worker).serve()
239
+ finally:
240
+ worker.stop()
241
+ return 0
242
+
243
+
244
+ def _parse(args) -> int:
245
+ page = args.file.read_text(encoding="utf-8", errors="replace")
246
+ verdict = classify("https://www.google.com/search", page)
247
+ out = {"page_verdict": verdict.kind, "verdict_reason": verdict.reason}
248
+ if verdict.served:
249
+ out.update(parse_serp(page))
250
+ json.dump(out, sys.stdout, ensure_ascii=False, indent=2)
251
+ sys.stdout.write("\n")
252
+ return 0 if verdict.served else 4
253
+
254
+
255
+ def _doctor(args) -> int:
256
+ from .doctor import FAIL, assess, read_fingerprint
257
+
258
+ browser_args = tuple(args.browser_args) + tuple(
259
+ shlex.split(os.environ.get("GBS_BROWSER_ARGS", ""))
260
+ )
261
+ fp = read_fingerprint(
262
+ engine=args.engine, headless=args.headless, channel=args.channel, browser_args=browser_args
263
+ )
264
+ checks = assess(fp)
265
+ width = max(len(c.name) for c in checks)
266
+ print(f"engine: {args.engine}")
267
+ for c in checks:
268
+ print(f"[{c.status:4}] {c.name:<{width}} {c.detail}")
269
+ return 1 if any(c.status == FAIL for c in checks) else 0
270
+
271
+
272
+ def _pair(text: str) -> tuple[float, float]:
273
+ try:
274
+ low, high = (float(x) for x in text.split(","))
275
+ except ValueError:
276
+ raise argparse.ArgumentTypeError("expected LOW,HIGH, e.g. 8,20") from None
277
+ return low, high
278
+
279
+
280
+ if __name__ == "__main__": # pragma: no cover
281
+ sys.exit(main())