google-browser-scraper 0.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. google_browser_scraper-0.0.1/LICENSE +21 -0
  2. google_browser_scraper-0.0.1/PKG-INFO +223 -0
  3. google_browser_scraper-0.0.1/README.md +192 -0
  4. google_browser_scraper-0.0.1/pyproject.toml +57 -0
  5. google_browser_scraper-0.0.1/setup.cfg +4 -0
  6. google_browser_scraper-0.0.1/src/google_browser_scraper/__init__.py +27 -0
  7. google_browser_scraper-0.0.1/src/google_browser_scraper/__main__.py +5 -0
  8. google_browser_scraper-0.0.1/src/google_browser_scraper/browser.py +203 -0
  9. google_browser_scraper-0.0.1/src/google_browser_scraper/classify.py +105 -0
  10. google_browser_scraper-0.0.1/src/google_browser_scraper/cli.py +281 -0
  11. google_browser_scraper-0.0.1/src/google_browser_scraper/doctor.py +155 -0
  12. google_browser_scraper-0.0.1/src/google_browser_scraper/engines.py +54 -0
  13. google_browser_scraper-0.0.1/src/google_browser_scraper/links.py +125 -0
  14. google_browser_scraper-0.0.1/src/google_browser_scraper/mcp.py +172 -0
  15. google_browser_scraper-0.0.1/src/google_browser_scraper/output.py +144 -0
  16. google_browser_scraper-0.0.1/src/google_browser_scraper/parse.py +270 -0
  17. google_browser_scraper-0.0.1/src/google_browser_scraper/proxy.py +120 -0
  18. google_browser_scraper-0.0.1/src/google_browser_scraper/relay.py +277 -0
  19. google_browser_scraper-0.0.1/src/google_browser_scraper/scraper.py +449 -0
  20. google_browser_scraper-0.0.1/src/google_browser_scraper/server.py +240 -0
  21. google_browser_scraper-0.0.1/src/google_browser_scraper/warmup.py +40 -0
  22. google_browser_scraper-0.0.1/src/google_browser_scraper.egg-info/PKG-INFO +223 -0
  23. google_browser_scraper-0.0.1/src/google_browser_scraper.egg-info/SOURCES.txt +35 -0
  24. google_browser_scraper-0.0.1/src/google_browser_scraper.egg-info/dependency_links.txt +1 -0
  25. google_browser_scraper-0.0.1/src/google_browser_scraper.egg-info/entry_points.txt +2 -0
  26. google_browser_scraper-0.0.1/src/google_browser_scraper.egg-info/requires.txt +16 -0
  27. google_browser_scraper-0.0.1/src/google_browser_scraper.egg-info/top_level.txt +1 -0
  28. google_browser_scraper-0.0.1/tests/test_browser_local.py +48 -0
  29. google_browser_scraper-0.0.1/tests/test_classify.py +58 -0
  30. google_browser_scraper-0.0.1/tests/test_cli_and_output.py +109 -0
  31. google_browser_scraper-0.0.1/tests/test_engines_local.py +85 -0
  32. google_browser_scraper-0.0.1/tests/test_links.py +88 -0
  33. google_browser_scraper-0.0.1/tests/test_parse.py +108 -0
  34. google_browser_scraper-0.0.1/tests/test_proxy.py +50 -0
  35. google_browser_scraper-0.0.1/tests/test_relay.py +172 -0
  36. google_browser_scraper-0.0.1/tests/test_scraper.py +333 -0
  37. google_browser_scraper-0.0.1/tests/test_serve_and_mcp.py +341 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 NodeMaven
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,223 @@
1
+ Metadata-Version: 2.4
2
+ Name: google-browser-scraper
3
+ Version: 0.0.1
4
+ Summary: Google Search results from a real browser through your own sticky proxy, with captchas and blocks reported explicitly.
5
+ Author: NodeMaven
6
+ License: MIT
7
+ Project-URL: Source, https://github.com/nodemaven/google-browser-scraper
8
+ Project-URL: Issues, https://github.com/nodemaven/google-browser-scraper/issues
9
+ Project-URL: Benchmark, https://github.com/nodemaven/proxy-benchmark
10
+ Keywords: google,serp,scraper,search,proxy,playwright,patchright
11
+ Classifier: Development Status :: 2 - Pre-Alpha
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
15
+ Requires-Python: >=3.10
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Requires-Dist: patchright==1.62.2
19
+ Requires-Dist: lxml>=5
20
+ Provides-Extra: nodemaven
21
+ Requires-Dist: nodemaven>=0.1.5; extra == "nodemaven"
22
+ Provides-Extra: cloak
23
+ Requires-Dist: cloakbrowser>=0.5; extra == "cloak"
24
+ Provides-Extra: all
25
+ Requires-Dist: nodemaven>=0.1.5; extra == "all"
26
+ Requires-Dist: cloakbrowser>=0.5; extra == "all"
27
+ Provides-Extra: dev
28
+ Requires-Dist: pytest>=8; extra == "dev"
29
+ Requires-Dist: ruff>=0.6; extra == "dev"
30
+ Dynamic: license-file
31
+
32
+ # google-browser-scraper
33
+
34
+ Google Search results from a real browser, through your own sticky proxy.
35
+
36
+ Plain HTTP is no longer a reliable way to scrape Google: it often expects a
37
+ browser that runs JavaScript, and a fresh browser on a fresh proxy exit can be
38
+ refused quickly. This tool takes a browser path instead. It warms a sticky proxy
39
+ session with ordinary browsing, types the query into Google's search box, checks
40
+ whether Google served results or refused, and writes one JSON record per results
41
+ page. Captchas, blocks and failures are reported as such, never as "no results".
42
+
43
+ ## Install
44
+
45
+ ```
46
+ pip install google-browser-scraper
47
+ patchright install chromium
48
+ ```
49
+
50
+ ## Quickstart
51
+
52
+ You need a residential proxy with **sticky sessions**. Put `{session}` where
53
+ your provider expects a session id in the username:
54
+
55
+ ```
56
+ google-browser-scraper search "best running shoes" \
57
+ --proxy "http://USER-session-{session}:PASS@gate.example.com:7000" \
58
+ -o results.jsonl
59
+ ```
60
+
61
+ Success is a `results.jsonl` with one JSON object per results page:
62
+
63
+ ```json
64
+ {
65
+ "search_metadata": {"status": "Success", "page_verdict": "ok"},
66
+ "search_parameters": {"engine": "google", "q": "best running shoes", "page": 1},
67
+ "organic_results": [
68
+ {"position": 1, "title": "...", "link": "https://...", "snippet": "..."}
69
+ ],
70
+ "related_questions": [{"question": "..."}],
71
+ "exit": {"session": "...", "queries_on_exit": 1, "bytes": "..."}
72
+ }
73
+ ```
74
+
75
+ The first query takes a few minutes, because a new exit is warmed up first, and
76
+ the warm-up costs noticeably more traffic than a search. Later queries on the
77
+ same exit take seconds. A summary of pages served and traffic used is printed at
78
+ the end of every run.
79
+
80
+ With NodeMaven, the session id is handled for you and the credentials come from
81
+ the environment:
82
+
83
+ ```
84
+ pip install "google-browser-scraper[nodemaven]"
85
+ export NODEMAVEN_LOGIN=... NODEMAVEN_PASSWORD=...
86
+ google-browser-scraper search -f queries.txt --nodemaven --country us -o results.jsonl
87
+ ```
88
+
89
+ From Python:
90
+
91
+ ```python
92
+ from google_browser_scraper import ProxyTemplate, Scraper, Settings
93
+
94
+ proxy = ProxyTemplate("http://USER-session-{session}:PASS@gate.example.com:7000")
95
+ for page in Scraper(proxy, Settings(pages=2)).run(["best running shoes"]):
96
+ for result in page.get("organic_results", []):
97
+ print(result["position"], result["title"], result["link"])
98
+ ```
99
+
100
+ ## What you get
101
+
102
+ The schema is intentionally SerpApi-like: `organic_results`, `ads`,
103
+ `related_questions`, `related_searches`, `pagination` and `search_metadata`, so
104
+ simple consumers can often switch with minimal mapping. Knowledge panels,
105
+ shopping, images and other special blocks are not parsed.
106
+
107
+ Each record also has an `exit` block: which proxy session answered and how many
108
+ bytes the page cost. A query that failed is a record too, with an `error`.
109
+ `--format csv` writes the organic results as a table, and `--price-per-gb` adds
110
+ the cost per 1000 pages to the summary.
111
+
112
+ ## How it works
113
+
114
+ 1. **New identity**: a sticky proxy session and a fresh browser profile.
115
+ 2. **Warm-up**: a few ordinary pages, then Google's front page.
116
+ 3. **Search**: the query is typed into the box and submitted.
117
+ 4. **Check before parsing**: the page is classified first, so a captcha or a
118
+ block is reported as one and never parsed as an empty result.
119
+ 5. **Keep or drop**: an exit Google answers is reused for the next queries; an
120
+ exit it refuses is dropped and the query is retried on a new one. If exit
121
+ after exit is refused, the run stops instead of spending more traffic.
122
+
123
+ Google wraps result links in encrypted `/goto` redirects. Each one is resolved
124
+ with a small request through a separate short-lived proxy session, so link
125
+ resolution does not touch the warmed browser session. `--no-resolve-links`
126
+ skips it.
127
+
128
+ Traffic goes through a small local relay that counts bytes and blocks a large
129
+ model download Chrome otherwise makes on every fresh profile.
130
+
131
+ The defaults come from measurements in
132
+ [proxy-benchmark](https://github.com/nodemaven/proxy-benchmark).
133
+
134
+ ## Proxies
135
+
136
+ Any provider with sticky sessions works through `--proxy` and a `{session}`
137
+ placeholder. A proxy without one is accepted with a warning: if the exit changes
138
+ on every request, warming it does nothing. SOCKS5 needs an IP-whitelisted
139
+ endpoint, because Chrome cannot send a SOCKS5 username and password.
140
+
141
+ ## Run it as an API
142
+
143
+ ```
144
+ google-browser-scraper serve --nodemaven --country us --prewarm
145
+ curl "http://127.0.0.1:8000/search.json?q=best+running+shoes"
146
+ ```
147
+
148
+ Same record format as above. Requests run one at a time on one held identity.
149
+ Set `GBS_API_KEY` to require a key; without one the server only listens on
150
+ localhost.
151
+
152
+ ## Use it from an AI agent (MCP)
153
+
154
+ ```json
155
+ {
156
+ "mcpServers": {
157
+ "google-search": {
158
+ "command": "google-browser-scraper",
159
+ "args": ["mcp", "--nodemaven", "--country", "us", "--prewarm"],
160
+ "env": {"NODEMAVEN_LOGIN": "...", "NODEMAVEN_PASSWORD": "..."}
161
+ }
162
+ }
163
+ }
164
+ ```
165
+
166
+ It exposes one tool, `google_search(query, pages)`.
167
+
168
+ ## Docker
169
+
170
+ ```
171
+ docker build -t google-browser-scraper .
172
+ docker run --rm google-browser-scraper doctor
173
+ docker run --rm -p 8000:8000 -e GBS_API_KEY=change-me \
174
+ -e GBS_PROXY='http://USER-session-{session}:PASS@gate.example.com:7000' \
175
+ google-browser-scraper serve --host 0.0.0.0
176
+ ```
177
+
178
+ ## Check your machine
179
+
180
+ ```
181
+ google-browser-scraper doctor
182
+ ```
183
+
184
+ Starts the browser on a blank page, sends nothing, and reports what a website
185
+ would see: whether the browser looks headless, whether WebGL works, the screen
186
+ size. Google answers some machines much less than others, so run this first.
187
+
188
+ ## What this is not
189
+
190
+ Not a free Google Search API, and not a guarantee that every proxy exit will
191
+ work. It is a browser-based collector that makes failures explicit, drops exits
192
+ Google refuses, and tells you what each page cost.
193
+
194
+ ## Status
195
+
196
+ Early. Tested on Windows. **Linux and Docker are experimental**: on a machine
197
+ without a GPU, Chrome has no WebGL, which a normal desktop always has; the
198
+ Docker image turns on a software renderer for it. The `cloak` engine
199
+ (`--engine cloak`) is experimental too; `patchright` is the default.
200
+
201
+ Google changes its pages often. If results come back empty or wrong, run with
202
+ `--save-html DIR` and open an issue with what you see.
203
+
204
+ You are responsible for using this within Google's terms and the law where you
205
+ run it.
206
+
207
+ ## Adding an engine
208
+
209
+ An engine joins this tool only after it has been measured on
210
+ [proxy-benchmark](https://github.com/nodemaven/proxy-benchmark): add it there,
211
+ run the warm-up ladder against Google -
212
+
213
+ ```
214
+ python scripts/run_ladder.py --engines your-engine
215
+ ```
216
+
217
+ - and open a pull request here with the run file. The ladder runs identities
218
+ with and without the warm-up in the same hours, so the result shows whether the
219
+ engine actually gets served, not whether it got lucky once.
220
+
221
+ ## License
222
+
223
+ MIT
@@ -0,0 +1,192 @@
1
+ # google-browser-scraper
2
+
3
+ Google Search results from a real browser, through your own sticky proxy.
4
+
5
+ Plain HTTP is no longer a reliable way to scrape Google: it often expects a
6
+ browser that runs JavaScript, and a fresh browser on a fresh proxy exit can be
7
+ refused quickly. This tool takes a browser path instead. It warms a sticky proxy
8
+ session with ordinary browsing, types the query into Google's search box, checks
9
+ whether Google served results or refused, and writes one JSON record per results
10
+ page. Captchas, blocks and failures are reported as such, never as "no results".
11
+
12
+ ## Install
13
+
14
+ ```
15
+ pip install google-browser-scraper
16
+ patchright install chromium
17
+ ```
18
+
19
+ ## Quickstart
20
+
21
+ You need a residential proxy with **sticky sessions**. Put `{session}` where
22
+ your provider expects a session id in the username:
23
+
24
+ ```
25
+ google-browser-scraper search "best running shoes" \
26
+ --proxy "http://USER-session-{session}:PASS@gate.example.com:7000" \
27
+ -o results.jsonl
28
+ ```
29
+
30
+ Success is a `results.jsonl` with one JSON object per results page:
31
+
32
+ ```json
33
+ {
34
+ "search_metadata": {"status": "Success", "page_verdict": "ok"},
35
+ "search_parameters": {"engine": "google", "q": "best running shoes", "page": 1},
36
+ "organic_results": [
37
+ {"position": 1, "title": "...", "link": "https://...", "snippet": "..."}
38
+ ],
39
+ "related_questions": [{"question": "..."}],
40
+ "exit": {"session": "...", "queries_on_exit": 1, "bytes": "..."}
41
+ }
42
+ ```
43
+
44
+ The first query takes a few minutes, because a new exit is warmed up first, and
45
+ the warm-up costs noticeably more traffic than a search. Later queries on the
46
+ same exit take seconds. A summary of pages served and traffic used is printed at
47
+ the end of every run.
48
+
49
+ With NodeMaven, the session id is handled for you and the credentials come from
50
+ the environment:
51
+
52
+ ```
53
+ pip install "google-browser-scraper[nodemaven]"
54
+ export NODEMAVEN_LOGIN=... NODEMAVEN_PASSWORD=...
55
+ google-browser-scraper search -f queries.txt --nodemaven --country us -o results.jsonl
56
+ ```
57
+
58
+ From Python:
59
+
60
+ ```python
61
+ from google_browser_scraper import ProxyTemplate, Scraper, Settings
62
+
63
+ proxy = ProxyTemplate("http://USER-session-{session}:PASS@gate.example.com:7000")
64
+ for page in Scraper(proxy, Settings(pages=2)).run(["best running shoes"]):
65
+ for result in page.get("organic_results", []):
66
+ print(result["position"], result["title"], result["link"])
67
+ ```
68
+
69
+ ## What you get
70
+
71
+ The schema is intentionally SerpApi-like: `organic_results`, `ads`,
72
+ `related_questions`, `related_searches`, `pagination` and `search_metadata`, so
73
+ simple consumers can often switch with minimal mapping. Knowledge panels,
74
+ shopping, images and other special blocks are not parsed.
75
+
76
+ Each record also has an `exit` block: which proxy session answered and how many
77
+ bytes the page cost. A query that failed is a record too, with an `error`.
78
+ `--format csv` writes the organic results as a table, and `--price-per-gb` adds
79
+ the cost per 1000 pages to the summary.
80
+
81
+ ## How it works
82
+
83
+ 1. **New identity**: a sticky proxy session and a fresh browser profile.
84
+ 2. **Warm-up**: a few ordinary pages, then Google's front page.
85
+ 3. **Search**: the query is typed into the box and submitted.
86
+ 4. **Check before parsing**: the page is classified first, so a captcha or a
87
+ block is reported as one and never parsed as an empty result.
88
+ 5. **Keep or drop**: an exit Google answers is reused for the next queries; an
89
+ exit it refuses is dropped and the query is retried on a new one. If exit
90
+ after exit is refused, the run stops instead of spending more traffic.
91
+
92
+ Google wraps result links in encrypted `/goto` redirects. Each one is resolved
93
+ with a small request through a separate short-lived proxy session, so link
94
+ resolution does not touch the warmed browser session. `--no-resolve-links`
95
+ skips it.
96
+
97
+ Traffic goes through a small local relay that counts bytes and blocks a large
98
+ model download Chrome otherwise makes on every fresh profile.
99
+
100
+ The defaults come from measurements in
101
+ [proxy-benchmark](https://github.com/nodemaven/proxy-benchmark).
102
+
103
+ ## Proxies
104
+
105
+ Any provider with sticky sessions works through `--proxy` and a `{session}`
106
+ placeholder. A proxy without one is accepted with a warning: if the exit changes
107
+ on every request, warming it does nothing. SOCKS5 needs an IP-whitelisted
108
+ endpoint, because Chrome cannot send a SOCKS5 username and password.
109
+
110
+ ## Run it as an API
111
+
112
+ ```
113
+ google-browser-scraper serve --nodemaven --country us --prewarm
114
+ curl "http://127.0.0.1:8000/search.json?q=best+running+shoes"
115
+ ```
116
+
117
+ Same record format as above. Requests run one at a time on one held identity.
118
+ Set `GBS_API_KEY` to require a key; without one the server only listens on
119
+ localhost.
120
+
121
+ ## Use it from an AI agent (MCP)
122
+
123
+ ```json
124
+ {
125
+ "mcpServers": {
126
+ "google-search": {
127
+ "command": "google-browser-scraper",
128
+ "args": ["mcp", "--nodemaven", "--country", "us", "--prewarm"],
129
+ "env": {"NODEMAVEN_LOGIN": "...", "NODEMAVEN_PASSWORD": "..."}
130
+ }
131
+ }
132
+ }
133
+ ```
134
+
135
+ It exposes one tool, `google_search(query, pages)`.
136
+
137
+ ## Docker
138
+
139
+ ```
140
+ docker build -t google-browser-scraper .
141
+ docker run --rm google-browser-scraper doctor
142
+ docker run --rm -p 8000:8000 -e GBS_API_KEY=change-me \
143
+ -e GBS_PROXY='http://USER-session-{session}:PASS@gate.example.com:7000' \
144
+ google-browser-scraper serve --host 0.0.0.0
145
+ ```
146
+
147
+ ## Check your machine
148
+
149
+ ```
150
+ google-browser-scraper doctor
151
+ ```
152
+
153
+ Starts the browser on a blank page, sends nothing, and reports what a website
154
+ would see: whether the browser looks headless, whether WebGL works, the screen
155
+ size. Google answers some machines much less than others, so run this first.
156
+
157
+ ## What this is not
158
+
159
+ Not a free Google Search API, and not a guarantee that every proxy exit will
160
+ work. It is a browser-based collector that makes failures explicit, drops exits
161
+ Google refuses, and tells you what each page cost.
162
+
163
+ ## Status
164
+
165
+ Early. Tested on Windows. **Linux and Docker are experimental**: on a machine
166
+ without a GPU, Chrome has no WebGL, which a normal desktop always has; the
167
+ Docker image turns on a software renderer for it. The `cloak` engine
168
+ (`--engine cloak`) is experimental too; `patchright` is the default.
169
+
170
+ Google changes its pages often. If results come back empty or wrong, run with
171
+ `--save-html DIR` and open an issue with what you see.
172
+
173
+ You are responsible for using this within Google's terms and the law where you
174
+ run it.
175
+
176
+ ## Adding an engine
177
+
178
+ An engine joins this tool only after it has been measured on
179
+ [proxy-benchmark](https://github.com/nodemaven/proxy-benchmark): add it there,
180
+ run the warm-up ladder against Google -
181
+
182
+ ```
183
+ python scripts/run_ladder.py --engines your-engine
184
+ ```
185
+
186
+ - and open a pull request here with the run file. The ladder runs identities
187
+ with and without the warm-up in the same hours, so the result shows whether the
188
+ engine actually gets served, not whether it got lucky once.
189
+
190
+ ## License
191
+
192
+ MIT
@@ -0,0 +1,57 @@
1
+ [build-system]
2
+ requires = ["setuptools>=69"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "google-browser-scraper"
7
+ version = "0.0.1"
8
+ description = "Google Search results from a real browser through your own sticky proxy, with captchas and blocks reported explicitly."
9
+ readme = "README.md"
10
+ license = { text = "MIT" }
11
+ requires-python = ">=3.10"
12
+ authors = [{ name = "NodeMaven" }]
13
+ keywords = ["google", "serp", "scraper", "search", "proxy", "playwright", "patchright"]
14
+ classifiers = [
15
+ "Development Status :: 2 - Pre-Alpha",
16
+ "License :: OSI Approved :: MIT License",
17
+ "Programming Language :: Python :: 3",
18
+ "Topic :: Internet :: WWW/HTTP :: Indexing/Search",
19
+ ]
20
+ dependencies = [
21
+ # Pinned, because each patchright release ships a different Chromium and
22
+ # the browser build is part of what Google sees. 1.62.2 is Chromium 151,
23
+ # the build every live run of this tool was made with.
24
+ "patchright==1.62.2",
25
+ "lxml>=5",
26
+ ]
27
+
28
+ [project.optional-dependencies]
29
+ # The NodeMaven SDK builds sticky-session usernames for its gateway. Any other
30
+ # provider works through a proxy URL with a `{session}` placeholder instead.
31
+ nodemaven = ["nodemaven>=0.1.5"]
32
+ # CloakBrowser's patched Chromium; downloads its own binary on first use.
33
+ cloak = ["cloakbrowser>=0.5"]
34
+ all = ["nodemaven>=0.1.5", "cloakbrowser>=0.5"]
35
+ dev = ["pytest>=8", "ruff>=0.6"]
36
+
37
+ [project.scripts]
38
+ google-browser-scraper = "google_browser_scraper.cli:main"
39
+
40
+ [project.urls]
41
+ Source = "https://github.com/nodemaven/google-browser-scraper"
42
+ Issues = "https://github.com/nodemaven/google-browser-scraper/issues"
43
+ Benchmark = "https://github.com/nodemaven/proxy-benchmark"
44
+
45
+ [tool.setuptools.packages.find]
46
+ where = ["src"]
47
+
48
+ [tool.pytest.ini_options]
49
+ testpaths = ["tests"]
50
+ pythonpath = ["src"]
51
+
52
+ [tool.ruff]
53
+ line-length = 100
54
+ target-version = "py310"
55
+
56
+ [tool.ruff.lint]
57
+ select = ["E", "F", "W", "I", "B", "UP"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,27 @@
1
+ """Google search results from a real browser through your own proxy.
2
+
3
+ from google_browser_scraper import Scraper, Settings, ProxyTemplate
4
+
5
+ proxy = ProxyTemplate("http://user-session-{session}:pass@gate.example.com:7000")
6
+ for record in Scraper(proxy, Settings(pages=2)).run(["best running shoes"]):
7
+ print(record["organic_results"])
8
+ """
9
+
10
+ __version__ = "0.0.1"
11
+
12
+ from .classify import Verdict, classify # noqa: E402
13
+ from .parse import parse_serp # noqa: E402
14
+ from .proxy import NodeMavenSource, ProxyTemplate # noqa: E402
15
+ from .scraper import ExitsRefused, Scraper, Settings # noqa: E402
16
+
17
+ __all__ = [
18
+ "ExitsRefused",
19
+ "NodeMavenSource",
20
+ "ProxyTemplate",
21
+ "Scraper",
22
+ "Settings",
23
+ "Verdict",
24
+ "__version__",
25
+ "classify",
26
+ "parse_serp",
27
+ ]
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())