google-browser-scraper 0.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- google_browser_scraper-0.0.1/LICENSE +21 -0
- google_browser_scraper-0.0.1/PKG-INFO +223 -0
- google_browser_scraper-0.0.1/README.md +192 -0
- google_browser_scraper-0.0.1/pyproject.toml +57 -0
- google_browser_scraper-0.0.1/setup.cfg +4 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/__init__.py +27 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/__main__.py +5 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/browser.py +203 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/classify.py +105 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/cli.py +281 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/doctor.py +155 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/engines.py +54 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/links.py +125 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/mcp.py +172 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/output.py +144 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/parse.py +270 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/proxy.py +120 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/relay.py +277 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/scraper.py +449 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/server.py +240 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper/warmup.py +40 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper.egg-info/PKG-INFO +223 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper.egg-info/SOURCES.txt +35 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper.egg-info/dependency_links.txt +1 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper.egg-info/entry_points.txt +2 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper.egg-info/requires.txt +16 -0
- google_browser_scraper-0.0.1/src/google_browser_scraper.egg-info/top_level.txt +1 -0
- google_browser_scraper-0.0.1/tests/test_browser_local.py +48 -0
- google_browser_scraper-0.0.1/tests/test_classify.py +58 -0
- google_browser_scraper-0.0.1/tests/test_cli_and_output.py +109 -0
- google_browser_scraper-0.0.1/tests/test_engines_local.py +85 -0
- google_browser_scraper-0.0.1/tests/test_links.py +88 -0
- google_browser_scraper-0.0.1/tests/test_parse.py +108 -0
- google_browser_scraper-0.0.1/tests/test_proxy.py +50 -0
- google_browser_scraper-0.0.1/tests/test_relay.py +172 -0
- google_browser_scraper-0.0.1/tests/test_scraper.py +333 -0
- google_browser_scraper-0.0.1/tests/test_serve_and_mcp.py +341 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 NodeMaven
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,223 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: google-browser-scraper
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: Google Search results from a real browser through your own sticky proxy, with captchas and blocks reported explicitly.
|
|
5
|
+
Author: NodeMaven
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Source, https://github.com/nodemaven/google-browser-scraper
|
|
8
|
+
Project-URL: Issues, https://github.com/nodemaven/google-browser-scraper/issues
|
|
9
|
+
Project-URL: Benchmark, https://github.com/nodemaven/proxy-benchmark
|
|
10
|
+
Keywords: google,serp,scraper,search,proxy,playwright,patchright
|
|
11
|
+
Classifier: Development Status :: 2 - Pre-Alpha
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
|
|
15
|
+
Requires-Python: >=3.10
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: patchright==1.62.2
|
|
19
|
+
Requires-Dist: lxml>=5
|
|
20
|
+
Provides-Extra: nodemaven
|
|
21
|
+
Requires-Dist: nodemaven>=0.1.5; extra == "nodemaven"
|
|
22
|
+
Provides-Extra: cloak
|
|
23
|
+
Requires-Dist: cloakbrowser>=0.5; extra == "cloak"
|
|
24
|
+
Provides-Extra: all
|
|
25
|
+
Requires-Dist: nodemaven>=0.1.5; extra == "all"
|
|
26
|
+
Requires-Dist: cloakbrowser>=0.5; extra == "all"
|
|
27
|
+
Provides-Extra: dev
|
|
28
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
29
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
30
|
+
Dynamic: license-file
|
|
31
|
+
|
|
32
|
+
# google-browser-scraper
|
|
33
|
+
|
|
34
|
+
Google Search results from a real browser, through your own sticky proxy.
|
|
35
|
+
|
|
36
|
+
Plain HTTP is no longer a reliable way to scrape Google: it often expects a
|
|
37
|
+
browser that runs JavaScript, and a fresh browser on a fresh proxy exit can be
|
|
38
|
+
refused quickly. This tool takes a browser path instead. It warms a sticky proxy
|
|
39
|
+
session with ordinary browsing, types the query into Google's search box, checks
|
|
40
|
+
whether Google served results or refused, and writes one JSON record per results
|
|
41
|
+
page. Captchas, blocks and failures are reported as such, never as "no results".
|
|
42
|
+
|
|
43
|
+
## Install
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
pip install google-browser-scraper
|
|
47
|
+
patchright install chromium
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
## Quickstart
|
|
51
|
+
|
|
52
|
+
You need a residential proxy with **sticky sessions**. Put `{session}` where
|
|
53
|
+
your provider expects a session id in the username:
|
|
54
|
+
|
|
55
|
+
```
|
|
56
|
+
google-browser-scraper search "best running shoes" \
|
|
57
|
+
--proxy "http://USER-session-{session}:PASS@gate.example.com:7000" \
|
|
58
|
+
-o results.jsonl
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Success is a `results.jsonl` with one JSON object per results page:
|
|
62
|
+
|
|
63
|
+
```json
|
|
64
|
+
{
|
|
65
|
+
"search_metadata": {"status": "Success", "page_verdict": "ok"},
|
|
66
|
+
"search_parameters": {"engine": "google", "q": "best running shoes", "page": 1},
|
|
67
|
+
"organic_results": [
|
|
68
|
+
{"position": 1, "title": "...", "link": "https://...", "snippet": "..."}
|
|
69
|
+
],
|
|
70
|
+
"related_questions": [{"question": "..."}],
|
|
71
|
+
"exit": {"session": "...", "queries_on_exit": 1, "bytes": "..."}
|
|
72
|
+
}
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
The first query takes a few minutes, because a new exit is warmed up first, and
|
|
76
|
+
the warm-up costs noticeably more traffic than a search. Later queries on the
|
|
77
|
+
same exit take seconds. A summary of pages served and traffic used is printed at
|
|
78
|
+
the end of every run.
|
|
79
|
+
|
|
80
|
+
With NodeMaven, the session id is handled for you and the credentials come from
|
|
81
|
+
the environment:
|
|
82
|
+
|
|
83
|
+
```
|
|
84
|
+
pip install "google-browser-scraper[nodemaven]"
|
|
85
|
+
export NODEMAVEN_LOGIN=... NODEMAVEN_PASSWORD=...
|
|
86
|
+
google-browser-scraper search -f queries.txt --nodemaven --country us -o results.jsonl
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
From Python:
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
from google_browser_scraper import ProxyTemplate, Scraper, Settings
|
|
93
|
+
|
|
94
|
+
proxy = ProxyTemplate("http://USER-session-{session}:PASS@gate.example.com:7000")
|
|
95
|
+
for page in Scraper(proxy, Settings(pages=2)).run(["best running shoes"]):
|
|
96
|
+
for result in page.get("organic_results", []):
|
|
97
|
+
print(result["position"], result["title"], result["link"])
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
## What you get
|
|
101
|
+
|
|
102
|
+
The schema is intentionally SerpApi-like: `organic_results`, `ads`,
|
|
103
|
+
`related_questions`, `related_searches`, `pagination` and `search_metadata`, so
|
|
104
|
+
simple consumers can often switch with minimal mapping. Knowledge panels,
|
|
105
|
+
shopping, images and other special blocks are not parsed.
|
|
106
|
+
|
|
107
|
+
Each record also has an `exit` block: which proxy session answered and how many
|
|
108
|
+
bytes the page cost. A query that failed is a record too, with an `error`.
|
|
109
|
+
`--format csv` writes the organic results as a table, and `--price-per-gb` adds
|
|
110
|
+
the cost per 1000 pages to the summary.
|
|
111
|
+
|
|
112
|
+
## How it works
|
|
113
|
+
|
|
114
|
+
1. **New identity**: a sticky proxy session and a fresh browser profile.
|
|
115
|
+
2. **Warm-up**: a few ordinary pages, then Google's front page.
|
|
116
|
+
3. **Search**: the query is typed into the box and submitted.
|
|
117
|
+
4. **Check before parsing**: the page is classified first, so a captcha or a
|
|
118
|
+
block is reported as one and never parsed as an empty result.
|
|
119
|
+
5. **Keep or drop**: an exit Google answers is reused for the next queries; an
|
|
120
|
+
exit it refuses is dropped and the query is retried on a new one. If exit
|
|
121
|
+
after exit is refused, the run stops instead of spending more traffic.
|
|
122
|
+
|
|
123
|
+
Google wraps result links in encrypted `/goto` redirects. Each one is resolved
|
|
124
|
+
with a small request through a separate short-lived proxy session, so link
|
|
125
|
+
resolution does not touch the warmed browser session. `--no-resolve-links`
|
|
126
|
+
skips it.
|
|
127
|
+
|
|
128
|
+
Traffic goes through a small local relay that counts bytes and blocks a large
|
|
129
|
+
model download Chrome otherwise makes on every fresh profile.
|
|
130
|
+
|
|
131
|
+
The defaults come from measurements in
|
|
132
|
+
[proxy-benchmark](https://github.com/nodemaven/proxy-benchmark).
|
|
133
|
+
|
|
134
|
+
## Proxies
|
|
135
|
+
|
|
136
|
+
Any provider with sticky sessions works through `--proxy` and a `{session}`
|
|
137
|
+
placeholder. A proxy without one is accepted with a warning: if the exit changes
|
|
138
|
+
on every request, warming it does nothing. SOCKS5 needs an IP-whitelisted
|
|
139
|
+
endpoint, because Chrome cannot send a SOCKS5 username and password.
|
|
140
|
+
|
|
141
|
+
## Run it as an API
|
|
142
|
+
|
|
143
|
+
```
|
|
144
|
+
google-browser-scraper serve --nodemaven --country us --prewarm
|
|
145
|
+
curl "http://127.0.0.1:8000/search.json?q=best+running+shoes"
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
Same record format as above. Requests run one at a time on one held identity.
|
|
149
|
+
Set `GBS_API_KEY` to require a key; without one the server only listens on
|
|
150
|
+
localhost.
|
|
151
|
+
|
|
152
|
+
## Use it from an AI agent (MCP)
|
|
153
|
+
|
|
154
|
+
```json
|
|
155
|
+
{
|
|
156
|
+
"mcpServers": {
|
|
157
|
+
"google-search": {
|
|
158
|
+
"command": "google-browser-scraper",
|
|
159
|
+
"args": ["mcp", "--nodemaven", "--country", "us", "--prewarm"],
|
|
160
|
+
"env": {"NODEMAVEN_LOGIN": "...", "NODEMAVEN_PASSWORD": "..."}
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
It exposes one tool, `google_search(query, pages)`.
|
|
167
|
+
|
|
168
|
+
## Docker
|
|
169
|
+
|
|
170
|
+
```
|
|
171
|
+
docker build -t google-browser-scraper .
|
|
172
|
+
docker run --rm google-browser-scraper doctor
|
|
173
|
+
docker run --rm -p 8000:8000 -e GBS_API_KEY=change-me \
|
|
174
|
+
-e GBS_PROXY='http://USER-session-{session}:PASS@gate.example.com:7000' \
|
|
175
|
+
google-browser-scraper serve --host 0.0.0.0
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
## Check your machine
|
|
179
|
+
|
|
180
|
+
```
|
|
181
|
+
google-browser-scraper doctor
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
Starts the browser on a blank page, sends nothing, and reports what a website
|
|
185
|
+
would see: whether the browser looks headless, whether WebGL works, the screen
|
|
186
|
+
size. Google answers some machines much less than others, so run this first.
|
|
187
|
+
|
|
188
|
+
## What this is not
|
|
189
|
+
|
|
190
|
+
Not a free Google Search API, and not a guarantee that every proxy exit will
|
|
191
|
+
work. It is a browser-based collector that makes failures explicit, drops exits
|
|
192
|
+
Google refuses, and tells you what each page cost.
|
|
193
|
+
|
|
194
|
+
## Status
|
|
195
|
+
|
|
196
|
+
Early. Tested on Windows. **Linux and Docker are experimental**: on a machine
|
|
197
|
+
without a GPU, Chrome has no WebGL, which a normal desktop always has; the
|
|
198
|
+
Docker image turns on a software renderer for it. The `cloak` engine
|
|
199
|
+
(`--engine cloak`) is experimental too; `patchright` is the default.
|
|
200
|
+
|
|
201
|
+
Google changes its pages often. If results come back empty or wrong, run with
|
|
202
|
+
`--save-html DIR` and open an issue with what you see.
|
|
203
|
+
|
|
204
|
+
You are responsible for using this within Google's terms and the law where you
|
|
205
|
+
run it.
|
|
206
|
+
|
|
207
|
+
## Adding an engine
|
|
208
|
+
|
|
209
|
+
An engine joins this tool only after it has been measured on
|
|
210
|
+
[proxy-benchmark](https://github.com/nodemaven/proxy-benchmark): add it there,
|
|
211
|
+
run the warm-up ladder against Google -
|
|
212
|
+
|
|
213
|
+
```
|
|
214
|
+
python scripts/run_ladder.py --engines your-engine
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
- and open a pull request here with the run file. The ladder runs identities
|
|
218
|
+
with and without the warm-up in the same hours, so the result shows whether the
|
|
219
|
+
engine actually gets served, not whether it got lucky once.
|
|
220
|
+
|
|
221
|
+
## License
|
|
222
|
+
|
|
223
|
+
MIT
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
# google-browser-scraper
|
|
2
|
+
|
|
3
|
+
Google Search results from a real browser, through your own sticky proxy.
|
|
4
|
+
|
|
5
|
+
Plain HTTP is no longer a reliable way to scrape Google: it often expects a
|
|
6
|
+
browser that runs JavaScript, and a fresh browser on a fresh proxy exit can be
|
|
7
|
+
refused quickly. This tool takes a browser path instead. It warms a sticky proxy
|
|
8
|
+
session with ordinary browsing, types the query into Google's search box, checks
|
|
9
|
+
whether Google served results or refused, and writes one JSON record per results
|
|
10
|
+
page. Captchas, blocks and failures are reported as such, never as "no results".
|
|
11
|
+
|
|
12
|
+
## Install
|
|
13
|
+
|
|
14
|
+
```
|
|
15
|
+
pip install google-browser-scraper
|
|
16
|
+
patchright install chromium
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
## Quickstart
|
|
20
|
+
|
|
21
|
+
You need a residential proxy with **sticky sessions**. Put `{session}` where
|
|
22
|
+
your provider expects a session id in the username:
|
|
23
|
+
|
|
24
|
+
```
|
|
25
|
+
google-browser-scraper search "best running shoes" \
|
|
26
|
+
--proxy "http://USER-session-{session}:PASS@gate.example.com:7000" \
|
|
27
|
+
-o results.jsonl
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
Success is a `results.jsonl` with one JSON object per results page:
|
|
31
|
+
|
|
32
|
+
```json
|
|
33
|
+
{
|
|
34
|
+
"search_metadata": {"status": "Success", "page_verdict": "ok"},
|
|
35
|
+
"search_parameters": {"engine": "google", "q": "best running shoes", "page": 1},
|
|
36
|
+
"organic_results": [
|
|
37
|
+
{"position": 1, "title": "...", "link": "https://...", "snippet": "..."}
|
|
38
|
+
],
|
|
39
|
+
"related_questions": [{"question": "..."}],
|
|
40
|
+
"exit": {"session": "...", "queries_on_exit": 1, "bytes": "..."}
|
|
41
|
+
}
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
The first query takes a few minutes, because a new exit is warmed up first, and
|
|
45
|
+
the warm-up costs noticeably more traffic than a search. Later queries on the
|
|
46
|
+
same exit take seconds. A summary of pages served and traffic used is printed at
|
|
47
|
+
the end of every run.
|
|
48
|
+
|
|
49
|
+
With NodeMaven, the session id is handled for you and the credentials come from
|
|
50
|
+
the environment:
|
|
51
|
+
|
|
52
|
+
```
|
|
53
|
+
pip install "google-browser-scraper[nodemaven]"
|
|
54
|
+
export NODEMAVEN_LOGIN=... NODEMAVEN_PASSWORD=...
|
|
55
|
+
google-browser-scraper search -f queries.txt --nodemaven --country us -o results.jsonl
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
From Python:
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
from google_browser_scraper import ProxyTemplate, Scraper, Settings
|
|
62
|
+
|
|
63
|
+
proxy = ProxyTemplate("http://USER-session-{session}:PASS@gate.example.com:7000")
|
|
64
|
+
for page in Scraper(proxy, Settings(pages=2)).run(["best running shoes"]):
|
|
65
|
+
for result in page.get("organic_results", []):
|
|
66
|
+
print(result["position"], result["title"], result["link"])
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
## What you get
|
|
70
|
+
|
|
71
|
+
The schema is intentionally SerpApi-like: `organic_results`, `ads`,
|
|
72
|
+
`related_questions`, `related_searches`, `pagination` and `search_metadata`, so
|
|
73
|
+
simple consumers can often switch with minimal mapping. Knowledge panels,
|
|
74
|
+
shopping, images and other special blocks are not parsed.
|
|
75
|
+
|
|
76
|
+
Each record also has an `exit` block: which proxy session answered and how many
|
|
77
|
+
bytes the page cost. A query that failed is a record too, with an `error`.
|
|
78
|
+
`--format csv` writes the organic results as a table, and `--price-per-gb` adds
|
|
79
|
+
the cost per 1000 pages to the summary.
|
|
80
|
+
|
|
81
|
+
## How it works
|
|
82
|
+
|
|
83
|
+
1. **New identity**: a sticky proxy session and a fresh browser profile.
|
|
84
|
+
2. **Warm-up**: a few ordinary pages, then Google's front page.
|
|
85
|
+
3. **Search**: the query is typed into the box and submitted.
|
|
86
|
+
4. **Check before parsing**: the page is classified first, so a captcha or a
|
|
87
|
+
block is reported as one and never parsed as an empty result.
|
|
88
|
+
5. **Keep or drop**: an exit Google answers is reused for the next queries; an
|
|
89
|
+
exit it refuses is dropped and the query is retried on a new one. If exit
|
|
90
|
+
after exit is refused, the run stops instead of spending more traffic.
|
|
91
|
+
|
|
92
|
+
Google wraps result links in encrypted `/goto` redirects. Each one is resolved
|
|
93
|
+
with a small request through a separate short-lived proxy session, so link
|
|
94
|
+
resolution does not touch the warmed browser session. `--no-resolve-links`
|
|
95
|
+
skips it.
|
|
96
|
+
|
|
97
|
+
Traffic goes through a small local relay that counts bytes and blocks a large
|
|
98
|
+
model download Chrome otherwise makes on every fresh profile.
|
|
99
|
+
|
|
100
|
+
The defaults come from measurements in
|
|
101
|
+
[proxy-benchmark](https://github.com/nodemaven/proxy-benchmark).
|
|
102
|
+
|
|
103
|
+
## Proxies
|
|
104
|
+
|
|
105
|
+
Any provider with sticky sessions works through `--proxy` and a `{session}`
|
|
106
|
+
placeholder. A proxy without one is accepted with a warning: if the exit changes
|
|
107
|
+
on every request, warming it does nothing. SOCKS5 needs an IP-whitelisted
|
|
108
|
+
endpoint, because Chrome cannot send a SOCKS5 username and password.
|
|
109
|
+
|
|
110
|
+
## Run it as an API
|
|
111
|
+
|
|
112
|
+
```
|
|
113
|
+
google-browser-scraper serve --nodemaven --country us --prewarm
|
|
114
|
+
curl "http://127.0.0.1:8000/search.json?q=best+running+shoes"
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
Same record format as above. Requests run one at a time on one held identity.
|
|
118
|
+
Set `GBS_API_KEY` to require a key; without one the server only listens on
|
|
119
|
+
localhost.
|
|
120
|
+
|
|
121
|
+
## Use it from an AI agent (MCP)
|
|
122
|
+
|
|
123
|
+
```json
|
|
124
|
+
{
|
|
125
|
+
"mcpServers": {
|
|
126
|
+
"google-search": {
|
|
127
|
+
"command": "google-browser-scraper",
|
|
128
|
+
"args": ["mcp", "--nodemaven", "--country", "us", "--prewarm"],
|
|
129
|
+
"env": {"NODEMAVEN_LOGIN": "...", "NODEMAVEN_PASSWORD": "..."}
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
It exposes one tool, `google_search(query, pages)`.
|
|
136
|
+
|
|
137
|
+
## Docker
|
|
138
|
+
|
|
139
|
+
```
|
|
140
|
+
docker build -t google-browser-scraper .
|
|
141
|
+
docker run --rm google-browser-scraper doctor
|
|
142
|
+
docker run --rm -p 8000:8000 -e GBS_API_KEY=change-me \
|
|
143
|
+
-e GBS_PROXY='http://USER-session-{session}:PASS@gate.example.com:7000' \
|
|
144
|
+
google-browser-scraper serve --host 0.0.0.0
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
## Check your machine
|
|
148
|
+
|
|
149
|
+
```
|
|
150
|
+
google-browser-scraper doctor
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
Starts the browser on a blank page, sends nothing, and reports what a website
|
|
154
|
+
would see: whether the browser looks headless, whether WebGL works, the screen
|
|
155
|
+
size. Google answers some machines much less than others, so run this first.
|
|
156
|
+
|
|
157
|
+
## What this is not
|
|
158
|
+
|
|
159
|
+
Not a free Google Search API, and not a guarantee that every proxy exit will
|
|
160
|
+
work. It is a browser-based collector that makes failures explicit, drops exits
|
|
161
|
+
Google refuses, and tells you what each page cost.
|
|
162
|
+
|
|
163
|
+
## Status
|
|
164
|
+
|
|
165
|
+
Early. Tested on Windows. **Linux and Docker are experimental**: on a machine
|
|
166
|
+
without a GPU, Chrome has no WebGL, which a normal desktop always has; the
|
|
167
|
+
Docker image turns on a software renderer for it. The `cloak` engine
|
|
168
|
+
(`--engine cloak`) is experimental too; `patchright` is the default.
|
|
169
|
+
|
|
170
|
+
Google changes its pages often. If results come back empty or wrong, run with
|
|
171
|
+
`--save-html DIR` and open an issue with what you see.
|
|
172
|
+
|
|
173
|
+
You are responsible for using this within Google's terms and the law where you
|
|
174
|
+
run it.
|
|
175
|
+
|
|
176
|
+
## Adding an engine
|
|
177
|
+
|
|
178
|
+
An engine joins this tool only after it has been measured on
|
|
179
|
+
[proxy-benchmark](https://github.com/nodemaven/proxy-benchmark): add it there,
|
|
180
|
+
run the warm-up ladder against Google -
|
|
181
|
+
|
|
182
|
+
```
|
|
183
|
+
python scripts/run_ladder.py --engines your-engine
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
- and open a pull request here with the run file. The ladder runs identities
|
|
187
|
+
with and without the warm-up in the same hours, so the result shows whether the
|
|
188
|
+
engine actually gets served, not whether it got lucky once.
|
|
189
|
+
|
|
190
|
+
## License
|
|
191
|
+
|
|
192
|
+
MIT
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=69"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "google-browser-scraper"
|
|
7
|
+
version = "0.0.1"
|
|
8
|
+
description = "Google Search results from a real browser through your own sticky proxy, with captchas and blocks reported explicitly."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { text = "MIT" }
|
|
11
|
+
requires-python = ">=3.10"
|
|
12
|
+
authors = [{ name = "NodeMaven" }]
|
|
13
|
+
keywords = ["google", "serp", "scraper", "search", "proxy", "playwright", "patchright"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 2 - Pre-Alpha",
|
|
16
|
+
"License :: OSI Approved :: MIT License",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Topic :: Internet :: WWW/HTTP :: Indexing/Search",
|
|
19
|
+
]
|
|
20
|
+
dependencies = [
|
|
21
|
+
# Pinned, because each patchright release ships a different Chromium and
|
|
22
|
+
# the browser build is part of what Google sees. 1.62.2 is Chromium 151,
|
|
23
|
+
# the build every live run of this tool was made with.
|
|
24
|
+
"patchright==1.62.2",
|
|
25
|
+
"lxml>=5",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
[project.optional-dependencies]
|
|
29
|
+
# The NodeMaven SDK builds sticky-session usernames for its gateway. Any other
|
|
30
|
+
# provider works through a proxy URL with a `{session}` placeholder instead.
|
|
31
|
+
nodemaven = ["nodemaven>=0.1.5"]
|
|
32
|
+
# CloakBrowser's patched Chromium; downloads its own binary on first use.
|
|
33
|
+
cloak = ["cloakbrowser>=0.5"]
|
|
34
|
+
all = ["nodemaven>=0.1.5", "cloakbrowser>=0.5"]
|
|
35
|
+
dev = ["pytest>=8", "ruff>=0.6"]
|
|
36
|
+
|
|
37
|
+
[project.scripts]
|
|
38
|
+
google-browser-scraper = "google_browser_scraper.cli:main"
|
|
39
|
+
|
|
40
|
+
[project.urls]
|
|
41
|
+
Source = "https://github.com/nodemaven/google-browser-scraper"
|
|
42
|
+
Issues = "https://github.com/nodemaven/google-browser-scraper/issues"
|
|
43
|
+
Benchmark = "https://github.com/nodemaven/proxy-benchmark"
|
|
44
|
+
|
|
45
|
+
[tool.setuptools.packages.find]
|
|
46
|
+
where = ["src"]
|
|
47
|
+
|
|
48
|
+
[tool.pytest.ini_options]
|
|
49
|
+
testpaths = ["tests"]
|
|
50
|
+
pythonpath = ["src"]
|
|
51
|
+
|
|
52
|
+
[tool.ruff]
|
|
53
|
+
line-length = 100
|
|
54
|
+
target-version = "py310"
|
|
55
|
+
|
|
56
|
+
[tool.ruff.lint]
|
|
57
|
+
select = ["E", "F", "W", "I", "B", "UP"]
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Google search results from a real browser through your own proxy.
|
|
2
|
+
|
|
3
|
+
from google_browser_scraper import Scraper, Settings, ProxyTemplate
|
|
4
|
+
|
|
5
|
+
proxy = ProxyTemplate("http://user-session-{session}:pass@gate.example.com:7000")
|
|
6
|
+
for record in Scraper(proxy, Settings(pages=2)).run(["best running shoes"]):
|
|
7
|
+
print(record["organic_results"])
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
__version__ = "0.0.1"
|
|
11
|
+
|
|
12
|
+
from .classify import Verdict, classify # noqa: E402
|
|
13
|
+
from .parse import parse_serp # noqa: E402
|
|
14
|
+
from .proxy import NodeMavenSource, ProxyTemplate # noqa: E402
|
|
15
|
+
from .scraper import ExitsRefused, Scraper, Settings # noqa: E402
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"ExitsRefused",
|
|
19
|
+
"NodeMavenSource",
|
|
20
|
+
"ProxyTemplate",
|
|
21
|
+
"Scraper",
|
|
22
|
+
"Settings",
|
|
23
|
+
"Verdict",
|
|
24
|
+
"__version__",
|
|
25
|
+
"classify",
|
|
26
|
+
"parse_serp",
|
|
27
|
+
]
|