gmbscraper 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gmbscraper/__init__.py +39 -0
- gmbscraper/_adaptive_search.py +204 -0
- gmbscraper/_matching.py +64 -0
- gmbscraper/_page.py +418 -0
- gmbscraper/_parallel.py +40 -0
- gmbscraper/_places.py +225 -0
- gmbscraper/_reviews.py +262 -0
- gmbscraper/_scraper.py +117 -0
- gmbscraper/_version.py +7 -0
- gmbscraper/models.py +89 -0
- gmbscraper/parsing.py +220 -0
- gmbscraper/py.typed +0 -0
- gmbscraper-0.1.0.dist-info/METADATA +108 -0
- gmbscraper-0.1.0.dist-info/RECORD +16 -0
- gmbscraper-0.1.0.dist-info/WHEEL +4 -0
- gmbscraper-0.1.0.dist-info/licenses/LICENSE +21 -0
gmbscraper/__init__.py
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Playwright-based scraper for Google Maps / Google Business Profile listings and reviews."""
|
|
2
|
+
|
|
3
|
+
from gmbscraper._matching import has_match_key, is_definitive_match, match_place, pick_best_match
|
|
4
|
+
from gmbscraper._parallel import chunk_items, run_chunked
|
|
5
|
+
from gmbscraper._reviews import GoogleReviewsUnavailableError, scrape_reviews
|
|
6
|
+
from gmbscraper._scraper import GoogleMapsScraper, google_maps_browser
|
|
7
|
+
from gmbscraper._version import __version__
|
|
8
|
+
from gmbscraper.models import (
|
|
9
|
+
BusinessIdentity,
|
|
10
|
+
MapsReview,
|
|
11
|
+
MapsSearchLink,
|
|
12
|
+
MapsView,
|
|
13
|
+
OutletIdentity,
|
|
14
|
+
PlaceProfile,
|
|
15
|
+
SearchWindow,
|
|
16
|
+
)
|
|
17
|
+
from gmbscraper.parsing import normalize_text
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"BusinessIdentity",
|
|
21
|
+
"GoogleMapsScraper",
|
|
22
|
+
"GoogleReviewsUnavailableError",
|
|
23
|
+
"MapsReview",
|
|
24
|
+
"MapsSearchLink",
|
|
25
|
+
"MapsView",
|
|
26
|
+
"OutletIdentity",
|
|
27
|
+
"PlaceProfile",
|
|
28
|
+
"SearchWindow",
|
|
29
|
+
"__version__",
|
|
30
|
+
"chunk_items",
|
|
31
|
+
"google_maps_browser",
|
|
32
|
+
"has_match_key",
|
|
33
|
+
"is_definitive_match",
|
|
34
|
+
"match_place",
|
|
35
|
+
"normalize_text",
|
|
36
|
+
"pick_best_match",
|
|
37
|
+
"run_chunked",
|
|
38
|
+
"scrape_reviews",
|
|
39
|
+
]
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import math
|
|
4
|
+
import re
|
|
5
|
+
from collections.abc import Callable
|
|
6
|
+
from typing import Any, Protocol
|
|
7
|
+
|
|
8
|
+
from gmbscraper.models import MapsSearchLink, MapsView, SearchWindow
|
|
9
|
+
|
|
10
|
+
GOOGLE_MAPS_QUERY_RESULT_LIMIT = 120
|
|
11
|
+
SATURATED_RESULT_COUNT = 110
|
|
12
|
+
MAX_ADAPTIVE_DEPTH = 3
|
|
13
|
+
MAX_ADAPTIVE_WINDOWS = 220
|
|
14
|
+
MAX_SPLIT_ZOOM = 17
|
|
15
|
+
MAP_VIEWPORT_WIDTH = 1366
|
|
16
|
+
MAP_VIEWPORT_HEIGHT = 900
|
|
17
|
+
FALLBACK_SEARCH_ZOOM = 12
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class SearchScraper(Protocol):
|
|
21
|
+
def new_page(self) -> Any: ...
|
|
22
|
+
def open_search(self, page: Any, query: str) -> None: ...
|
|
23
|
+
def open_search_at(
|
|
24
|
+
self, page: Any, query: str, *, lat: float, lng: float, zoom: int
|
|
25
|
+
) -> None: ...
|
|
26
|
+
def collect_place_links(self, page: Any, limit: int) -> list[tuple[str, str]]: ...
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def collect_links_for_geo(
|
|
30
|
+
*,
|
|
31
|
+
scraper: SearchScraper,
|
|
32
|
+
query: str,
|
|
33
|
+
latitude: float | None = None,
|
|
34
|
+
longitude: float | None = None,
|
|
35
|
+
log: Callable[[str], None] | None = None,
|
|
36
|
+
) -> list[MapsSearchLink]:
|
|
37
|
+
"""Search a query and, if results saturate Google's ~120 cap, recursively
|
|
38
|
+
split the viewport into quadrants to pull results beyond that cap.
|
|
39
|
+
"""
|
|
40
|
+
seed_links, seed_view = _collect_seed_links(scraper, query)
|
|
41
|
+
seen: set[str] = set()
|
|
42
|
+
links: list[MapsSearchLink] = []
|
|
43
|
+
_add_unique_links(links, seen, seed_links, keep_search_rank=True)
|
|
44
|
+
if log:
|
|
45
|
+
log(f"seed found={len(seed_links)} total={len(links)}")
|
|
46
|
+
if len(seed_links) < SATURATED_RESULT_COUNT:
|
|
47
|
+
return links
|
|
48
|
+
|
|
49
|
+
root_view = seed_view
|
|
50
|
+
if root_view is None and latitude is not None and longitude is not None:
|
|
51
|
+
root_view = MapsView(lat=latitude, lng=longitude, zoom=FALLBACK_SEARCH_ZOOM)
|
|
52
|
+
if root_view is None:
|
|
53
|
+
if log:
|
|
54
|
+
log("missing search bounds; seed query only")
|
|
55
|
+
return links
|
|
56
|
+
|
|
57
|
+
max_depth = _max_adaptive_depth(root_view.zoom)
|
|
58
|
+
queue: list[tuple[int, SearchWindow]] = [
|
|
59
|
+
(1, window)
|
|
60
|
+
for window in _split_window(
|
|
61
|
+
_search_window_from_view(root_view, label="root"),
|
|
62
|
+
next_zoom=root_view.zoom + 1,
|
|
63
|
+
)
|
|
64
|
+
]
|
|
65
|
+
searched = 1
|
|
66
|
+
while queue and searched < MAX_ADAPTIVE_WINDOWS:
|
|
67
|
+
depth, window = queue.pop(0)
|
|
68
|
+
searched += 1
|
|
69
|
+
page = scraper.new_page()
|
|
70
|
+
try:
|
|
71
|
+
scraper.open_search_at(
|
|
72
|
+
page,
|
|
73
|
+
query,
|
|
74
|
+
lat=window.center_lat,
|
|
75
|
+
lng=window.center_lng,
|
|
76
|
+
zoom=window.zoom,
|
|
77
|
+
)
|
|
78
|
+
window_links = scraper.collect_place_links(page, GOOGLE_MAPS_QUERY_RESULT_LIMIT)
|
|
79
|
+
finally:
|
|
80
|
+
page.close()
|
|
81
|
+
added = _add_unique_links(links, seen, window_links, keep_search_rank=False)
|
|
82
|
+
if log:
|
|
83
|
+
log(
|
|
84
|
+
f"adaptive d{depth} z{window.zoom} {window.label} "
|
|
85
|
+
f"found={len(window_links)} new={added} total={len(links)}"
|
|
86
|
+
)
|
|
87
|
+
if _should_split_window(window_links, depth=depth, zoom=window.zoom, max_depth=max_depth):
|
|
88
|
+
queue.extend(
|
|
89
|
+
(depth + 1, child) for child in _split_window(window, next_zoom=window.zoom + 1)
|
|
90
|
+
)
|
|
91
|
+
return links
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _max_adaptive_depth(root_zoom: int) -> int:
|
|
95
|
+
zoom_headroom = max(0, MAX_SPLIT_ZOOM - root_zoom)
|
|
96
|
+
return min(zoom_headroom, MAX_ADAPTIVE_DEPTH)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _should_split_window(
|
|
100
|
+
window_links: list[tuple[str, str]],
|
|
101
|
+
*,
|
|
102
|
+
depth: int,
|
|
103
|
+
zoom: int,
|
|
104
|
+
max_depth: int,
|
|
105
|
+
) -> bool:
|
|
106
|
+
if len(window_links) < SATURATED_RESULT_COUNT:
|
|
107
|
+
return False
|
|
108
|
+
if depth >= max_depth:
|
|
109
|
+
return False
|
|
110
|
+
return zoom < MAX_SPLIT_ZOOM
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _collect_seed_links(
|
|
114
|
+
scraper: SearchScraper, query: str
|
|
115
|
+
) -> tuple[list[tuple[str, str]], MapsView | None]:
|
|
116
|
+
page = scraper.new_page()
|
|
117
|
+
try:
|
|
118
|
+
scraper.open_search(page, query)
|
|
119
|
+
links = scraper.collect_place_links(page, GOOGLE_MAPS_QUERY_RESULT_LIMIT)
|
|
120
|
+
return links, _parse_maps_view(page.url)
|
|
121
|
+
finally:
|
|
122
|
+
page.close()
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _add_unique_links(
|
|
126
|
+
links: list[MapsSearchLink],
|
|
127
|
+
seen: set[str],
|
|
128
|
+
new_links: list[tuple[str, str]],
|
|
129
|
+
*,
|
|
130
|
+
keep_search_rank: bool,
|
|
131
|
+
) -> int:
|
|
132
|
+
added = 0
|
|
133
|
+
for search_position, (maps_url, fallback_name) in enumerate(new_links, start=1):
|
|
134
|
+
if maps_url in seen:
|
|
135
|
+
continue
|
|
136
|
+
seen.add(maps_url)
|
|
137
|
+
links.append(
|
|
138
|
+
MapsSearchLink(
|
|
139
|
+
maps_url=maps_url,
|
|
140
|
+
fallback_name=fallback_name,
|
|
141
|
+
search_rank=search_position if keep_search_rank else None,
|
|
142
|
+
)
|
|
143
|
+
)
|
|
144
|
+
added += 1
|
|
145
|
+
return added
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _parse_maps_view(url: str) -> MapsView | None:
|
|
149
|
+
match = re.search(r"@(-?\d+(?:\.\d+)?),(-?\d+(?:\.\d+)?),(\d+(?:\.\d+)?)z", url)
|
|
150
|
+
if match is None:
|
|
151
|
+
return None
|
|
152
|
+
return MapsView(
|
|
153
|
+
lat=float(match.group(1)),
|
|
154
|
+
lng=float(match.group(2)),
|
|
155
|
+
zoom=max(1, round(float(match.group(3)))),
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _search_window_from_view(view: MapsView, *, label: str) -> SearchWindow:
|
|
160
|
+
lat_span, lng_span = _viewport_span_degrees(view.lat, view.zoom)
|
|
161
|
+
return SearchWindow(
|
|
162
|
+
south=max(-85.0, view.lat - lat_span / 2),
|
|
163
|
+
west=_normalize_lng(view.lng - lng_span / 2),
|
|
164
|
+
north=min(85.0, view.lat + lat_span / 2),
|
|
165
|
+
east=_normalize_lng(view.lng + lng_span / 2),
|
|
166
|
+
zoom=view.zoom,
|
|
167
|
+
label=label,
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _viewport_span_degrees(center_lat: float, zoom: int) -> tuple[float, float]:
|
|
172
|
+
world_pixels = 256 * (2**zoom)
|
|
173
|
+
lat_rad = math.radians(_clamp(center_lat, -85.0, 85.0))
|
|
174
|
+
meters_per_pixel = 156543.03392 * math.cos(lat_rad) / (2**zoom)
|
|
175
|
+
width_m = MAP_VIEWPORT_WIDTH * meters_per_pixel
|
|
176
|
+
height_m = MAP_VIEWPORT_HEIGHT * meters_per_pixel
|
|
177
|
+
lat_span = height_m / 110_574
|
|
178
|
+
lng_span = width_m / max(1.0, 111_320 * math.cos(lat_rad))
|
|
179
|
+
if world_pixels <= 0:
|
|
180
|
+
return 0.0, 0.0
|
|
181
|
+
return max(lat_span, 0.001), max(lng_span, 0.001)
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _split_window(window: SearchWindow, *, next_zoom: int) -> list[SearchWindow]:
|
|
185
|
+
mid_lat = (window.south + window.north) / 2
|
|
186
|
+
mid_lng = (window.west + window.east) / 2
|
|
187
|
+
return [
|
|
188
|
+
SearchWindow(window.south, window.west, mid_lat, mid_lng, next_zoom, f"{window.label}.sw"),
|
|
189
|
+
SearchWindow(window.south, mid_lng, mid_lat, window.east, next_zoom, f"{window.label}.se"),
|
|
190
|
+
SearchWindow(mid_lat, window.west, window.north, mid_lng, next_zoom, f"{window.label}.nw"),
|
|
191
|
+
SearchWindow(mid_lat, mid_lng, window.north, window.east, next_zoom, f"{window.label}.ne"),
|
|
192
|
+
]
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _normalize_lng(value: float) -> float:
|
|
196
|
+
while value < -180:
|
|
197
|
+
value += 360
|
|
198
|
+
while value > 180:
|
|
199
|
+
value -= 360
|
|
200
|
+
return value
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _clamp(value: float, lower: float, upper: float) -> float:
|
|
204
|
+
return min(max(value, lower), upper)
|
gmbscraper/_matching.py
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from gmbscraper.models import BusinessIdentity, OutletIdentity, PlaceProfile
|
|
4
|
+
from gmbscraper.parsing import normalize_domain, normalize_name, normalize_phone
|
|
5
|
+
|
|
6
|
+
_REASON_RANK = {
|
|
7
|
+
"phone_and_domain_match": 0,
|
|
8
|
+
"phone_match": 1,
|
|
9
|
+
"domain_match": 2,
|
|
10
|
+
"exact_name_match": 3,
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def has_match_key(outlet: OutletIdentity, business: BusinessIdentity) -> bool:
|
|
15
|
+
"""Whether there's enough identity signal to attempt a match at all."""
|
|
16
|
+
return bool(
|
|
17
|
+
normalize_phone(outlet.phone)
|
|
18
|
+
or normalize_domain(business.website)
|
|
19
|
+
or normalize_name(business.name)
|
|
20
|
+
or normalize_name(outlet.name)
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def match_place(
|
|
25
|
+
place: PlaceProfile,
|
|
26
|
+
outlet: OutletIdentity,
|
|
27
|
+
business: BusinessIdentity,
|
|
28
|
+
) -> PlaceProfile | None:
|
|
29
|
+
"""Score a scraped place against known identity, tagging `place.match_reason` on success."""
|
|
30
|
+
outlet_phone = normalize_phone(outlet.phone)
|
|
31
|
+
place_phone = normalize_phone(place.phone)
|
|
32
|
+
business_domain = normalize_domain(business.website)
|
|
33
|
+
place_domain = normalize_domain(place.website)
|
|
34
|
+
place_name = normalize_name(place.name)
|
|
35
|
+
business_name = normalize_name(business.name)
|
|
36
|
+
outlet_name = normalize_name(outlet.name)
|
|
37
|
+
phone_match = bool(outlet_phone and place_phone and outlet_phone == place_phone)
|
|
38
|
+
domain_match = bool(business_domain and place_domain and business_domain == place_domain)
|
|
39
|
+
exact_name_match = bool(place_name and place_name in {business_name, outlet_name})
|
|
40
|
+
if not phone_match and not domain_match and not exact_name_match:
|
|
41
|
+
return None
|
|
42
|
+
if phone_match and domain_match:
|
|
43
|
+
place.match_reason = "phone_and_domain_match"
|
|
44
|
+
elif phone_match:
|
|
45
|
+
place.match_reason = "phone_match"
|
|
46
|
+
elif domain_match:
|
|
47
|
+
place.match_reason = "domain_match"
|
|
48
|
+
else:
|
|
49
|
+
place.match_reason = "exact_name_match"
|
|
50
|
+
return place
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def is_definitive_match(place: PlaceProfile) -> bool:
|
|
54
|
+
"""True if the match reason is strong enough to stop searching further candidates."""
|
|
55
|
+
return place.match_reason in _REASON_RANK
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def pick_best_match(places: list[PlaceProfile]) -> PlaceProfile | None:
|
|
59
|
+
if not places:
|
|
60
|
+
return None
|
|
61
|
+
return sorted(
|
|
62
|
+
places,
|
|
63
|
+
key=lambda item: (_REASON_RANK.get(item.match_reason or "", 99), item.rank),
|
|
64
|
+
)[0]
|