untappd-scraper 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- untappd_scraper/__init__.py +1 -0
- untappd_scraper/beer.py +193 -0
- untappd_scraper/html_session.py +231 -0
- untappd_scraper/logging_config.py +78 -0
- untappd_scraper/logs.yml +59 -0
- untappd_scraper/py.typed +0 -0
- untappd_scraper/structs/__init__.py +0 -0
- untappd_scraper/structs/mixins.py +63 -0
- untappd_scraper/structs/other.py +27 -0
- untappd_scraper/structs/web.py +183 -0
- untappd_scraper/user.py +173 -0
- untappd_scraper/user_beer_history.py +112 -0
- untappd_scraper/user_lists.py +268 -0
- untappd_scraper/user_lists_details.py +117 -0
- untappd_scraper/user_utils.py +21 -0
- untappd_scraper/user_venue_history.py +67 -0
- untappd_scraper/venue.py +168 -0
- untappd_scraper/venue_menus.py +321 -0
- untappd_scraper/venue_utils.py +14 -0
- untappd_scraper/web.py +113 -0
- untappd_scraper-0.4.0.dist-info/METADATA +124 -0
- untappd_scraper-0.4.0.dist-info/RECORD +24 -0
- untappd_scraper-0.4.0.dist-info/WHEEL +4 -0
- untappd_scraper-0.4.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Untappd Scraper functions."""
|
untappd_scraper/beer.py
ADDED
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
"""Untappd beers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import TYPE_CHECKING, Any
|
|
6
|
+
|
|
7
|
+
from core.mixins import SimpleRepr
|
|
8
|
+
from dateutil.parser import parse as parse_date
|
|
9
|
+
|
|
10
|
+
from untappd_scraper.html_session import get
|
|
11
|
+
from untappd_scraper.structs.web import WebActivityBeer, WebBeerDetails
|
|
12
|
+
from untappd_scraper.web import id_from_href, parsed_value, slug_from_href
|
|
13
|
+
|
|
14
|
+
if TYPE_CHECKING: # pragma: no cover
|
|
15
|
+
from collections.abc import Iterator
|
|
16
|
+
|
|
17
|
+
from requests_html import Element, HTMLResponse
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class Beer(SimpleRepr):
|
|
21
|
+
"""Untappd beer."""
|
|
22
|
+
|
|
23
|
+
def __init__(self, beer_id: int) -> None:
|
|
24
|
+
"""Initiate a Beer object, storing the beer ID and loading details.
|
|
25
|
+
|
|
26
|
+
Raises:
|
|
27
|
+
ValueError: invalid beer ID
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
beer_id (int): beer ID
|
|
31
|
+
"""
|
|
32
|
+
self.beer_id = beer_id
|
|
33
|
+
|
|
34
|
+
self._page = get(url_of(beer_id))
|
|
35
|
+
if not self._page.ok:
|
|
36
|
+
msg = f"Invalid beer ID {beer_id} ({self._page})"
|
|
37
|
+
raise ValueError(msg)
|
|
38
|
+
self._beer_details: WebBeerDetails = beer_details(resp=self._page)
|
|
39
|
+
|
|
40
|
+
def __getattr__(self, name: str) -> Any:
|
|
41
|
+
"""Return unknown attributes from beer details.
|
|
42
|
+
|
|
43
|
+
Args:
|
|
44
|
+
name (str): attribute to lookup
|
|
45
|
+
|
|
46
|
+
Returns:
|
|
47
|
+
Any: attribute value
|
|
48
|
+
"""
|
|
49
|
+
return getattr(self._beer_details, name)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
# ----- utils -----
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def url_of(beer_id: int) -> str:
|
|
56
|
+
"""Return the URL for a beer's main page.
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
beer_id (int): beer ID
|
|
60
|
+
|
|
61
|
+
Returns:
|
|
62
|
+
str: url to load to get beer's main page
|
|
63
|
+
"""
|
|
64
|
+
return f"https://untappd.com/beer/{beer_id}"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
# ----- beer details processing -----
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def beer_details(resp: HTMLResponse) -> WebBeerDetails:
|
|
71
|
+
"""Parse a user's main page into user details.
|
|
72
|
+
|
|
73
|
+
Args:
|
|
74
|
+
resp (HTMLResponse): beer's main page loaded
|
|
75
|
+
|
|
76
|
+
Returns:
|
|
77
|
+
WebBeerDetails: general beer details
|
|
78
|
+
"""
|
|
79
|
+
content_el = resp.html.find(".main .content", first=True)
|
|
80
|
+
description = "".join(
|
|
81
|
+
content_el.find(".desc .beer-descrption-read-less", first=True).xpath("//div/text()")
|
|
82
|
+
).strip()
|
|
83
|
+
|
|
84
|
+
return WebBeerDetails(
|
|
85
|
+
beer_id=id_from_href(content_el.find("a.check", first=True)),
|
|
86
|
+
name=content_el.find(".name h1", first=True).text,
|
|
87
|
+
description=description.strip(),
|
|
88
|
+
brewery=content_el.find(".name .brewery a", first=True).text,
|
|
89
|
+
brewery_slug=slug_from_href(content_el.find(".name .brewery a", first=True)),
|
|
90
|
+
style=content_el.find(".name p.style", first=True).text,
|
|
91
|
+
url=resp.url,
|
|
92
|
+
global_rating=content_el.find(".details [data-rating]", first=True).attrs[
|
|
93
|
+
"data-rating"
|
|
94
|
+
],
|
|
95
|
+
num_ratings=parsed_value(
|
|
96
|
+
"{:d} Rat", content_el.find(".details p.raters", first=True).text
|
|
97
|
+
),
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def checkin_activity(resp: HTMLResponse) -> Iterator[WebActivityBeer]:
|
|
102
|
+
"""Parse all available recent checkins for a user or in a venue.
|
|
103
|
+
|
|
104
|
+
Args:
|
|
105
|
+
resp (HTMLResponse): user's main page or venue's activity page
|
|
106
|
+
|
|
107
|
+
Returns:
|
|
108
|
+
Iterator[WebActivityBeer]: user's visible recent checkins
|
|
109
|
+
"""
|
|
110
|
+
return (checkin_details(checkin) for checkin in resp.html.find(".activity .item"))
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def checkin_details(checkin_item: Element) -> WebActivityBeer:
|
|
114
|
+
"""Extract beer details from a checkin.
|
|
115
|
+
|
|
116
|
+
Args:
|
|
117
|
+
checkin_item (Element): single checkin
|
|
118
|
+
|
|
119
|
+
Returns:
|
|
120
|
+
WebActivityBeer: Interesting details for a beer
|
|
121
|
+
"""
|
|
122
|
+
user_el, beer_el, brewery_el, location_el = extract_checkin_elements(checkin_item)
|
|
123
|
+
|
|
124
|
+
checkin_time_el = checkin_item.find(".bottom .time", first=True)
|
|
125
|
+
checkin_time = checkin_time_el.attrs.get("data-gregtime", checkin_time_el.text)
|
|
126
|
+
checkin_time = parse_date(checkin_time)
|
|
127
|
+
assert checkin_time.tzinfo and checkin_time.tzinfo.utcoffset(checkin_time) is not None, (
|
|
128
|
+
f"Naive datetime from {checkin_time_el.html=}"
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
purchased_at = checkin_item.find(".purchased a", first=True)
|
|
132
|
+
try:
|
|
133
|
+
comment = checkin_item.find("p.comment-text", first=True).text
|
|
134
|
+
except AttributeError:
|
|
135
|
+
comment = None
|
|
136
|
+
serving = checkin_item.find(".serving", first=True)
|
|
137
|
+
|
|
138
|
+
data_rating_element = checkin_item.find("[data-rating]", first=True)
|
|
139
|
+
if data_rating_element:
|
|
140
|
+
data_rating: float | None = float(data_rating_element.attrs["data-rating"])
|
|
141
|
+
else:
|
|
142
|
+
data_rating = None # pragma: no cover
|
|
143
|
+
|
|
144
|
+
try:
|
|
145
|
+
friends: list[str] | None = [
|
|
146
|
+
slug_from_href(href) for href in checkin_item.find(".tagged-friends a")
|
|
147
|
+
]
|
|
148
|
+
except AttributeError: # pragma: no cover
|
|
149
|
+
friends = None
|
|
150
|
+
|
|
151
|
+
return WebActivityBeer(
|
|
152
|
+
checkin_id=int(checkin_item.attrs["data-checkin-id"]),
|
|
153
|
+
checkin=checkin_time,
|
|
154
|
+
user_name=slug_from_href(user_el),
|
|
155
|
+
name=beer_el.text,
|
|
156
|
+
beer_id=id_from_href(beer_el),
|
|
157
|
+
brewery=brewery_el.text,
|
|
158
|
+
brewery_slug=slug_from_href(brewery_el),
|
|
159
|
+
location=location_el.text if location_el else None,
|
|
160
|
+
location_id=id_from_href(location_el) if location_el else None,
|
|
161
|
+
purchased_at=purchased_at.text if purchased_at else None,
|
|
162
|
+
purchased_id=id_from_href(purchased_at) if purchased_at else None,
|
|
163
|
+
comment=comment,
|
|
164
|
+
serving=serving.text if serving else None,
|
|
165
|
+
user_rating=data_rating,
|
|
166
|
+
friends=friends,
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def extract_checkin_elements(
|
|
171
|
+
element: Element,
|
|
172
|
+
) -> tuple[Element, Element, Element, Element | None]:
|
|
173
|
+
"""Extract four linked elements in a checkin.
|
|
174
|
+
|
|
175
|
+
Args:
|
|
176
|
+
element (Element): checkin element
|
|
177
|
+
|
|
178
|
+
Raises:
|
|
179
|
+
ValueError: element passed didn't contain 3-4 <a> tags
|
|
180
|
+
|
|
181
|
+
Returns:
|
|
182
|
+
tuple[Element, Element, Element, Element]: user, beer, brewery, location
|
|
183
|
+
"""
|
|
184
|
+
elements = element.find(".top .text a")
|
|
185
|
+
|
|
186
|
+
if len(elements) == 3:
|
|
187
|
+
return elements + [None] # pragma: no cover
|
|
188
|
+
if len(elements) == 4:
|
|
189
|
+
return elements
|
|
190
|
+
|
|
191
|
+
raise ValueError(
|
|
192
|
+
f"Wanted 3 or 4 <a> elements (not {len(elements)}) in {element.html}"
|
|
193
|
+
) # pragma: no cover
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""HTML session to be shared across all modules."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from datetime import timedelta
|
|
7
|
+
from typing import TYPE_CHECKING, Final
|
|
8
|
+
|
|
9
|
+
import ratelim
|
|
10
|
+
import requests
|
|
11
|
+
from requests_cache import CacheMixin
|
|
12
|
+
from requests_html import HTMLResponse, HTMLSession
|
|
13
|
+
from tenacity import (
|
|
14
|
+
RetryCallState,
|
|
15
|
+
_utils,
|
|
16
|
+
retry,
|
|
17
|
+
retry_base,
|
|
18
|
+
stop_after_attempt,
|
|
19
|
+
wait_exponential,
|
|
20
|
+
)
|
|
21
|
+
from tenacity.after import after_log
|
|
22
|
+
from tenacity.wait import wait_base
|
|
23
|
+
|
|
24
|
+
if TYPE_CHECKING: # pragma: no cover
|
|
25
|
+
from collections.abc import Callable
|
|
26
|
+
|
|
27
|
+
logger = logging.getLogger(__name__)
|
|
28
|
+
|
|
29
|
+
CACHE_EXPIRY: Final[timedelta] = timedelta(hours=1)
|
|
30
|
+
MAX_GET_MIN1: Final[int] = 40 # max number of web GETs in a minute
|
|
31
|
+
MAX_RETRY_SECS: Final[int] = 180 # never wait more than this for a retry
|
|
32
|
+
MAX_RETRY_ATTEMPTS: Final[int] = 9 # give up after this many retries
|
|
33
|
+
MIN1: Final[int] = 60 # seconds
|
|
34
|
+
|
|
35
|
+
# Allow user to check these. Don't raise an exception here
|
|
36
|
+
ACCEPTABLE_HTTP_STATUS: Final[frozenset[int]] = frozenset((requests.codes["not_found"],))
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class CachedHTMLSession(CacheMixin, HTMLSession): # pyright: ignore[reportIncompatibleMethodOverride]
|
|
40
|
+
"""Session with features from both CachedSession and HTMLSession."""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
_html_session = CachedHTMLSession(cache_name="html", expire_after=CACHE_EXPIRY)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@ratelim.greedy(MAX_GET_MIN1, MIN1)
|
|
47
|
+
def get(url: str, *, emulate_404: bool = False, **kwargs: str) -> requests.Response:
|
|
48
|
+
"""Get a URL.
|
|
49
|
+
|
|
50
|
+
Handles too many requests errors, and retries after waiting
|
|
51
|
+
|
|
52
|
+
Args:
|
|
53
|
+
url (str): URL to get
|
|
54
|
+
emulate_404 (bool): if True, return a 404 response
|
|
55
|
+
kwargs (dict): extra requests options, eg, params and headers
|
|
56
|
+
|
|
57
|
+
Returns:
|
|
58
|
+
requests.Response: response to get
|
|
59
|
+
"""
|
|
60
|
+
if emulate_404:
|
|
61
|
+
url = "https://httpbin.org/status/404"
|
|
62
|
+
resp = _get(url, **kwargs)
|
|
63
|
+
logger.debug(
|
|
64
|
+
"GET %s (%s) received %s\tExpires: %s, Headers: %s",
|
|
65
|
+
url,
|
|
66
|
+
kwargs,
|
|
67
|
+
resp,
|
|
68
|
+
resp.expires, # pyright: ignore[reportAttributeAccessIssue]
|
|
69
|
+
resp.headers,
|
|
70
|
+
)
|
|
71
|
+
return resp
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
# ----- Tenacity -----
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class RetryAfter(wait_base):
|
|
78
|
+
"""Strategy that tries to wait as per Retry-After header.
|
|
79
|
+
|
|
80
|
+
Tries to wait for the length specified by the Retry-After header,
|
|
81
|
+
or the underlying wait / fallback strategy if not.
|
|
82
|
+
See RFC 6585 § 4.
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
def __init__(self, fallback: wait_base) -> None:
|
|
86
|
+
"""Store fallback strategy in case we can't work out retry wait time.
|
|
87
|
+
|
|
88
|
+
Args:
|
|
89
|
+
fallback (wait_base): fallback wait strategy if no Retry-After found
|
|
90
|
+
"""
|
|
91
|
+
self.fallback = fallback
|
|
92
|
+
|
|
93
|
+
def __call__(self, retry_state: RetryCallState) -> int: # pragma: no cover
|
|
94
|
+
"""Return seconds to wait until retry.
|
|
95
|
+
|
|
96
|
+
Args:
|
|
97
|
+
retry_state (RetryState): State of retry, with .outcome property
|
|
98
|
+
storing exception.
|
|
99
|
+
|
|
100
|
+
Returns:
|
|
101
|
+
int: seconds to wait
|
|
102
|
+
"""
|
|
103
|
+
assert retry_state.outcome
|
|
104
|
+
exc = retry_state.outcome.exception()
|
|
105
|
+
if isinstance(exc, requests.HTTPError):
|
|
106
|
+
retry_after = exc.response.headers.get("Retry-After")
|
|
107
|
+
logger.debug("Searching response header and found Retry-After of %s", retry_after)
|
|
108
|
+
|
|
109
|
+
try:
|
|
110
|
+
return int(retry_after) # pyright: ignore[reportArgumentType]
|
|
111
|
+
except (TypeError, ValueError):
|
|
112
|
+
return int(self.fallback(retry_state))
|
|
113
|
+
|
|
114
|
+
return int(self.fallback(retry_state))
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def my_before_sleep_log(
|
|
118
|
+
user_logger: logging.Logger, *, exc_info: bool = False
|
|
119
|
+
) -> Callable[[RetryCallState], None]:
|
|
120
|
+
"""Before call strategy that logs to some logger the attempt.
|
|
121
|
+
|
|
122
|
+
Logging level is determined by the number of retries.
|
|
123
|
+
|
|
124
|
+
Lifted from Tenacity function and removed hard-coded log level
|
|
125
|
+
|
|
126
|
+
Args:
|
|
127
|
+
user_logger (Logger): logger to use
|
|
128
|
+
exc_info (bool, optional): Is there an exception. Defaults to False.
|
|
129
|
+
|
|
130
|
+
Returns:
|
|
131
|
+
logging function
|
|
132
|
+
"""
|
|
133
|
+
|
|
134
|
+
def log_it(retry_state: RetryCallState) -> None: # pragma: no cover
|
|
135
|
+
if retry_state.attempt_number < 1:
|
|
136
|
+
log_level = logging.DEBUG
|
|
137
|
+
elif retry_state.attempt_number == 1:
|
|
138
|
+
log_level = logging.INFO
|
|
139
|
+
else:
|
|
140
|
+
log_level = logging.WARNING
|
|
141
|
+
|
|
142
|
+
assert retry_state.outcome
|
|
143
|
+
|
|
144
|
+
if retry_state.outcome.failed:
|
|
145
|
+
ex = retry_state.outcome.exception()
|
|
146
|
+
verb, retry_value = "raised", f"{type(ex).__name__}: {ex}"
|
|
147
|
+
|
|
148
|
+
if exc_info and retry_state.outcome:
|
|
149
|
+
local_exc_info = retry_state.outcome.exception()
|
|
150
|
+
else:
|
|
151
|
+
local_exc_info = False
|
|
152
|
+
else:
|
|
153
|
+
verb, retry_value = "returned", retry_state.outcome.result()
|
|
154
|
+
local_exc_info = False # exc_info does not apply when no exception
|
|
155
|
+
|
|
156
|
+
user_logger.log(
|
|
157
|
+
log_level,
|
|
158
|
+
"Retrying %s (%s attempt) in %s seconds as it %s %s.",
|
|
159
|
+
_utils.get_callback_name(retry_state.fn), # pyright: ignore[reportArgumentType]
|
|
160
|
+
_utils.to_ordinal(retry_state.attempt_number),
|
|
161
|
+
retry_state.next_action.sleep, # pyright: ignore[reportOptionalMemberAccess]
|
|
162
|
+
verb,
|
|
163
|
+
retry_value,
|
|
164
|
+
exc_info=local_exc_info,
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
return log_it
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def is_throttling_related_exception(excp: Exception) -> bool: # pragma: no cover
|
|
171
|
+
"""Check is the exception is a requests one and throttling related.
|
|
172
|
+
|
|
173
|
+
Args:
|
|
174
|
+
excp (Exception): exception raised
|
|
175
|
+
|
|
176
|
+
Returns:
|
|
177
|
+
bool: was it a throttle
|
|
178
|
+
"""
|
|
179
|
+
return (
|
|
180
|
+
isinstance(excp, requests.HTTPError)
|
|
181
|
+
and excp.response.status_code == requests.codes.too_many_requests
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
class RetryIfThrottling(retry_base):
|
|
186
|
+
"""Retry class which only retries if a throttling exception occured.
|
|
187
|
+
|
|
188
|
+
From https://www.seelk.co/blog/efficient-client-side-handling-of-api-throttling-in-python-with-tenacity/
|
|
189
|
+
"""
|
|
190
|
+
|
|
191
|
+
def __call__(self, retry_state: RetryCallState) -> bool: # pragma: no cover
|
|
192
|
+
"""Return if the call raised an exception and it's a throttle.
|
|
193
|
+
|
|
194
|
+
Args:
|
|
195
|
+
retry_state (RetryCallState): info about current retry invocation
|
|
196
|
+
|
|
197
|
+
Returns:
|
|
198
|
+
bool: is it throttling related
|
|
199
|
+
"""
|
|
200
|
+
if (
|
|
201
|
+
retry_state.outcome
|
|
202
|
+
and retry_state.outcome.failed
|
|
203
|
+
and (exception := retry_state.outcome.exception())
|
|
204
|
+
):
|
|
205
|
+
return is_throttling_related_exception(exception) # pyright: ignore[reportArgumentType]
|
|
206
|
+
return False
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
# ---- Main part of tenacity retry ----
|
|
210
|
+
|
|
211
|
+
# Retry too-many-requests after a delay
|
|
212
|
+
|
|
213
|
+
# - Retries if throttle response received
|
|
214
|
+
# - Checks for Retry-After header and wait that long
|
|
215
|
+
# - If no header, waits expontially longer each retry
|
|
216
|
+
# - logs retries, with increasing severity as attempts increase
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
@retry(
|
|
220
|
+
reraise=True,
|
|
221
|
+
retry=RetryIfThrottling(),
|
|
222
|
+
wait=RetryAfter(fallback=wait_exponential(min=5, max=MAX_RETRY_SECS)),
|
|
223
|
+
stop=stop_after_attempt(MAX_RETRY_ATTEMPTS),
|
|
224
|
+
before_sleep=my_before_sleep_log(logger),
|
|
225
|
+
after=after_log(logger, logging.DEBUG),
|
|
226
|
+
)
|
|
227
|
+
def _get(url: str, **kwargs: str) -> HTMLResponse:
|
|
228
|
+
resp = _html_session.get(url, **kwargs)
|
|
229
|
+
if resp.status_code not in ACCEPTABLE_HTTP_STATUS:
|
|
230
|
+
resp.raise_for_status() # pragma: no cover
|
|
231
|
+
return resp # pyright: ignore[reportReturnType]
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Set up logging using the Loguru library."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import inspect
|
|
6
|
+
import logging
|
|
7
|
+
import sys
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Final
|
|
10
|
+
|
|
11
|
+
from loguru import logger
|
|
12
|
+
|
|
13
|
+
# ----- Constants -----
|
|
14
|
+
|
|
15
|
+
# Log formats
|
|
16
|
+
STANDARD: Final = "[{time:HH:mm:ss}] {level} - {message}"
|
|
17
|
+
DETAIL: Final = "{time} {file:>25}:{line:<4} {level:<8} {message}"
|
|
18
|
+
ROTATION: Final = "1 hour"
|
|
19
|
+
RETENTION: Final = "2 days"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def configure_logging(
|
|
23
|
+
log_filename: str | Path,
|
|
24
|
+
*,
|
|
25
|
+
standard_format: str = STANDARD,
|
|
26
|
+
detail_format: str = DETAIL,
|
|
27
|
+
log_rotation: str = ROTATION,
|
|
28
|
+
log_retention: str = RETENTION,
|
|
29
|
+
) -> None:
|
|
30
|
+
"""Setup logging for the application."""
|
|
31
|
+
# Capture things like Hishel logging
|
|
32
|
+
intercept_logging()
|
|
33
|
+
# Replace the default StdErr handler.
|
|
34
|
+
logger.remove()
|
|
35
|
+
logger.add(sys.stderr, level="WARNING", format=standard_format)
|
|
36
|
+
|
|
37
|
+
log_filename = Path(log_filename)
|
|
38
|
+
if log_filename.suffix != ".log":
|
|
39
|
+
log_filename = log_filename.with_suffix(".log")
|
|
40
|
+
|
|
41
|
+
# Add a rotating file handler.
|
|
42
|
+
logger.add(
|
|
43
|
+
log_filename,
|
|
44
|
+
level="DEBUG",
|
|
45
|
+
format=detail_format,
|
|
46
|
+
rotation=log_rotation,
|
|
47
|
+
retention=log_retention,
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
# ----- Interface to the standard logging module -----
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class InterceptHandler(logging.Handler):
|
|
55
|
+
"""Send logs to Loguru."""
|
|
56
|
+
|
|
57
|
+
def emit(self, record: logging.LogRecord) -> None:
|
|
58
|
+
"""Emit a log record."""
|
|
59
|
+
# Get corresponding Loguru level if it exists.
|
|
60
|
+
level: str | int
|
|
61
|
+
try:
|
|
62
|
+
level = logger.level(record.levelname).name
|
|
63
|
+
except ValueError:
|
|
64
|
+
level = record.levelno
|
|
65
|
+
|
|
66
|
+
# Find caller from where originated the logged message.
|
|
67
|
+
frame, depth = inspect.currentframe(), 0
|
|
68
|
+
while frame and (depth == 0 or frame.f_code.co_filename == logging.__file__):
|
|
69
|
+
frame = frame.f_back
|
|
70
|
+
depth += 1
|
|
71
|
+
|
|
72
|
+
logger.opt(depth=depth, exception=record.exc_info).log(level, record.getMessage())
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def intercept_logging() -> None:
|
|
76
|
+
"""Intercept standard logging and send it to Loguru."""
|
|
77
|
+
# Configure the root logger
|
|
78
|
+
logging.basicConfig(handlers=[InterceptHandler()], level=0, force=True)
|
untappd_scraper/logs.yml
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# Inspired from https://towardsdatascience.com/
|
|
2
|
+
# ultimate-guide-to-python-debugging-854dea731e1b
|
|
3
|
+
|
|
4
|
+
version: 1
|
|
5
|
+
disable_existing_loggers: false
|
|
6
|
+
|
|
7
|
+
formatters:
|
|
8
|
+
standard:
|
|
9
|
+
format: "[%(asctime)s] %(levelname)s - %(message)s"
|
|
10
|
+
datefmt: "%H:%M:%S"
|
|
11
|
+
detail:
|
|
12
|
+
format: "%(asctime)s %(filename)25s:%(lineno)-4d %(levelname)-8s %(message)s"
|
|
13
|
+
|
|
14
|
+
handlers:
|
|
15
|
+
console: # handler which will log into stdout
|
|
16
|
+
class: logging.StreamHandler
|
|
17
|
+
level: WARNING
|
|
18
|
+
formatter: standard # Use formatter defined above
|
|
19
|
+
stream: ext://sys.stdout
|
|
20
|
+
filesize: # handler which will log to file up to a size
|
|
21
|
+
class: logging.handlers.RotatingFileHandler
|
|
22
|
+
level: DEBUG
|
|
23
|
+
formatter: detail # Use formatter defined above
|
|
24
|
+
filename: ext://os.devnull
|
|
25
|
+
maxBytes: 10485760 # 10 Mb
|
|
26
|
+
backupCount: 10
|
|
27
|
+
encoding: utf8
|
|
28
|
+
filetime: # handler which will switch files regularly
|
|
29
|
+
class: logging.handlers.TimedRotatingFileHandler
|
|
30
|
+
level: INFO
|
|
31
|
+
formatter: detail # Use formatter defined above
|
|
32
|
+
filename: ext://os.devnull
|
|
33
|
+
when: W0
|
|
34
|
+
# interval: 0
|
|
35
|
+
# maxBytes: 10485760
|
|
36
|
+
# backupCount:
|
|
37
|
+
encoding: utf8
|
|
38
|
+
email: # email handler for error messages
|
|
39
|
+
class: logging.handlers.SMTPHandler
|
|
40
|
+
level: ERROR
|
|
41
|
+
formatter: standard # Use formatter defined above
|
|
42
|
+
mailhost: tbd # ext://util.config.log.mailhost
|
|
43
|
+
# credentials: ext://util.config.log.credentials
|
|
44
|
+
# secure: ext://util.config.log.secure
|
|
45
|
+
fromaddr: tbd # ext://util.config.log.fromaddr
|
|
46
|
+
toaddrs: tbd # ext://util.config.log.toaddrs
|
|
47
|
+
# - support_team@domain.tld
|
|
48
|
+
# - dev_team@domain.tld
|
|
49
|
+
subject: tbd # ext://util.config.log.subject # Houston, we have a problem.
|
|
50
|
+
|
|
51
|
+
root: # Loggers are organized in hierarchy - this is the root logger config
|
|
52
|
+
level: DEBUG
|
|
53
|
+
handlers: [console] # , file, email] # Attaches both handler defined above
|
|
54
|
+
|
|
55
|
+
# loggers: # Defines descendants of root logger
|
|
56
|
+
# mymodule: # Logger for "mymodule"
|
|
57
|
+
# level: INFO
|
|
58
|
+
# handlers: [file] # Will only use "file" handler defined above
|
|
59
|
+
# propagate: no # Will not propagate logs to "root" logger
|
untappd_scraper/py.typed
ADDED
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""Mixin classes."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from typing import Protocol
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class HasBeerId(Protocol):
|
|
8
|
+
"""Check a class has a beer id."""
|
|
9
|
+
|
|
10
|
+
beer_id: int
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class HasVenueId(Protocol):
|
|
14
|
+
"""Check a class has a venue id."""
|
|
15
|
+
|
|
16
|
+
venue_id: int
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class IdStrMixin:
|
|
20
|
+
"""Provide string versions of integer IDs."""
|
|
21
|
+
|
|
22
|
+
@property
|
|
23
|
+
def beer_id_str(self: HasBeerId) -> str:
|
|
24
|
+
"""Return str version of Beer ID.
|
|
25
|
+
|
|
26
|
+
JSON keys are str but dict can be int. Can get dupe JSON keys.
|
|
27
|
+
|
|
28
|
+
Returns:
|
|
29
|
+
str: string version of beer ID
|
|
30
|
+
"""
|
|
31
|
+
return str(self.beer_id)
|
|
32
|
+
|
|
33
|
+
@property
|
|
34
|
+
def venue_id_str(self: HasVenueId) -> str:
|
|
35
|
+
"""Return str version of Venue ID.
|
|
36
|
+
|
|
37
|
+
JSON keys are str but dict can be int. Can get dupe JSON keys.
|
|
38
|
+
|
|
39
|
+
Returns:
|
|
40
|
+
str: string version of venue ID
|
|
41
|
+
"""
|
|
42
|
+
return str(self.venue_id)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class HasBeerDetails(Protocol):
|
|
46
|
+
"""Check if a class has beer details."""
|
|
47
|
+
|
|
48
|
+
name: str
|
|
49
|
+
brewery: str
|
|
50
|
+
style: str
|
|
51
|
+
global_rating: float
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class BeerStrMixin:
|
|
55
|
+
"""Provide a nice description of a beer."""
|
|
56
|
+
|
|
57
|
+
def __str__(self: HasBeerDetails) -> str:
|
|
58
|
+
"""Create a summary description of a beer.
|
|
59
|
+
|
|
60
|
+
Returns:
|
|
61
|
+
str: beer description
|
|
62
|
+
"""
|
|
63
|
+
return f"{self.name} by {self.brewery}, {self.style} ({self.global_rating})"
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Structures used to represent data not scraped."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from dataclasses import astuple, dataclass
|
|
5
|
+
from functools import lru_cache
|
|
6
|
+
|
|
7
|
+
from haversine import haversine
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True)
|
|
11
|
+
class Location:
|
|
12
|
+
"""Store latitude and longiture and calculate haversine distance between them."""
|
|
13
|
+
|
|
14
|
+
lat: float
|
|
15
|
+
lng: float
|
|
16
|
+
|
|
17
|
+
@lru_cache
|
|
18
|
+
def distance_from(self, other: Location) -> float:
|
|
19
|
+
"""Return km between two Location objects.
|
|
20
|
+
|
|
21
|
+
Args:
|
|
22
|
+
other (Location): where to measure distance to
|
|
23
|
+
|
|
24
|
+
Returns:
|
|
25
|
+
float: distance between points in km
|
|
26
|
+
"""
|
|
27
|
+
return haversine(astuple(self), astuple(other))
|