untappd-scraper 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1 @@
1
+ """Untappd Scraper functions."""
@@ -0,0 +1,193 @@
1
+ """Untappd beers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import TYPE_CHECKING, Any
6
+
7
+ from core.mixins import SimpleRepr
8
+ from dateutil.parser import parse as parse_date
9
+
10
+ from untappd_scraper.html_session import get
11
+ from untappd_scraper.structs.web import WebActivityBeer, WebBeerDetails
12
+ from untappd_scraper.web import id_from_href, parsed_value, slug_from_href
13
+
14
+ if TYPE_CHECKING: # pragma: no cover
15
+ from collections.abc import Iterator
16
+
17
+ from requests_html import Element, HTMLResponse
18
+
19
+
20
+ class Beer(SimpleRepr):
21
+ """Untappd beer."""
22
+
23
+ def __init__(self, beer_id: int) -> None:
24
+ """Initiate a Beer object, storing the beer ID and loading details.
25
+
26
+ Raises:
27
+ ValueError: invalid beer ID
28
+
29
+ Args:
30
+ beer_id (int): beer ID
31
+ """
32
+ self.beer_id = beer_id
33
+
34
+ self._page = get(url_of(beer_id))
35
+ if not self._page.ok:
36
+ msg = f"Invalid beer ID {beer_id} ({self._page})"
37
+ raise ValueError(msg)
38
+ self._beer_details: WebBeerDetails = beer_details(resp=self._page)
39
+
40
+ def __getattr__(self, name: str) -> Any:
41
+ """Return unknown attributes from beer details.
42
+
43
+ Args:
44
+ name (str): attribute to lookup
45
+
46
+ Returns:
47
+ Any: attribute value
48
+ """
49
+ return getattr(self._beer_details, name)
50
+
51
+
52
+ # ----- utils -----
53
+
54
+
55
+ def url_of(beer_id: int) -> str:
56
+ """Return the URL for a beer's main page.
57
+
58
+ Args:
59
+ beer_id (int): beer ID
60
+
61
+ Returns:
62
+ str: url to load to get beer's main page
63
+ """
64
+ return f"https://untappd.com/beer/{beer_id}"
65
+
66
+
67
+ # ----- beer details processing -----
68
+
69
+
70
+ def beer_details(resp: HTMLResponse) -> WebBeerDetails:
71
+ """Parse a user's main page into user details.
72
+
73
+ Args:
74
+ resp (HTMLResponse): beer's main page loaded
75
+
76
+ Returns:
77
+ WebBeerDetails: general beer details
78
+ """
79
+ content_el = resp.html.find(".main .content", first=True)
80
+ description = "".join(
81
+ content_el.find(".desc .beer-descrption-read-less", first=True).xpath("//div/text()")
82
+ ).strip()
83
+
84
+ return WebBeerDetails(
85
+ beer_id=id_from_href(content_el.find("a.check", first=True)),
86
+ name=content_el.find(".name h1", first=True).text,
87
+ description=description.strip(),
88
+ brewery=content_el.find(".name .brewery a", first=True).text,
89
+ brewery_slug=slug_from_href(content_el.find(".name .brewery a", first=True)),
90
+ style=content_el.find(".name p.style", first=True).text,
91
+ url=resp.url,
92
+ global_rating=content_el.find(".details [data-rating]", first=True).attrs[
93
+ "data-rating"
94
+ ],
95
+ num_ratings=parsed_value(
96
+ "{:d} Rat", content_el.find(".details p.raters", first=True).text
97
+ ),
98
+ )
99
+
100
+
101
+ def checkin_activity(resp: HTMLResponse) -> Iterator[WebActivityBeer]:
102
+ """Parse all available recent checkins for a user or in a venue.
103
+
104
+ Args:
105
+ resp (HTMLResponse): user's main page or venue's activity page
106
+
107
+ Returns:
108
+ Iterator[WebActivityBeer]: user's visible recent checkins
109
+ """
110
+ return (checkin_details(checkin) for checkin in resp.html.find(".activity .item"))
111
+
112
+
113
+ def checkin_details(checkin_item: Element) -> WebActivityBeer:
114
+ """Extract beer details from a checkin.
115
+
116
+ Args:
117
+ checkin_item (Element): single checkin
118
+
119
+ Returns:
120
+ WebActivityBeer: Interesting details for a beer
121
+ """
122
+ user_el, beer_el, brewery_el, location_el = extract_checkin_elements(checkin_item)
123
+
124
+ checkin_time_el = checkin_item.find(".bottom .time", first=True)
125
+ checkin_time = checkin_time_el.attrs.get("data-gregtime", checkin_time_el.text)
126
+ checkin_time = parse_date(checkin_time)
127
+ assert checkin_time.tzinfo and checkin_time.tzinfo.utcoffset(checkin_time) is not None, (
128
+ f"Naive datetime from {checkin_time_el.html=}"
129
+ )
130
+
131
+ purchased_at = checkin_item.find(".purchased a", first=True)
132
+ try:
133
+ comment = checkin_item.find("p.comment-text", first=True).text
134
+ except AttributeError:
135
+ comment = None
136
+ serving = checkin_item.find(".serving", first=True)
137
+
138
+ data_rating_element = checkin_item.find("[data-rating]", first=True)
139
+ if data_rating_element:
140
+ data_rating: float | None = float(data_rating_element.attrs["data-rating"])
141
+ else:
142
+ data_rating = None # pragma: no cover
143
+
144
+ try:
145
+ friends: list[str] | None = [
146
+ slug_from_href(href) for href in checkin_item.find(".tagged-friends a")
147
+ ]
148
+ except AttributeError: # pragma: no cover
149
+ friends = None
150
+
151
+ return WebActivityBeer(
152
+ checkin_id=int(checkin_item.attrs["data-checkin-id"]),
153
+ checkin=checkin_time,
154
+ user_name=slug_from_href(user_el),
155
+ name=beer_el.text,
156
+ beer_id=id_from_href(beer_el),
157
+ brewery=brewery_el.text,
158
+ brewery_slug=slug_from_href(brewery_el),
159
+ location=location_el.text if location_el else None,
160
+ location_id=id_from_href(location_el) if location_el else None,
161
+ purchased_at=purchased_at.text if purchased_at else None,
162
+ purchased_id=id_from_href(purchased_at) if purchased_at else None,
163
+ comment=comment,
164
+ serving=serving.text if serving else None,
165
+ user_rating=data_rating,
166
+ friends=friends,
167
+ )
168
+
169
+
170
+ def extract_checkin_elements(
171
+ element: Element,
172
+ ) -> tuple[Element, Element, Element, Element | None]:
173
+ """Extract four linked elements in a checkin.
174
+
175
+ Args:
176
+ element (Element): checkin element
177
+
178
+ Raises:
179
+ ValueError: element passed didn't contain 3-4 <a> tags
180
+
181
+ Returns:
182
+ tuple[Element, Element, Element, Element]: user, beer, brewery, location
183
+ """
184
+ elements = element.find(".top .text a")
185
+
186
+ if len(elements) == 3:
187
+ return elements + [None] # pragma: no cover
188
+ if len(elements) == 4:
189
+ return elements
190
+
191
+ raise ValueError(
192
+ f"Wanted 3 or 4 <a> elements (not {len(elements)}) in {element.html}"
193
+ ) # pragma: no cover
@@ -0,0 +1,231 @@
1
+ """HTML session to be shared across all modules."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from datetime import timedelta
7
+ from typing import TYPE_CHECKING, Final
8
+
9
+ import ratelim
10
+ import requests
11
+ from requests_cache import CacheMixin
12
+ from requests_html import HTMLResponse, HTMLSession
13
+ from tenacity import (
14
+ RetryCallState,
15
+ _utils,
16
+ retry,
17
+ retry_base,
18
+ stop_after_attempt,
19
+ wait_exponential,
20
+ )
21
+ from tenacity.after import after_log
22
+ from tenacity.wait import wait_base
23
+
24
+ if TYPE_CHECKING: # pragma: no cover
25
+ from collections.abc import Callable
26
+
27
+ logger = logging.getLogger(__name__)
28
+
29
+ CACHE_EXPIRY: Final[timedelta] = timedelta(hours=1)
30
+ MAX_GET_MIN1: Final[int] = 40 # max number of web GETs in a minute
31
+ MAX_RETRY_SECS: Final[int] = 180 # never wait more than this for a retry
32
+ MAX_RETRY_ATTEMPTS: Final[int] = 9 # give up after this many retries
33
+ MIN1: Final[int] = 60 # seconds
34
+
35
+ # Allow user to check these. Don't raise an exception here
36
+ ACCEPTABLE_HTTP_STATUS: Final[frozenset[int]] = frozenset((requests.codes["not_found"],))
37
+
38
+
39
+ class CachedHTMLSession(CacheMixin, HTMLSession): # pyright: ignore[reportIncompatibleMethodOverride]
40
+ """Session with features from both CachedSession and HTMLSession."""
41
+
42
+
43
+ _html_session = CachedHTMLSession(cache_name="html", expire_after=CACHE_EXPIRY)
44
+
45
+
46
+ @ratelim.greedy(MAX_GET_MIN1, MIN1)
47
+ def get(url: str, *, emulate_404: bool = False, **kwargs: str) -> requests.Response:
48
+ """Get a URL.
49
+
50
+ Handles too many requests errors, and retries after waiting
51
+
52
+ Args:
53
+ url (str): URL to get
54
+ emulate_404 (bool): if True, return a 404 response
55
+ kwargs (dict): extra requests options, eg, params and headers
56
+
57
+ Returns:
58
+ requests.Response: response to get
59
+ """
60
+ if emulate_404:
61
+ url = "https://httpbin.org/status/404"
62
+ resp = _get(url, **kwargs)
63
+ logger.debug(
64
+ "GET %s (%s) received %s\tExpires: %s, Headers: %s",
65
+ url,
66
+ kwargs,
67
+ resp,
68
+ resp.expires, # pyright: ignore[reportAttributeAccessIssue]
69
+ resp.headers,
70
+ )
71
+ return resp
72
+
73
+
74
+ # ----- Tenacity -----
75
+
76
+
77
+ class RetryAfter(wait_base):
78
+ """Strategy that tries to wait as per Retry-After header.
79
+
80
+ Tries to wait for the length specified by the Retry-After header,
81
+ or the underlying wait / fallback strategy if not.
82
+ See RFC 6585 § 4.
83
+ """
84
+
85
+ def __init__(self, fallback: wait_base) -> None:
86
+ """Store fallback strategy in case we can't work out retry wait time.
87
+
88
+ Args:
89
+ fallback (wait_base): fallback wait strategy if no Retry-After found
90
+ """
91
+ self.fallback = fallback
92
+
93
+ def __call__(self, retry_state: RetryCallState) -> int: # pragma: no cover
94
+ """Return seconds to wait until retry.
95
+
96
+ Args:
97
+ retry_state (RetryState): State of retry, with .outcome property
98
+ storing exception.
99
+
100
+ Returns:
101
+ int: seconds to wait
102
+ """
103
+ assert retry_state.outcome
104
+ exc = retry_state.outcome.exception()
105
+ if isinstance(exc, requests.HTTPError):
106
+ retry_after = exc.response.headers.get("Retry-After")
107
+ logger.debug("Searching response header and found Retry-After of %s", retry_after)
108
+
109
+ try:
110
+ return int(retry_after) # pyright: ignore[reportArgumentType]
111
+ except (TypeError, ValueError):
112
+ return int(self.fallback(retry_state))
113
+
114
+ return int(self.fallback(retry_state))
115
+
116
+
117
+ def my_before_sleep_log(
118
+ user_logger: logging.Logger, *, exc_info: bool = False
119
+ ) -> Callable[[RetryCallState], None]:
120
+ """Before call strategy that logs to some logger the attempt.
121
+
122
+ Logging level is determined by the number of retries.
123
+
124
+ Lifted from Tenacity function and removed hard-coded log level
125
+
126
+ Args:
127
+ user_logger (Logger): logger to use
128
+ exc_info (bool, optional): Is there an exception. Defaults to False.
129
+
130
+ Returns:
131
+ logging function
132
+ """
133
+
134
+ def log_it(retry_state: RetryCallState) -> None: # pragma: no cover
135
+ if retry_state.attempt_number < 1:
136
+ log_level = logging.DEBUG
137
+ elif retry_state.attempt_number == 1:
138
+ log_level = logging.INFO
139
+ else:
140
+ log_level = logging.WARNING
141
+
142
+ assert retry_state.outcome
143
+
144
+ if retry_state.outcome.failed:
145
+ ex = retry_state.outcome.exception()
146
+ verb, retry_value = "raised", f"{type(ex).__name__}: {ex}"
147
+
148
+ if exc_info and retry_state.outcome:
149
+ local_exc_info = retry_state.outcome.exception()
150
+ else:
151
+ local_exc_info = False
152
+ else:
153
+ verb, retry_value = "returned", retry_state.outcome.result()
154
+ local_exc_info = False # exc_info does not apply when no exception
155
+
156
+ user_logger.log(
157
+ log_level,
158
+ "Retrying %s (%s attempt) in %s seconds as it %s %s.",
159
+ _utils.get_callback_name(retry_state.fn), # pyright: ignore[reportArgumentType]
160
+ _utils.to_ordinal(retry_state.attempt_number),
161
+ retry_state.next_action.sleep, # pyright: ignore[reportOptionalMemberAccess]
162
+ verb,
163
+ retry_value,
164
+ exc_info=local_exc_info,
165
+ )
166
+
167
+ return log_it
168
+
169
+
170
+ def is_throttling_related_exception(excp: Exception) -> bool: # pragma: no cover
171
+ """Check is the exception is a requests one and throttling related.
172
+
173
+ Args:
174
+ excp (Exception): exception raised
175
+
176
+ Returns:
177
+ bool: was it a throttle
178
+ """
179
+ return (
180
+ isinstance(excp, requests.HTTPError)
181
+ and excp.response.status_code == requests.codes.too_many_requests
182
+ )
183
+
184
+
185
+ class RetryIfThrottling(retry_base):
186
+ """Retry class which only retries if a throttling exception occured.
187
+
188
+ From https://www.seelk.co/blog/efficient-client-side-handling-of-api-throttling-in-python-with-tenacity/
189
+ """
190
+
191
+ def __call__(self, retry_state: RetryCallState) -> bool: # pragma: no cover
192
+ """Return if the call raised an exception and it's a throttle.
193
+
194
+ Args:
195
+ retry_state (RetryCallState): info about current retry invocation
196
+
197
+ Returns:
198
+ bool: is it throttling related
199
+ """
200
+ if (
201
+ retry_state.outcome
202
+ and retry_state.outcome.failed
203
+ and (exception := retry_state.outcome.exception())
204
+ ):
205
+ return is_throttling_related_exception(exception) # pyright: ignore[reportArgumentType]
206
+ return False
207
+
208
+
209
+ # ---- Main part of tenacity retry ----
210
+
211
+ # Retry too-many-requests after a delay
212
+
213
+ # - Retries if throttle response received
214
+ # - Checks for Retry-After header and wait that long
215
+ # - If no header, waits expontially longer each retry
216
+ # - logs retries, with increasing severity as attempts increase
217
+
218
+
219
+ @retry(
220
+ reraise=True,
221
+ retry=RetryIfThrottling(),
222
+ wait=RetryAfter(fallback=wait_exponential(min=5, max=MAX_RETRY_SECS)),
223
+ stop=stop_after_attempt(MAX_RETRY_ATTEMPTS),
224
+ before_sleep=my_before_sleep_log(logger),
225
+ after=after_log(logger, logging.DEBUG),
226
+ )
227
+ def _get(url: str, **kwargs: str) -> HTMLResponse:
228
+ resp = _html_session.get(url, **kwargs)
229
+ if resp.status_code not in ACCEPTABLE_HTTP_STATUS:
230
+ resp.raise_for_status() # pragma: no cover
231
+ return resp # pyright: ignore[reportReturnType]
@@ -0,0 +1,78 @@
1
+ """Set up logging using the Loguru library."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import inspect
6
+ import logging
7
+ import sys
8
+ from pathlib import Path
9
+ from typing import Final
10
+
11
+ from loguru import logger
12
+
13
+ # ----- Constants -----
14
+
15
+ # Log formats
16
+ STANDARD: Final = "[{time:HH:mm:ss}] {level} - {message}"
17
+ DETAIL: Final = "{time} {file:>25}:{line:<4} {level:<8} {message}"
18
+ ROTATION: Final = "1 hour"
19
+ RETENTION: Final = "2 days"
20
+
21
+
22
+ def configure_logging(
23
+ log_filename: str | Path,
24
+ *,
25
+ standard_format: str = STANDARD,
26
+ detail_format: str = DETAIL,
27
+ log_rotation: str = ROTATION,
28
+ log_retention: str = RETENTION,
29
+ ) -> None:
30
+ """Setup logging for the application."""
31
+ # Capture things like Hishel logging
32
+ intercept_logging()
33
+ # Replace the default StdErr handler.
34
+ logger.remove()
35
+ logger.add(sys.stderr, level="WARNING", format=standard_format)
36
+
37
+ log_filename = Path(log_filename)
38
+ if log_filename.suffix != ".log":
39
+ log_filename = log_filename.with_suffix(".log")
40
+
41
+ # Add a rotating file handler.
42
+ logger.add(
43
+ log_filename,
44
+ level="DEBUG",
45
+ format=detail_format,
46
+ rotation=log_rotation,
47
+ retention=log_retention,
48
+ )
49
+
50
+
51
+ # ----- Interface to the standard logging module -----
52
+
53
+
54
+ class InterceptHandler(logging.Handler):
55
+ """Send logs to Loguru."""
56
+
57
+ def emit(self, record: logging.LogRecord) -> None:
58
+ """Emit a log record."""
59
+ # Get corresponding Loguru level if it exists.
60
+ level: str | int
61
+ try:
62
+ level = logger.level(record.levelname).name
63
+ except ValueError:
64
+ level = record.levelno
65
+
66
+ # Find caller from where originated the logged message.
67
+ frame, depth = inspect.currentframe(), 0
68
+ while frame and (depth == 0 or frame.f_code.co_filename == logging.__file__):
69
+ frame = frame.f_back
70
+ depth += 1
71
+
72
+ logger.opt(depth=depth, exception=record.exc_info).log(level, record.getMessage())
73
+
74
+
75
+ def intercept_logging() -> None:
76
+ """Intercept standard logging and send it to Loguru."""
77
+ # Configure the root logger
78
+ logging.basicConfig(handlers=[InterceptHandler()], level=0, force=True)
@@ -0,0 +1,59 @@
1
+ # Inspired from https://towardsdatascience.com/
2
+ # ultimate-guide-to-python-debugging-854dea731e1b
3
+
4
+ version: 1
5
+ disable_existing_loggers: false
6
+
7
+ formatters:
8
+ standard:
9
+ format: "[%(asctime)s] %(levelname)s - %(message)s"
10
+ datefmt: "%H:%M:%S"
11
+ detail:
12
+ format: "%(asctime)s %(filename)25s:%(lineno)-4d %(levelname)-8s %(message)s"
13
+
14
+ handlers:
15
+ console: # handler which will log into stdout
16
+ class: logging.StreamHandler
17
+ level: WARNING
18
+ formatter: standard # Use formatter defined above
19
+ stream: ext://sys.stdout
20
+ filesize: # handler which will log to file up to a size
21
+ class: logging.handlers.RotatingFileHandler
22
+ level: DEBUG
23
+ formatter: detail # Use formatter defined above
24
+ filename: ext://os.devnull
25
+ maxBytes: 10485760 # 10 Mb
26
+ backupCount: 10
27
+ encoding: utf8
28
+ filetime: # handler which will switch files regularly
29
+ class: logging.handlers.TimedRotatingFileHandler
30
+ level: INFO
31
+ formatter: detail # Use formatter defined above
32
+ filename: ext://os.devnull
33
+ when: W0
34
+ # interval: 0
35
+ # maxBytes: 10485760
36
+ # backupCount:
37
+ encoding: utf8
38
+ email: # email handler for error messages
39
+ class: logging.handlers.SMTPHandler
40
+ level: ERROR
41
+ formatter: standard # Use formatter defined above
42
+ mailhost: tbd # ext://util.config.log.mailhost
43
+ # credentials: ext://util.config.log.credentials
44
+ # secure: ext://util.config.log.secure
45
+ fromaddr: tbd # ext://util.config.log.fromaddr
46
+ toaddrs: tbd # ext://util.config.log.toaddrs
47
+ # - support_team@domain.tld
48
+ # - dev_team@domain.tld
49
+ subject: tbd # ext://util.config.log.subject # Houston, we have a problem.
50
+
51
+ root: # Loggers are organized in hierarchy - this is the root logger config
52
+ level: DEBUG
53
+ handlers: [console] # , file, email] # Attaches both handler defined above
54
+
55
+ # loggers: # Defines descendants of root logger
56
+ # mymodule: # Logger for "mymodule"
57
+ # level: INFO
58
+ # handlers: [file] # Will only use "file" handler defined above
59
+ # propagate: no # Will not propagate logs to "root" logger
File without changes
File without changes
@@ -0,0 +1,63 @@
1
+ """Mixin classes."""
2
+ from __future__ import annotations
3
+
4
+ from typing import Protocol
5
+
6
+
7
+ class HasBeerId(Protocol):
8
+ """Check a class has a beer id."""
9
+
10
+ beer_id: int
11
+
12
+
13
+ class HasVenueId(Protocol):
14
+ """Check a class has a venue id."""
15
+
16
+ venue_id: int
17
+
18
+
19
+ class IdStrMixin:
20
+ """Provide string versions of integer IDs."""
21
+
22
+ @property
23
+ def beer_id_str(self: HasBeerId) -> str:
24
+ """Return str version of Beer ID.
25
+
26
+ JSON keys are str but dict can be int. Can get dupe JSON keys.
27
+
28
+ Returns:
29
+ str: string version of beer ID
30
+ """
31
+ return str(self.beer_id)
32
+
33
+ @property
34
+ def venue_id_str(self: HasVenueId) -> str:
35
+ """Return str version of Venue ID.
36
+
37
+ JSON keys are str but dict can be int. Can get dupe JSON keys.
38
+
39
+ Returns:
40
+ str: string version of venue ID
41
+ """
42
+ return str(self.venue_id)
43
+
44
+
45
+ class HasBeerDetails(Protocol):
46
+ """Check if a class has beer details."""
47
+
48
+ name: str
49
+ brewery: str
50
+ style: str
51
+ global_rating: float
52
+
53
+
54
+ class BeerStrMixin:
55
+ """Provide a nice description of a beer."""
56
+
57
+ def __str__(self: HasBeerDetails) -> str:
58
+ """Create a summary description of a beer.
59
+
60
+ Returns:
61
+ str: beer description
62
+ """
63
+ return f"{self.name} by {self.brewery}, {self.style} ({self.global_rating})"
@@ -0,0 +1,27 @@
1
+ """Structures used to represent data not scraped."""
2
+ from __future__ import annotations
3
+
4
+ from dataclasses import astuple, dataclass
5
+ from functools import lru_cache
6
+
7
+ from haversine import haversine
8
+
9
+
10
+ @dataclass(frozen=True)
11
+ class Location:
12
+ """Store latitude and longiture and calculate haversine distance between them."""
13
+
14
+ lat: float
15
+ lng: float
16
+
17
+ @lru_cache
18
+ def distance_from(self, other: Location) -> float:
19
+ """Return km between two Location objects.
20
+
21
+ Args:
22
+ other (Location): where to measure distance to
23
+
24
+ Returns:
25
+ float: distance between points in km
26
+ """
27
+ return haversine(astuple(self), astuple(other))