postfinder 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
postfinder/__init__.py ADDED
@@ -0,0 +1,87 @@
1
+ """PostFinder: post offices, parcel lockers and post boxes, as an API.
2
+
3
+ Free, keyless and cached at the edge. The directory behind postfinder.io.
4
+
5
+ >>> from postfinder import PostFinder
6
+ >>> pf = PostFinder()
7
+ >>> near = pf.nearby(-37.7404, 144.9633, category="post-offices",
8
+ ... country="australia")
9
+ >>> near[0].name, near[0].distance_m
10
+ ('Coburg Post Office', 420)
11
+
12
+ Data from OpenStreetMap (ODbL) and GeoNames (CC BY 4.0). Publishing what you
13
+ get back means carrying those credits: postfinder.io/en/legal/.
14
+ """
15
+
16
+ from ._client import DEFAULT_BASE_URL, MIN_QUERY, PostFinder, Transport, __version__
17
+ from ._errors import BadRequest, NotFound, PostFinderError, RateLimited
18
+ from ._models import (
19
+ CATEGORIES,
20
+ CategoryHub,
21
+ Country,
22
+ CountryDetail,
23
+ CountrySummary,
24
+ Facet,
25
+ Locality,
26
+ LocalityDetail,
27
+ LocalityLink,
28
+ LocalitySummary,
29
+ NearbyPlace,
30
+ Photo,
31
+ Place,
32
+ PlaceDetail,
33
+ PlaceLink,
34
+ PostcodeDetail,
35
+ PostcodeEntry,
36
+ PostcodeIndex,
37
+ Region,
38
+ RegionDetail,
39
+ RegionPostcodes,
40
+ RegionSummary,
41
+ Review,
42
+ Reviews,
43
+ ReviewSummary,
44
+ SearchHit,
45
+ locality_path,
46
+ place_path,
47
+ )
48
+
49
+ __all__ = [
50
+ "CATEGORIES",
51
+ "DEFAULT_BASE_URL",
52
+ "MIN_QUERY",
53
+ "BadRequest",
54
+ "CategoryHub",
55
+ "Country",
56
+ "CountryDetail",
57
+ "CountrySummary",
58
+ "Facet",
59
+ "Locality",
60
+ "LocalityDetail",
61
+ "LocalityLink",
62
+ "LocalitySummary",
63
+ "NearbyPlace",
64
+ "NotFound",
65
+ "Photo",
66
+ "Place",
67
+ "PlaceDetail",
68
+ "PlaceLink",
69
+ "PostFinder",
70
+ "PostFinderError",
71
+ "PostcodeDetail",
72
+ "PostcodeEntry",
73
+ "PostcodeIndex",
74
+ "RateLimited",
75
+ "Region",
76
+ "RegionDetail",
77
+ "RegionPostcodes",
78
+ "RegionSummary",
79
+ "Review",
80
+ "ReviewSummary",
81
+ "Reviews",
82
+ "SearchHit",
83
+ "Transport",
84
+ "__version__",
85
+ "locality_path",
86
+ "place_path",
87
+ ]
postfinder/_client.py ADDED
@@ -0,0 +1,374 @@
1
+ """The client itself."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import ipaddress
6
+ import json
7
+ import urllib.error
8
+ import urllib.parse
9
+ import urllib.request
10
+ from typing import Any, Callable, Sequence
11
+
12
+ from ._errors import PostFinderError, error_for
13
+ from ._models import (
14
+ CATEGORIES,
15
+ CategoryHub,
16
+ CountryDetail,
17
+ CountrySummary,
18
+ LocalityDetail,
19
+ NearbyPlace,
20
+ PlaceDetail,
21
+ PostcodeDetail,
22
+ PostcodeIndex,
23
+ RegionDetail,
24
+ SearchHit,
25
+ )
26
+
27
+ __version__ = "0.1.0"
28
+
29
+ DEFAULT_BASE_URL = "https://api.postfinder.io"
30
+
31
+ #: The site answers an empty list below this, so there is nothing to ask for.
32
+ MIN_QUERY = 2
33
+
34
+ #: (method, url, headers, timeout) -> (status, body bytes)
35
+ Transport = Callable[[str, str, dict, float], "tuple[int, bytes]"]
36
+
37
+ # Bounded reads. A country's postcode index is the largest answer here and runs
38
+ # to a few megabytes; past that, something on the other end of the socket is
39
+ # not this API and a client should not be talked into reading it all.
40
+ _MAX_BODY = 32 << 20
41
+
42
+
43
+ class _SameOriginRedirects(urllib.request.HTTPRedirectHandler):
44
+ """Follow a redirect only to the same host over https.
45
+
46
+ Nothing here carries a credential, but a query string does carry what
47
+ somebody typed -- which is often their own street address. A redirect to
48
+ another host, or down to plaintext, would hand that to whoever answered.
49
+ """
50
+
51
+ def redirect_request(self, req, fp, code, msg, headers, newurl):
52
+ old = urllib.parse.urlparse(req.full_url)
53
+ new = urllib.parse.urlparse(newurl)
54
+ if new.scheme != old.scheme or new.netloc != old.netloc:
55
+ raise urllib.error.HTTPError(
56
+ req.full_url, code,
57
+ f"postfinder: refused a redirect to {new.scheme}://{new.netloc}",
58
+ headers, fp,
59
+ )
60
+ return super().redirect_request(req, fp, code, msg, headers, newurl)
61
+
62
+
63
+ _opener = urllib.request.build_opener(_SameOriginRedirects)
64
+
65
+
66
+ def _urllib_transport(method: str, url: str, headers: dict, timeout: float):
67
+ req = urllib.request.Request(url, method=method, headers=headers)
68
+ try:
69
+ with _opener.open(req, timeout=timeout) as res:
70
+ return res.status, res.read(_MAX_BODY)
71
+ except urllib.error.HTTPError as err:
72
+ return err.code, err.read(_MAX_BODY)
73
+
74
+
75
+ def _check_base_url(raw: str) -> str:
76
+ """Refuse a base URL that would put the request somewhere it should not go.
77
+
78
+ https always, and plaintext http only to loopback, which is what a local
79
+ proxy and a test server need. The queries this client sends are what a
80
+ person typed into a search box, and in this product that is frequently
81
+ their own address.
82
+ """
83
+ if not raw:
84
+ raise ValueError("postfinder: no base URL")
85
+
86
+ parsed = urllib.parse.urlparse(raw)
87
+ if not parsed.scheme or not parsed.hostname:
88
+ raise ValueError(f"postfinder: base URL {raw!r} is not absolute")
89
+ # https://api.postfinder.io:pass@evil.example reads as the real host to
90
+ # anyone skimming a config file, and is a different host to urllib.
91
+ # Nothing here needs userinfo, so a URL carrying it is a mistake or a trick.
92
+ if parsed.username or parsed.password:
93
+ raise ValueError(
94
+ f"postfinder: base URL {raw!r} carries credentials, and the host it "
95
+ "would reach is not the one it reads as"
96
+ )
97
+ if parsed.scheme == "https":
98
+ return raw.rstrip("/")
99
+ if parsed.scheme == "http" and _is_loopback(parsed.hostname):
100
+ return raw.rstrip("/")
101
+ if parsed.scheme == "http":
102
+ raise ValueError(
103
+ f"postfinder: base URL {raw!r} is plaintext http to a public host, "
104
+ "which would send what people type in the clear"
105
+ )
106
+ raise ValueError(
107
+ f"postfinder: base URL {raw!r} has scheme {parsed.scheme!r}, want https"
108
+ )
109
+
110
+
111
+ def _is_loopback(host: str) -> bool:
112
+ if host == "localhost":
113
+ return True
114
+ try:
115
+ return ipaddress.ip_address(host).is_loopback
116
+ except ValueError:
117
+ return False
118
+
119
+
120
+ def _segment(value: str, what: str) -> str:
121
+ """One path segment, encoded.
122
+
123
+ ``safe=""`` so a slug carrying a slash or a dot cannot walk up the path and
124
+ reach an endpoint on another prefix. The API separates its products by path
125
+ prefix, which makes this the boundary rather than a nicety.
126
+ """
127
+ text = str(value or "").strip()
128
+ if not text:
129
+ raise ValueError(f"postfinder: no {what}")
130
+ return urllib.parse.quote(text, safe="")
131
+
132
+
133
+ class PostFinder:
134
+ """A client for the PostFinder directory API.
135
+
136
+ Post offices, post boxes, express post boxes, parcel lockers, drop off
137
+ points and collection points, plus the suburbs and postcodes they sit in.
138
+
139
+ >>> from postfinder import PostFinder
140
+ >>> pf = PostFinder()
141
+ >>> for p in pf.nearby(-37.7404, 144.9633, category="post-offices",
142
+ ... country="australia"):
143
+ ... print(p.name, p.distance_m, "m")
144
+
145
+ No key and nothing to buy: the API is free, anonymous and cached at the
146
+ edge. It does ask callers to be gentle -- roughly a thousand requests a
147
+ month from one address, and results kept rather than re-fetched. Pass
148
+ ``contact`` with a URL or an email and it travels in the User-Agent, so a
149
+ conversation can start before a rate limit does.
150
+
151
+ Attribution travels with the data: locations from OpenStreetMap (ODbL),
152
+ localities and postcodes from GeoNames (CC BY 4.0). Publishing what you get
153
+ back means carrying those credits. See postfinder.io/en/legal/.
154
+
155
+ Thread safe. Make one and keep it.
156
+ """
157
+
158
+ def __init__(
159
+ self,
160
+ *,
161
+ base_url: str = DEFAULT_BASE_URL,
162
+ contact: str = "",
163
+ timeout: float = 15.0,
164
+ transport: Transport | None = None,
165
+ ):
166
+ self._base_url = _check_base_url(base_url)
167
+ self._timeout = timeout
168
+ self._transport = transport or _urllib_transport
169
+ agent = f"postfinder-python/{__version__}"
170
+ contact = (contact or "").strip()
171
+ if contact:
172
+ # contact lands in a header. A newline there is a header the caller
173
+ # did not write, and a control character is not part of a URL or an
174
+ # email address either way.
175
+ if any(c in contact for c in "\r\n\0") or not contact.isprintable():
176
+ raise ValueError(
177
+ "postfinder: contact must be one line of printable text, "
178
+ "a URL or an email address"
179
+ )
180
+ agent = f"{agent} (+{contact})"
181
+ self._agent = agent
182
+
183
+ # -- finding something ------------------------------------------------
184
+
185
+ def search(self, term: str, limit: int | None = None) -> "list[SearchHit]":
186
+ """Suburbs and locations matching what somebody has typed.
187
+
188
+ The typeahead. Rows come back most useful first and each carries enough
189
+ to link to the page it names. Debounce by at least 150ms: firing on
190
+ every keystroke spends bandwidth for no better answer.
191
+
192
+ A term shorter than two characters returns an empty list without a
193
+ request, which is what the service would answer anyway.
194
+ """
195
+ term = (term or "").strip()
196
+ if len(term) < MIN_QUERY:
197
+ return []
198
+ body = self._get("/v1/search", {"q": term, "limit": limit})
199
+ return [SearchHit.from_json(h) for h in body.get("data") or []]
200
+
201
+ def nearby(
202
+ self,
203
+ lat: float,
204
+ lng: float,
205
+ *,
206
+ category: str,
207
+ country: str,
208
+ ) -> "list[NearbyPlace]":
209
+ """The closest locations of one category to a coordinate, nearest first.
210
+
211
+ Within 50km and at most 30 rows, which is the question "where is the
212
+ nearest one" rather than "list everything in the state".
213
+
214
+ ``category`` is one of :data:`CATEGORIES`. ``country`` is the slug the
215
+ site's URLs use: ``australia``, ``new-zealand``, ``united-kingdom``,
216
+ ``united-states``.
217
+ """
218
+ if category not in CATEGORIES:
219
+ raise ValueError(
220
+ f"postfinder: {category!r} is not a category; use one of "
221
+ + ", ".join(CATEGORIES)
222
+ )
223
+ lat, lng = self._coordinate(lat, lng)
224
+ body = self._get(
225
+ "/v1/nearby",
226
+ {"lat": lat, "lng": lng, "category": category, "country": self._slug(country, "country")},
227
+ )
228
+ return [NearbyPlace.from_json(p) for p in body.get("data") or []]
229
+
230
+ def place(self, public_id: str) -> PlaceDetail:
231
+ """One location by its public id, with what is near it.
232
+
233
+ The id is permanent: it is minted once and never derived from a source
234
+ record, so a feed that renumbers its rows does not change it. Store
235
+ this rather than a name or a path.
236
+
237
+ Raises :class:`NotFound` when nothing is filed under it.
238
+ """
239
+ body = self._get(f"/v1/places/{_segment(public_id, 'place id')}", None)
240
+ return PlaceDetail.from_json(body.get("data") or {})
241
+
242
+ # -- browsing ---------------------------------------------------------
243
+
244
+ def countries(self) -> "list[CountrySummary]":
245
+ """Every country with pages, and how much is in each."""
246
+ body = self._get("/v1/countries", None)
247
+ return [CountrySummary.from_json(c) for c in body.get("data") or []]
248
+
249
+ def country(self, country: str) -> CountryDetail:
250
+ """One country and its states or regions."""
251
+ body = self._get(f"/v1/countries/{self._slug(country, 'country')}", None)
252
+ return CountryDetail.from_json(body.get("data") or {})
253
+
254
+ def region(
255
+ self, country: str, region: str, *, limit: int | None = None, offset: int | None = None
256
+ ) -> RegionDetail:
257
+ """One state or region and its localities, a page at a time.
258
+
259
+ ``locality_total`` on the answer is every locality in the region, so a
260
+ caller can page without asking twice.
261
+ """
262
+ body = self._get(
263
+ f"/v1/countries/{self._slug(country, 'country')}"
264
+ f"/regions/{self._slug(region, 'region')}",
265
+ {"limit": limit, "offset": offset},
266
+ )
267
+ return RegionDetail.from_json(body.get("data") or {})
268
+
269
+ def locality(self, country: str, region: str, locality: str) -> LocalityDetail:
270
+ """One suburb: its locations, its neighbours, and what to filter by.
271
+
272
+ Raises :class:`NotFound` for a suburb with no locations: those have no
273
+ page, deliberately.
274
+ """
275
+ body = self._get(
276
+ f"/v1/localities/{self._slug(country, 'country')}"
277
+ f"/{self._slug(region, 'region')}"
278
+ f"/{self._slug(locality, 'locality')}",
279
+ None,
280
+ )
281
+ return LocalityDetail.from_json(body.get("data") or {})
282
+
283
+ def category(self, country: str, category: str, *, region: str = "") -> CategoryHub:
284
+ """One category across a country: counts per state, busiest suburbs.
285
+
286
+ Not a national list of every post box, which would be tens of thousands
287
+ of rows and useful to nobody.
288
+ """
289
+ body = self._get(
290
+ f"/v1/countries/{self._slug(country, 'country')}"
291
+ f"/categories/{self._slug(category, 'category')}",
292
+ {"region": region},
293
+ )
294
+ return CategoryHub.from_json(body.get("data") or {})
295
+
296
+ # -- postcodes --------------------------------------------------------
297
+
298
+ def postcodes(self, country: str) -> PostcodeIndex:
299
+ """Every postcode in a country, grouped by state.
300
+
301
+ One response, meant to be kept: it is the whole index, and
302
+ :meth:`PostcodeIndex.suburbs_in` answers from it without another call.
303
+ """
304
+ body = self._get(f"/v1/countries/{self._slug(country, 'country')}/postcodes", None)
305
+ return PostcodeIndex.from_json(body.get("data") or {})
306
+
307
+ def postcode(self, country: str, postcode: str) -> PostcodeDetail:
308
+ """The suburbs one postcode covers, busiest first.
309
+
310
+ A postcode is not a suburb: 3058 is Coburg, Coburg North and
311
+ Merlynston, and an address in any of them is written with the same four
312
+ digits.
313
+
314
+ Three to five digits. Letters are a different problem -- the UK and
315
+ Canada -- and the service does not take them, so neither does this.
316
+ """
317
+ code = str(postcode or "").strip()
318
+ if not (3 <= len(code) <= 5) or not code.isdigit():
319
+ raise ValueError(
320
+ f"postfinder: {postcode!r} is not a postcode this API takes; "
321
+ "three to five digits"
322
+ )
323
+ body = self._get(
324
+ f"/v1/countries/{self._slug(country, 'country')}/postcodes/{code}", None
325
+ )
326
+ return PostcodeDetail.from_json(body.get("data") or {})
327
+
328
+ # -- plumbing ---------------------------------------------------------
329
+
330
+ @staticmethod
331
+ def _coordinate(lat: Any, lng: Any) -> "tuple[float, float]":
332
+ try:
333
+ lat, lng = float(lat), float(lng)
334
+ except (TypeError, ValueError):
335
+ raise ValueError("postfinder: lat and lng must be numbers") from None
336
+ # NaN fails every comparison, which is why this is written as a range
337
+ # check that it cannot pass.
338
+ if not (-90 <= lat <= 90) or not (-180 <= lng <= 180):
339
+ raise ValueError(
340
+ f"postfinder: ({lat}, {lng}) is not a point on the globe; "
341
+ "lat -90..90, lng -180..180"
342
+ )
343
+ return lat, lng
344
+
345
+ @staticmethod
346
+ def _slug(value: str, what: str) -> str:
347
+ return _segment(value, what)
348
+
349
+ def _get(self, path: str, params: "dict | None") -> dict:
350
+ url = self._base_url + path
351
+ query = {k: v for k, v in (params or {}).items() if v not in (None, "")}
352
+ if query:
353
+ url += "?" + urllib.parse.urlencode(query)
354
+
355
+ status, raw = self._transport(
356
+ "GET",
357
+ url,
358
+ {
359
+ "Accept": "application/json",
360
+ "User-Agent": self._agent,
361
+ },
362
+ self._timeout,
363
+ )
364
+
365
+ try:
366
+ body = json.loads(raw or b"{}")
367
+ except ValueError:
368
+ body = {}
369
+ if not isinstance(body, dict):
370
+ body = {}
371
+
372
+ if not 200 <= status < 300:
373
+ raise error_for(status, body.get("title", ""), body.get("detail", ""))
374
+ return body
postfinder/_errors.py ADDED
@@ -0,0 +1,63 @@
1
+ """What the API said when it refused."""
2
+
3
+ from __future__ import annotations
4
+
5
+
6
+ class PostFinderError(Exception):
7
+ """The API refused, and said why.
8
+
9
+ The service answers RFC 7807 problem documents: a title and a detail
10
+ written for a person to read and act on. Collapsing that into "HTTP 400"
11
+ throws away the only part of the answer that says what to do about it.
12
+ """
13
+
14
+ def __init__(self, status: int, title: str = "", detail: str = ""):
15
+ self.status = status
16
+ self.title = title
17
+ self.detail = detail
18
+ message = title or f"HTTP {status}"
19
+ if detail:
20
+ message = f"{message}: {detail}"
21
+ super().__init__(f"{status} {message}")
22
+
23
+
24
+ class NotFound(PostFinderError):
25
+ """Nothing is filed under that id, slug or postcode.
26
+
27
+ Its own class because it is an ordinary thing to hit rather than a failure.
28
+ A suburb with no locations has no page (§12.2: never a 200 with an empty
29
+ page), a place that closed is retired, and a postcode outside the seeded
30
+ countries was never there. Code that walks a list of ids will meet this.
31
+ """
32
+
33
+
34
+ class BadRequest(PostFinderError):
35
+ """The request could not be served as written: a missing coordinate, a
36
+ category that does not exist, a query of one character."""
37
+
38
+
39
+ class RateLimited(PostFinderError):
40
+ """Too many requests from here.
41
+
42
+ The API is free and keyless and asks callers to be gentle: around a
43
+ thousand requests a month from one address, responses cached, typing
44
+ debounced. Meeting this means backing off, and, if the volume is real,
45
+ saying what you are building at postfinder.io/en/contact/ -- the docs say
46
+ they will very likely help.
47
+
48
+ ``retry_after`` is the seconds the service asked for, when it said.
49
+ """
50
+
51
+ def __init__(self, status: int, title: str = "", detail: str = "", retry_after: float | None = None):
52
+ super().__init__(status, title, detail)
53
+ self.retry_after = retry_after
54
+
55
+
56
+ def error_for(status: int, title: str = "", detail: str = "", retry_after: float | None = None) -> PostFinderError:
57
+ if status == 404:
58
+ return NotFound(status, title, detail)
59
+ if status == 429:
60
+ return RateLimited(status, title, detail, retry_after)
61
+ if 400 <= status < 500:
62
+ return BadRequest(status, title, detail)
63
+ return PostFinderError(status, title, detail)