publicdata-au 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,436 @@
1
+ """Query and download Australian government open data from publicdata.au.
2
+
3
+ Every function works for every dataset the site serves, named by its slug, so a dataset added to
4
+ the site needs no new release of this package.
5
+
6
+ >>> import publicdata_au as pd_au
7
+ >>> pd_au.datasets("road crashes")
8
+ >>> pd_au.rows("au-road-deaths", {"state": "QLD", "year": pd_au.gte(2020)}, limit=5)
9
+ >>> pd_au.aggregate("au-road-deaths", group="state")
10
+ >>> pd_au.read("au-road-deaths") # a pandas DataFrame, with the [pandas] extra
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import os
17
+ import shutil
18
+ import tempfile
19
+ import time
20
+ import urllib.error
21
+ import urllib.parse
22
+ import urllib.request
23
+ from collections.abc import Mapping
24
+ from pathlib import Path
25
+ from typing import Any
26
+
27
+ __version__ = "0.1.0"
28
+ __all__ = [
29
+ "Client",
30
+ "Filter",
31
+ "PublicDataError",
32
+ "Rows",
33
+ "aggregate",
34
+ "dataset",
35
+ "datasets",
36
+ "download",
37
+ "eq",
38
+ "gt",
39
+ "gte",
40
+ "ilike",
41
+ "in_",
42
+ "is_null",
43
+ "like",
44
+ "lt",
45
+ "lte",
46
+ "neq",
47
+ "not_",
48
+ "read",
49
+ "rows",
50
+ "versions",
51
+ ]
52
+
53
+ SITE = "https://publicdata.au"
54
+ FORMATS = (
55
+ "parquet",
56
+ "csv",
57
+ "csv.gz",
58
+ "json",
59
+ "ndjson",
60
+ "sqlite",
61
+ "xlsx",
62
+ "arrow",
63
+ "geojson",
64
+ "gpkg",
65
+ )
66
+ PAGE_MAX = 10_000
67
+
68
+
69
+ class PublicDataError(Exception):
70
+ """An answer from publicdata.au that was not a success."""
71
+
72
+ def __init__(self, status: int, message: str, body: Any = None, url: str = ""):
73
+ super().__init__(f"{status}: {message}" + (f" ({url})" if url else ""))
74
+ self.status = status
75
+ self.body = body
76
+ self.url = url
77
+
78
+
79
+ class Filter:
80
+ """One condition on a field, in the query API's `operator.value` form."""
81
+
82
+ def __init__(self, expr: str):
83
+ self.expr = expr
84
+
85
+ def __str__(self) -> str:
86
+ return self.expr
87
+
88
+ def __repr__(self) -> str:
89
+ return f"Filter({self.expr!r})"
90
+
91
+ def __eq__(self, other) -> bool:
92
+ return isinstance(other, Filter) and other.expr == self.expr
93
+
94
+ def __hash__(self) -> int:
95
+ return hash(self.expr)
96
+
97
+
98
+ def _v(value) -> str:
99
+ if isinstance(value, bool):
100
+ return "true" if value else "false"
101
+ return str(value)
102
+
103
+
104
+ def eq(value) -> Filter:
105
+ return Filter(f"eq.{_v(value)}")
106
+
107
+
108
+ def neq(value) -> Filter:
109
+ return Filter(f"neq.{_v(value)}")
110
+
111
+
112
+ def gt(value) -> Filter:
113
+ return Filter(f"gt.{_v(value)}")
114
+
115
+
116
+ def gte(value) -> Filter:
117
+ return Filter(f"gte.{_v(value)}")
118
+
119
+
120
+ def lt(value) -> Filter:
121
+ return Filter(f"lt.{_v(value)}")
122
+
123
+
124
+ def lte(value) -> Filter:
125
+ return Filter(f"lte.{_v(value)}")
126
+
127
+
128
+ def like(pattern: str) -> Filter:
129
+ """`*` stands for any run of characters. Case-sensitive."""
130
+ return Filter(f"like.{pattern}")
131
+
132
+
133
+ def ilike(pattern: str) -> Filter:
134
+ """As `like`, ignoring case."""
135
+ return Filter(f"ilike.{pattern}")
136
+
137
+
138
+ def in_(*values) -> Filter:
139
+ if len(values) == 1 and isinstance(values[0], (list, tuple, set, frozenset)):
140
+ values = tuple(values[0])
141
+ if any("," in _v(v) for v in values):
142
+ raise ValueError("a value in in_() cannot contain a comma")
143
+ return Filter(f"in.({','.join(_v(v) for v in values)})")
144
+
145
+
146
+ def is_null() -> Filter:
147
+ """Blank in the source, or suppressed by the publisher."""
148
+ return Filter("is.null")
149
+
150
+
151
+ def not_(f: Filter) -> Filter:
152
+ return Filter(f"not.{f.expr}")
153
+
154
+
155
+ def _filter(value) -> str:
156
+ if isinstance(value, Filter):
157
+ return value.expr
158
+ if value is None:
159
+ return "is.null"
160
+ if isinstance(value, (list, tuple, set, frozenset)):
161
+ return in_(tuple(value)).expr
162
+ return eq(value).expr
163
+
164
+
165
+ class Rows(list):
166
+ """A list of row dicts that also carries where they came from.
167
+
168
+ `version`, `attribution`, `cite` and `licence` come from the answer itself, so they always
169
+ name the version the rows were read from."""
170
+
171
+ def __init__(self, rows=(), meta: Mapping | None = None, page: Mapping | None = None):
172
+ super().__init__(rows)
173
+ self.meta = dict(meta or {})
174
+ self.page = dict(page or {})
175
+
176
+ @property
177
+ def version(self) -> str | None:
178
+ return self.meta.get("version")
179
+
180
+ @property
181
+ def attribution(self) -> str | None:
182
+ return self.meta.get("attribution")
183
+
184
+ @property
185
+ def cite(self) -> str | None:
186
+ return self.meta.get("cite")
187
+
188
+ @property
189
+ def licence(self) -> dict | None:
190
+ return self.meta.get("licence")
191
+
192
+ @property
193
+ def version_page(self) -> str | None:
194
+ return self.page.get("version_page")
195
+
196
+ def to_pandas(self):
197
+ import pandas as pd
198
+
199
+ df = pd.DataFrame(list(self))
200
+ df.attrs["publicdata"] = self.meta
201
+ return df
202
+
203
+
204
+ class Client:
205
+ """Talks to one publicdata.au site. The module-level functions use a shared default."""
206
+
207
+ def __init__(
208
+ self,
209
+ site: str | None = None,
210
+ *,
211
+ timeout: float = 60,
212
+ retries: int = 3,
213
+ user_agent: str | None = None,
214
+ ):
215
+ self.site = (site or os.environ.get("PUBLICDATA_SITE") or SITE).rstrip("/")
216
+ self.timeout = timeout
217
+ self.retries = retries
218
+ self.user_agent = user_agent or f"publicdata-au-python/{__version__}"
219
+
220
+ def _open(self, url: str):
221
+ req = urllib.request.Request(url, headers={"User-Agent": self.user_agent})
222
+ attempt = 0
223
+ while True:
224
+ try:
225
+ return urllib.request.urlopen(req, timeout=self.timeout)
226
+ except urllib.error.HTTPError as err:
227
+ # The API allows a burst per address, then answers 429 with how long to wait.
228
+ if err.code == 429 and attempt < self.retries:
229
+ attempt += 1
230
+ wait = err.headers.get("Retry-After") or "10"
231
+ time.sleep(min(float(wait) if wait.isdigit() else 10.0, 60.0))
232
+ continue
233
+ raw = err.read()
234
+ try:
235
+ body = json.loads(raw)
236
+ except ValueError:
237
+ body = raw.decode("utf-8", "replace")
238
+ msg = body.get("error") if isinstance(body, dict) else str(body)[:200]
239
+ raise PublicDataError(err.code, msg or err.reason, body, url) from None
240
+ except (urllib.error.URLError, TimeoutError) as err:
241
+ reason = getattr(err, "reason", err)
242
+ raise PublicDataError(0, f"could not reach the site: {reason}", None, url) from err
243
+
244
+ def _json(self, path_or_url: str, params: Mapping | None = None):
245
+ url = path_or_url if "://" in path_or_url else self.site + path_or_url
246
+ if params:
247
+ q = urllib.parse.urlencode(
248
+ {k: v for k, v in params.items() if v is not None},
249
+ safe="*(),.:",
250
+ quote_via=urllib.parse.quote,
251
+ )
252
+ if q:
253
+ url += ("&" if "?" in url else "?") + q
254
+ with self._open(url) as r:
255
+ return json.loads(r.read())
256
+
257
+ def datasets(self, q: str | None = None) -> list[dict]:
258
+ """Datasets the site serves, each with slug, title, publisher, licence and page URL.
259
+ `q` searches titles, summaries, publishers, keywords and field names."""
260
+ return self._json("/api/v1/datasets", {"q": q} if q else None)["results"]
261
+
262
+ def dataset(self, slug: str) -> dict:
263
+ """The dataset's Frictionless data package: title, licence, attribution, fields and
264
+ every file of the newest version."""
265
+ return self._json(f"/d/{_slug(slug)}/datapackage.json")
266
+
267
+ def versions(self, slug: str) -> list[dict]:
268
+ """Every version kept, newest first, each with its date, rows, fields and source hash."""
269
+ return self._json(f"/d/{_slug(slug)}/versions.json")["versions"]
270
+
271
+ def _query(self, slug, kind, where, version, params) -> dict:
272
+ base = f"/api/v1/datasets/{_slug(slug)}/"
273
+ if version:
274
+ base += f"versions/{_date(version)}/"
275
+ params = dict(params)
276
+ for field, value in (where or {}).items():
277
+ params[field] = _filter(value)
278
+ return self._json(base + kind, params)
279
+
280
+ def rows(
281
+ self,
282
+ slug: str,
283
+ where: Mapping[str, Any] | None = None,
284
+ *,
285
+ select: str | list[str] | None = None,
286
+ order: str | list[str] | None = None,
287
+ limit: int | None = None,
288
+ offset: int | None = None,
289
+ version: str | None = None,
290
+ all: bool = False,
291
+ ) -> Rows:
292
+ """Rows of a dataset from the query API.
293
+
294
+ `where` maps a field to a value, which must match exactly, or to a filter such as
295
+ `gte(2020)`, `in_("QLD", "NSW")` or `is_null()`. A list means any of its values and None
296
+ means blank. Every condition must match.
297
+
298
+ Without `version` the answer comes from the newest version and changes when the
299
+ publisher releases again; with a date from `versions()` it never changes. `all=True`
300
+ follows every page. For a whole table, `read()` or `download()` is faster and has no
301
+ rate limit."""
302
+ if all and limit is None:
303
+ limit = PAGE_MAX
304
+ params = {
305
+ "select": _list(select),
306
+ "order": _list(order),
307
+ "limit": limit,
308
+ "offset": offset,
309
+ }
310
+ body = self._query(slug, "rows", where, version, params)
311
+ out = Rows(body.get("rows", ()), body.get("publicdata"), _page(body))
312
+ nxt = body.get("next")
313
+ while all and nxt:
314
+ body = self._json(nxt)
315
+ out.extend(body.get("rows", ()))
316
+ nxt = body.get("next")
317
+ out.page["next"] = nxt
318
+ return out
319
+
320
+ def aggregate(
321
+ self,
322
+ slug: str,
323
+ group: str | list[str] | None = None,
324
+ metric: str | list[str] = "count",
325
+ where: Mapping[str, Any] | None = None,
326
+ *,
327
+ version: str | None = None,
328
+ ) -> Rows:
329
+ """Counts, sums, averages, minimums and maximums by group, from the query API.
330
+
331
+ `metric` is `count`, `sum.<field>`, `avg.<field>`, `min.<field>` or `max.<field>`, or a
332
+ list of them. `where` works as it does for `rows()`."""
333
+ params = {"group": _list(group), "metric": _list(metric)}
334
+ body = self._query(slug, "aggregate", where, version, params)
335
+ return Rows(body.get("rows", ()), body.get("publicdata"), _page(body))
336
+
337
+ def file_url(self, slug: str, format: str = "parquet", version: str | None = None) -> str:
338
+ if format not in FORMATS:
339
+ raise ValueError(f"format must be one of {', '.join(FORMATS)}")
340
+ at = f"v/{_date(version)}" if version else "latest"
341
+ return f"{self.site}/d/{_slug(slug)}/{at}/data.{format}"
342
+
343
+ def _save(self, slug, format, version, path) -> tuple[Path, str]:
344
+ with self._open(self.file_url(slug, format, version)) as r:
345
+ final = r.geturl()
346
+ got = final.split("/v/", 1)[1].split("/", 1)[0] if "/v/" in final else "latest"
347
+ dest = Path(path) if path else Path(f"{slug}-{got}.{format}")
348
+ with open(dest, "wb") as f:
349
+ shutil.copyfileobj(r, f, 1 << 20)
350
+ return dest, got
351
+
352
+ def download(
353
+ self,
354
+ slug: str,
355
+ format: str = "parquet",
356
+ version: str | None = None,
357
+ path: str | os.PathLike | None = None,
358
+ ) -> Path:
359
+ """Saves one version's file, the newest by default, and returns where. Without `path`
360
+ the file is named `<slug>-<version>.<format>` in the working directory. Files have no
361
+ rate limit."""
362
+ return self._save(slug, format, version, path)[0]
363
+
364
+ def read(self, slug: str, version: str | None = None):
365
+ """The whole table as a pandas DataFrame, read from the version's Parquet file.
366
+ `df.attrs["publicdata"]` is the provenance header the file itself carries: version,
367
+ licence, attribution, citation and source."""
368
+ import pyarrow.parquet as pq
369
+
370
+ with tempfile.TemporaryDirectory() as d:
371
+ p, _ = self._save(slug, "parquet", version, Path(d) / "data.parquet")
372
+ table = pq.read_table(p)
373
+ header = (table.schema.metadata or {}).get(b"publicdata")
374
+ df = table.to_pandas()
375
+ df.attrs["publicdata"] = json.loads(header) if header else {}
376
+ return df
377
+
378
+
379
+ def _slug(slug: str) -> str:
380
+ if not slug or not all(c.isalnum() or c == "-" for c in slug):
381
+ raise ValueError(f"not a dataset slug: {slug!r}")
382
+ return slug
383
+
384
+
385
+ def _date(version: str) -> str:
386
+ v = str(version)
387
+ if len(v) != 10 or v[4] != "-" or v[7] != "-" or not (v[:4] + v[5:7] + v[8:]).isdigit():
388
+ raise ValueError(f"a version is a date such as 2026-08-07, not {version!r}")
389
+ return v
390
+
391
+
392
+ def _list(value) -> str | None:
393
+ if value is None:
394
+ return None
395
+ return value if isinstance(value, str) else ",".join(value)
396
+
397
+
398
+ def _page(body: Mapping) -> dict:
399
+ return {
400
+ k: body.get(k) for k in ("dataset_page", "version_page", "this_version", "manifest", "next")
401
+ }
402
+
403
+
404
+ _default = Client()
405
+
406
+
407
+ def datasets(q: str | None = None) -> list[dict]:
408
+ return _default.datasets(q)
409
+
410
+
411
+ def dataset(slug: str) -> dict:
412
+ return _default.dataset(slug)
413
+
414
+
415
+ def versions(slug: str) -> list[dict]:
416
+ return _default.versions(slug)
417
+
418
+
419
+ def rows(slug: str, where: Mapping[str, Any] | None = None, **kw) -> Rows:
420
+ return _default.rows(slug, where, **kw)
421
+
422
+
423
+ def aggregate(slug: str, group=None, metric="count", where=None, **kw) -> Rows:
424
+ return _default.aggregate(slug, group, metric, where, **kw)
425
+
426
+
427
+ def download(slug: str, format: str = "parquet", version: str | None = None, path=None) -> Path:
428
+ return _default.download(slug, format, version, path)
429
+
430
+
431
+ def read(slug: str, version: str | None = None):
432
+ return _default.read(slug, version)
433
+
434
+
435
+ for _f in (datasets, dataset, versions, rows, aggregate, download, read):
436
+ _f.__doc__ = getattr(Client, _f.__name__).__doc__
@@ -0,0 +1,93 @@
1
+ Metadata-Version: 2.5
2
+ Name: publicdata-au
3
+ Version: 0.1.0
4
+ Summary: Query and download Australian government open data from publicdata.au.
5
+ Project-URL: Homepage, https://publicdata.au/
6
+ Project-URL: Documentation, https://publicdata.au/agents/
7
+ Project-URL: Source, https://github.com/National-Digital/publicdata.au/tree/main/clients/python
8
+ Author: National Digital
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: australia,government data,open data,publicdata.au
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
16
+ Requires-Python: >=3.10
17
+ Provides-Extra: dev
18
+ Requires-Dist: pytest>=8; extra == 'dev'
19
+ Requires-Dist: ruff>=0.6; extra == 'dev'
20
+ Provides-Extra: pandas
21
+ Requires-Dist: pandas>=2.0; extra == 'pandas'
22
+ Requires-Dist: pyarrow>=14; extra == 'pandas'
23
+ Description-Content-Type: text/markdown
24
+
25
+ # publicdata-au
26
+
27
+ Query and download Australian government open data from [publicdata.au](https://publicdata.au/).
28
+ publicdata.au republishes datasets that governments already publish under open licences, keeps
29
+ every version at a URL that never changes, and serves each one in ten formats with a query API.
30
+
31
+ This package works for every dataset the site serves, named by its slug, so a dataset added to
32
+ the site needs no new release.
33
+
34
+ ```
35
+ pip install publicdata-au # queries and downloads, no dependencies
36
+ pip install "publicdata-au[pandas]" # adds read() into a pandas DataFrame
37
+ ```
38
+
39
+ ## Find a dataset
40
+
41
+ ```python
42
+ import publicdata_au as pd_au
43
+
44
+ pd_au.datasets("road crashes") # slug, title, publisher, licence and page of each match
45
+ pd_au.versions("au-road-deaths") # every version kept, newest first
46
+ ```
47
+
48
+ ## Query rows and totals
49
+
50
+ ```python
51
+ from publicdata_au import gte, in_
52
+
53
+ deaths = pd_au.rows(
54
+ "au-road-deaths",
55
+ {"state": in_("QLD", "NSW"), "year": gte(2020)},
56
+ select=["state", "year", "road_user"],
57
+ order="year.desc",
58
+ all=True,
59
+ )
60
+ deaths.version # the version the rows came from
61
+ deaths.attribution # the attribution the publisher's licence requires
62
+ deaths.to_pandas()
63
+
64
+ pd_au.aggregate("au-road-deaths", group="state", metric="count", where={"year": 2025})
65
+ ```
66
+
67
+ A plain value must match exactly, a list matches any of its values and None matches a blank or
68
+ suppressed cell. The filters are `eq`, `neq`, `gt`, `gte`, `lt`, `lte`, `like`, `ilike`, `in_`,
69
+ `is_null` and `not_`.
70
+
71
+ Without `version=` an answer comes from the newest version and changes when the publisher
72
+ releases again. Pass a date from `versions()` for an answer that never changes.
73
+
74
+ ## Whole tables
75
+
76
+ ```python
77
+ df = pd_au.read("au-road-deaths") # needs the [pandas] extra
78
+ df.attrs["publicdata"] # the version, licence and attribution the file itself carries
79
+ pd_au.download(
80
+ "au-road-deaths", "csv"
81
+ ) # parquet, csv, csv.gz, json, ndjson, sqlite, xlsx, arrow, geojson, gpkg
82
+ ```
83
+
84
+ Files have no rate limit. The query API allows 60 requests in 10 seconds from one address, and
85
+ this package waits and retries when it answers 429.
86
+
87
+ ## Licence and attribution
88
+
89
+ The data is under each publisher's own licence, which requires the attribution string that every
90
+ answer carries. Please also name publicdata.au and link to the version you used. publicdata.au is
91
+ an independent republication, and the publishers have not endorsed it.
92
+
93
+ The package itself is under the MIT licence.
@@ -0,0 +1,5 @@
1
+ publicdata_au/__init__.py,sha256=RkLwUoPZZu4LU9a7ueZ4QFDIRaLzmuUJjeeKooQ195o,13850
2
+ publicdata_au-0.1.0.dist-info/METADATA,sha256=964WlAFp8Di7stbZV1jNa1jmoHvd_Zb-rR9r8VpdkiY,3433
3
+ publicdata_au-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
4
+ publicdata_au-0.1.0.dist-info/licenses/LICENSE,sha256=NTchXD9LNCpXrl0BvkqTZwLU-cPvpxJAH7kfdhrRlwI,1073
5
+ publicdata_au-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 National Digital
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.