publicdata-au 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,24 @@
1
+ __pycache__/
2
+ *.egg-info/
3
+ .venv/
4
+ .pytest_cache/
5
+ .ruff_cache/
6
+ dist/
7
+ snapshots/
8
+ raw/
9
+ .wrangler/
10
+ .DS_Store
11
+ .env
12
+ .env.*
13
+ store/*/*/source.*
14
+ dist-large/
15
+ # Lighthouse and Chrome write beside the working directory during local audits.
16
+ lh-*.json
17
+ C:*
18
+ node_modules/
19
+
20
+ # Written at publish time by python -m publicdata.api_text server.json.
21
+ /server.json
22
+ .build-cache/
23
+ .venv*/
24
+ clients/*/dist/
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 National Digital
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,93 @@
1
+ Metadata-Version: 2.5
2
+ Name: publicdata-au
3
+ Version: 0.1.0
4
+ Summary: Query and download Australian government open data from publicdata.au.
5
+ Project-URL: Homepage, https://publicdata.au/
6
+ Project-URL: Documentation, https://publicdata.au/agents/
7
+ Project-URL: Source, https://github.com/National-Digital/publicdata.au/tree/main/clients/python
8
+ Author: National Digital
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: australia,government data,open data,publicdata.au
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
16
+ Requires-Python: >=3.10
17
+ Provides-Extra: dev
18
+ Requires-Dist: pytest>=8; extra == 'dev'
19
+ Requires-Dist: ruff>=0.6; extra == 'dev'
20
+ Provides-Extra: pandas
21
+ Requires-Dist: pandas>=2.0; extra == 'pandas'
22
+ Requires-Dist: pyarrow>=14; extra == 'pandas'
23
+ Description-Content-Type: text/markdown
24
+
25
+ # publicdata-au
26
+
27
+ Query and download Australian government open data from [publicdata.au](https://publicdata.au/).
28
+ publicdata.au republishes datasets that governments already publish under open licences, keeps
29
+ every version at a URL that never changes, and serves each one in ten formats with a query API.
30
+
31
+ This package works for every dataset the site serves, named by its slug, so a dataset added to
32
+ the site needs no new release.
33
+
34
+ ```
35
+ pip install publicdata-au # queries and downloads, no dependencies
36
+ pip install "publicdata-au[pandas]" # adds read() into a pandas DataFrame
37
+ ```
38
+
39
+ ## Find a dataset
40
+
41
+ ```python
42
+ import publicdata_au as pd_au
43
+
44
+ pd_au.datasets("road crashes") # slug, title, publisher, licence and page of each match
45
+ pd_au.versions("au-road-deaths") # every version kept, newest first
46
+ ```
47
+
48
+ ## Query rows and totals
49
+
50
+ ```python
51
+ from publicdata_au import gte, in_
52
+
53
+ deaths = pd_au.rows(
54
+ "au-road-deaths",
55
+ {"state": in_("QLD", "NSW"), "year": gte(2020)},
56
+ select=["state", "year", "road_user"],
57
+ order="year.desc",
58
+ all=True,
59
+ )
60
+ deaths.version # the version the rows came from
61
+ deaths.attribution # the attribution the publisher's licence requires
62
+ deaths.to_pandas()
63
+
64
+ pd_au.aggregate("au-road-deaths", group="state", metric="count", where={"year": 2025})
65
+ ```
66
+
67
+ A plain value must match exactly, a list matches any of its values and None matches a blank or
68
+ suppressed cell. The filters are `eq`, `neq`, `gt`, `gte`, `lt`, `lte`, `like`, `ilike`, `in_`,
69
+ `is_null` and `not_`.
70
+
71
+ Without `version=` an answer comes from the newest version and changes when the publisher
72
+ releases again. Pass a date from `versions()` for an answer that never changes.
73
+
74
+ ## Whole tables
75
+
76
+ ```python
77
+ df = pd_au.read("au-road-deaths") # needs the [pandas] extra
78
+ df.attrs["publicdata"] # the version, licence and attribution the file itself carries
79
+ pd_au.download(
80
+ "au-road-deaths", "csv"
81
+ ) # parquet, csv, csv.gz, json, ndjson, sqlite, xlsx, arrow, geojson, gpkg
82
+ ```
83
+
84
+ Files have no rate limit. The query API allows 60 requests in 10 seconds from one address, and
85
+ this package waits and retries when it answers 429.
86
+
87
+ ## Licence and attribution
88
+
89
+ The data is under each publisher's own licence, which requires the attribution string that every
90
+ answer carries. Please also name publicdata.au and link to the version you used. publicdata.au is
91
+ an independent republication, and the publishers have not endorsed it.
92
+
93
+ The package itself is under the MIT licence.
@@ -0,0 +1,69 @@
1
+ # publicdata-au
2
+
3
+ Query and download Australian government open data from [publicdata.au](https://publicdata.au/).
4
+ publicdata.au republishes datasets that governments already publish under open licences, keeps
5
+ every version at a URL that never changes, and serves each one in ten formats with a query API.
6
+
7
+ This package works for every dataset the site serves, named by its slug, so a dataset added to
8
+ the site needs no new release.
9
+
10
+ ```
11
+ pip install publicdata-au # queries and downloads, no dependencies
12
+ pip install "publicdata-au[pandas]" # adds read() into a pandas DataFrame
13
+ ```
14
+
15
+ ## Find a dataset
16
+
17
+ ```python
18
+ import publicdata_au as pd_au
19
+
20
+ pd_au.datasets("road crashes") # slug, title, publisher, licence and page of each match
21
+ pd_au.versions("au-road-deaths") # every version kept, newest first
22
+ ```
23
+
24
+ ## Query rows and totals
25
+
26
+ ```python
27
+ from publicdata_au import gte, in_
28
+
29
+ deaths = pd_au.rows(
30
+ "au-road-deaths",
31
+ {"state": in_("QLD", "NSW"), "year": gte(2020)},
32
+ select=["state", "year", "road_user"],
33
+ order="year.desc",
34
+ all=True,
35
+ )
36
+ deaths.version # the version the rows came from
37
+ deaths.attribution # the attribution the publisher's licence requires
38
+ deaths.to_pandas()
39
+
40
+ pd_au.aggregate("au-road-deaths", group="state", metric="count", where={"year": 2025})
41
+ ```
42
+
43
+ A plain value must match exactly, a list matches any of its values and None matches a blank or
44
+ suppressed cell. The filters are `eq`, `neq`, `gt`, `gte`, `lt`, `lte`, `like`, `ilike`, `in_`,
45
+ `is_null` and `not_`.
46
+
47
+ Without `version=` an answer comes from the newest version and changes when the publisher
48
+ releases again. Pass a date from `versions()` for an answer that never changes.
49
+
50
+ ## Whole tables
51
+
52
+ ```python
53
+ df = pd_au.read("au-road-deaths") # needs the [pandas] extra
54
+ df.attrs["publicdata"] # the version, licence and attribution the file itself carries
55
+ pd_au.download(
56
+ "au-road-deaths", "csv"
57
+ ) # parquet, csv, csv.gz, json, ndjson, sqlite, xlsx, arrow, geojson, gpkg
58
+ ```
59
+
60
+ Files have no rate limit. The query API allows 60 requests in 10 seconds from one address, and
61
+ this package waits and retries when it answers 429.
62
+
63
+ ## Licence and attribution
64
+
65
+ The data is under each publisher's own licence, which requires the attribution string that every
66
+ answer carries. Please also name publicdata.au and link to the version you used. publicdata.au is
67
+ an independent republication, and the publishers have not endorsed it.
68
+
69
+ The package itself is under the MIT licence.
@@ -0,0 +1,41 @@
1
+ [project]
2
+ name = "publicdata-au"
3
+ version = "0.1.0"
4
+ description = "Query and download Australian government open data from publicdata.au."
5
+ readme = "README.md"
6
+ requires-python = ">=3.10"
7
+ license = "MIT"
8
+ license-files = ["LICENSE"]
9
+ authors = [{ name = "National Digital" }]
10
+ keywords = ["australia", "open data", "government data", "publicdata.au"]
11
+ classifiers = [
12
+ "Development Status :: 4 - Beta",
13
+ "Intended Audience :: Science/Research",
14
+ "Programming Language :: Python :: 3",
15
+ "Topic :: Scientific/Engineering :: Information Analysis",
16
+ ]
17
+ dependencies = []
18
+
19
+ [project.optional-dependencies]
20
+ pandas = ["pandas>=2.0", "pyarrow>=14"]
21
+ dev = ["pytest>=8", "ruff>=0.6"]
22
+
23
+ [project.urls]
24
+ Homepage = "https://publicdata.au/"
25
+ Documentation = "https://publicdata.au/agents/"
26
+ Source = "https://github.com/National-Digital/publicdata.au/tree/main/clients/python"
27
+
28
+ [build-system]
29
+ requires = ["hatchling>=1.27"]
30
+ build-backend = "hatchling.build"
31
+
32
+ [tool.hatch.build.targets.wheel]
33
+ packages = ["src/publicdata_au"]
34
+
35
+ [tool.ruff]
36
+ line-length = 100
37
+ target-version = "py310"
38
+
39
+ [tool.ruff.lint]
40
+ select = ["E", "F", "I", "B", "UP", "W"]
41
+ ignore = ["E501"]
@@ -0,0 +1,436 @@
1
+ """Query and download Australian government open data from publicdata.au.
2
+
3
+ Every function works for every dataset the site serves, named by its slug, so a dataset added to
4
+ the site needs no new release of this package.
5
+
6
+ >>> import publicdata_au as pd_au
7
+ >>> pd_au.datasets("road crashes")
8
+ >>> pd_au.rows("au-road-deaths", {"state": "QLD", "year": pd_au.gte(2020)}, limit=5)
9
+ >>> pd_au.aggregate("au-road-deaths", group="state")
10
+ >>> pd_au.read("au-road-deaths") # a pandas DataFrame, with the [pandas] extra
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import os
17
+ import shutil
18
+ import tempfile
19
+ import time
20
+ import urllib.error
21
+ import urllib.parse
22
+ import urllib.request
23
+ from collections.abc import Mapping
24
+ from pathlib import Path
25
+ from typing import Any
26
+
27
+ __version__ = "0.1.0"
28
+ __all__ = [
29
+ "Client",
30
+ "Filter",
31
+ "PublicDataError",
32
+ "Rows",
33
+ "aggregate",
34
+ "dataset",
35
+ "datasets",
36
+ "download",
37
+ "eq",
38
+ "gt",
39
+ "gte",
40
+ "ilike",
41
+ "in_",
42
+ "is_null",
43
+ "like",
44
+ "lt",
45
+ "lte",
46
+ "neq",
47
+ "not_",
48
+ "read",
49
+ "rows",
50
+ "versions",
51
+ ]
52
+
53
+ SITE = "https://publicdata.au"
54
+ FORMATS = (
55
+ "parquet",
56
+ "csv",
57
+ "csv.gz",
58
+ "json",
59
+ "ndjson",
60
+ "sqlite",
61
+ "xlsx",
62
+ "arrow",
63
+ "geojson",
64
+ "gpkg",
65
+ )
66
+ PAGE_MAX = 10_000
67
+
68
+
69
+ class PublicDataError(Exception):
70
+ """An answer from publicdata.au that was not a success."""
71
+
72
+ def __init__(self, status: int, message: str, body: Any = None, url: str = ""):
73
+ super().__init__(f"{status}: {message}" + (f" ({url})" if url else ""))
74
+ self.status = status
75
+ self.body = body
76
+ self.url = url
77
+
78
+
79
+ class Filter:
80
+ """One condition on a field, in the query API's `operator.value` form."""
81
+
82
+ def __init__(self, expr: str):
83
+ self.expr = expr
84
+
85
+ def __str__(self) -> str:
86
+ return self.expr
87
+
88
+ def __repr__(self) -> str:
89
+ return f"Filter({self.expr!r})"
90
+
91
+ def __eq__(self, other) -> bool:
92
+ return isinstance(other, Filter) and other.expr == self.expr
93
+
94
+ def __hash__(self) -> int:
95
+ return hash(self.expr)
96
+
97
+
98
+ def _v(value) -> str:
99
+ if isinstance(value, bool):
100
+ return "true" if value else "false"
101
+ return str(value)
102
+
103
+
104
+ def eq(value) -> Filter:
105
+ return Filter(f"eq.{_v(value)}")
106
+
107
+
108
+ def neq(value) -> Filter:
109
+ return Filter(f"neq.{_v(value)}")
110
+
111
+
112
+ def gt(value) -> Filter:
113
+ return Filter(f"gt.{_v(value)}")
114
+
115
+
116
+ def gte(value) -> Filter:
117
+ return Filter(f"gte.{_v(value)}")
118
+
119
+
120
+ def lt(value) -> Filter:
121
+ return Filter(f"lt.{_v(value)}")
122
+
123
+
124
+ def lte(value) -> Filter:
125
+ return Filter(f"lte.{_v(value)}")
126
+
127
+
128
+ def like(pattern: str) -> Filter:
129
+ """`*` stands for any run of characters. Case-sensitive."""
130
+ return Filter(f"like.{pattern}")
131
+
132
+
133
+ def ilike(pattern: str) -> Filter:
134
+ """As `like`, ignoring case."""
135
+ return Filter(f"ilike.{pattern}")
136
+
137
+
138
+ def in_(*values) -> Filter:
139
+ if len(values) == 1 and isinstance(values[0], (list, tuple, set, frozenset)):
140
+ values = tuple(values[0])
141
+ if any("," in _v(v) for v in values):
142
+ raise ValueError("a value in in_() cannot contain a comma")
143
+ return Filter(f"in.({','.join(_v(v) for v in values)})")
144
+
145
+
146
+ def is_null() -> Filter:
147
+ """Blank in the source, or suppressed by the publisher."""
148
+ return Filter("is.null")
149
+
150
+
151
+ def not_(f: Filter) -> Filter:
152
+ return Filter(f"not.{f.expr}")
153
+
154
+
155
+ def _filter(value) -> str:
156
+ if isinstance(value, Filter):
157
+ return value.expr
158
+ if value is None:
159
+ return "is.null"
160
+ if isinstance(value, (list, tuple, set, frozenset)):
161
+ return in_(tuple(value)).expr
162
+ return eq(value).expr
163
+
164
+
165
+ class Rows(list):
166
+ """A list of row dicts that also carries where they came from.
167
+
168
+ `version`, `attribution`, `cite` and `licence` come from the answer itself, so they always
169
+ name the version the rows were read from."""
170
+
171
+ def __init__(self, rows=(), meta: Mapping | None = None, page: Mapping | None = None):
172
+ super().__init__(rows)
173
+ self.meta = dict(meta or {})
174
+ self.page = dict(page or {})
175
+
176
+ @property
177
+ def version(self) -> str | None:
178
+ return self.meta.get("version")
179
+
180
+ @property
181
+ def attribution(self) -> str | None:
182
+ return self.meta.get("attribution")
183
+
184
+ @property
185
+ def cite(self) -> str | None:
186
+ return self.meta.get("cite")
187
+
188
+ @property
189
+ def licence(self) -> dict | None:
190
+ return self.meta.get("licence")
191
+
192
+ @property
193
+ def version_page(self) -> str | None:
194
+ return self.page.get("version_page")
195
+
196
+ def to_pandas(self):
197
+ import pandas as pd
198
+
199
+ df = pd.DataFrame(list(self))
200
+ df.attrs["publicdata"] = self.meta
201
+ return df
202
+
203
+
204
+ class Client:
205
+ """Talks to one publicdata.au site. The module-level functions use a shared default."""
206
+
207
+ def __init__(
208
+ self,
209
+ site: str | None = None,
210
+ *,
211
+ timeout: float = 60,
212
+ retries: int = 3,
213
+ user_agent: str | None = None,
214
+ ):
215
+ self.site = (site or os.environ.get("PUBLICDATA_SITE") or SITE).rstrip("/")
216
+ self.timeout = timeout
217
+ self.retries = retries
218
+ self.user_agent = user_agent or f"publicdata-au-python/{__version__}"
219
+
220
+ def _open(self, url: str):
221
+ req = urllib.request.Request(url, headers={"User-Agent": self.user_agent})
222
+ attempt = 0
223
+ while True:
224
+ try:
225
+ return urllib.request.urlopen(req, timeout=self.timeout)
226
+ except urllib.error.HTTPError as err:
227
+ # The API allows a burst per address, then answers 429 with how long to wait.
228
+ if err.code == 429 and attempt < self.retries:
229
+ attempt += 1
230
+ wait = err.headers.get("Retry-After") or "10"
231
+ time.sleep(min(float(wait) if wait.isdigit() else 10.0, 60.0))
232
+ continue
233
+ raw = err.read()
234
+ try:
235
+ body = json.loads(raw)
236
+ except ValueError:
237
+ body = raw.decode("utf-8", "replace")
238
+ msg = body.get("error") if isinstance(body, dict) else str(body)[:200]
239
+ raise PublicDataError(err.code, msg or err.reason, body, url) from None
240
+ except (urllib.error.URLError, TimeoutError) as err:
241
+ reason = getattr(err, "reason", err)
242
+ raise PublicDataError(0, f"could not reach the site: {reason}", None, url) from err
243
+
244
+ def _json(self, path_or_url: str, params: Mapping | None = None):
245
+ url = path_or_url if "://" in path_or_url else self.site + path_or_url
246
+ if params:
247
+ q = urllib.parse.urlencode(
248
+ {k: v for k, v in params.items() if v is not None},
249
+ safe="*(),.:",
250
+ quote_via=urllib.parse.quote,
251
+ )
252
+ if q:
253
+ url += ("&" if "?" in url else "?") + q
254
+ with self._open(url) as r:
255
+ return json.loads(r.read())
256
+
257
+ def datasets(self, q: str | None = None) -> list[dict]:
258
+ """Datasets the site serves, each with slug, title, publisher, licence and page URL.
259
+ `q` searches titles, summaries, publishers, keywords and field names."""
260
+ return self._json("/api/v1/datasets", {"q": q} if q else None)["results"]
261
+
262
+ def dataset(self, slug: str) -> dict:
263
+ """The dataset's Frictionless data package: title, licence, attribution, fields and
264
+ every file of the newest version."""
265
+ return self._json(f"/d/{_slug(slug)}/datapackage.json")
266
+
267
+ def versions(self, slug: str) -> list[dict]:
268
+ """Every version kept, newest first, each with its date, rows, fields and source hash."""
269
+ return self._json(f"/d/{_slug(slug)}/versions.json")["versions"]
270
+
271
+ def _query(self, slug, kind, where, version, params) -> dict:
272
+ base = f"/api/v1/datasets/{_slug(slug)}/"
273
+ if version:
274
+ base += f"versions/{_date(version)}/"
275
+ params = dict(params)
276
+ for field, value in (where or {}).items():
277
+ params[field] = _filter(value)
278
+ return self._json(base + kind, params)
279
+
280
+ def rows(
281
+ self,
282
+ slug: str,
283
+ where: Mapping[str, Any] | None = None,
284
+ *,
285
+ select: str | list[str] | None = None,
286
+ order: str | list[str] | None = None,
287
+ limit: int | None = None,
288
+ offset: int | None = None,
289
+ version: str | None = None,
290
+ all: bool = False,
291
+ ) -> Rows:
292
+ """Rows of a dataset from the query API.
293
+
294
+ `where` maps a field to a value, which must match exactly, or to a filter such as
295
+ `gte(2020)`, `in_("QLD", "NSW")` or `is_null()`. A list means any of its values and None
296
+ means blank. Every condition must match.
297
+
298
+ Without `version` the answer comes from the newest version and changes when the
299
+ publisher releases again; with a date from `versions()` it never changes. `all=True`
300
+ follows every page. For a whole table, `read()` or `download()` is faster and has no
301
+ rate limit."""
302
+ if all and limit is None:
303
+ limit = PAGE_MAX
304
+ params = {
305
+ "select": _list(select),
306
+ "order": _list(order),
307
+ "limit": limit,
308
+ "offset": offset,
309
+ }
310
+ body = self._query(slug, "rows", where, version, params)
311
+ out = Rows(body.get("rows", ()), body.get("publicdata"), _page(body))
312
+ nxt = body.get("next")
313
+ while all and nxt:
314
+ body = self._json(nxt)
315
+ out.extend(body.get("rows", ()))
316
+ nxt = body.get("next")
317
+ out.page["next"] = nxt
318
+ return out
319
+
320
+ def aggregate(
321
+ self,
322
+ slug: str,
323
+ group: str | list[str] | None = None,
324
+ metric: str | list[str] = "count",
325
+ where: Mapping[str, Any] | None = None,
326
+ *,
327
+ version: str | None = None,
328
+ ) -> Rows:
329
+ """Counts, sums, averages, minimums and maximums by group, from the query API.
330
+
331
+ `metric` is `count`, `sum.<field>`, `avg.<field>`, `min.<field>` or `max.<field>`, or a
332
+ list of them. `where` works as it does for `rows()`."""
333
+ params = {"group": _list(group), "metric": _list(metric)}
334
+ body = self._query(slug, "aggregate", where, version, params)
335
+ return Rows(body.get("rows", ()), body.get("publicdata"), _page(body))
336
+
337
+ def file_url(self, slug: str, format: str = "parquet", version: str | None = None) -> str:
338
+ if format not in FORMATS:
339
+ raise ValueError(f"format must be one of {', '.join(FORMATS)}")
340
+ at = f"v/{_date(version)}" if version else "latest"
341
+ return f"{self.site}/d/{_slug(slug)}/{at}/data.{format}"
342
+
343
+ def _save(self, slug, format, version, path) -> tuple[Path, str]:
344
+ with self._open(self.file_url(slug, format, version)) as r:
345
+ final = r.geturl()
346
+ got = final.split("/v/", 1)[1].split("/", 1)[0] if "/v/" in final else "latest"
347
+ dest = Path(path) if path else Path(f"{slug}-{got}.{format}")
348
+ with open(dest, "wb") as f:
349
+ shutil.copyfileobj(r, f, 1 << 20)
350
+ return dest, got
351
+
352
+ def download(
353
+ self,
354
+ slug: str,
355
+ format: str = "parquet",
356
+ version: str | None = None,
357
+ path: str | os.PathLike | None = None,
358
+ ) -> Path:
359
+ """Saves one version's file, the newest by default, and returns where. Without `path`
360
+ the file is named `<slug>-<version>.<format>` in the working directory. Files have no
361
+ rate limit."""
362
+ return self._save(slug, format, version, path)[0]
363
+
364
+ def read(self, slug: str, version: str | None = None):
365
+ """The whole table as a pandas DataFrame, read from the version's Parquet file.
366
+ `df.attrs["publicdata"]` is the provenance header the file itself carries: version,
367
+ licence, attribution, citation and source."""
368
+ import pyarrow.parquet as pq
369
+
370
+ with tempfile.TemporaryDirectory() as d:
371
+ p, _ = self._save(slug, "parquet", version, Path(d) / "data.parquet")
372
+ table = pq.read_table(p)
373
+ header = (table.schema.metadata or {}).get(b"publicdata")
374
+ df = table.to_pandas()
375
+ df.attrs["publicdata"] = json.loads(header) if header else {}
376
+ return df
377
+
378
+
379
+ def _slug(slug: str) -> str:
380
+ if not slug or not all(c.isalnum() or c == "-" for c in slug):
381
+ raise ValueError(f"not a dataset slug: {slug!r}")
382
+ return slug
383
+
384
+
385
+ def _date(version: str) -> str:
386
+ v = str(version)
387
+ if len(v) != 10 or v[4] != "-" or v[7] != "-" or not (v[:4] + v[5:7] + v[8:]).isdigit():
388
+ raise ValueError(f"a version is a date such as 2026-08-07, not {version!r}")
389
+ return v
390
+
391
+
392
+ def _list(value) -> str | None:
393
+ if value is None:
394
+ return None
395
+ return value if isinstance(value, str) else ",".join(value)
396
+
397
+
398
+ def _page(body: Mapping) -> dict:
399
+ return {
400
+ k: body.get(k) for k in ("dataset_page", "version_page", "this_version", "manifest", "next")
401
+ }
402
+
403
+
404
+ _default = Client()
405
+
406
+
407
+ def datasets(q: str | None = None) -> list[dict]:
408
+ return _default.datasets(q)
409
+
410
+
411
+ def dataset(slug: str) -> dict:
412
+ return _default.dataset(slug)
413
+
414
+
415
+ def versions(slug: str) -> list[dict]:
416
+ return _default.versions(slug)
417
+
418
+
419
+ def rows(slug: str, where: Mapping[str, Any] | None = None, **kw) -> Rows:
420
+ return _default.rows(slug, where, **kw)
421
+
422
+
423
+ def aggregate(slug: str, group=None, metric="count", where=None, **kw) -> Rows:
424
+ return _default.aggregate(slug, group, metric, where, **kw)
425
+
426
+
427
+ def download(slug: str, format: str = "parquet", version: str | None = None, path=None) -> Path:
428
+ return _default.download(slug, format, version, path)
429
+
430
+
431
+ def read(slug: str, version: str | None = None):
432
+ return _default.read(slug, version)
433
+
434
+
435
+ for _f in (datasets, dataset, versions, rows, aggregate, download, read):
436
+ _f.__doc__ = getattr(Client, _f.__name__).__doc__
@@ -0,0 +1,244 @@
1
+ import io
2
+ import json
3
+ import os
4
+ import threading
5
+ import urllib.parse
6
+ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
7
+
8
+ import pytest
9
+
10
+ import publicdata_au as pd_au
11
+
12
+ META = {
13
+ "version": "2026-08-07",
14
+ "attribution": "Publisher, CC BY 4.0.",
15
+ "cite": "Cite.",
16
+ "licence": {"id": "CC-BY-4.0"},
17
+ }
18
+
19
+
20
+ class Handler(BaseHTTPRequestHandler):
21
+ hits: list = []
22
+ throttle = 0
23
+
24
+ def log_message(self, *a):
25
+ pass
26
+
27
+ def send(self, status, body, ctype="application/json", headers=()):
28
+ raw = body if isinstance(body, bytes) else json.dumps(body).encode()
29
+ self.send_response(status)
30
+ self.send_header("Content-Type", ctype)
31
+ for k, v in headers:
32
+ self.send_header(k, v)
33
+ self.send_header("Content-Length", str(len(raw)))
34
+ self.end_headers()
35
+ self.wfile.write(raw)
36
+
37
+ def do_GET(self):
38
+ u = urllib.parse.urlsplit(self.path)
39
+ q = dict(urllib.parse.parse_qsl(u.query))
40
+ Handler.hits.append((u.path, q, self.headers.get("User-Agent")))
41
+ base = f"http://{self.headers['Host']}"
42
+ if Handler.throttle:
43
+ Handler.throttle -= 1
44
+ return self.send(429, {"error": "slow down"}, headers=[("Retry-After", "0")])
45
+ if u.path == "/api/v1/datasets":
46
+ return self.send(200, {"results": [{"slug": "a", "q": q.get("q")}]})
47
+ if u.path in ("/api/v1/datasets/a/rows", "/api/v1/datasets/a/versions/2026-08-07/rows"):
48
+ off = int(q.get("offset", 0))
49
+ data = [{"n": i} for i in range(5)]
50
+ lim = int(q.get("limit", 100))
51
+ page = data[off : off + lim]
52
+ more = off + lim < len(data)
53
+ nxt = (
54
+ f"{base}{u.path}?{urllib.parse.urlencode({**q, 'offset': off + lim})}"
55
+ if more
56
+ else None
57
+ )
58
+ return self.send(
59
+ 200, {"publicdata": META, "version_page": "vp", "rows": page, "next": nxt}
60
+ )
61
+ if u.path == "/api/v1/datasets/a/aggregate":
62
+ return self.send(
63
+ 200, {"publicdata": META, "rows": [{"g": 1, "count": 2}], "next": None}
64
+ )
65
+ if u.path == "/api/v1/datasets/nope/rows":
66
+ return self.send(404, {"error": "No such dataset in the query API"})
67
+ if u.path == "/d/a/versions.json":
68
+ return self.send(200, {"versions": [{"version": "2026-08-07"}]})
69
+ if u.path == "/d/a/datapackage.json":
70
+ return self.send(
71
+ 200, {"licenses": [{"name": "CC-BY-4.0"}], "publicdata:attribution": "Publisher."}
72
+ )
73
+ if u.path.startswith("/d/a/latest/"):
74
+ self.send_response(302)
75
+ self.send_header("Location", u.path.replace("/latest/", "/v/2026-08-07/"))
76
+ self.send_header("Content-Length", "0")
77
+ self.end_headers()
78
+ return
79
+ if u.path == "/d/a/v/2026-08-07/data.csv":
80
+ return self.send(200, b"n\n1\n", "text/csv")
81
+ if u.path == "/d/a/v/2026-08-07/data.parquet":
82
+ import pyarrow as pa
83
+ import pyarrow.parquet as pq
84
+
85
+ buf = io.BytesIO()
86
+ header = {
87
+ **META,
88
+ "version": "2026-08-07",
89
+ "not_endorsed": "The publisher has not endorsed this site.",
90
+ }
91
+ t = pa.table({"n": [1, 2, 3]}).replace_schema_metadata(
92
+ {"publicdata": json.dumps(header)}
93
+ )
94
+ pq.write_table(t, buf)
95
+ return self.send(200, buf.getvalue(), "application/vnd.apache.parquet")
96
+ self.send(404, {"error": "not here"})
97
+
98
+
99
+ @pytest.fixture
100
+ def client():
101
+ Handler.hits = []
102
+ Handler.throttle = 0
103
+ srv = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
104
+ t = threading.Thread(target=srv.serve_forever, daemon=True)
105
+ t.start()
106
+ yield pd_au.Client(f"http://127.0.0.1:{srv.server_port}", retries=2)
107
+ srv.shutdown()
108
+
109
+
110
+ def test_filters_render_in_the_apis_operator_form():
111
+ assert str(pd_au.gte(2020)) == "gte.2020"
112
+ assert str(pd_au.in_("QLD", "NSW")) == "in.(QLD,NSW)"
113
+ assert str(pd_au.in_(["a", "b"])) == "in.(a,b)"
114
+ assert str(pd_au.not_(pd_au.eq("x"))) == "not.eq.x"
115
+ assert str(pd_au.is_null()) == "is.null"
116
+ assert str(pd_au.ilike("*rider*")) == "ilike.*rider*"
117
+ assert str(pd_au.eq(True)) == "eq.true"
118
+ with pytest.raises(ValueError, match="comma"):
119
+ pd_au.in_("a,b")
120
+
121
+
122
+ def test_rows_sends_where_select_and_order(client):
123
+ r = client.rows(
124
+ "a",
125
+ {"state": "QLD", "year": pd_au.gte(2020), "lga": None, "sex": ["F", "M"]},
126
+ select=["n", "m"],
127
+ order="n.desc",
128
+ limit=2,
129
+ )
130
+ path, q, ua = Handler.hits[-1]
131
+ assert path == "/api/v1/datasets/a/rows"
132
+ assert q == {
133
+ "state": "eq.QLD",
134
+ "year": "gte.2020",
135
+ "lga": "is.null",
136
+ "sex": "in.(F,M)",
137
+ "select": "n,m",
138
+ "order": "n.desc",
139
+ "limit": "2",
140
+ }
141
+ assert ua.startswith("publicdata-au-python/")
142
+ assert r == [{"n": 0}, {"n": 1}] and r.page["next"]
143
+
144
+
145
+ def test_rows_carries_provenance(client):
146
+ r = client.rows("a")
147
+ assert (
148
+ r.version == "2026-08-07" and r.attribution and r.cite and r.licence == {"id": "CC-BY-4.0"}
149
+ )
150
+ assert r.version_page == "vp"
151
+
152
+
153
+ def test_all_follows_every_page(client):
154
+ r = client.rows("a", all=True, limit=2)
155
+ assert [x["n"] for x in r] == [0, 1, 2, 3, 4] and r.page["next"] is None
156
+ assert len([h for h in Handler.hits if h[0].endswith("/rows")]) == 3
157
+
158
+
159
+ def test_a_dated_version_goes_on_the_path(client):
160
+ client.rows("a", version="2026-08-07")
161
+ assert Handler.hits[-1][0] == "/api/v1/datasets/a/versions/2026-08-07/rows"
162
+ with pytest.raises(ValueError, match="date"):
163
+ client.rows("a", version="latest")
164
+
165
+
166
+ def test_aggregate(client):
167
+ a = client.aggregate("a", group=["g", "h"], metric=["count", "sum.n"], where={"x": 1})
168
+ assert Handler.hits[-1][1] == {"group": "g,h", "metric": "count,sum.n", "x": "eq.1"}
169
+ assert a == [{"g": 1, "count": 2}] and a.version == "2026-08-07"
170
+
171
+
172
+ def test_429_waits_then_retries(client):
173
+ Handler.throttle = 2
174
+ assert client.datasets("crash") == [{"slug": "a", "q": "crash"}]
175
+
176
+
177
+ def test_429_past_the_retries_raises(client):
178
+ Handler.throttle = 5
179
+ with pytest.raises(pd_au.PublicDataError) as err:
180
+ client.datasets()
181
+ assert err.value.status == 429
182
+
183
+
184
+ def test_an_error_names_the_status_and_the_apis_message(client):
185
+ with pytest.raises(pd_au.PublicDataError) as err:
186
+ client.rows("nope")
187
+ assert err.value.status == 404 and "No such dataset" in str(err.value)
188
+ assert err.value.body == {"error": "No such dataset in the query API"}
189
+
190
+
191
+ def test_a_bad_slug_never_reaches_the_network(client):
192
+ with pytest.raises(ValueError):
193
+ client.rows("../etc")
194
+ assert Handler.hits == []
195
+
196
+
197
+ def test_download_follows_latest_and_names_the_version(client, tmp_path, monkeypatch):
198
+ monkeypatch.chdir(tmp_path)
199
+ p = client.download("a", "csv")
200
+ assert p.name == "a-2026-08-07.csv" and p.read_bytes() == b"n\n1\n"
201
+ with pytest.raises(ValueError, match="format"):
202
+ client.download("a", "docx")
203
+
204
+
205
+ def test_read_takes_provenance_from_the_file_it_read(client):
206
+ pytest.importorskip("pandas")
207
+ df = client.read("a", version="2026-08-07")
208
+ assert list(df["n"]) == [1, 2, 3]
209
+ assert df.attrs["publicdata"]["version"] == "2026-08-07"
210
+ assert df.attrs["publicdata"]["attribution"] == "Publisher, CC BY 4.0."
211
+ assert "has not endorsed" in df.attrs["publicdata"]["not_endorsed"]
212
+ assert not any(h[0].endswith("datapackage.json") for h in Handler.hits)
213
+
214
+
215
+ def test_an_unreachable_site_raises_the_packages_own_error():
216
+ import socket
217
+
218
+ s = socket.socket()
219
+ s.bind(("127.0.0.1", 0))
220
+ port = s.getsockname()[1]
221
+ s.close()
222
+ with pytest.raises(pd_au.PublicDataError) as err:
223
+ pd_au.Client(f"http://127.0.0.1:{port}", timeout=2).datasets()
224
+ assert err.value.status == 0 and "could not reach" in str(err.value)
225
+
226
+
227
+ def test_versions_and_dataset(client):
228
+ assert client.versions("a") == [{"version": "2026-08-07"}]
229
+ assert client.dataset("a")["licenses"][0]["name"] == "CC-BY-4.0"
230
+
231
+
232
+ def test_site_can_come_from_the_environment(monkeypatch):
233
+ monkeypatch.setenv("PUBLICDATA_SITE", "http://example.test/")
234
+ assert pd_au.Client().site == "http://example.test"
235
+
236
+
237
+ @pytest.mark.skipif(
238
+ not os.environ.get("PUBLICDATA_LIVE"), reason="set PUBLICDATA_LIVE=1 to query the live site"
239
+ )
240
+ def test_live_site():
241
+ slugs = [d["slug"] for d in pd_au.datasets()]
242
+ assert "au-road-deaths" in slugs
243
+ r = pd_au.rows("au-road-deaths", {"state": "QLD"}, limit=1)
244
+ assert len(r) == 1 and r.attribution