publicdata-au 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
__pycache__/
|
|
2
|
+
*.egg-info/
|
|
3
|
+
.venv/
|
|
4
|
+
.pytest_cache/
|
|
5
|
+
.ruff_cache/
|
|
6
|
+
dist/
|
|
7
|
+
snapshots/
|
|
8
|
+
raw/
|
|
9
|
+
.wrangler/
|
|
10
|
+
.DS_Store
|
|
11
|
+
.env
|
|
12
|
+
.env.*
|
|
13
|
+
store/*/*/source.*
|
|
14
|
+
dist-large/
|
|
15
|
+
# Lighthouse and Chrome write beside the working directory during local audits.
|
|
16
|
+
lh-*.json
|
|
17
|
+
C:*
|
|
18
|
+
node_modules/
|
|
19
|
+
|
|
20
|
+
# Written at publish time by python -m publicdata.api_text server.json.
|
|
21
|
+
/server.json
|
|
22
|
+
.build-cache/
|
|
23
|
+
.venv*/
|
|
24
|
+
clients/*/dist/
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 National Digital
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: publicdata-au
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Query and download Australian government open data from publicdata.au.
|
|
5
|
+
Project-URL: Homepage, https://publicdata.au/
|
|
6
|
+
Project-URL: Documentation, https://publicdata.au/agents/
|
|
7
|
+
Project-URL: Source, https://github.com/National-Digital/publicdata.au/tree/main/clients/python
|
|
8
|
+
Author: National Digital
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: australia,government data,open data,publicdata.au
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Provides-Extra: dev
|
|
18
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
19
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
20
|
+
Provides-Extra: pandas
|
|
21
|
+
Requires-Dist: pandas>=2.0; extra == 'pandas'
|
|
22
|
+
Requires-Dist: pyarrow>=14; extra == 'pandas'
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
|
|
25
|
+
# publicdata-au
|
|
26
|
+
|
|
27
|
+
Query and download Australian government open data from [publicdata.au](https://publicdata.au/).
|
|
28
|
+
publicdata.au republishes datasets that governments already publish under open licences, keeps
|
|
29
|
+
every version at a URL that never changes, and serves each one in ten formats with a query API.
|
|
30
|
+
|
|
31
|
+
This package works for every dataset the site serves, named by its slug, so a dataset added to
|
|
32
|
+
the site needs no new release.
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
pip install publicdata-au # queries and downloads, no dependencies
|
|
36
|
+
pip install "publicdata-au[pandas]" # adds read() into a pandas DataFrame
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Find a dataset
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
import publicdata_au as pd_au
|
|
43
|
+
|
|
44
|
+
pd_au.datasets("road crashes") # slug, title, publisher, licence and page of each match
|
|
45
|
+
pd_au.versions("au-road-deaths") # every version kept, newest first
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
## Query rows and totals
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
from publicdata_au import gte, in_
|
|
52
|
+
|
|
53
|
+
deaths = pd_au.rows(
|
|
54
|
+
"au-road-deaths",
|
|
55
|
+
{"state": in_("QLD", "NSW"), "year": gte(2020)},
|
|
56
|
+
select=["state", "year", "road_user"],
|
|
57
|
+
order="year.desc",
|
|
58
|
+
all=True,
|
|
59
|
+
)
|
|
60
|
+
deaths.version # the version the rows came from
|
|
61
|
+
deaths.attribution # the attribution the publisher's licence requires
|
|
62
|
+
deaths.to_pandas()
|
|
63
|
+
|
|
64
|
+
pd_au.aggregate("au-road-deaths", group="state", metric="count", where={"year": 2025})
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
A plain value must match exactly, a list matches any of its values and None matches a blank or
|
|
68
|
+
suppressed cell. The filters are `eq`, `neq`, `gt`, `gte`, `lt`, `lte`, `like`, `ilike`, `in_`,
|
|
69
|
+
`is_null` and `not_`.
|
|
70
|
+
|
|
71
|
+
Without `version=` an answer comes from the newest version and changes when the publisher
|
|
72
|
+
releases again. Pass a date from `versions()` for an answer that never changes.
|
|
73
|
+
|
|
74
|
+
## Whole tables
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
df = pd_au.read("au-road-deaths") # needs the [pandas] extra
|
|
78
|
+
df.attrs["publicdata"] # the version, licence and attribution the file itself carries
|
|
79
|
+
pd_au.download(
|
|
80
|
+
"au-road-deaths", "csv"
|
|
81
|
+
) # parquet, csv, csv.gz, json, ndjson, sqlite, xlsx, arrow, geojson, gpkg
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Files have no rate limit. The query API allows 60 requests in 10 seconds from one address, and
|
|
85
|
+
this package waits and retries when it answers 429.
|
|
86
|
+
|
|
87
|
+
## Licence and attribution
|
|
88
|
+
|
|
89
|
+
The data is under each publisher's own licence, which requires the attribution string that every
|
|
90
|
+
answer carries. Please also name publicdata.au and link to the version you used. publicdata.au is
|
|
91
|
+
an independent republication, and the publishers have not endorsed it.
|
|
92
|
+
|
|
93
|
+
The package itself is under the MIT licence.
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# publicdata-au
|
|
2
|
+
|
|
3
|
+
Query and download Australian government open data from [publicdata.au](https://publicdata.au/).
|
|
4
|
+
publicdata.au republishes datasets that governments already publish under open licences, keeps
|
|
5
|
+
every version at a URL that never changes, and serves each one in ten formats with a query API.
|
|
6
|
+
|
|
7
|
+
This package works for every dataset the site serves, named by its slug, so a dataset added to
|
|
8
|
+
the site needs no new release.
|
|
9
|
+
|
|
10
|
+
```
|
|
11
|
+
pip install publicdata-au # queries and downloads, no dependencies
|
|
12
|
+
pip install "publicdata-au[pandas]" # adds read() into a pandas DataFrame
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
## Find a dataset
|
|
16
|
+
|
|
17
|
+
```python
|
|
18
|
+
import publicdata_au as pd_au
|
|
19
|
+
|
|
20
|
+
pd_au.datasets("road crashes") # slug, title, publisher, licence and page of each match
|
|
21
|
+
pd_au.versions("au-road-deaths") # every version kept, newest first
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
## Query rows and totals
|
|
25
|
+
|
|
26
|
+
```python
|
|
27
|
+
from publicdata_au import gte, in_
|
|
28
|
+
|
|
29
|
+
deaths = pd_au.rows(
|
|
30
|
+
"au-road-deaths",
|
|
31
|
+
{"state": in_("QLD", "NSW"), "year": gte(2020)},
|
|
32
|
+
select=["state", "year", "road_user"],
|
|
33
|
+
order="year.desc",
|
|
34
|
+
all=True,
|
|
35
|
+
)
|
|
36
|
+
deaths.version # the version the rows came from
|
|
37
|
+
deaths.attribution # the attribution the publisher's licence requires
|
|
38
|
+
deaths.to_pandas()
|
|
39
|
+
|
|
40
|
+
pd_au.aggregate("au-road-deaths", group="state", metric="count", where={"year": 2025})
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
A plain value must match exactly, a list matches any of its values and None matches a blank or
|
|
44
|
+
suppressed cell. The filters are `eq`, `neq`, `gt`, `gte`, `lt`, `lte`, `like`, `ilike`, `in_`,
|
|
45
|
+
`is_null` and `not_`.
|
|
46
|
+
|
|
47
|
+
Without `version=` an answer comes from the newest version and changes when the publisher
|
|
48
|
+
releases again. Pass a date from `versions()` for an answer that never changes.
|
|
49
|
+
|
|
50
|
+
## Whole tables
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
df = pd_au.read("au-road-deaths") # needs the [pandas] extra
|
|
54
|
+
df.attrs["publicdata"] # the version, licence and attribution the file itself carries
|
|
55
|
+
pd_au.download(
|
|
56
|
+
"au-road-deaths", "csv"
|
|
57
|
+
) # parquet, csv, csv.gz, json, ndjson, sqlite, xlsx, arrow, geojson, gpkg
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
Files have no rate limit. The query API allows 60 requests in 10 seconds from one address, and
|
|
61
|
+
this package waits and retries when it answers 429.
|
|
62
|
+
|
|
63
|
+
## Licence and attribution
|
|
64
|
+
|
|
65
|
+
The data is under each publisher's own licence, which requires the attribution string that every
|
|
66
|
+
answer carries. Please also name publicdata.au and link to the version you used. publicdata.au is
|
|
67
|
+
an independent republication, and the publishers have not endorsed it.
|
|
68
|
+
|
|
69
|
+
The package itself is under the MIT licence.
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "publicdata-au"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Query and download Australian government open data from publicdata.au."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
license = "MIT"
|
|
8
|
+
license-files = ["LICENSE"]
|
|
9
|
+
authors = [{ name = "National Digital" }]
|
|
10
|
+
keywords = ["australia", "open data", "government data", "publicdata.au"]
|
|
11
|
+
classifiers = [
|
|
12
|
+
"Development Status :: 4 - Beta",
|
|
13
|
+
"Intended Audience :: Science/Research",
|
|
14
|
+
"Programming Language :: Python :: 3",
|
|
15
|
+
"Topic :: Scientific/Engineering :: Information Analysis",
|
|
16
|
+
]
|
|
17
|
+
dependencies = []
|
|
18
|
+
|
|
19
|
+
[project.optional-dependencies]
|
|
20
|
+
pandas = ["pandas>=2.0", "pyarrow>=14"]
|
|
21
|
+
dev = ["pytest>=8", "ruff>=0.6"]
|
|
22
|
+
|
|
23
|
+
[project.urls]
|
|
24
|
+
Homepage = "https://publicdata.au/"
|
|
25
|
+
Documentation = "https://publicdata.au/agents/"
|
|
26
|
+
Source = "https://github.com/National-Digital/publicdata.au/tree/main/clients/python"
|
|
27
|
+
|
|
28
|
+
[build-system]
|
|
29
|
+
requires = ["hatchling>=1.27"]
|
|
30
|
+
build-backend = "hatchling.build"
|
|
31
|
+
|
|
32
|
+
[tool.hatch.build.targets.wheel]
|
|
33
|
+
packages = ["src/publicdata_au"]
|
|
34
|
+
|
|
35
|
+
[tool.ruff]
|
|
36
|
+
line-length = 100
|
|
37
|
+
target-version = "py310"
|
|
38
|
+
|
|
39
|
+
[tool.ruff.lint]
|
|
40
|
+
select = ["E", "F", "I", "B", "UP", "W"]
|
|
41
|
+
ignore = ["E501"]
|
|
@@ -0,0 +1,436 @@
|
|
|
1
|
+
"""Query and download Australian government open data from publicdata.au.
|
|
2
|
+
|
|
3
|
+
Every function works for every dataset the site serves, named by its slug, so a dataset added to
|
|
4
|
+
the site needs no new release of this package.
|
|
5
|
+
|
|
6
|
+
>>> import publicdata_au as pd_au
|
|
7
|
+
>>> pd_au.datasets("road crashes")
|
|
8
|
+
>>> pd_au.rows("au-road-deaths", {"state": "QLD", "year": pd_au.gte(2020)}, limit=5)
|
|
9
|
+
>>> pd_au.aggregate("au-road-deaths", group="state")
|
|
10
|
+
>>> pd_au.read("au-road-deaths") # a pandas DataFrame, with the [pandas] extra
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
import shutil
|
|
18
|
+
import tempfile
|
|
19
|
+
import time
|
|
20
|
+
import urllib.error
|
|
21
|
+
import urllib.parse
|
|
22
|
+
import urllib.request
|
|
23
|
+
from collections.abc import Mapping
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
__version__ = "0.1.0"
|
|
28
|
+
__all__ = [
|
|
29
|
+
"Client",
|
|
30
|
+
"Filter",
|
|
31
|
+
"PublicDataError",
|
|
32
|
+
"Rows",
|
|
33
|
+
"aggregate",
|
|
34
|
+
"dataset",
|
|
35
|
+
"datasets",
|
|
36
|
+
"download",
|
|
37
|
+
"eq",
|
|
38
|
+
"gt",
|
|
39
|
+
"gte",
|
|
40
|
+
"ilike",
|
|
41
|
+
"in_",
|
|
42
|
+
"is_null",
|
|
43
|
+
"like",
|
|
44
|
+
"lt",
|
|
45
|
+
"lte",
|
|
46
|
+
"neq",
|
|
47
|
+
"not_",
|
|
48
|
+
"read",
|
|
49
|
+
"rows",
|
|
50
|
+
"versions",
|
|
51
|
+
]
|
|
52
|
+
|
|
53
|
+
SITE = "https://publicdata.au"
|
|
54
|
+
FORMATS = (
|
|
55
|
+
"parquet",
|
|
56
|
+
"csv",
|
|
57
|
+
"csv.gz",
|
|
58
|
+
"json",
|
|
59
|
+
"ndjson",
|
|
60
|
+
"sqlite",
|
|
61
|
+
"xlsx",
|
|
62
|
+
"arrow",
|
|
63
|
+
"geojson",
|
|
64
|
+
"gpkg",
|
|
65
|
+
)
|
|
66
|
+
PAGE_MAX = 10_000
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class PublicDataError(Exception):
|
|
70
|
+
"""An answer from publicdata.au that was not a success."""
|
|
71
|
+
|
|
72
|
+
def __init__(self, status: int, message: str, body: Any = None, url: str = ""):
|
|
73
|
+
super().__init__(f"{status}: {message}" + (f" ({url})" if url else ""))
|
|
74
|
+
self.status = status
|
|
75
|
+
self.body = body
|
|
76
|
+
self.url = url
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class Filter:
|
|
80
|
+
"""One condition on a field, in the query API's `operator.value` form."""
|
|
81
|
+
|
|
82
|
+
def __init__(self, expr: str):
|
|
83
|
+
self.expr = expr
|
|
84
|
+
|
|
85
|
+
def __str__(self) -> str:
|
|
86
|
+
return self.expr
|
|
87
|
+
|
|
88
|
+
def __repr__(self) -> str:
|
|
89
|
+
return f"Filter({self.expr!r})"
|
|
90
|
+
|
|
91
|
+
def __eq__(self, other) -> bool:
|
|
92
|
+
return isinstance(other, Filter) and other.expr == self.expr
|
|
93
|
+
|
|
94
|
+
def __hash__(self) -> int:
|
|
95
|
+
return hash(self.expr)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _v(value) -> str:
|
|
99
|
+
if isinstance(value, bool):
|
|
100
|
+
return "true" if value else "false"
|
|
101
|
+
return str(value)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def eq(value) -> Filter:
|
|
105
|
+
return Filter(f"eq.{_v(value)}")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def neq(value) -> Filter:
|
|
109
|
+
return Filter(f"neq.{_v(value)}")
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def gt(value) -> Filter:
|
|
113
|
+
return Filter(f"gt.{_v(value)}")
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def gte(value) -> Filter:
|
|
117
|
+
return Filter(f"gte.{_v(value)}")
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def lt(value) -> Filter:
|
|
121
|
+
return Filter(f"lt.{_v(value)}")
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def lte(value) -> Filter:
|
|
125
|
+
return Filter(f"lte.{_v(value)}")
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def like(pattern: str) -> Filter:
|
|
129
|
+
"""`*` stands for any run of characters. Case-sensitive."""
|
|
130
|
+
return Filter(f"like.{pattern}")
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def ilike(pattern: str) -> Filter:
|
|
134
|
+
"""As `like`, ignoring case."""
|
|
135
|
+
return Filter(f"ilike.{pattern}")
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def in_(*values) -> Filter:
|
|
139
|
+
if len(values) == 1 and isinstance(values[0], (list, tuple, set, frozenset)):
|
|
140
|
+
values = tuple(values[0])
|
|
141
|
+
if any("," in _v(v) for v in values):
|
|
142
|
+
raise ValueError("a value in in_() cannot contain a comma")
|
|
143
|
+
return Filter(f"in.({','.join(_v(v) for v in values)})")
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def is_null() -> Filter:
|
|
147
|
+
"""Blank in the source, or suppressed by the publisher."""
|
|
148
|
+
return Filter("is.null")
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def not_(f: Filter) -> Filter:
|
|
152
|
+
return Filter(f"not.{f.expr}")
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _filter(value) -> str:
|
|
156
|
+
if isinstance(value, Filter):
|
|
157
|
+
return value.expr
|
|
158
|
+
if value is None:
|
|
159
|
+
return "is.null"
|
|
160
|
+
if isinstance(value, (list, tuple, set, frozenset)):
|
|
161
|
+
return in_(tuple(value)).expr
|
|
162
|
+
return eq(value).expr
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
class Rows(list):
|
|
166
|
+
"""A list of row dicts that also carries where they came from.
|
|
167
|
+
|
|
168
|
+
`version`, `attribution`, `cite` and `licence` come from the answer itself, so they always
|
|
169
|
+
name the version the rows were read from."""
|
|
170
|
+
|
|
171
|
+
def __init__(self, rows=(), meta: Mapping | None = None, page: Mapping | None = None):
|
|
172
|
+
super().__init__(rows)
|
|
173
|
+
self.meta = dict(meta or {})
|
|
174
|
+
self.page = dict(page or {})
|
|
175
|
+
|
|
176
|
+
@property
|
|
177
|
+
def version(self) -> str | None:
|
|
178
|
+
return self.meta.get("version")
|
|
179
|
+
|
|
180
|
+
@property
|
|
181
|
+
def attribution(self) -> str | None:
|
|
182
|
+
return self.meta.get("attribution")
|
|
183
|
+
|
|
184
|
+
@property
|
|
185
|
+
def cite(self) -> str | None:
|
|
186
|
+
return self.meta.get("cite")
|
|
187
|
+
|
|
188
|
+
@property
|
|
189
|
+
def licence(self) -> dict | None:
|
|
190
|
+
return self.meta.get("licence")
|
|
191
|
+
|
|
192
|
+
@property
|
|
193
|
+
def version_page(self) -> str | None:
|
|
194
|
+
return self.page.get("version_page")
|
|
195
|
+
|
|
196
|
+
def to_pandas(self):
|
|
197
|
+
import pandas as pd
|
|
198
|
+
|
|
199
|
+
df = pd.DataFrame(list(self))
|
|
200
|
+
df.attrs["publicdata"] = self.meta
|
|
201
|
+
return df
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
class Client:
|
|
205
|
+
"""Talks to one publicdata.au site. The module-level functions use a shared default."""
|
|
206
|
+
|
|
207
|
+
def __init__(
|
|
208
|
+
self,
|
|
209
|
+
site: str | None = None,
|
|
210
|
+
*,
|
|
211
|
+
timeout: float = 60,
|
|
212
|
+
retries: int = 3,
|
|
213
|
+
user_agent: str | None = None,
|
|
214
|
+
):
|
|
215
|
+
self.site = (site or os.environ.get("PUBLICDATA_SITE") or SITE).rstrip("/")
|
|
216
|
+
self.timeout = timeout
|
|
217
|
+
self.retries = retries
|
|
218
|
+
self.user_agent = user_agent or f"publicdata-au-python/{__version__}"
|
|
219
|
+
|
|
220
|
+
def _open(self, url: str):
|
|
221
|
+
req = urllib.request.Request(url, headers={"User-Agent": self.user_agent})
|
|
222
|
+
attempt = 0
|
|
223
|
+
while True:
|
|
224
|
+
try:
|
|
225
|
+
return urllib.request.urlopen(req, timeout=self.timeout)
|
|
226
|
+
except urllib.error.HTTPError as err:
|
|
227
|
+
# The API allows a burst per address, then answers 429 with how long to wait.
|
|
228
|
+
if err.code == 429 and attempt < self.retries:
|
|
229
|
+
attempt += 1
|
|
230
|
+
wait = err.headers.get("Retry-After") or "10"
|
|
231
|
+
time.sleep(min(float(wait) if wait.isdigit() else 10.0, 60.0))
|
|
232
|
+
continue
|
|
233
|
+
raw = err.read()
|
|
234
|
+
try:
|
|
235
|
+
body = json.loads(raw)
|
|
236
|
+
except ValueError:
|
|
237
|
+
body = raw.decode("utf-8", "replace")
|
|
238
|
+
msg = body.get("error") if isinstance(body, dict) else str(body)[:200]
|
|
239
|
+
raise PublicDataError(err.code, msg or err.reason, body, url) from None
|
|
240
|
+
except (urllib.error.URLError, TimeoutError) as err:
|
|
241
|
+
reason = getattr(err, "reason", err)
|
|
242
|
+
raise PublicDataError(0, f"could not reach the site: {reason}", None, url) from err
|
|
243
|
+
|
|
244
|
+
def _json(self, path_or_url: str, params: Mapping | None = None):
|
|
245
|
+
url = path_or_url if "://" in path_or_url else self.site + path_or_url
|
|
246
|
+
if params:
|
|
247
|
+
q = urllib.parse.urlencode(
|
|
248
|
+
{k: v for k, v in params.items() if v is not None},
|
|
249
|
+
safe="*(),.:",
|
|
250
|
+
quote_via=urllib.parse.quote,
|
|
251
|
+
)
|
|
252
|
+
if q:
|
|
253
|
+
url += ("&" if "?" in url else "?") + q
|
|
254
|
+
with self._open(url) as r:
|
|
255
|
+
return json.loads(r.read())
|
|
256
|
+
|
|
257
|
+
def datasets(self, q: str | None = None) -> list[dict]:
|
|
258
|
+
"""Datasets the site serves, each with slug, title, publisher, licence and page URL.
|
|
259
|
+
`q` searches titles, summaries, publishers, keywords and field names."""
|
|
260
|
+
return self._json("/api/v1/datasets", {"q": q} if q else None)["results"]
|
|
261
|
+
|
|
262
|
+
def dataset(self, slug: str) -> dict:
|
|
263
|
+
"""The dataset's Frictionless data package: title, licence, attribution, fields and
|
|
264
|
+
every file of the newest version."""
|
|
265
|
+
return self._json(f"/d/{_slug(slug)}/datapackage.json")
|
|
266
|
+
|
|
267
|
+
def versions(self, slug: str) -> list[dict]:
|
|
268
|
+
"""Every version kept, newest first, each with its date, rows, fields and source hash."""
|
|
269
|
+
return self._json(f"/d/{_slug(slug)}/versions.json")["versions"]
|
|
270
|
+
|
|
271
|
+
def _query(self, slug, kind, where, version, params) -> dict:
|
|
272
|
+
base = f"/api/v1/datasets/{_slug(slug)}/"
|
|
273
|
+
if version:
|
|
274
|
+
base += f"versions/{_date(version)}/"
|
|
275
|
+
params = dict(params)
|
|
276
|
+
for field, value in (where or {}).items():
|
|
277
|
+
params[field] = _filter(value)
|
|
278
|
+
return self._json(base + kind, params)
|
|
279
|
+
|
|
280
|
+
def rows(
|
|
281
|
+
self,
|
|
282
|
+
slug: str,
|
|
283
|
+
where: Mapping[str, Any] | None = None,
|
|
284
|
+
*,
|
|
285
|
+
select: str | list[str] | None = None,
|
|
286
|
+
order: str | list[str] | None = None,
|
|
287
|
+
limit: int | None = None,
|
|
288
|
+
offset: int | None = None,
|
|
289
|
+
version: str | None = None,
|
|
290
|
+
all: bool = False,
|
|
291
|
+
) -> Rows:
|
|
292
|
+
"""Rows of a dataset from the query API.
|
|
293
|
+
|
|
294
|
+
`where` maps a field to a value, which must match exactly, or to a filter such as
|
|
295
|
+
`gte(2020)`, `in_("QLD", "NSW")` or `is_null()`. A list means any of its values and None
|
|
296
|
+
means blank. Every condition must match.
|
|
297
|
+
|
|
298
|
+
Without `version` the answer comes from the newest version and changes when the
|
|
299
|
+
publisher releases again; with a date from `versions()` it never changes. `all=True`
|
|
300
|
+
follows every page. For a whole table, `read()` or `download()` is faster and has no
|
|
301
|
+
rate limit."""
|
|
302
|
+
if all and limit is None:
|
|
303
|
+
limit = PAGE_MAX
|
|
304
|
+
params = {
|
|
305
|
+
"select": _list(select),
|
|
306
|
+
"order": _list(order),
|
|
307
|
+
"limit": limit,
|
|
308
|
+
"offset": offset,
|
|
309
|
+
}
|
|
310
|
+
body = self._query(slug, "rows", where, version, params)
|
|
311
|
+
out = Rows(body.get("rows", ()), body.get("publicdata"), _page(body))
|
|
312
|
+
nxt = body.get("next")
|
|
313
|
+
while all and nxt:
|
|
314
|
+
body = self._json(nxt)
|
|
315
|
+
out.extend(body.get("rows", ()))
|
|
316
|
+
nxt = body.get("next")
|
|
317
|
+
out.page["next"] = nxt
|
|
318
|
+
return out
|
|
319
|
+
|
|
320
|
+
def aggregate(
|
|
321
|
+
self,
|
|
322
|
+
slug: str,
|
|
323
|
+
group: str | list[str] | None = None,
|
|
324
|
+
metric: str | list[str] = "count",
|
|
325
|
+
where: Mapping[str, Any] | None = None,
|
|
326
|
+
*,
|
|
327
|
+
version: str | None = None,
|
|
328
|
+
) -> Rows:
|
|
329
|
+
"""Counts, sums, averages, minimums and maximums by group, from the query API.
|
|
330
|
+
|
|
331
|
+
`metric` is `count`, `sum.<field>`, `avg.<field>`, `min.<field>` or `max.<field>`, or a
|
|
332
|
+
list of them. `where` works as it does for `rows()`."""
|
|
333
|
+
params = {"group": _list(group), "metric": _list(metric)}
|
|
334
|
+
body = self._query(slug, "aggregate", where, version, params)
|
|
335
|
+
return Rows(body.get("rows", ()), body.get("publicdata"), _page(body))
|
|
336
|
+
|
|
337
|
+
def file_url(self, slug: str, format: str = "parquet", version: str | None = None) -> str:
|
|
338
|
+
if format not in FORMATS:
|
|
339
|
+
raise ValueError(f"format must be one of {', '.join(FORMATS)}")
|
|
340
|
+
at = f"v/{_date(version)}" if version else "latest"
|
|
341
|
+
return f"{self.site}/d/{_slug(slug)}/{at}/data.{format}"
|
|
342
|
+
|
|
343
|
+
def _save(self, slug, format, version, path) -> tuple[Path, str]:
|
|
344
|
+
with self._open(self.file_url(slug, format, version)) as r:
|
|
345
|
+
final = r.geturl()
|
|
346
|
+
got = final.split("/v/", 1)[1].split("/", 1)[0] if "/v/" in final else "latest"
|
|
347
|
+
dest = Path(path) if path else Path(f"{slug}-{got}.{format}")
|
|
348
|
+
with open(dest, "wb") as f:
|
|
349
|
+
shutil.copyfileobj(r, f, 1 << 20)
|
|
350
|
+
return dest, got
|
|
351
|
+
|
|
352
|
+
def download(
|
|
353
|
+
self,
|
|
354
|
+
slug: str,
|
|
355
|
+
format: str = "parquet",
|
|
356
|
+
version: str | None = None,
|
|
357
|
+
path: str | os.PathLike | None = None,
|
|
358
|
+
) -> Path:
|
|
359
|
+
"""Saves one version's file, the newest by default, and returns where. Without `path`
|
|
360
|
+
the file is named `<slug>-<version>.<format>` in the working directory. Files have no
|
|
361
|
+
rate limit."""
|
|
362
|
+
return self._save(slug, format, version, path)[0]
|
|
363
|
+
|
|
364
|
+
def read(self, slug: str, version: str | None = None):
|
|
365
|
+
"""The whole table as a pandas DataFrame, read from the version's Parquet file.
|
|
366
|
+
`df.attrs["publicdata"]` is the provenance header the file itself carries: version,
|
|
367
|
+
licence, attribution, citation and source."""
|
|
368
|
+
import pyarrow.parquet as pq
|
|
369
|
+
|
|
370
|
+
with tempfile.TemporaryDirectory() as d:
|
|
371
|
+
p, _ = self._save(slug, "parquet", version, Path(d) / "data.parquet")
|
|
372
|
+
table = pq.read_table(p)
|
|
373
|
+
header = (table.schema.metadata or {}).get(b"publicdata")
|
|
374
|
+
df = table.to_pandas()
|
|
375
|
+
df.attrs["publicdata"] = json.loads(header) if header else {}
|
|
376
|
+
return df
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def _slug(slug: str) -> str:
|
|
380
|
+
if not slug or not all(c.isalnum() or c == "-" for c in slug):
|
|
381
|
+
raise ValueError(f"not a dataset slug: {slug!r}")
|
|
382
|
+
return slug
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _date(version: str) -> str:
|
|
386
|
+
v = str(version)
|
|
387
|
+
if len(v) != 10 or v[4] != "-" or v[7] != "-" or not (v[:4] + v[5:7] + v[8:]).isdigit():
|
|
388
|
+
raise ValueError(f"a version is a date such as 2026-08-07, not {version!r}")
|
|
389
|
+
return v
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
def _list(value) -> str | None:
|
|
393
|
+
if value is None:
|
|
394
|
+
return None
|
|
395
|
+
return value if isinstance(value, str) else ",".join(value)
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def _page(body: Mapping) -> dict:
|
|
399
|
+
return {
|
|
400
|
+
k: body.get(k) for k in ("dataset_page", "version_page", "this_version", "manifest", "next")
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
_default = Client()
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def datasets(q: str | None = None) -> list[dict]:
|
|
408
|
+
return _default.datasets(q)
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
def dataset(slug: str) -> dict:
|
|
412
|
+
return _default.dataset(slug)
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def versions(slug: str) -> list[dict]:
|
|
416
|
+
return _default.versions(slug)
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def rows(slug: str, where: Mapping[str, Any] | None = None, **kw) -> Rows:
|
|
420
|
+
return _default.rows(slug, where, **kw)
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def aggregate(slug: str, group=None, metric="count", where=None, **kw) -> Rows:
|
|
424
|
+
return _default.aggregate(slug, group, metric, where, **kw)
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
def download(slug: str, format: str = "parquet", version: str | None = None, path=None) -> Path:
|
|
428
|
+
return _default.download(slug, format, version, path)
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def read(slug: str, version: str | None = None):
|
|
432
|
+
return _default.read(slug, version)
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
for _f in (datasets, dataset, versions, rows, aggregate, download, read):
|
|
436
|
+
_f.__doc__ = getattr(Client, _f.__name__).__doc__
|
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
import io
|
|
2
|
+
import json
|
|
3
|
+
import os
|
|
4
|
+
import threading
|
|
5
|
+
import urllib.parse
|
|
6
|
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
7
|
+
|
|
8
|
+
import pytest
|
|
9
|
+
|
|
10
|
+
import publicdata_au as pd_au
|
|
11
|
+
|
|
12
|
+
META = {
|
|
13
|
+
"version": "2026-08-07",
|
|
14
|
+
"attribution": "Publisher, CC BY 4.0.",
|
|
15
|
+
"cite": "Cite.",
|
|
16
|
+
"licence": {"id": "CC-BY-4.0"},
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class Handler(BaseHTTPRequestHandler):
|
|
21
|
+
hits: list = []
|
|
22
|
+
throttle = 0
|
|
23
|
+
|
|
24
|
+
def log_message(self, *a):
|
|
25
|
+
pass
|
|
26
|
+
|
|
27
|
+
def send(self, status, body, ctype="application/json", headers=()):
|
|
28
|
+
raw = body if isinstance(body, bytes) else json.dumps(body).encode()
|
|
29
|
+
self.send_response(status)
|
|
30
|
+
self.send_header("Content-Type", ctype)
|
|
31
|
+
for k, v in headers:
|
|
32
|
+
self.send_header(k, v)
|
|
33
|
+
self.send_header("Content-Length", str(len(raw)))
|
|
34
|
+
self.end_headers()
|
|
35
|
+
self.wfile.write(raw)
|
|
36
|
+
|
|
37
|
+
def do_GET(self):
|
|
38
|
+
u = urllib.parse.urlsplit(self.path)
|
|
39
|
+
q = dict(urllib.parse.parse_qsl(u.query))
|
|
40
|
+
Handler.hits.append((u.path, q, self.headers.get("User-Agent")))
|
|
41
|
+
base = f"http://{self.headers['Host']}"
|
|
42
|
+
if Handler.throttle:
|
|
43
|
+
Handler.throttle -= 1
|
|
44
|
+
return self.send(429, {"error": "slow down"}, headers=[("Retry-After", "0")])
|
|
45
|
+
if u.path == "/api/v1/datasets":
|
|
46
|
+
return self.send(200, {"results": [{"slug": "a", "q": q.get("q")}]})
|
|
47
|
+
if u.path in ("/api/v1/datasets/a/rows", "/api/v1/datasets/a/versions/2026-08-07/rows"):
|
|
48
|
+
off = int(q.get("offset", 0))
|
|
49
|
+
data = [{"n": i} for i in range(5)]
|
|
50
|
+
lim = int(q.get("limit", 100))
|
|
51
|
+
page = data[off : off + lim]
|
|
52
|
+
more = off + lim < len(data)
|
|
53
|
+
nxt = (
|
|
54
|
+
f"{base}{u.path}?{urllib.parse.urlencode({**q, 'offset': off + lim})}"
|
|
55
|
+
if more
|
|
56
|
+
else None
|
|
57
|
+
)
|
|
58
|
+
return self.send(
|
|
59
|
+
200, {"publicdata": META, "version_page": "vp", "rows": page, "next": nxt}
|
|
60
|
+
)
|
|
61
|
+
if u.path == "/api/v1/datasets/a/aggregate":
|
|
62
|
+
return self.send(
|
|
63
|
+
200, {"publicdata": META, "rows": [{"g": 1, "count": 2}], "next": None}
|
|
64
|
+
)
|
|
65
|
+
if u.path == "/api/v1/datasets/nope/rows":
|
|
66
|
+
return self.send(404, {"error": "No such dataset in the query API"})
|
|
67
|
+
if u.path == "/d/a/versions.json":
|
|
68
|
+
return self.send(200, {"versions": [{"version": "2026-08-07"}]})
|
|
69
|
+
if u.path == "/d/a/datapackage.json":
|
|
70
|
+
return self.send(
|
|
71
|
+
200, {"licenses": [{"name": "CC-BY-4.0"}], "publicdata:attribution": "Publisher."}
|
|
72
|
+
)
|
|
73
|
+
if u.path.startswith("/d/a/latest/"):
|
|
74
|
+
self.send_response(302)
|
|
75
|
+
self.send_header("Location", u.path.replace("/latest/", "/v/2026-08-07/"))
|
|
76
|
+
self.send_header("Content-Length", "0")
|
|
77
|
+
self.end_headers()
|
|
78
|
+
return
|
|
79
|
+
if u.path == "/d/a/v/2026-08-07/data.csv":
|
|
80
|
+
return self.send(200, b"n\n1\n", "text/csv")
|
|
81
|
+
if u.path == "/d/a/v/2026-08-07/data.parquet":
|
|
82
|
+
import pyarrow as pa
|
|
83
|
+
import pyarrow.parquet as pq
|
|
84
|
+
|
|
85
|
+
buf = io.BytesIO()
|
|
86
|
+
header = {
|
|
87
|
+
**META,
|
|
88
|
+
"version": "2026-08-07",
|
|
89
|
+
"not_endorsed": "The publisher has not endorsed this site.",
|
|
90
|
+
}
|
|
91
|
+
t = pa.table({"n": [1, 2, 3]}).replace_schema_metadata(
|
|
92
|
+
{"publicdata": json.dumps(header)}
|
|
93
|
+
)
|
|
94
|
+
pq.write_table(t, buf)
|
|
95
|
+
return self.send(200, buf.getvalue(), "application/vnd.apache.parquet")
|
|
96
|
+
self.send(404, {"error": "not here"})
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
@pytest.fixture
|
|
100
|
+
def client():
|
|
101
|
+
Handler.hits = []
|
|
102
|
+
Handler.throttle = 0
|
|
103
|
+
srv = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
|
|
104
|
+
t = threading.Thread(target=srv.serve_forever, daemon=True)
|
|
105
|
+
t.start()
|
|
106
|
+
yield pd_au.Client(f"http://127.0.0.1:{srv.server_port}", retries=2)
|
|
107
|
+
srv.shutdown()
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def test_filters_render_in_the_apis_operator_form():
|
|
111
|
+
assert str(pd_au.gte(2020)) == "gte.2020"
|
|
112
|
+
assert str(pd_au.in_("QLD", "NSW")) == "in.(QLD,NSW)"
|
|
113
|
+
assert str(pd_au.in_(["a", "b"])) == "in.(a,b)"
|
|
114
|
+
assert str(pd_au.not_(pd_au.eq("x"))) == "not.eq.x"
|
|
115
|
+
assert str(pd_au.is_null()) == "is.null"
|
|
116
|
+
assert str(pd_au.ilike("*rider*")) == "ilike.*rider*"
|
|
117
|
+
assert str(pd_au.eq(True)) == "eq.true"
|
|
118
|
+
with pytest.raises(ValueError, match="comma"):
|
|
119
|
+
pd_au.in_("a,b")
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def test_rows_sends_where_select_and_order(client):
|
|
123
|
+
r = client.rows(
|
|
124
|
+
"a",
|
|
125
|
+
{"state": "QLD", "year": pd_au.gte(2020), "lga": None, "sex": ["F", "M"]},
|
|
126
|
+
select=["n", "m"],
|
|
127
|
+
order="n.desc",
|
|
128
|
+
limit=2,
|
|
129
|
+
)
|
|
130
|
+
path, q, ua = Handler.hits[-1]
|
|
131
|
+
assert path == "/api/v1/datasets/a/rows"
|
|
132
|
+
assert q == {
|
|
133
|
+
"state": "eq.QLD",
|
|
134
|
+
"year": "gte.2020",
|
|
135
|
+
"lga": "is.null",
|
|
136
|
+
"sex": "in.(F,M)",
|
|
137
|
+
"select": "n,m",
|
|
138
|
+
"order": "n.desc",
|
|
139
|
+
"limit": "2",
|
|
140
|
+
}
|
|
141
|
+
assert ua.startswith("publicdata-au-python/")
|
|
142
|
+
assert r == [{"n": 0}, {"n": 1}] and r.page["next"]
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def test_rows_carries_provenance(client):
|
|
146
|
+
r = client.rows("a")
|
|
147
|
+
assert (
|
|
148
|
+
r.version == "2026-08-07" and r.attribution and r.cite and r.licence == {"id": "CC-BY-4.0"}
|
|
149
|
+
)
|
|
150
|
+
assert r.version_page == "vp"
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def test_all_follows_every_page(client):
|
|
154
|
+
r = client.rows("a", all=True, limit=2)
|
|
155
|
+
assert [x["n"] for x in r] == [0, 1, 2, 3, 4] and r.page["next"] is None
|
|
156
|
+
assert len([h for h in Handler.hits if h[0].endswith("/rows")]) == 3
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def test_a_dated_version_goes_on_the_path(client):
|
|
160
|
+
client.rows("a", version="2026-08-07")
|
|
161
|
+
assert Handler.hits[-1][0] == "/api/v1/datasets/a/versions/2026-08-07/rows"
|
|
162
|
+
with pytest.raises(ValueError, match="date"):
|
|
163
|
+
client.rows("a", version="latest")
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def test_aggregate(client):
|
|
167
|
+
a = client.aggregate("a", group=["g", "h"], metric=["count", "sum.n"], where={"x": 1})
|
|
168
|
+
assert Handler.hits[-1][1] == {"group": "g,h", "metric": "count,sum.n", "x": "eq.1"}
|
|
169
|
+
assert a == [{"g": 1, "count": 2}] and a.version == "2026-08-07"
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def test_429_waits_then_retries(client):
|
|
173
|
+
Handler.throttle = 2
|
|
174
|
+
assert client.datasets("crash") == [{"slug": "a", "q": "crash"}]
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def test_429_past_the_retries_raises(client):
|
|
178
|
+
Handler.throttle = 5
|
|
179
|
+
with pytest.raises(pd_au.PublicDataError) as err:
|
|
180
|
+
client.datasets()
|
|
181
|
+
assert err.value.status == 429
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def test_an_error_names_the_status_and_the_apis_message(client):
|
|
185
|
+
with pytest.raises(pd_au.PublicDataError) as err:
|
|
186
|
+
client.rows("nope")
|
|
187
|
+
assert err.value.status == 404 and "No such dataset" in str(err.value)
|
|
188
|
+
assert err.value.body == {"error": "No such dataset in the query API"}
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def test_a_bad_slug_never_reaches_the_network(client):
|
|
192
|
+
with pytest.raises(ValueError):
|
|
193
|
+
client.rows("../etc")
|
|
194
|
+
assert Handler.hits == []
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def test_download_follows_latest_and_names_the_version(client, tmp_path, monkeypatch):
|
|
198
|
+
monkeypatch.chdir(tmp_path)
|
|
199
|
+
p = client.download("a", "csv")
|
|
200
|
+
assert p.name == "a-2026-08-07.csv" and p.read_bytes() == b"n\n1\n"
|
|
201
|
+
with pytest.raises(ValueError, match="format"):
|
|
202
|
+
client.download("a", "docx")
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def test_read_takes_provenance_from_the_file_it_read(client):
|
|
206
|
+
pytest.importorskip("pandas")
|
|
207
|
+
df = client.read("a", version="2026-08-07")
|
|
208
|
+
assert list(df["n"]) == [1, 2, 3]
|
|
209
|
+
assert df.attrs["publicdata"]["version"] == "2026-08-07"
|
|
210
|
+
assert df.attrs["publicdata"]["attribution"] == "Publisher, CC BY 4.0."
|
|
211
|
+
assert "has not endorsed" in df.attrs["publicdata"]["not_endorsed"]
|
|
212
|
+
assert not any(h[0].endswith("datapackage.json") for h in Handler.hits)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def test_an_unreachable_site_raises_the_packages_own_error():
|
|
216
|
+
import socket
|
|
217
|
+
|
|
218
|
+
s = socket.socket()
|
|
219
|
+
s.bind(("127.0.0.1", 0))
|
|
220
|
+
port = s.getsockname()[1]
|
|
221
|
+
s.close()
|
|
222
|
+
with pytest.raises(pd_au.PublicDataError) as err:
|
|
223
|
+
pd_au.Client(f"http://127.0.0.1:{port}", timeout=2).datasets()
|
|
224
|
+
assert err.value.status == 0 and "could not reach" in str(err.value)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def test_versions_and_dataset(client):
|
|
228
|
+
assert client.versions("a") == [{"version": "2026-08-07"}]
|
|
229
|
+
assert client.dataset("a")["licenses"][0]["name"] == "CC-BY-4.0"
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def test_site_can_come_from_the_environment(monkeypatch):
|
|
233
|
+
monkeypatch.setenv("PUBLICDATA_SITE", "http://example.test/")
|
|
234
|
+
assert pd_au.Client().site == "http://example.test"
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
@pytest.mark.skipif(
|
|
238
|
+
not os.environ.get("PUBLICDATA_LIVE"), reason="set PUBLICDATA_LIVE=1 to query the live site"
|
|
239
|
+
)
|
|
240
|
+
def test_live_site():
|
|
241
|
+
slugs = [d["slug"] for d in pd_au.datasets()]
|
|
242
|
+
assert "au-road-deaths" in slugs
|
|
243
|
+
r = pd_au.rows("au-road-deaths", {"state": "QLD"}, limit=1)
|
|
244
|
+
assert len(r) == 1 and r.attribution
|