dclimate-tabular-py 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dclimate_tabular_py-0.1.0.dist-info/METADATA +126 -0
- dclimate_tabular_py-0.1.0.dist-info/RECORD +15 -0
- dclimate_tabular_py-0.1.0.dist-info/WHEEL +4 -0
- tabular_py/__init__.py +245 -0
- tabular_py/codec.py +114 -0
- tabular_py/errors.py +85 -0
- tabular_py/geo.py +167 -0
- tabular_py/geo_index.py +463 -0
- tabular_py/model.py +422 -0
- tabular_py/predicate.py +317 -0
- tabular_py/reader.py +691 -0
- tabular_py/source.py +212 -0
- tabular_py/station_dataset.py +571 -0
- tabular_py/station_index.py +161 -0
- tabular_py/wire.py +550 -0
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: dclimate-tabular-py
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Content-addressed tabular/telemetry manifest for climate data on IPFS (dclimate-tabular/0)
|
|
5
|
+
Project-URL: Homepage, https://github.com/dClimate/tabular-py
|
|
6
|
+
Project-URL: Repository, https://github.com/dClimate/tabular-py
|
|
7
|
+
License: MIT
|
|
8
|
+
Requires-Python: >=3.12
|
|
9
|
+
Requires-Dist: blake3>=0.4.1
|
|
10
|
+
Requires-Dist: dag-cbor>=0.3.3
|
|
11
|
+
Requires-Dist: httpx>=0.28.1
|
|
12
|
+
Requires-Dist: multiformats[full]>=0.3.1.post4
|
|
13
|
+
Requires-Dist: py-hamt>=3.6.0
|
|
14
|
+
Requires-Dist: pyarrow>=18.0.0
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
|
|
17
|
+
# tabular-py
|
|
18
|
+
|
|
19
|
+
Python reader for `dclimate-tabular/0` — content-addressed station/telemetry data on IPFS.
|
|
20
|
+
|
|
21
|
+
The Python counterpart to [`tabular-js`](https://github.com/dClimate/tabular-js), reading the
|
|
22
|
+
same format from the same CIDs. Built on [`py-hamt`](https://github.com/dClimate/py-hamt),
|
|
23
|
+
which supplies the HAMT and the content-addressed store the same way
|
|
24
|
+
`@dclimate/ipld-index` does for JS.
|
|
25
|
+
|
|
26
|
+
```
|
|
27
|
+
tabular-js -> @dclimate/ipld-index (HAMT, CAS, range reads)
|
|
28
|
+
tabular-py -> py-hamt (same, already existed)
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## Status
|
|
32
|
+
|
|
33
|
+
**Reader only.** Publishing, compaction, and rollup stay in `tabular-js` — the ETL that
|
|
34
|
+
writes these datasets is JS, and a second writer would be a second thing to keep
|
|
35
|
+
byte-identical for no current gain. Everything needed to *read* a dataset published by
|
|
36
|
+
`tabular-js` is here.
|
|
37
|
+
|
|
38
|
+
## Install
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
uv pip install dclimate-tabular-py
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## Usage
|
|
45
|
+
|
|
46
|
+
```python
|
|
47
|
+
import asyncio
|
|
48
|
+
from tabular_py import GatewayRangeSource, StationDataset
|
|
49
|
+
|
|
50
|
+
async def main():
|
|
51
|
+
source = GatewayRangeSource("https://ipfs-gateway.dclimate.net")
|
|
52
|
+
ds = await StationDataset.open(source, "bafyr4if2wbttslbxpzmro427j4l4nvcrxqo4tufuffqqmqz7afj2pxyu4a")
|
|
53
|
+
|
|
54
|
+
# nearest station to a point, then a year of readings
|
|
55
|
+
near = await ds.nearest(34.05, -118.24, max_km=100)
|
|
56
|
+
rows = await near.time_range("2024-01-01", "2024-12-31").elements("PRCP").rows()
|
|
57
|
+
for row in rows[:5]:
|
|
58
|
+
print(row.station_id, row.ts, row.values)
|
|
59
|
+
|
|
60
|
+
asyncio.run(main())
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
The chainable selection API mirrors `tabular-js`:
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
ds.select("USW00023174") # explicit station ids
|
|
67
|
+
ds.circle(34.05, -118.24, 50) # within 50 km
|
|
68
|
+
ds.rectangle(33.0, -119.0, 35.0, -117.0)
|
|
69
|
+
ds.polygon([[(lon, lat), ...]])
|
|
70
|
+
await ds.nearest(lat, lon) # async: reads the geo index
|
|
71
|
+
ds.time_range(start, end)
|
|
72
|
+
ds.elements("PRCP", "TMAX")
|
|
73
|
+
ds.where(gt("TMAX", 300)) # pushed down to fragment statistics
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Selections are immutable — each call returns a new `StationDataset`, so a base dataset can
|
|
77
|
+
be reused across queries.
|
|
78
|
+
|
|
79
|
+
Terminal operations:
|
|
80
|
+
|
|
81
|
+
| Call | Returns |
|
|
82
|
+
|---|---|
|
|
83
|
+
| `await ds.rows()` | `list[ResultRow]` |
|
|
84
|
+
| `await ds.to_records()` | `list[dict]` — `station_id`, `time`, `values` |
|
|
85
|
+
| `await ds.to_records("TMAX")` | `list[dict]` — `station_id`, `time`, `value` |
|
|
86
|
+
| `await ds.to_arrow()` | `pyarrow.Table` |
|
|
87
|
+
| `await ds.plan()` | `QueryPlan` — what *would* be fetched, without fetching |
|
|
88
|
+
| `await ds.list_stations()` | `list[StationInfo]` |
|
|
89
|
+
|
|
90
|
+
## How reads stay small
|
|
91
|
+
|
|
92
|
+
A query never scans the dataset. Three things prune before any Parquet byte is fetched:
|
|
93
|
+
|
|
94
|
+
1. **The station index** (a HAMT keyed by station id) resolves named stations directly.
|
|
95
|
+
2. **The geo projection** answers region queries by reading one or two shard blocks
|
|
96
|
+
instead of walking all 132k stations.
|
|
97
|
+
3. **Fragment statistics in the manifest** — per-column min/max and null counts — let a
|
|
98
|
+
predicate skip whole fragments unread.
|
|
99
|
+
|
|
100
|
+
What survives is fetched with HTTP range requests against the exact column-chunk byte
|
|
101
|
+
ranges the manifest records, so a query for one column of one year moves kilobytes.
|
|
102
|
+
|
|
103
|
+
### One deviation from `tabular-js`, and why
|
|
104
|
+
|
|
105
|
+
`tabular-js` synthesizes Parquet `FileMetaData` client-side from the manifest and reads a
|
|
106
|
+
fragment with **zero** footer fetches. PyArrow exposes no public `FileMetaData`
|
|
107
|
+
constructor, so that trick does not transfer.
|
|
108
|
+
|
|
109
|
+
Instead this reader fetches the footer by its manifest-recorded
|
|
110
|
+
`footer_offset`/`footer_length` in a single ranged GET, verifies it against the
|
|
111
|
+
manifest's `footer_digest`, and hands the parsed metadata to PyArrow. Cost is one extra
|
|
112
|
+
range request per fragment — and it buys a corruption check `tabular-js` does not
|
|
113
|
+
perform. Footers are cached per fragment CID, so a repeated query pays it once.
|
|
114
|
+
|
|
115
|
+
## Development
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
uv sync
|
|
119
|
+
uv run pytest # unit tests, no network
|
|
120
|
+
uv run pytest -m network # conformance against the live gateway
|
|
121
|
+
uv run ruff check . && uv run mypy tabular_py
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
`tests/test_conformance.py` reads a real published GHCNd dataset
|
|
125
|
+
(`bafyr4if2wbttslbxpzmro427j4l4nvcrxqo4tufuffqqmqz7afj2pxyu4a`, 132,437 stations,
|
|
126
|
+
1.15 B rows) and asserts this implementation agrees with `tabular-js` on decoded values.
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
tabular_py/__init__.py,sha256=srZln4BumpfdFIy4m2k37VLy1aNiyOflQxURMMAVNPk,5071
|
|
2
|
+
tabular_py/codec.py,sha256=R7pZCr_3X_8fIxDbRDcxcQGUifmLyerjUyNhh6AiYUQ,4284
|
|
3
|
+
tabular_py/errors.py,sha256=ZEKqfXRCcoWPnX8QHQwGM9B34f7HNCMYAwVeQgbH_p4,2709
|
|
4
|
+
tabular_py/geo.py,sha256=ehCu5uRRzhyzaT4DdwwaN6iCqjfoF7UD8Ey_NicdPdI,5771
|
|
5
|
+
tabular_py/geo_index.py,sha256=wI2957JzSszjjmydmsV5-OrDtlH7dFpvREesE6FDZ7I,16747
|
|
6
|
+
tabular_py/model.py,sha256=2qstncwWNCLRp_OPm2giCiaxG5DB3x9cq9ibPwKCXcU,13022
|
|
7
|
+
tabular_py/predicate.py,sha256=OzBXRoWCAXqsiv-iLsrHBnbFrusdEbGHaQb-Ioq_CVQ,11105
|
|
8
|
+
tabular_py/reader.py,sha256=rEDB62L_bDFIe5UZ-3-y534m66dFVmz6uWVWTWcCSM0,26466
|
|
9
|
+
tabular_py/source.py,sha256=bRw2EwknrAx92I2Fix5KZQ4pJ2HE1Rm3CoAlHhadCpk,7965
|
|
10
|
+
tabular_py/station_dataset.py,sha256=c-onhTzzcCdtv45CEVarZgAMpLJab6Sni3VoWnZ2SIM,23192
|
|
11
|
+
tabular_py/station_index.py,sha256=0ckMC5cu6o8CCCxMFwOolMEPUPp6CFGpqw93_YJpH4g,5901
|
|
12
|
+
tabular_py/wire.py,sha256=WKDmzVQJkDHvfucKVzaL2gt7FWDMRIC8tp3uYCXLqdY,19788
|
|
13
|
+
dclimate_tabular_py-0.1.0.dist-info/METADATA,sha256=Zpu909niwS3h6iG5HsPx2zz_vkDkwLXvG6X4qUGiFWY,4780
|
|
14
|
+
dclimate_tabular_py-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
15
|
+
dclimate_tabular_py-0.1.0.dist-info/RECORD,,
|
tabular_py/__init__.py
ADDED
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
"""tabular-py -- Python reader for ``dclimate-tabular/0``.
|
|
2
|
+
|
|
3
|
+
The Python counterpart to ``@dclimate/tabular``, reading the same content-addressed
|
|
4
|
+
station datasets from the same CIDs, built on ``py-hamt``.
|
|
5
|
+
|
|
6
|
+
from tabular_py import GatewayRangeSource, StationDataset
|
|
7
|
+
|
|
8
|
+
source = GatewayRangeSource("https://ipfs-gateway.dclimate.net")
|
|
9
|
+
ds = await StationDataset.open(source, root_cid)
|
|
10
|
+
rows = await ds.select("USW00023174").time_range("2024-01-01", "2024-12-31").rows()
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from .codec import (
|
|
16
|
+
BLOCK_HARD_LIMIT_BYTES,
|
|
17
|
+
BLOCK_TARGET_BYTES,
|
|
18
|
+
blake3_digest,
|
|
19
|
+
cid_for_bytes,
|
|
20
|
+
cid_str,
|
|
21
|
+
decode_block,
|
|
22
|
+
encode_block,
|
|
23
|
+
make_block,
|
|
24
|
+
verify_cid,
|
|
25
|
+
)
|
|
26
|
+
from .errors import (
|
|
27
|
+
BlockSizeError,
|
|
28
|
+
CidMismatchError,
|
|
29
|
+
CodecError,
|
|
30
|
+
DatasetIntegrityError,
|
|
31
|
+
DatasetReaderError,
|
|
32
|
+
DclimateTabularError,
|
|
33
|
+
GeoFilterError,
|
|
34
|
+
GeoIndexError,
|
|
35
|
+
PredicateError,
|
|
36
|
+
RangeSourceError,
|
|
37
|
+
StationIndexError,
|
|
38
|
+
StationSelectionError,
|
|
39
|
+
WireError,
|
|
40
|
+
)
|
|
41
|
+
from .geo import (
|
|
42
|
+
BBox,
|
|
43
|
+
Circle,
|
|
44
|
+
GeoFilter,
|
|
45
|
+
Polygon,
|
|
46
|
+
haversine_meters,
|
|
47
|
+
matches_geo_filter,
|
|
48
|
+
validate_geo_filter,
|
|
49
|
+
)
|
|
50
|
+
from .geo_index import (
|
|
51
|
+
bounding_boxes_for_filter,
|
|
52
|
+
first_accepted_in_order,
|
|
53
|
+
lookup_geo_stations,
|
|
54
|
+
nearest_station,
|
|
55
|
+
nearest_station_where,
|
|
56
|
+
)
|
|
57
|
+
from .model import (
|
|
58
|
+
SPEC_VERSION,
|
|
59
|
+
CellValue,
|
|
60
|
+
ColumnStat,
|
|
61
|
+
DatasetRoot,
|
|
62
|
+
FragmentEntry,
|
|
63
|
+
GeoProjection,
|
|
64
|
+
GeoShardRef,
|
|
65
|
+
HamtStationIndex,
|
|
66
|
+
InlineStationIndex,
|
|
67
|
+
StationEntry,
|
|
68
|
+
StationIndex,
|
|
69
|
+
StationSummary,
|
|
70
|
+
TableField,
|
|
71
|
+
TableSchema,
|
|
72
|
+
columns_from_fragments,
|
|
73
|
+
correction_schema,
|
|
74
|
+
default_column_key,
|
|
75
|
+
)
|
|
76
|
+
from .predicate import (
|
|
77
|
+
And,
|
|
78
|
+
Comparison,
|
|
79
|
+
Not,
|
|
80
|
+
NullCheck,
|
|
81
|
+
Or,
|
|
82
|
+
Predicate,
|
|
83
|
+
and_,
|
|
84
|
+
eq,
|
|
85
|
+
evaluate_row_predicate,
|
|
86
|
+
evaluate_stats_predicate,
|
|
87
|
+
ge,
|
|
88
|
+
gt,
|
|
89
|
+
is_null,
|
|
90
|
+
le,
|
|
91
|
+
lt,
|
|
92
|
+
ne,
|
|
93
|
+
not_,
|
|
94
|
+
not_null,
|
|
95
|
+
or_,
|
|
96
|
+
predicate_elements,
|
|
97
|
+
)
|
|
98
|
+
from .reader import (
|
|
99
|
+
DatasetReader,
|
|
100
|
+
PlanRange,
|
|
101
|
+
Query,
|
|
102
|
+
QueryPlan,
|
|
103
|
+
QueryPlanFragment,
|
|
104
|
+
ResultRow,
|
|
105
|
+
)
|
|
106
|
+
from .source import (
|
|
107
|
+
CasRangeSource,
|
|
108
|
+
CountingSource,
|
|
109
|
+
GatewayRangeSource,
|
|
110
|
+
MemoryRangeSource,
|
|
111
|
+
RangeSource,
|
|
112
|
+
)
|
|
113
|
+
from .station_dataset import NearestStation, StationDataset, StationInfo, WithinRange
|
|
114
|
+
from .station_index import (
|
|
115
|
+
list_station_entries,
|
|
116
|
+
list_stations,
|
|
117
|
+
lookup_station,
|
|
118
|
+
station_count,
|
|
119
|
+
)
|
|
120
|
+
from .wire import (
|
|
121
|
+
fragment_from_wire,
|
|
122
|
+
fragment_to_wire,
|
|
123
|
+
geo_projection_from_wire,
|
|
124
|
+
geo_projection_to_wire,
|
|
125
|
+
geo_shard_from_wire,
|
|
126
|
+
geo_shard_to_wire,
|
|
127
|
+
root_from_wire,
|
|
128
|
+
root_to_wire,
|
|
129
|
+
station_entry_from_wire,
|
|
130
|
+
station_entry_to_wire,
|
|
131
|
+
station_summary_from_wire,
|
|
132
|
+
station_summary_to_wire,
|
|
133
|
+
table_schema_from_wire,
|
|
134
|
+
table_schema_to_wire,
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
__version__ = "0.1.0"
|
|
138
|
+
|
|
139
|
+
__all__ = [
|
|
140
|
+
"BLOCK_HARD_LIMIT_BYTES",
|
|
141
|
+
"BLOCK_TARGET_BYTES",
|
|
142
|
+
"SPEC_VERSION",
|
|
143
|
+
"And",
|
|
144
|
+
"BBox",
|
|
145
|
+
"BlockSizeError",
|
|
146
|
+
"CasRangeSource",
|
|
147
|
+
"CellValue",
|
|
148
|
+
"CidMismatchError",
|
|
149
|
+
"Circle",
|
|
150
|
+
"CodecError",
|
|
151
|
+
"ColumnStat",
|
|
152
|
+
"Comparison",
|
|
153
|
+
"CountingSource",
|
|
154
|
+
"DatasetIntegrityError",
|
|
155
|
+
"DatasetReader",
|
|
156
|
+
"DatasetReaderError",
|
|
157
|
+
"DatasetRoot",
|
|
158
|
+
"DclimateTabularError",
|
|
159
|
+
"FragmentEntry",
|
|
160
|
+
"GatewayRangeSource",
|
|
161
|
+
"GeoFilter",
|
|
162
|
+
"GeoFilterError",
|
|
163
|
+
"GeoIndexError",
|
|
164
|
+
"GeoProjection",
|
|
165
|
+
"GeoShardRef",
|
|
166
|
+
"HamtStationIndex",
|
|
167
|
+
"InlineStationIndex",
|
|
168
|
+
"MemoryRangeSource",
|
|
169
|
+
"NearestStation",
|
|
170
|
+
"Not",
|
|
171
|
+
"NullCheck",
|
|
172
|
+
"Or",
|
|
173
|
+
"PlanRange",
|
|
174
|
+
"Polygon",
|
|
175
|
+
"Predicate",
|
|
176
|
+
"PredicateError",
|
|
177
|
+
"Query",
|
|
178
|
+
"QueryPlan",
|
|
179
|
+
"QueryPlanFragment",
|
|
180
|
+
"RangeSource",
|
|
181
|
+
"RangeSourceError",
|
|
182
|
+
"ResultRow",
|
|
183
|
+
"StationDataset",
|
|
184
|
+
"StationEntry",
|
|
185
|
+
"StationIndex",
|
|
186
|
+
"StationIndexError",
|
|
187
|
+
"StationInfo",
|
|
188
|
+
"StationSelectionError",
|
|
189
|
+
"StationSummary",
|
|
190
|
+
"TableField",
|
|
191
|
+
"TableSchema",
|
|
192
|
+
"WireError",
|
|
193
|
+
"WithinRange",
|
|
194
|
+
"__version__",
|
|
195
|
+
"and_",
|
|
196
|
+
"blake3_digest",
|
|
197
|
+
"bounding_boxes_for_filter",
|
|
198
|
+
"cid_for_bytes",
|
|
199
|
+
"cid_str",
|
|
200
|
+
"columns_from_fragments",
|
|
201
|
+
"correction_schema",
|
|
202
|
+
"decode_block",
|
|
203
|
+
"default_column_key",
|
|
204
|
+
"encode_block",
|
|
205
|
+
"eq",
|
|
206
|
+
"first_accepted_in_order",
|
|
207
|
+
"evaluate_row_predicate",
|
|
208
|
+
"evaluate_stats_predicate",
|
|
209
|
+
"fragment_from_wire",
|
|
210
|
+
"fragment_to_wire",
|
|
211
|
+
"ge",
|
|
212
|
+
"geo_projection_from_wire",
|
|
213
|
+
"geo_projection_to_wire",
|
|
214
|
+
"geo_shard_from_wire",
|
|
215
|
+
"geo_shard_to_wire",
|
|
216
|
+
"gt",
|
|
217
|
+
"haversine_meters",
|
|
218
|
+
"is_null",
|
|
219
|
+
"le",
|
|
220
|
+
"list_station_entries",
|
|
221
|
+
"list_stations",
|
|
222
|
+
"lookup_geo_stations",
|
|
223
|
+
"lookup_station",
|
|
224
|
+
"lt",
|
|
225
|
+
"make_block",
|
|
226
|
+
"matches_geo_filter",
|
|
227
|
+
"ne",
|
|
228
|
+
"nearest_station",
|
|
229
|
+
"nearest_station_where",
|
|
230
|
+
"not_",
|
|
231
|
+
"not_null",
|
|
232
|
+
"or_",
|
|
233
|
+
"predicate_elements",
|
|
234
|
+
"root_from_wire",
|
|
235
|
+
"root_to_wire",
|
|
236
|
+
"station_count",
|
|
237
|
+
"station_entry_from_wire",
|
|
238
|
+
"station_entry_to_wire",
|
|
239
|
+
"station_summary_from_wire",
|
|
240
|
+
"station_summary_to_wire",
|
|
241
|
+
"table_schema_from_wire",
|
|
242
|
+
"table_schema_to_wire",
|
|
243
|
+
"validate_geo_filter",
|
|
244
|
+
"verify_cid",
|
|
245
|
+
]
|
tabular_py/codec.py
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""Canonical DAG-CBOR blocks and blake3 CIDs.
|
|
2
|
+
|
|
3
|
+
Ported from ``tabular-js`` ``src/codec.ts``.
|
|
4
|
+
|
|
5
|
+
Every block in the format is strict canonical DAG-CBOR addressed by a CIDv1 with
|
|
6
|
+
a 32-byte blake3 multihash. Decoding re-encodes and compares, so a block that
|
|
7
|
+
round-trips to different bytes is rejected rather than silently accepted -- two
|
|
8
|
+
encodings of the same value would otherwise produce two CIDs for one logical
|
|
9
|
+
block, and structural dedup across commits depends on that not happening.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from typing import Any, Literal
|
|
15
|
+
|
|
16
|
+
import dag_cbor
|
|
17
|
+
from blake3 import blake3
|
|
18
|
+
from multiformats import CID
|
|
19
|
+
|
|
20
|
+
from .errors import BlockSizeError, CodecError
|
|
21
|
+
|
|
22
|
+
# Blocks target 512 KiB and must never exceed 1 MiB. The target is what the
|
|
23
|
+
# builders split on; the hard limit is what the format guarantees.
|
|
24
|
+
BLOCK_TARGET_BYTES = 512 * 1024
|
|
25
|
+
BLOCK_HARD_LIMIT_BYTES = 1024 * 1024
|
|
26
|
+
|
|
27
|
+
BLAKE3_CODE = 0x1E
|
|
28
|
+
DAG_CBOR_CODE = 0x71
|
|
29
|
+
RAW_CODE = 0x55
|
|
30
|
+
DAG_PB_CODE = 0x70
|
|
31
|
+
|
|
32
|
+
Codec = Literal["dag-cbor", "raw"]
|
|
33
|
+
|
|
34
|
+
_CODEC_NAMES: dict[str, int] = {"dag-cbor": DAG_CBOR_CODE, "raw": RAW_CODE}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def blake3_digest(data: bytes) -> bytes:
|
|
38
|
+
"""32-byte blake3, the only hash this format uses."""
|
|
39
|
+
return blake3(data).digest(length=32)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def encode_block(value: Any) -> bytes:
|
|
43
|
+
"""Encode a wire value to canonical DAG-CBOR."""
|
|
44
|
+
try:
|
|
45
|
+
return dag_cbor.encode(value)
|
|
46
|
+
except Exception as error: # noqa: BLE001 - re-raised with context below
|
|
47
|
+
raise CodecError(f"DAG-CBOR encode failed: {error}") from error
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def decode_block(data: bytes) -> Any:
|
|
51
|
+
"""Decode canonical DAG-CBOR, rejecting any non-canonical encoding.
|
|
52
|
+
|
|
53
|
+
The re-encode check is the reason this exists rather than a bare
|
|
54
|
+
``dag_cbor.decode``. A block whose bytes are not the canonical encoding of
|
|
55
|
+
their own value cannot be re-derived from that value, so its CID could never
|
|
56
|
+
be reproduced -- accepting it would let a dataset carry blocks this
|
|
57
|
+
implementation can read but not rewrite.
|
|
58
|
+
"""
|
|
59
|
+
try:
|
|
60
|
+
value = dag_cbor.decode(data)
|
|
61
|
+
except Exception as error: # noqa: BLE001 - re-raised with context below
|
|
62
|
+
raise CodecError(f"DAG-CBOR decode failed: {error}") from error
|
|
63
|
+
|
|
64
|
+
try:
|
|
65
|
+
canonical = dag_cbor.encode(value)
|
|
66
|
+
except Exception as error: # noqa: BLE001 - re-raised with context below
|
|
67
|
+
raise CodecError(f"DAG-CBOR decode failed: re-encode failed: {error}") from error
|
|
68
|
+
|
|
69
|
+
if canonical != data:
|
|
70
|
+
raise CodecError("DAG-CBOR decode failed: input is not canonical DAG-CBOR")
|
|
71
|
+
return value
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def cid_for_bytes(data: bytes, codec: Codec) -> CID:
|
|
75
|
+
"""The CIDv1 that addresses ``data`` under ``codec``."""
|
|
76
|
+
code = _CODEC_NAMES.get(codec)
|
|
77
|
+
if code is None:
|
|
78
|
+
raise CodecError(f"Unsupported codec: {codec}")
|
|
79
|
+
digest = blake3_digest(data)
|
|
80
|
+
# multiformats wants the full multihash (code + length prefix + digest).
|
|
81
|
+
multihash = bytes([BLAKE3_CODE, len(digest)]) + digest
|
|
82
|
+
return CID("base32", 1, code, multihash)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def make_block(value: Any) -> tuple[CID, bytes]:
|
|
86
|
+
"""Encode a wire value and address it, enforcing the hard block limit."""
|
|
87
|
+
data = encode_block(value)
|
|
88
|
+
if len(data) > BLOCK_HARD_LIMIT_BYTES:
|
|
89
|
+
raise BlockSizeError(
|
|
90
|
+
f"Block is {len(data)} bytes, above the {BLOCK_HARD_LIMIT_BYTES}-byte hard limit"
|
|
91
|
+
)
|
|
92
|
+
return cid_for_bytes(data, "dag-cbor"), data
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def cid_str(cid: CID) -> str:
|
|
96
|
+
"""Base32 string form, the spelling used for gateway paths and dict keys."""
|
|
97
|
+
return str(cid.set(base="base32"))
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def verify_cid(cid: CID, data: bytes) -> None:
|
|
101
|
+
"""Raise unless ``data`` hashes to ``cid``.
|
|
102
|
+
|
|
103
|
+
Only meaningful for whole blocks. Ranged reads return a slice, which by
|
|
104
|
+
construction does not hash to the block's CID; those are covered by the
|
|
105
|
+
fragment footer digest instead.
|
|
106
|
+
"""
|
|
107
|
+
if cid.codec.code == DAG_PB_CODE:
|
|
108
|
+
# UnixFS: the CID addresses a dag-pb node whose digest covers the DAG
|
|
109
|
+
# structure, not the concatenated file bytes, so this check does not
|
|
110
|
+
# apply. The caller gets the same guarantee from the footer digest.
|
|
111
|
+
return
|
|
112
|
+
# `.raw_digest` is the hash bytes; `.digest` carries the multihash prefix.
|
|
113
|
+
if blake3_digest(data) != bytes(cid.raw_digest):
|
|
114
|
+
raise CodecError(f"Block bytes do not match {cid_str(cid)}")
|
tabular_py/errors.py
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""Error hierarchy, ported from ``tabular-js`` ``src/errors.ts``.
|
|
2
|
+
|
|
3
|
+
The shape matters as much as the names: ``dclimate-client-py`` will translate
|
|
4
|
+
these into its own error types the same way ``dclimate-client-js`` does, so a
|
|
5
|
+
class moving between branches of this tree changes how a caller's ``except``
|
|
6
|
+
behaves.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class DclimateTabularError(Exception):
|
|
13
|
+
"""Base for everything this package raises."""
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class CodecError(DclimateTabularError):
|
|
17
|
+
"""DAG-CBOR encode/decode failed, or the bytes were not canonical."""
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class BlockSizeError(DclimateTabularError):
|
|
21
|
+
"""A block exceeded the hard limit."""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class WireError(DclimateTabularError):
|
|
25
|
+
"""A wire value did not match the schema for its position in the manifest."""
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class CidMismatchError(DclimateTabularError):
|
|
29
|
+
"""Fetched bytes did not hash to the CID that addressed them."""
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class RangeSourceError(DclimateTabularError):
|
|
33
|
+
"""A range request could not be satisfied."""
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class StationIndexError(DclimateTabularError):
|
|
37
|
+
"""The station index could not be read or built."""
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class GeoIndexError(DclimateTabularError):
|
|
41
|
+
"""The geo projection could not be read or built."""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class GeoFilterError(DclimateTabularError):
|
|
45
|
+
"""A geo filter was malformed."""
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class PredicateError(DclimateTabularError):
|
|
49
|
+
"""A predicate was malformed."""
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class DatasetReaderError(DclimateTabularError):
|
|
53
|
+
"""The caller asked for something this dataset cannot answer.
|
|
54
|
+
|
|
55
|
+
A bad request: an unknown column, an incoherent predicate, a selection that
|
|
56
|
+
names no station. Consumers translate this wholesale into their own
|
|
57
|
+
"invalid request" type.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class DatasetIntegrityError(DclimateTabularError):
|
|
62
|
+
"""The dataset contradicts itself.
|
|
63
|
+
|
|
64
|
+
A root naming a schema that is not there, a correction fragment whose commit
|
|
65
|
+
sequence contradicts the entry pointing at it, a fragment whose bytes do not
|
|
66
|
+
match the manifest that describes them.
|
|
67
|
+
|
|
68
|
+
Deliberately **not** a subclass of :class:`DatasetReaderError`: consumers
|
|
69
|
+
translate that type wholesale into their own "invalid request" error, so
|
|
70
|
+
inheriting would report a corrupt dataset as the caller's mistake and send
|
|
71
|
+
them looking for a bug they did not write.
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class StationSelectionError(DatasetReaderError):
|
|
76
|
+
"""A station selection could not be resolved.
|
|
77
|
+
|
|
78
|
+
``reason`` distinguishes "this station is not in this dataset" from "this
|
|
79
|
+
selection is malformed", because the client maps the first to a not-found
|
|
80
|
+
error and the second to an invalid-selection error.
|
|
81
|
+
"""
|
|
82
|
+
|
|
83
|
+
def __init__(self, message: str, reason: str = "invalid") -> None:
|
|
84
|
+
super().__init__(message)
|
|
85
|
+
self.reason = reason
|