crc-sdk 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crc_sdk/__init__.py +31 -0
- crc_sdk/connectors/__init__.py +35 -0
- crc_sdk/connectors/adapters.py +285 -0
- crc_sdk/connectors/duckdb/__init__.py +35 -0
- crc_sdk/connectors/duckdb/connection.py +366 -0
- crc_sdk/connectors/duckdb/geotiff.py +617 -0
- crc_sdk/connectors/duckdb/zarr.py +461 -0
- crc_sdk/connectors/parquet.py +323 -0
- crc_sdk/connectors/protocols.py +13 -0
- crc_sdk/core/__init__.py +95 -0
- crc_sdk/fitting/__init__.py +31 -0
- crc_sdk/fitting/workflows.py +4 -0
- crc_sdk/geometry/__init__.py +117 -0
- crc_sdk/geometry/admin.py +462 -0
- crc_sdk/geometry/coverage.py +1201 -0
- crc_sdk/geometry/formats.py +91 -0
- crc_sdk/geometry/h3.py +418 -0
- crc_sdk/geometry/pmtiles/__init__.py +49 -0
- crc_sdk/geometry/pmtiles/_build.py +185 -0
- crc_sdk/geometry/pmtiles/_geojson_sql.py +221 -0
- crc_sdk/geometry/pmtiles/_process.py +195 -0
- crc_sdk/geometry/pmtiles/archive.py +153 -0
- crc_sdk/geometry/pmtiles/binaries.py +38 -0
- crc_sdk/geometry/pmtiles/budget.py +243 -0
- crc_sdk/geometry/pmtiles/presets.py +231 -0
- crc_sdk/geometry/vector.py +288 -0
- crc_sdk/impacts/__init__.py +25 -0
- crc_sdk/impacts/custom.py +5 -0
- crc_sdk/providers/__init__.py +21 -0
- crc_sdk/providers/local.py +36 -0
- crc_sdk/providers/metadata.py +1 -0
- crc_sdk/providers/os_climate.py +270 -0
- crc_sdk/providers/protocol.py +22 -0
- crc_sdk/py.typed +1 -0
- crc_sdk/schema/__init__.py +17 -0
- crc_sdk/schema/hazard_field.py +41 -0
- crc_sdk/types/__init__.py +20 -0
- crc_sdk/types/dataset.py +75 -0
- crc_sdk/types/geometry.py +12 -0
- crc_sdk/types/hazard.py +87 -0
- crc_sdk/types/storage.py +11 -0
- crc_sdk/workflows/__init__.py +55 -0
- crc_sdk/workflows/_portfolio.py +407 -0
- crc_sdk/workflows/distributions.py +317 -0
- crc_sdk/workflows/portfolio.py +386 -0
- crc_sdk/workflows/tiling.py +322 -0
- crc_sdk-0.1.0.dist-info/METADATA +374 -0
- crc_sdk-0.1.0.dist-info/RECORD +51 -0
- crc_sdk-0.1.0.dist-info/WHEEL +5 -0
- crc_sdk-0.1.0.dist-info/licenses/LICENSE +18 -0
- crc_sdk-0.1.0.dist-info/top_level.txt +1 -0
crc_sdk/__init__.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""Higher-level Python SDK for Climate Risk Commons."""
|
|
2
|
+
|
|
3
|
+
from .core import (
|
|
4
|
+
Distribution,
|
|
5
|
+
EmpiricalDistribution,
|
|
6
|
+
FittedDistribution,
|
|
7
|
+
HurdleDistribution,
|
|
8
|
+
HurdleQuantileFitResult,
|
|
9
|
+
QuantileFitResult,
|
|
10
|
+
TabulatedDistribution,
|
|
11
|
+
fit_distribution,
|
|
12
|
+
fit_hurdle_quantiles,
|
|
13
|
+
fit_quantiles,
|
|
14
|
+
)
|
|
15
|
+
from .providers import LocalProvider, OSClimateProvider, Provider
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"Distribution",
|
|
19
|
+
"EmpiricalDistribution",
|
|
20
|
+
"FittedDistribution",
|
|
21
|
+
"HurdleDistribution",
|
|
22
|
+
"HurdleQuantileFitResult",
|
|
23
|
+
"LocalProvider",
|
|
24
|
+
"OSClimateProvider",
|
|
25
|
+
"Provider",
|
|
26
|
+
"QuantileFitResult",
|
|
27
|
+
"TabulatedDistribution",
|
|
28
|
+
"fit_distribution",
|
|
29
|
+
"fit_hurdle_quantiles",
|
|
30
|
+
"fit_quantiles",
|
|
31
|
+
]
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""External format and query-engine connectors."""
|
|
2
|
+
|
|
3
|
+
from .adapters import (
|
|
4
|
+
CanonicalHazardBatch,
|
|
5
|
+
CanonicalHazardStream,
|
|
6
|
+
HurdleFitPolicy,
|
|
7
|
+
OSClimateIngestPolicy,
|
|
8
|
+
canonicalize_os_climate,
|
|
9
|
+
)
|
|
10
|
+
from .parquet import (
|
|
11
|
+
hazard_arrow_schema,
|
|
12
|
+
read_hazard_dataset,
|
|
13
|
+
read_hazard_metadata,
|
|
14
|
+
sort_hazard_table,
|
|
15
|
+
validate_hazard_table,
|
|
16
|
+
write_hazard_dataset,
|
|
17
|
+
write_hazard_stream,
|
|
18
|
+
)
|
|
19
|
+
from .protocols import HazardReader
|
|
20
|
+
|
|
21
|
+
__all__ = [
|
|
22
|
+
"HazardReader",
|
|
23
|
+
"CanonicalHazardBatch",
|
|
24
|
+
"CanonicalHazardStream",
|
|
25
|
+
"HurdleFitPolicy",
|
|
26
|
+
"OSClimateIngestPolicy",
|
|
27
|
+
"canonicalize_os_climate",
|
|
28
|
+
"hazard_arrow_schema",
|
|
29
|
+
"read_hazard_dataset",
|
|
30
|
+
"read_hazard_metadata",
|
|
31
|
+
"sort_hazard_table",
|
|
32
|
+
"validate_hazard_table",
|
|
33
|
+
"write_hazard_dataset",
|
|
34
|
+
"write_hazard_stream",
|
|
35
|
+
]
|
|
@@ -0,0 +1,285 @@
|
|
|
1
|
+
"""Adapters from external connector results to canonical hazard rows."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterator
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from hashlib import sha256
|
|
8
|
+
from typing import Any, Literal, get_args
|
|
9
|
+
|
|
10
|
+
import numpy as np
|
|
11
|
+
import pyarrow as pa # type: ignore[import-untyped]
|
|
12
|
+
from crc_framework import (
|
|
13
|
+
FittedDistribution,
|
|
14
|
+
HurdleDistribution,
|
|
15
|
+
QuantileFitDiagnostics,
|
|
16
|
+
TabulatedDistribution,
|
|
17
|
+
fit_hurdle_quantiles,
|
|
18
|
+
fit_quantiles,
|
|
19
|
+
)
|
|
20
|
+
from crc_framework.distributions import DistributionFamily
|
|
21
|
+
|
|
22
|
+
from crc_sdk.connectors.duckdb.zarr import Bounds, RasterCurve, ZarrRaster
|
|
23
|
+
from crc_sdk.connectors.parquet import (
|
|
24
|
+
hazard_arrow_schema,
|
|
25
|
+
validate_hazard_table,
|
|
26
|
+
)
|
|
27
|
+
from crc_sdk.geometry import intersecting_cells
|
|
28
|
+
from crc_sdk.types import HazardDatasetMetadata, SourceProvenance
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True)
|
|
32
|
+
class HurdleFitPolicy:
|
|
33
|
+
"""Explicit point-mass policy for one external quantile dataset."""
|
|
34
|
+
|
|
35
|
+
atom_probability: float
|
|
36
|
+
atom_location: float = 0.0
|
|
37
|
+
|
|
38
|
+
def __post_init__(self) -> None:
|
|
39
|
+
if not 0.0 < self.atom_probability < 1.0:
|
|
40
|
+
raise ValueError("atom_probability must be strictly between zero and one")
|
|
41
|
+
if not np.isfinite(self.atom_location):
|
|
42
|
+
raise ValueError("atom_location must be finite")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass(frozen=True)
|
|
46
|
+
class OSClimateIngestPolicy:
|
|
47
|
+
"""Explicit policy controlling raster-to-canonical conversion."""
|
|
48
|
+
|
|
49
|
+
h3_resolution: int
|
|
50
|
+
family: DistributionFamily
|
|
51
|
+
producer: str
|
|
52
|
+
creation_version: str
|
|
53
|
+
tail: Literal["upper", "lower"] = "upper"
|
|
54
|
+
batch_rows: int = 65_536
|
|
55
|
+
value_semantics: str | None = None
|
|
56
|
+
source_version: str | None = None
|
|
57
|
+
hurdle: HurdleFitPolicy | None = None
|
|
58
|
+
maximum_normalized_rmse: float | None = None
|
|
59
|
+
maximum_absolute_residual: float | None = None
|
|
60
|
+
# Most pixels in an area (as opposed to a single known-exposed point) never
|
|
61
|
+
# exceed the hazard threshold and carry a constant, unfittable curve;
|
|
62
|
+
# "skip" drops those rather than aborting the whole area ingest.
|
|
63
|
+
on_fit_failure: Literal["raise", "skip"] = "raise"
|
|
64
|
+
|
|
65
|
+
def __post_init__(self) -> None:
|
|
66
|
+
if not 0 <= self.h3_resolution <= 15:
|
|
67
|
+
raise ValueError("H3 resolution must be between 0 and 15")
|
|
68
|
+
if self.family not in get_args(DistributionFamily):
|
|
69
|
+
raise ValueError(f"unknown distribution family {self.family!r}")
|
|
70
|
+
if not self.producer or not self.creation_version:
|
|
71
|
+
raise ValueError("producer and creation_version must be non-empty")
|
|
72
|
+
if self.batch_rows < 1:
|
|
73
|
+
raise ValueError("batch_rows must be positive")
|
|
74
|
+
if self.on_fit_failure not in ("raise", "skip"):
|
|
75
|
+
raise ValueError("on_fit_failure must be 'raise' or 'skip'")
|
|
76
|
+
for name, value in (
|
|
77
|
+
("maximum_normalized_rmse", self.maximum_normalized_rmse),
|
|
78
|
+
("maximum_absolute_residual", self.maximum_absolute_residual),
|
|
79
|
+
):
|
|
80
|
+
if value is not None and (not np.isfinite(value) or value < 0.0):
|
|
81
|
+
raise ValueError(f"{name} must be finite and non-negative")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
@dataclass(frozen=True)
|
|
85
|
+
class CanonicalHazardBatch:
|
|
86
|
+
"""One batch of canonical H3-expanded hazard rows."""
|
|
87
|
+
|
|
88
|
+
hazard_rows: Any
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclass
|
|
92
|
+
class CanonicalHazardStream:
|
|
93
|
+
"""Canonical metadata and one-shot Arrow batches."""
|
|
94
|
+
|
|
95
|
+
metadata: HazardDatasetMetadata
|
|
96
|
+
batches: Iterator[CanonicalHazardBatch]
|
|
97
|
+
|
|
98
|
+
def read_all(self) -> Any:
|
|
99
|
+
"""Consume this stream into one canonical Arrow table."""
|
|
100
|
+
batches = list(self.batches)
|
|
101
|
+
if batches:
|
|
102
|
+
return pa.concat_tables(
|
|
103
|
+
[batch.hazard_rows for batch in batches],
|
|
104
|
+
promote_options="none",
|
|
105
|
+
)
|
|
106
|
+
return pa.Table.from_batches(
|
|
107
|
+
[],
|
|
108
|
+
schema=hazard_arrow_schema(self.metadata),
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _source_id(raster: ZarrRaster, curve: RasterCurve) -> str:
|
|
113
|
+
identity = (
|
|
114
|
+
f"os-climate\0{raster.metadata.path}\0{curve.row}\0{curve.column}"
|
|
115
|
+
).encode()
|
|
116
|
+
return sha256(identity).hexdigest()
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _metadata(
|
|
120
|
+
raster: ZarrRaster, policy: OSClimateIngestPolicy
|
|
121
|
+
) -> HazardDatasetMetadata:
|
|
122
|
+
values = raster.metadata
|
|
123
|
+
return HazardDatasetMetadata(
|
|
124
|
+
h3_resolution=policy.h3_resolution,
|
|
125
|
+
value_unit=values.units,
|
|
126
|
+
value_semantics=policy.value_semantics or values.indicator_id,
|
|
127
|
+
producer=policy.producer,
|
|
128
|
+
creation_version=policy.creation_version,
|
|
129
|
+
source=SourceProvenance(
|
|
130
|
+
provider="os-climate",
|
|
131
|
+
dataset=f"{values.hazard_type}:{values.indicator_id}",
|
|
132
|
+
uri=values.path,
|
|
133
|
+
version=policy.source_version,
|
|
134
|
+
),
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _fit_curve(
|
|
139
|
+
tabulated: TabulatedDistribution,
|
|
140
|
+
policy: OSClimateIngestPolicy,
|
|
141
|
+
) -> tuple[Any, Any]:
|
|
142
|
+
distribution: FittedDistribution | HurdleDistribution
|
|
143
|
+
diagnostics: QuantileFitDiagnostics
|
|
144
|
+
if policy.hurdle is None:
|
|
145
|
+
quantile_result = fit_quantiles(tabulated, family=policy.family)
|
|
146
|
+
distribution = quantile_result.distribution
|
|
147
|
+
diagnostics = quantile_result.diagnostics
|
|
148
|
+
else:
|
|
149
|
+
hurdle_result = fit_hurdle_quantiles(
|
|
150
|
+
tabulated,
|
|
151
|
+
family=policy.family,
|
|
152
|
+
atom_probability=policy.hurdle.atom_probability,
|
|
153
|
+
atom_location=policy.hurdle.atom_location,
|
|
154
|
+
)
|
|
155
|
+
distribution = hurdle_result.distribution
|
|
156
|
+
diagnostics = hurdle_result.diagnostics.tail
|
|
157
|
+
if not diagnostics.converged:
|
|
158
|
+
raise ValueError("quantile optimizer did not converge")
|
|
159
|
+
if (
|
|
160
|
+
policy.maximum_normalized_rmse is not None
|
|
161
|
+
and diagnostics.normalized_rmse > policy.maximum_normalized_rmse
|
|
162
|
+
):
|
|
163
|
+
raise ValueError(
|
|
164
|
+
f"normalized RMSE {diagnostics.normalized_rmse} exceeds policy "
|
|
165
|
+
f"{policy.maximum_normalized_rmse}"
|
|
166
|
+
)
|
|
167
|
+
if (
|
|
168
|
+
policy.maximum_absolute_residual is not None
|
|
169
|
+
and diagnostics.maximum_absolute_residual
|
|
170
|
+
> policy.maximum_absolute_residual
|
|
171
|
+
):
|
|
172
|
+
raise ValueError(
|
|
173
|
+
"maximum absolute residual "
|
|
174
|
+
f"{diagnostics.maximum_absolute_residual} exceeds policy "
|
|
175
|
+
f"{policy.maximum_absolute_residual}"
|
|
176
|
+
)
|
|
177
|
+
base = (
|
|
178
|
+
distribution.base
|
|
179
|
+
if isinstance(distribution, HurdleDistribution)
|
|
180
|
+
else distribution
|
|
181
|
+
)
|
|
182
|
+
return distribution, base
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _canonical_batches(
|
|
186
|
+
raster: ZarrRaster,
|
|
187
|
+
policy: OSClimateIngestPolicy,
|
|
188
|
+
metadata: HazardDatasetMetadata,
|
|
189
|
+
bounds: Bounds | None,
|
|
190
|
+
) -> Iterator[CanonicalHazardBatch]:
|
|
191
|
+
try:
|
|
192
|
+
from shapely.geometry import Polygon # type: ignore[import-untyped]
|
|
193
|
+
except ImportError as error:
|
|
194
|
+
raise ImportError(
|
|
195
|
+
"OS-Climate ingest requires `pip install crc-sdk[geometry]`"
|
|
196
|
+
) from error
|
|
197
|
+
|
|
198
|
+
hazard_schema = hazard_arrow_schema(metadata)
|
|
199
|
+
hazard_rows: list[dict[str, Any]] = []
|
|
200
|
+
if "return period" not in raster.axis_name.lower():
|
|
201
|
+
raise ValueError(
|
|
202
|
+
f"{raster.metadata.path} has axis {raster.axis_name!r}, "
|
|
203
|
+
"not return periods"
|
|
204
|
+
)
|
|
205
|
+
for curve in raster.iter_curves(bounds):
|
|
206
|
+
valid = np.isfinite(curve.axis_values) & np.isfinite(curve.values)
|
|
207
|
+
periods = curve.axis_values[valid]
|
|
208
|
+
values = curve.values[valid]
|
|
209
|
+
if len(values) < 4:
|
|
210
|
+
continue
|
|
211
|
+
tabulated = TabulatedDistribution.from_return_periods(
|
|
212
|
+
periods,
|
|
213
|
+
values,
|
|
214
|
+
tail=policy.tail,
|
|
215
|
+
)
|
|
216
|
+
try:
|
|
217
|
+
distribution, base = _fit_curve(tabulated, policy)
|
|
218
|
+
except ValueError as error:
|
|
219
|
+
if policy.on_fit_failure == "skip":
|
|
220
|
+
continue
|
|
221
|
+
raise ValueError(
|
|
222
|
+
f"failed to fit source pixel row={curve.row}, "
|
|
223
|
+
f"column={curve.column}: {error}"
|
|
224
|
+
) from error
|
|
225
|
+
geometry = Polygon(curve.boundary)
|
|
226
|
+
source_id = _source_id(raster, curve)
|
|
227
|
+
cells = intersecting_cells(geometry, policy.h3_resolution)
|
|
228
|
+
if not cells:
|
|
229
|
+
continue
|
|
230
|
+
curve_kind = (
|
|
231
|
+
"hurdle" if isinstance(distribution, HurdleDistribution) else "fitted"
|
|
232
|
+
)
|
|
233
|
+
for cell_index in cells:
|
|
234
|
+
hazard_rows.append(
|
|
235
|
+
{
|
|
236
|
+
"cell_index": cell_index,
|
|
237
|
+
"source_id": source_id,
|
|
238
|
+
"source_geometry": geometry.wkb,
|
|
239
|
+
"hazard_name": raster.metadata.hazard_type,
|
|
240
|
+
"horizon": raster.metadata.year,
|
|
241
|
+
"pathway": raster.metadata.scenario,
|
|
242
|
+
"curve_kind": curve_kind,
|
|
243
|
+
"curve_type": base.family,
|
|
244
|
+
"curve_shape": base.shape,
|
|
245
|
+
"curve_location": base.location,
|
|
246
|
+
"curve_scale": base.scale,
|
|
247
|
+
"curve_atom_probability": (
|
|
248
|
+
distribution.atom_probability
|
|
249
|
+
if isinstance(distribution, HurdleDistribution)
|
|
250
|
+
else None
|
|
251
|
+
),
|
|
252
|
+
"curve_atom_location": (
|
|
253
|
+
distribution.atom_location
|
|
254
|
+
if isinstance(distribution, HurdleDistribution)
|
|
255
|
+
else None
|
|
256
|
+
),
|
|
257
|
+
}
|
|
258
|
+
)
|
|
259
|
+
if len(hazard_rows) >= policy.batch_rows:
|
|
260
|
+
hazards = validate_hazard_table(
|
|
261
|
+
pa.Table.from_pylist(hazard_rows, schema=hazard_schema),
|
|
262
|
+
metadata=metadata,
|
|
263
|
+
)
|
|
264
|
+
yield CanonicalHazardBatch(hazard_rows=hazards)
|
|
265
|
+
hazard_rows.clear()
|
|
266
|
+
if hazard_rows:
|
|
267
|
+
hazards = validate_hazard_table(
|
|
268
|
+
pa.Table.from_pylist(hazard_rows, schema=hazard_schema),
|
|
269
|
+
metadata=metadata,
|
|
270
|
+
)
|
|
271
|
+
yield CanonicalHazardBatch(hazard_rows=hazards)
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def canonicalize_os_climate(
|
|
275
|
+
raster: ZarrRaster,
|
|
276
|
+
policy: OSClimateIngestPolicy,
|
|
277
|
+
*,
|
|
278
|
+
bounds: Bounds | None = None,
|
|
279
|
+
) -> CanonicalHazardStream:
|
|
280
|
+
"""Return a lazy canonical stream for one selected raster."""
|
|
281
|
+
metadata = _metadata(raster, policy)
|
|
282
|
+
return CanonicalHazardStream(
|
|
283
|
+
metadata=metadata,
|
|
284
|
+
batches=_canonical_batches(raster, policy, metadata, bounds),
|
|
285
|
+
)
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""DuckDB-backed connector helpers."""
|
|
2
|
+
|
|
3
|
+
from .connection import (
|
|
4
|
+
DuckDBConnection,
|
|
5
|
+
DuckDBStreamEngine,
|
|
6
|
+
RuntimeResources,
|
|
7
|
+
default_work_dir,
|
|
8
|
+
detected_cpu_count,
|
|
9
|
+
ensure_extensions,
|
|
10
|
+
sql_identifier,
|
|
11
|
+
sql_quote,
|
|
12
|
+
)
|
|
13
|
+
from .geotiff import GeoTiffH3Scan, GeoTiffRaster, GeoTiffScan, trim_cache_dir
|
|
14
|
+
from .zarr import Bounds, Point, RasterCurve, RasterMetadata, ZarrRaster, ZarrScan
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"Bounds",
|
|
18
|
+
"DuckDBConnection",
|
|
19
|
+
"GeoTiffH3Scan",
|
|
20
|
+
"GeoTiffRaster",
|
|
21
|
+
"GeoTiffScan",
|
|
22
|
+
"Point",
|
|
23
|
+
"RasterCurve",
|
|
24
|
+
"RasterMetadata",
|
|
25
|
+
"DuckDBStreamEngine",
|
|
26
|
+
"RuntimeResources",
|
|
27
|
+
"ZarrRaster",
|
|
28
|
+
"ZarrScan",
|
|
29
|
+
"default_work_dir",
|
|
30
|
+
"detected_cpu_count",
|
|
31
|
+
"ensure_extensions",
|
|
32
|
+
"sql_identifier",
|
|
33
|
+
"sql_quote",
|
|
34
|
+
"trim_cache_dir",
|
|
35
|
+
]
|