crc-sdk 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. crc_sdk/__init__.py +31 -0
  2. crc_sdk/connectors/__init__.py +35 -0
  3. crc_sdk/connectors/adapters.py +285 -0
  4. crc_sdk/connectors/duckdb/__init__.py +35 -0
  5. crc_sdk/connectors/duckdb/connection.py +366 -0
  6. crc_sdk/connectors/duckdb/geotiff.py +617 -0
  7. crc_sdk/connectors/duckdb/zarr.py +461 -0
  8. crc_sdk/connectors/parquet.py +323 -0
  9. crc_sdk/connectors/protocols.py +13 -0
  10. crc_sdk/core/__init__.py +95 -0
  11. crc_sdk/fitting/__init__.py +31 -0
  12. crc_sdk/fitting/workflows.py +4 -0
  13. crc_sdk/geometry/__init__.py +117 -0
  14. crc_sdk/geometry/admin.py +462 -0
  15. crc_sdk/geometry/coverage.py +1201 -0
  16. crc_sdk/geometry/formats.py +91 -0
  17. crc_sdk/geometry/h3.py +418 -0
  18. crc_sdk/geometry/pmtiles/__init__.py +49 -0
  19. crc_sdk/geometry/pmtiles/_build.py +185 -0
  20. crc_sdk/geometry/pmtiles/_geojson_sql.py +221 -0
  21. crc_sdk/geometry/pmtiles/_process.py +195 -0
  22. crc_sdk/geometry/pmtiles/archive.py +153 -0
  23. crc_sdk/geometry/pmtiles/binaries.py +38 -0
  24. crc_sdk/geometry/pmtiles/budget.py +243 -0
  25. crc_sdk/geometry/pmtiles/presets.py +231 -0
  26. crc_sdk/geometry/vector.py +288 -0
  27. crc_sdk/impacts/__init__.py +25 -0
  28. crc_sdk/impacts/custom.py +5 -0
  29. crc_sdk/providers/__init__.py +21 -0
  30. crc_sdk/providers/local.py +36 -0
  31. crc_sdk/providers/metadata.py +1 -0
  32. crc_sdk/providers/os_climate.py +270 -0
  33. crc_sdk/providers/protocol.py +22 -0
  34. crc_sdk/py.typed +1 -0
  35. crc_sdk/schema/__init__.py +17 -0
  36. crc_sdk/schema/hazard_field.py +41 -0
  37. crc_sdk/types/__init__.py +20 -0
  38. crc_sdk/types/dataset.py +75 -0
  39. crc_sdk/types/geometry.py +12 -0
  40. crc_sdk/types/hazard.py +87 -0
  41. crc_sdk/types/storage.py +11 -0
  42. crc_sdk/workflows/__init__.py +55 -0
  43. crc_sdk/workflows/_portfolio.py +407 -0
  44. crc_sdk/workflows/distributions.py +317 -0
  45. crc_sdk/workflows/portfolio.py +386 -0
  46. crc_sdk/workflows/tiling.py +322 -0
  47. crc_sdk-0.1.0.dist-info/METADATA +374 -0
  48. crc_sdk-0.1.0.dist-info/RECORD +51 -0
  49. crc_sdk-0.1.0.dist-info/WHEEL +5 -0
  50. crc_sdk-0.1.0.dist-info/licenses/LICENSE +18 -0
  51. crc_sdk-0.1.0.dist-info/top_level.txt +1 -0
crc_sdk/__init__.py ADDED
@@ -0,0 +1,31 @@
1
+ """Higher-level Python SDK for Climate Risk Commons."""
2
+
3
+ from .core import (
4
+ Distribution,
5
+ EmpiricalDistribution,
6
+ FittedDistribution,
7
+ HurdleDistribution,
8
+ HurdleQuantileFitResult,
9
+ QuantileFitResult,
10
+ TabulatedDistribution,
11
+ fit_distribution,
12
+ fit_hurdle_quantiles,
13
+ fit_quantiles,
14
+ )
15
+ from .providers import LocalProvider, OSClimateProvider, Provider
16
+
17
+ __all__ = [
18
+ "Distribution",
19
+ "EmpiricalDistribution",
20
+ "FittedDistribution",
21
+ "HurdleDistribution",
22
+ "HurdleQuantileFitResult",
23
+ "LocalProvider",
24
+ "OSClimateProvider",
25
+ "Provider",
26
+ "QuantileFitResult",
27
+ "TabulatedDistribution",
28
+ "fit_distribution",
29
+ "fit_hurdle_quantiles",
30
+ "fit_quantiles",
31
+ ]
@@ -0,0 +1,35 @@
1
+ """External format and query-engine connectors."""
2
+
3
+ from .adapters import (
4
+ CanonicalHazardBatch,
5
+ CanonicalHazardStream,
6
+ HurdleFitPolicy,
7
+ OSClimateIngestPolicy,
8
+ canonicalize_os_climate,
9
+ )
10
+ from .parquet import (
11
+ hazard_arrow_schema,
12
+ read_hazard_dataset,
13
+ read_hazard_metadata,
14
+ sort_hazard_table,
15
+ validate_hazard_table,
16
+ write_hazard_dataset,
17
+ write_hazard_stream,
18
+ )
19
+ from .protocols import HazardReader
20
+
21
+ __all__ = [
22
+ "HazardReader",
23
+ "CanonicalHazardBatch",
24
+ "CanonicalHazardStream",
25
+ "HurdleFitPolicy",
26
+ "OSClimateIngestPolicy",
27
+ "canonicalize_os_climate",
28
+ "hazard_arrow_schema",
29
+ "read_hazard_dataset",
30
+ "read_hazard_metadata",
31
+ "sort_hazard_table",
32
+ "validate_hazard_table",
33
+ "write_hazard_dataset",
34
+ "write_hazard_stream",
35
+ ]
@@ -0,0 +1,285 @@
1
+ """Adapters from external connector results to canonical hazard rows."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Iterator
6
+ from dataclasses import dataclass
7
+ from hashlib import sha256
8
+ from typing import Any, Literal, get_args
9
+
10
+ import numpy as np
11
+ import pyarrow as pa # type: ignore[import-untyped]
12
+ from crc_framework import (
13
+ FittedDistribution,
14
+ HurdleDistribution,
15
+ QuantileFitDiagnostics,
16
+ TabulatedDistribution,
17
+ fit_hurdle_quantiles,
18
+ fit_quantiles,
19
+ )
20
+ from crc_framework.distributions import DistributionFamily
21
+
22
+ from crc_sdk.connectors.duckdb.zarr import Bounds, RasterCurve, ZarrRaster
23
+ from crc_sdk.connectors.parquet import (
24
+ hazard_arrow_schema,
25
+ validate_hazard_table,
26
+ )
27
+ from crc_sdk.geometry import intersecting_cells
28
+ from crc_sdk.types import HazardDatasetMetadata, SourceProvenance
29
+
30
+
31
+ @dataclass(frozen=True)
32
+ class HurdleFitPolicy:
33
+ """Explicit point-mass policy for one external quantile dataset."""
34
+
35
+ atom_probability: float
36
+ atom_location: float = 0.0
37
+
38
+ def __post_init__(self) -> None:
39
+ if not 0.0 < self.atom_probability < 1.0:
40
+ raise ValueError("atom_probability must be strictly between zero and one")
41
+ if not np.isfinite(self.atom_location):
42
+ raise ValueError("atom_location must be finite")
43
+
44
+
45
+ @dataclass(frozen=True)
46
+ class OSClimateIngestPolicy:
47
+ """Explicit policy controlling raster-to-canonical conversion."""
48
+
49
+ h3_resolution: int
50
+ family: DistributionFamily
51
+ producer: str
52
+ creation_version: str
53
+ tail: Literal["upper", "lower"] = "upper"
54
+ batch_rows: int = 65_536
55
+ value_semantics: str | None = None
56
+ source_version: str | None = None
57
+ hurdle: HurdleFitPolicy | None = None
58
+ maximum_normalized_rmse: float | None = None
59
+ maximum_absolute_residual: float | None = None
60
+ # Most pixels in an area (as opposed to a single known-exposed point) never
61
+ # exceed the hazard threshold and carry a constant, unfittable curve;
62
+ # "skip" drops those rather than aborting the whole area ingest.
63
+ on_fit_failure: Literal["raise", "skip"] = "raise"
64
+
65
+ def __post_init__(self) -> None:
66
+ if not 0 <= self.h3_resolution <= 15:
67
+ raise ValueError("H3 resolution must be between 0 and 15")
68
+ if self.family not in get_args(DistributionFamily):
69
+ raise ValueError(f"unknown distribution family {self.family!r}")
70
+ if not self.producer or not self.creation_version:
71
+ raise ValueError("producer and creation_version must be non-empty")
72
+ if self.batch_rows < 1:
73
+ raise ValueError("batch_rows must be positive")
74
+ if self.on_fit_failure not in ("raise", "skip"):
75
+ raise ValueError("on_fit_failure must be 'raise' or 'skip'")
76
+ for name, value in (
77
+ ("maximum_normalized_rmse", self.maximum_normalized_rmse),
78
+ ("maximum_absolute_residual", self.maximum_absolute_residual),
79
+ ):
80
+ if value is not None and (not np.isfinite(value) or value < 0.0):
81
+ raise ValueError(f"{name} must be finite and non-negative")
82
+
83
+
84
+ @dataclass(frozen=True)
85
+ class CanonicalHazardBatch:
86
+ """One batch of canonical H3-expanded hazard rows."""
87
+
88
+ hazard_rows: Any
89
+
90
+
91
+ @dataclass
92
+ class CanonicalHazardStream:
93
+ """Canonical metadata and one-shot Arrow batches."""
94
+
95
+ metadata: HazardDatasetMetadata
96
+ batches: Iterator[CanonicalHazardBatch]
97
+
98
+ def read_all(self) -> Any:
99
+ """Consume this stream into one canonical Arrow table."""
100
+ batches = list(self.batches)
101
+ if batches:
102
+ return pa.concat_tables(
103
+ [batch.hazard_rows for batch in batches],
104
+ promote_options="none",
105
+ )
106
+ return pa.Table.from_batches(
107
+ [],
108
+ schema=hazard_arrow_schema(self.metadata),
109
+ )
110
+
111
+
112
+ def _source_id(raster: ZarrRaster, curve: RasterCurve) -> str:
113
+ identity = (
114
+ f"os-climate\0{raster.metadata.path}\0{curve.row}\0{curve.column}"
115
+ ).encode()
116
+ return sha256(identity).hexdigest()
117
+
118
+
119
+ def _metadata(
120
+ raster: ZarrRaster, policy: OSClimateIngestPolicy
121
+ ) -> HazardDatasetMetadata:
122
+ values = raster.metadata
123
+ return HazardDatasetMetadata(
124
+ h3_resolution=policy.h3_resolution,
125
+ value_unit=values.units,
126
+ value_semantics=policy.value_semantics or values.indicator_id,
127
+ producer=policy.producer,
128
+ creation_version=policy.creation_version,
129
+ source=SourceProvenance(
130
+ provider="os-climate",
131
+ dataset=f"{values.hazard_type}:{values.indicator_id}",
132
+ uri=values.path,
133
+ version=policy.source_version,
134
+ ),
135
+ )
136
+
137
+
138
+ def _fit_curve(
139
+ tabulated: TabulatedDistribution,
140
+ policy: OSClimateIngestPolicy,
141
+ ) -> tuple[Any, Any]:
142
+ distribution: FittedDistribution | HurdleDistribution
143
+ diagnostics: QuantileFitDiagnostics
144
+ if policy.hurdle is None:
145
+ quantile_result = fit_quantiles(tabulated, family=policy.family)
146
+ distribution = quantile_result.distribution
147
+ diagnostics = quantile_result.diagnostics
148
+ else:
149
+ hurdle_result = fit_hurdle_quantiles(
150
+ tabulated,
151
+ family=policy.family,
152
+ atom_probability=policy.hurdle.atom_probability,
153
+ atom_location=policy.hurdle.atom_location,
154
+ )
155
+ distribution = hurdle_result.distribution
156
+ diagnostics = hurdle_result.diagnostics.tail
157
+ if not diagnostics.converged:
158
+ raise ValueError("quantile optimizer did not converge")
159
+ if (
160
+ policy.maximum_normalized_rmse is not None
161
+ and diagnostics.normalized_rmse > policy.maximum_normalized_rmse
162
+ ):
163
+ raise ValueError(
164
+ f"normalized RMSE {diagnostics.normalized_rmse} exceeds policy "
165
+ f"{policy.maximum_normalized_rmse}"
166
+ )
167
+ if (
168
+ policy.maximum_absolute_residual is not None
169
+ and diagnostics.maximum_absolute_residual
170
+ > policy.maximum_absolute_residual
171
+ ):
172
+ raise ValueError(
173
+ "maximum absolute residual "
174
+ f"{diagnostics.maximum_absolute_residual} exceeds policy "
175
+ f"{policy.maximum_absolute_residual}"
176
+ )
177
+ base = (
178
+ distribution.base
179
+ if isinstance(distribution, HurdleDistribution)
180
+ else distribution
181
+ )
182
+ return distribution, base
183
+
184
+
185
+ def _canonical_batches(
186
+ raster: ZarrRaster,
187
+ policy: OSClimateIngestPolicy,
188
+ metadata: HazardDatasetMetadata,
189
+ bounds: Bounds | None,
190
+ ) -> Iterator[CanonicalHazardBatch]:
191
+ try:
192
+ from shapely.geometry import Polygon # type: ignore[import-untyped]
193
+ except ImportError as error:
194
+ raise ImportError(
195
+ "OS-Climate ingest requires `pip install crc-sdk[geometry]`"
196
+ ) from error
197
+
198
+ hazard_schema = hazard_arrow_schema(metadata)
199
+ hazard_rows: list[dict[str, Any]] = []
200
+ if "return period" not in raster.axis_name.lower():
201
+ raise ValueError(
202
+ f"{raster.metadata.path} has axis {raster.axis_name!r}, "
203
+ "not return periods"
204
+ )
205
+ for curve in raster.iter_curves(bounds):
206
+ valid = np.isfinite(curve.axis_values) & np.isfinite(curve.values)
207
+ periods = curve.axis_values[valid]
208
+ values = curve.values[valid]
209
+ if len(values) < 4:
210
+ continue
211
+ tabulated = TabulatedDistribution.from_return_periods(
212
+ periods,
213
+ values,
214
+ tail=policy.tail,
215
+ )
216
+ try:
217
+ distribution, base = _fit_curve(tabulated, policy)
218
+ except ValueError as error:
219
+ if policy.on_fit_failure == "skip":
220
+ continue
221
+ raise ValueError(
222
+ f"failed to fit source pixel row={curve.row}, "
223
+ f"column={curve.column}: {error}"
224
+ ) from error
225
+ geometry = Polygon(curve.boundary)
226
+ source_id = _source_id(raster, curve)
227
+ cells = intersecting_cells(geometry, policy.h3_resolution)
228
+ if not cells:
229
+ continue
230
+ curve_kind = (
231
+ "hurdle" if isinstance(distribution, HurdleDistribution) else "fitted"
232
+ )
233
+ for cell_index in cells:
234
+ hazard_rows.append(
235
+ {
236
+ "cell_index": cell_index,
237
+ "source_id": source_id,
238
+ "source_geometry": geometry.wkb,
239
+ "hazard_name": raster.metadata.hazard_type,
240
+ "horizon": raster.metadata.year,
241
+ "pathway": raster.metadata.scenario,
242
+ "curve_kind": curve_kind,
243
+ "curve_type": base.family,
244
+ "curve_shape": base.shape,
245
+ "curve_location": base.location,
246
+ "curve_scale": base.scale,
247
+ "curve_atom_probability": (
248
+ distribution.atom_probability
249
+ if isinstance(distribution, HurdleDistribution)
250
+ else None
251
+ ),
252
+ "curve_atom_location": (
253
+ distribution.atom_location
254
+ if isinstance(distribution, HurdleDistribution)
255
+ else None
256
+ ),
257
+ }
258
+ )
259
+ if len(hazard_rows) >= policy.batch_rows:
260
+ hazards = validate_hazard_table(
261
+ pa.Table.from_pylist(hazard_rows, schema=hazard_schema),
262
+ metadata=metadata,
263
+ )
264
+ yield CanonicalHazardBatch(hazard_rows=hazards)
265
+ hazard_rows.clear()
266
+ if hazard_rows:
267
+ hazards = validate_hazard_table(
268
+ pa.Table.from_pylist(hazard_rows, schema=hazard_schema),
269
+ metadata=metadata,
270
+ )
271
+ yield CanonicalHazardBatch(hazard_rows=hazards)
272
+
273
+
274
+ def canonicalize_os_climate(
275
+ raster: ZarrRaster,
276
+ policy: OSClimateIngestPolicy,
277
+ *,
278
+ bounds: Bounds | None = None,
279
+ ) -> CanonicalHazardStream:
280
+ """Return a lazy canonical stream for one selected raster."""
281
+ metadata = _metadata(raster, policy)
282
+ return CanonicalHazardStream(
283
+ metadata=metadata,
284
+ batches=_canonical_batches(raster, policy, metadata, bounds),
285
+ )
@@ -0,0 +1,35 @@
1
+ """DuckDB-backed connector helpers."""
2
+
3
+ from .connection import (
4
+ DuckDBConnection,
5
+ DuckDBStreamEngine,
6
+ RuntimeResources,
7
+ default_work_dir,
8
+ detected_cpu_count,
9
+ ensure_extensions,
10
+ sql_identifier,
11
+ sql_quote,
12
+ )
13
+ from .geotiff import GeoTiffH3Scan, GeoTiffRaster, GeoTiffScan, trim_cache_dir
14
+ from .zarr import Bounds, Point, RasterCurve, RasterMetadata, ZarrRaster, ZarrScan
15
+
16
+ __all__ = [
17
+ "Bounds",
18
+ "DuckDBConnection",
19
+ "GeoTiffH3Scan",
20
+ "GeoTiffRaster",
21
+ "GeoTiffScan",
22
+ "Point",
23
+ "RasterCurve",
24
+ "RasterMetadata",
25
+ "DuckDBStreamEngine",
26
+ "RuntimeResources",
27
+ "ZarrRaster",
28
+ "ZarrScan",
29
+ "default_work_dir",
30
+ "detected_cpu_count",
31
+ "ensure_extensions",
32
+ "sql_identifier",
33
+ "sql_quote",
34
+ "trim_cache_dir",
35
+ ]