cng-datasets 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cng_datasets/__init__.py +13 -0
- cng_datasets/cli.py +463 -0
- cng_datasets/hex_checks.py +63 -0
- cng_datasets/k8s/__init__.py +25 -0
- cng_datasets/k8s/armada.py +226 -0
- cng_datasets/k8s/jobs.py +228 -0
- cng_datasets/k8s/profiles/nrp.yaml +10 -0
- cng_datasets/k8s/workflows.py +1835 -0
- cng_datasets/raster/__init__.py +5 -0
- cng_datasets/raster/cog.py +1817 -0
- cng_datasets/storage/__init__.py +14 -0
- cng_datasets/storage/rclone.py +272 -0
- cng_datasets/storage/s3.py +236 -0
- cng_datasets/storage/setup_bucket.py +284 -0
- cng_datasets/vector/__init__.py +6 -0
- cng_datasets/vector/convert_to_parquet.py +1237 -0
- cng_datasets/vector/h3_tiling.py +1033 -0
- cng_datasets/vector/repartition.py +239 -0
- cng_datasets-0.3.0.dist-info/METADATA +227 -0
- cng_datasets-0.3.0.dist-info/RECORD +24 -0
- cng_datasets-0.3.0.dist-info/WHEEL +5 -0
- cng_datasets-0.3.0.dist-info/entry_points.txt +3 -0
- cng_datasets-0.3.0.dist-info/licenses/LICENSE +194 -0
- cng_datasets-0.3.0.dist-info/top_level.txt +1 -0
cng_datasets/__init__.py
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Cloud-Native Geospatial Datasets Processing Toolkit
|
|
3
|
+
|
|
4
|
+
A toolkit for processing large geospatial datasets into cloud-native formats
|
|
5
|
+
(COG, GeoParquet, PMTiles) with H3 hexagonal indexing.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
__version__ = "0.3.0"
|
|
9
|
+
|
|
10
|
+
# Lazy imports can be handled by users as needed,
|
|
11
|
+
# or we can keep these commented out to allow core usage without all dependencies.
|
|
12
|
+
# from . import vector, raster, k8s, storage
|
|
13
|
+
# __all__ = ["vector", "raster", "k8s", "storage", "__version__"]
|
cng_datasets/cli.py
ADDED
|
@@ -0,0 +1,463 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Command-line interface for cng-datasets toolkit.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
|
|
8
|
+
from cng_datasets.raster.cog import VALID_HEX_REDUCERS
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def main():
|
|
12
|
+
"""Main CLI entry point."""
|
|
13
|
+
parser = argparse.ArgumentParser(
|
|
14
|
+
description="Cloud-native geospatial dataset processing toolkit",
|
|
15
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
subparsers = parser.add_subparsers(dest="command", help="Available commands")
|
|
19
|
+
|
|
20
|
+
# Vector processing command
|
|
21
|
+
vector_parser = subparsers.add_parser("vector", help="Process vector datasets")
|
|
22
|
+
vector_parser.add_argument("--input", required=True, help="Input file URL")
|
|
23
|
+
vector_parser.add_argument("--output", required=True, help="Output directory URL")
|
|
24
|
+
vector_res_group = vector_parser.add_mutually_exclusive_group()
|
|
25
|
+
vector_res_group.add_argument("--resolution", type=int, default=10, help="H3 resolution")
|
|
26
|
+
vector_res_group.add_argument(
|
|
27
|
+
"--resolution-by-area", type=str, default=None,
|
|
28
|
+
help="Variable resolution by polygon area (issue #98). Comma-separated "
|
|
29
|
+
"'threshold:resolution' bins (planar deg2) plus a trailing catch-all "
|
|
30
|
+
"resolution, e.g. '12:8,600:6,5' (area<=12 -> res 8, <=600 -> res 6, "
|
|
31
|
+
"else res 5). Output carries a union schema + native_res column. "
|
|
32
|
+
"Include 0 in --parent-resolutions so the h0 partition column exists.")
|
|
33
|
+
vector_parser.add_argument("--chunk-size", type=int, default=500, help="Number of rows to process in pass 1 (geometry to H3 arrays)")
|
|
34
|
+
vector_parser.add_argument("--intermediate-chunk-size", type=int, default=10, help="Number of rows to process in pass 2 (unnesting arrays) - reduce if hitting OOM")
|
|
35
|
+
vector_parser.add_argument("--chunk-id", type=int, help="Process specific chunk")
|
|
36
|
+
vector_parser.add_argument("--parent-resolutions", type=str, default="9,8,0", help="Comma-separated parent H3 resolutions (default: '9,8,0')")
|
|
37
|
+
vector_parser.add_argument("--id-column", help="ID column name (auto-detected if not specified)")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
# Raster processing command
|
|
41
|
+
raster_parser = subparsers.add_parser("raster", help="Process raster datasets")
|
|
42
|
+
raster_parser.add_argument("--input", required=True, action="append", dest="inputs",
|
|
43
|
+
help="Input raster file (local or /vsicurl/ URL). Repeat for multiple tiles to mosaic.")
|
|
44
|
+
raster_parser.add_argument("--output-cog", help="Output COG file path")
|
|
45
|
+
raster_parser.add_argument("--output-parquet", help="Output parquet directory (e.g., s3://bucket/dataset/hex/)")
|
|
46
|
+
raster_parser.add_argument("--resolution", type=int, help="H3 resolution (auto-detected if not specified)")
|
|
47
|
+
raster_parser.add_argument("--parent-resolutions", type=str, default="0", help="Comma-separated parent H3 resolutions (default: '0')")
|
|
48
|
+
raster_parser.add_argument("--h0-index", type=int, help="Process specific h0 region (0-121), or omit to process all")
|
|
49
|
+
raster_parser.add_argument("--value-column", default="value", help="Name for raster value column (default: 'value')")
|
|
50
|
+
raster_parser.add_argument("--nodata", type=str,
|
|
51
|
+
help="NoData value(s) to exclude. Accepts a single value or a "
|
|
52
|
+
"comma-separated list for categorical products with multiple "
|
|
53
|
+
"fill codes, e.g. '-9999,-1111,32767' (all collapsed to the "
|
|
54
|
+
"first in the COG and excluded from hex tiling).")
|
|
55
|
+
raster_parser.add_argument("--compression", default="deflate", help="COG compression (deflate, lzw, zstd)")
|
|
56
|
+
raster_parser.add_argument("--blocksize", type=int, default=512, help="COG block size (default: 512)")
|
|
57
|
+
raster_parser.add_argument("--resampling", default="nearest", help="Resampling method for COG creation (default: nearest)")
|
|
58
|
+
raster_parser.add_argument("--hex-resampling", default="mean",
|
|
59
|
+
help="Reducer for aggregating source pixels into each "
|
|
60
|
+
"H3 cell. With --method=exact-extract (default), "
|
|
61
|
+
"one of: sum/mean/mode/fractions/max/min (max/min for "
|
|
62
|
+
"peak/richness rasters; 'fractions' emits per-class "
|
|
63
|
+
"coverage rows for categorical area accounting, #142). "
|
|
64
|
+
"With --method=warp-centroid, any GDAL resampleAlg "
|
|
65
|
+
"(average, sum, mode, near, bilinear, cubic, ...). "
|
|
66
|
+
"Default: mean.")
|
|
67
|
+
raster_parser.add_argument("--method", default="exact-extract",
|
|
68
|
+
choices=("exact-extract", "warp-centroid"),
|
|
69
|
+
help="Raster→hex algorithm. 'exact-extract' (default): "
|
|
70
|
+
"area-weighted per-cell, one row per cell, mass-conserving. "
|
|
71
|
+
"'warp-centroid': older gdal.Warp→XYZ→centroid path; "
|
|
72
|
+
"fast and low-memory but emits one row per warped pixel "
|
|
73
|
+
"(consumers GROUP BY h<res>) and is mass-conserving only "
|
|
74
|
+
"when hex pitch is finer than source pixel pitch (see #84).")
|
|
75
|
+
raster_parser.add_argument("--target-crs", default="EPSG:4326", help="Output CRS for mosaic (default: EPSG:4326)")
|
|
76
|
+
raster_parser.add_argument("--target-extent", help="Clip bbox 'xmin,ymin,xmax,ymax' in target CRS (mosaic only)")
|
|
77
|
+
raster_parser.add_argument("--target-resolution", type=float, help="Output pixel size in target CRS units (mosaic only)")
|
|
78
|
+
raster_parser.add_argument("--band", type=int, help="Extract single band from multi-band sources, 1-indexed (mosaic only)")
|
|
79
|
+
raster_parser.add_argument("--local-cache-dir", default="/tmp/cng-raster-cache",
|
|
80
|
+
help="Directory to copy remote input rasters into before processing "
|
|
81
|
+
"(default: /tmp/cng-raster-cache). Reading a remote COG via "
|
|
82
|
+
"/vsis3/ pays per-pixel HTTP latency that dominates wall time "
|
|
83
|
+
"on dense h0 cells — local-cache gives ~12x speedup at the "
|
|
84
|
+
"cost of one upfront copy. Use --no-local-cache to stream.")
|
|
85
|
+
raster_parser.add_argument("--no-local-cache", dest="local_cache_dir", action="store_const",
|
|
86
|
+
const=None, help="Stream the input via /vsis3/ instead of "
|
|
87
|
+
"copying to local disk first.")
|
|
88
|
+
|
|
89
|
+
# Repartition command
|
|
90
|
+
repartition_parser = subparsers.add_parser("repartition", help="Repartition chunks by h0")
|
|
91
|
+
repartition_parser.add_argument("--chunks-dir", required=True, help="Input chunks directory URL")
|
|
92
|
+
repartition_parser.add_argument("--output-dir", required=True, help="Output directory URL")
|
|
93
|
+
repartition_parser.add_argument("--source-parquet", required=True, help="Source parquet with full attributes")
|
|
94
|
+
repartition_parser.add_argument("--cleanup", action="store_true", default=True, help="Remove chunks after repartitioning")
|
|
95
|
+
repartition_parser.add_argument("--memory-limit", type=str, default=None, help="DuckDB memory limit (e.g. '27GiB'). Overrides DUCKDB_MEMORY_LIMIT env var.")
|
|
96
|
+
|
|
97
|
+
# K8s job generation command
|
|
98
|
+
k8s_parser = subparsers.add_parser("k8s", help="Generate Kubernetes job")
|
|
99
|
+
k8s_parser.add_argument("--job-name", required=True, help="Job name")
|
|
100
|
+
k8s_parser.add_argument("--cmd", nargs="+", required=True, help="Container command", dest="container_command")
|
|
101
|
+
k8s_parser.add_argument("--output", default="job.yaml", help="Output YAML file")
|
|
102
|
+
k8s_parser.add_argument("--chunks", type=int, help="Number of chunks for indexed job")
|
|
103
|
+
k8s_parser.add_argument("--namespace", default="biodiversity", help="Kubernetes namespace (default: biodiversity)")
|
|
104
|
+
|
|
105
|
+
# Workflow generation command
|
|
106
|
+
workflow_parser = subparsers.add_parser("workflow", help="Generate complete dataset workflow")
|
|
107
|
+
workflow_parser.add_argument("--dataset", required=True, help="Dataset name (e.g., redlining)")
|
|
108
|
+
workflow_parser.add_argument("--source-url", action="append", required=True, dest="source_urls", help="Source data URL (can be specified multiple times for multiple inputs)")
|
|
109
|
+
workflow_parser.add_argument("--bucket", required=True, help="S3 bucket for outputs")
|
|
110
|
+
workflow_parser.add_argument("--output-dir", default="k8s", help="Output directory for YAML files")
|
|
111
|
+
workflow_parser.add_argument("--namespace", default="biodiversity", help="Kubernetes namespace (default: biodiversity)")
|
|
112
|
+
workflow_parser.add_argument("--h3-resolution", type=int, default=None, help="Target H3 resolution (default: auto — 10 for polygons/points, 8 for lines)")
|
|
113
|
+
workflow_parser.add_argument(
|
|
114
|
+
"--resolution-by-area", type=str, default=None,
|
|
115
|
+
help="Variable resolution by polygon area (issue #98), e.g. '12:8,600:6,5'. "
|
|
116
|
+
"Mutually exclusive with --h3-resolution; emits the same flag into the "
|
|
117
|
+
"hex job. Include 0 in --parent-resolutions for the h0 partition column.")
|
|
118
|
+
workflow_parser.add_argument("--parent-resolutions", type=str, default="9,8,0", help="Comma-separated parent H3 resolutions (default: '9,8,0')")
|
|
119
|
+
workflow_parser.add_argument("--id-column", help="ID column name (auto-detected if not specified)")
|
|
120
|
+
workflow_parser.add_argument("--layer", help="Layer name for multi-layer datasets (e.g., GDB files)")
|
|
121
|
+
workflow_parser.add_argument("--hex-memory", type=str, default="8Gi", help="Memory per hex job pod (default: 8Gi)")
|
|
122
|
+
workflow_parser.add_argument("--max-parallelism", type=int, default=50, help="Maximum parallel hex jobs (default: 50)")
|
|
123
|
+
workflow_parser.add_argument("--max-completions", type=int, default=200, help="Maximum hex job completions (default: 200, increase to reduce chunk size/memory)")
|
|
124
|
+
workflow_parser.add_argument("--intermediate-chunk-size", type=int, default=10, help="Number of rows to process in pass 2 (unnesting arrays) - reduce if hitting OOM")
|
|
125
|
+
workflow_parser.add_argument("--row-group-size", type=int, default=100000, help="Number of rows per group in convert job (default: 100000)")
|
|
126
|
+
workflow_parser.add_argument("--hex-storage", type=str, default="10Gi", help="Ephemeral storage request/limit per hex job pod (default: 10Gi)")
|
|
127
|
+
workflow_parser.add_argument("--repartition-storage", type=str, default="50Gi", help="Ephemeral storage request/limit for repartition job pod (default: 50Gi)")
|
|
128
|
+
workflow_parser.add_argument("--repartition-memory", type=str, default="32Gi", help="Memory request/limit for repartition job pod (default: 32Gi)")
|
|
129
|
+
workflow_parser.add_argument("--backend", choices=["k8s", "armada"], default="k8s", help="Job backend: 'k8s' for standard Kubernetes Jobs (default), 'armada' for Armada queue submission")
|
|
130
|
+
# Cluster/storage configuration flags
|
|
131
|
+
workflow_parser.add_argument("--profile", default=None, metavar="NAME_OR_PATH", help="Cluster profile name (e.g. 'nrp') or path to a YAML profile file. Explicit flags below override profile values.")
|
|
132
|
+
workflow_parser.add_argument("--s3-endpoint", default=None, metavar="HOST", help="Internal S3 endpoint for jobs (default from profile, or rook-ceph-rgw-nautiluss3.rook)")
|
|
133
|
+
workflow_parser.add_argument("--s3-public-endpoint", default=None, metavar="HOST", help="Public S3 endpoint (default from profile, or s3-west.nrp-nautilus.io)")
|
|
134
|
+
workflow_parser.add_argument("--s3-secret-name", default=None, metavar="SECRET", help="Kubernetes secret name for S3 credentials (default from profile, or 'aws')")
|
|
135
|
+
workflow_parser.add_argument("--rclone-secret-name", default=None, metavar="SECRET", help="Kubernetes secret name for rclone config (default from profile, or 'rclone-config')")
|
|
136
|
+
workflow_parser.add_argument("--rclone-remote", default=None, metavar="REMOTE", help="Rclone remote name for setup-bucket and pmtiles (default from profile, or 'nrp')")
|
|
137
|
+
workflow_parser.add_argument("--priority-class", default=None, metavar="CLASS", help="Kubernetes priorityClassName; '' to omit (default from profile, or 'opportunistic')")
|
|
138
|
+
workflow_parser.add_argument("--node-affinity", default=None, choices=["gpu-avoid", "none"], help="Node affinity: 'gpu-avoid' (NRP GPU avoidance) or 'none' to omit (default from profile)")
|
|
139
|
+
|
|
140
|
+
# Raster workflow generation command
|
|
141
|
+
raster_workflow_parser = subparsers.add_parser("raster-workflow", help="Generate complete raster dataset workflow")
|
|
142
|
+
raster_workflow_parser.add_argument("--dataset", required=True, help="Dataset name")
|
|
143
|
+
raster_workflow_parser.add_argument("--source-url", required=True, action="append", dest="source_urls",
|
|
144
|
+
help="Source raster URL. Repeat for multiple tiles to mosaic.")
|
|
145
|
+
raster_workflow_parser.add_argument("--bucket", required=True, help="S3 bucket for outputs")
|
|
146
|
+
raster_workflow_parser.add_argument("--output-dir", default="k8s", help="Output directory for YAML files")
|
|
147
|
+
raster_workflow_parser.add_argument("--namespace", default="biodiversity", help="Kubernetes namespace")
|
|
148
|
+
raster_workflow_parser.add_argument("--h3-resolution", type=int, default=8, help="Target H3 resolution (default: 8)")
|
|
149
|
+
raster_workflow_parser.add_argument("--parent-resolutions", type=str, default="0", help="Comma-separated parent H3 resolutions (default: '0')")
|
|
150
|
+
raster_workflow_parser.add_argument("--value-column", default="value", help="Name for raster value column")
|
|
151
|
+
raster_workflow_parser.add_argument("--nodata", type=str,
|
|
152
|
+
help="NoData value(s) to exclude. Accepts a single value or "
|
|
153
|
+
"a comma-separated list for categorical products with "
|
|
154
|
+
"multiple fill codes, e.g. '-9999,-1111,32767'.")
|
|
155
|
+
raster_workflow_parser.add_argument("--hex-resampling", default="mean",
|
|
156
|
+
choices=VALID_HEX_REDUCERS,
|
|
157
|
+
help="Reducer for H3 aggregation. 'sum' for "
|
|
158
|
+
"counts (population, carbon); 'mean' for intensities; "
|
|
159
|
+
"'mode' for categorical (single dominant class); "
|
|
160
|
+
"'fractions' for categorical area accounting (one "
|
|
161
|
+
"(value, frac) row per class per cell); 'max'/'min' "
|
|
162
|
+
"for peak/extremum (species richness). Default: mean.")
|
|
163
|
+
raster_workflow_parser.add_argument("--hex-memory", type=str, default="32Gi", help="Memory per hex job pod (default: 32Gi)")
|
|
164
|
+
raster_workflow_parser.add_argument("--max-parallelism", type=int, default=61, help="Maximum parallel hex jobs (default: 61)")
|
|
165
|
+
raster_workflow_parser.add_argument("--hex-storage", type=str, default="20Gi", help="Ephemeral storage request/limit per hex job pod (default: 20Gi)")
|
|
166
|
+
raster_workflow_parser.add_argument("--cog-storage", type=str, default="50Gi", help="Ephemeral storage request/limit for COG preprocess job pod (default: 50Gi)")
|
|
167
|
+
raster_workflow_parser.add_argument("--target-extent", help="Clip bbox 'xmin,ymin,xmax,ymax' in EPSG:4326 (multi-tile only)")
|
|
168
|
+
raster_workflow_parser.add_argument("--target-resolution", type=float, help="Output pixel size in degrees (multi-tile only)")
|
|
169
|
+
raster_workflow_parser.add_argument("--band", type=int, help="Extract single band from multi-band sources, 1-indexed (multi-tile only)")
|
|
170
|
+
raster_workflow_parser.add_argument("--output-cog-name", help="S3 key for intermediate COG (default: {dataset}-cog.tif)")
|
|
171
|
+
raster_workflow_parser.add_argument("--backend", choices=["k8s", "armada"], default="k8s", help="Job backend: 'k8s' for standard Kubernetes Jobs (default), 'armada' for Armada queue submission")
|
|
172
|
+
# Cluster/storage configuration flags
|
|
173
|
+
raster_workflow_parser.add_argument("--profile", default=None, metavar="NAME_OR_PATH", help="Cluster profile name (e.g. 'nrp') or path to a YAML profile file. Explicit flags below override profile values.")
|
|
174
|
+
raster_workflow_parser.add_argument("--s3-endpoint", default=None, metavar="HOST", help="Internal S3 endpoint for jobs (default from profile, or rook-ceph-rgw-nautiluss3.rook)")
|
|
175
|
+
raster_workflow_parser.add_argument("--s3-public-endpoint", default=None, metavar="HOST", help="Public S3 endpoint (default from profile, or s3-west.nrp-nautilus.io)")
|
|
176
|
+
raster_workflow_parser.add_argument("--s3-secret-name", default=None, metavar="SECRET", help="Kubernetes secret name for S3 credentials (default from profile, or 'aws')")
|
|
177
|
+
raster_workflow_parser.add_argument("--rclone-secret-name", default=None, metavar="SECRET", help="Kubernetes secret name for rclone config (default from profile, or 'rclone-config')")
|
|
178
|
+
raster_workflow_parser.add_argument("--rclone-remote", default=None, metavar="REMOTE", help="Rclone remote name for setup-bucket (default from profile, or 'nrp')")
|
|
179
|
+
raster_workflow_parser.add_argument("--priority-class", default=None, metavar="CLASS", help="Kubernetes priorityClassName; '' to omit (default from profile, or 'opportunistic')")
|
|
180
|
+
raster_workflow_parser.add_argument("--node-affinity", default=None, choices=["gpu-avoid", "none"], help="Node affinity: 'gpu-avoid' (NRP GPU avoidance) or 'none' to omit (default from profile)")
|
|
181
|
+
|
|
182
|
+
# Sync job generation command
|
|
183
|
+
sync_job_parser = subparsers.add_parser("sync-job", help="Generate Kubernetes job for syncing between S3 locations")
|
|
184
|
+
sync_job_parser.add_argument("--job-name", required=True, help="Job name")
|
|
185
|
+
sync_job_parser.add_argument("--source", required=True, help="Source path (e.g., 'remote1:bucket/path')")
|
|
186
|
+
sync_job_parser.add_argument("--destination", required=True, help="Destination path (e.g., 'remote2:bucket/path')")
|
|
187
|
+
sync_job_parser.add_argument("--output", default="sync-job.yaml", help="Output YAML file (default: sync-job.yaml)")
|
|
188
|
+
sync_job_parser.add_argument("--namespace", default="biodiversity", help="Kubernetes namespace (default: biodiversity)")
|
|
189
|
+
sync_job_parser.add_argument("--cpu", default="2", help="CPU request/limit (default: 2)")
|
|
190
|
+
sync_job_parser.add_argument("--memory", default="4Gi", help="Memory request/limit (default: 4Gi)")
|
|
191
|
+
sync_job_parser.add_argument("--dry-run", action="store_true", help="Dry run mode (show what would be synced)")
|
|
192
|
+
|
|
193
|
+
# Storage management command
|
|
194
|
+
storage_parser = subparsers.add_parser("storage", help="Manage cloud storage")
|
|
195
|
+
storage_subparsers = storage_parser.add_subparsers(dest="storage_command")
|
|
196
|
+
|
|
197
|
+
cors_parser = storage_subparsers.add_parser("cors", help="Configure bucket CORS")
|
|
198
|
+
cors_parser.add_argument("--bucket", required=True, help="Bucket name")
|
|
199
|
+
cors_parser.add_argument("--endpoint", help="S3 endpoint URL")
|
|
200
|
+
|
|
201
|
+
sync_parser = storage_subparsers.add_parser("sync", help="Sync with rclone")
|
|
202
|
+
sync_parser.add_argument("--source", required=True, help="Source path")
|
|
203
|
+
sync_parser.add_argument("--destination", required=True, help="Destination path")
|
|
204
|
+
sync_parser.add_argument("--dry-run", action="store_true", help="Dry run mode")
|
|
205
|
+
|
|
206
|
+
setup_bucket_parser = storage_subparsers.add_parser("setup-bucket", help="Setup public bucket with CORS")
|
|
207
|
+
setup_bucket_parser.add_argument("--bucket", required=True, help="Bucket name")
|
|
208
|
+
setup_bucket_parser.add_argument("--remote", default="nrp", help="Rclone remote name (default: nrp)")
|
|
209
|
+
setup_bucket_parser.add_argument("--endpoint", help="S3 endpoint URL (defaults to AWS_PUBLIC_ENDPOINT env var)")
|
|
210
|
+
setup_bucket_parser.add_argument("--no-cors", action="store_true", help="Skip CORS configuration")
|
|
211
|
+
setup_bucket_parser.add_argument("--verify", action="store_true", help="Verify bucket configuration after setup")
|
|
212
|
+
|
|
213
|
+
args = parser.parse_args()
|
|
214
|
+
|
|
215
|
+
if not args.command:
|
|
216
|
+
parser.print_help()
|
|
217
|
+
sys.exit(1)
|
|
218
|
+
|
|
219
|
+
try:
|
|
220
|
+
_dispatch(args)
|
|
221
|
+
except ValueError as e:
|
|
222
|
+
print(f"Error: {e}", file=sys.stderr)
|
|
223
|
+
sys.exit(1)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _dispatch(args):
|
|
227
|
+
if args.command == "vector":
|
|
228
|
+
from .vector import process_vector_chunks
|
|
229
|
+
from .vector.h3_tiling import parse_resolution_by_area
|
|
230
|
+
# Parse parent resolutions from comma-separated string
|
|
231
|
+
parent_res = [int(x.strip()) for x in args.parent_resolutions.split(',') if x.strip()]
|
|
232
|
+
resolution_by_area = (
|
|
233
|
+
parse_resolution_by_area(args.resolution_by_area)
|
|
234
|
+
if args.resolution_by_area else None
|
|
235
|
+
)
|
|
236
|
+
process_vector_chunks(
|
|
237
|
+
input_url=args.input,
|
|
238
|
+
output_url=args.output,
|
|
239
|
+
chunk_id=args.chunk_id,
|
|
240
|
+
h3_resolution=args.resolution,
|
|
241
|
+
parent_resolutions=parent_res,
|
|
242
|
+
chunk_size=args.chunk_size,
|
|
243
|
+
intermediate_chunk_size=args.intermediate_chunk_size,
|
|
244
|
+
id_column=args.id_column,
|
|
245
|
+
resolution_by_area=resolution_by_area,
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
elif args.command == "raster":
|
|
249
|
+
from .raster import RasterProcessor, create_mosaic_cog
|
|
250
|
+
|
|
251
|
+
# Parse parent resolutions
|
|
252
|
+
parent_res = [int(x.strip()) for x in args.parent_resolutions.split(',') if x.strip()]
|
|
253
|
+
|
|
254
|
+
# Parse optional mosaic parameters
|
|
255
|
+
target_extent = None
|
|
256
|
+
if getattr(args, 'target_extent', None):
|
|
257
|
+
parts = [float(x) for x in args.target_extent.split(',')]
|
|
258
|
+
target_extent = tuple(parts)
|
|
259
|
+
|
|
260
|
+
input_path = args.inputs if len(args.inputs) > 1 else args.inputs[0]
|
|
261
|
+
|
|
262
|
+
# If multiple inputs and only --output-cog requested, use create_mosaic_cog directly
|
|
263
|
+
if isinstance(input_path, list) and args.output_cog and not args.output_parquet:
|
|
264
|
+
# Categorical sources (--hex-resampling mode/fractions) must not
|
|
265
|
+
# average class codes in the COG overviews (issue #108).
|
|
266
|
+
overview_resampling = (
|
|
267
|
+
"mode" if args.hex_resampling in ("mode", "fractions") else "average"
|
|
268
|
+
)
|
|
269
|
+
create_mosaic_cog(
|
|
270
|
+
source_urls=input_path,
|
|
271
|
+
output_path=args.output_cog,
|
|
272
|
+
target_crs=getattr(args, 'target_crs', 'EPSG:4326'),
|
|
273
|
+
target_extent=target_extent,
|
|
274
|
+
target_resolution=getattr(args, 'target_resolution', None),
|
|
275
|
+
band=getattr(args, 'band', None),
|
|
276
|
+
nodata=args.nodata,
|
|
277
|
+
resampling=args.resampling,
|
|
278
|
+
compression=args.compression,
|
|
279
|
+
overview_resampling=overview_resampling,
|
|
280
|
+
)
|
|
281
|
+
else:
|
|
282
|
+
processor = RasterProcessor(
|
|
283
|
+
input_path=input_path,
|
|
284
|
+
output_cog_path=args.output_cog,
|
|
285
|
+
output_parquet_path=args.output_parquet,
|
|
286
|
+
h3_resolution=args.resolution,
|
|
287
|
+
parent_resolutions=parent_res,
|
|
288
|
+
h0_index=args.h0_index,
|
|
289
|
+
value_column=args.value_column,
|
|
290
|
+
nodata_value=args.nodata,
|
|
291
|
+
compression=args.compression,
|
|
292
|
+
blocksize=args.blocksize,
|
|
293
|
+
resampling=args.resampling,
|
|
294
|
+
hex_resampling=args.hex_resampling,
|
|
295
|
+
method=getattr(args, 'method', 'exact-extract'),
|
|
296
|
+
target_crs=getattr(args, 'target_crs', 'EPSG:4326'),
|
|
297
|
+
target_extent=target_extent,
|
|
298
|
+
target_resolution=getattr(args, 'target_resolution', None),
|
|
299
|
+
band=getattr(args, 'band', None),
|
|
300
|
+
local_cache_dir=getattr(args, 'local_cache_dir', '/tmp/cng-raster-cache'),
|
|
301
|
+
)
|
|
302
|
+
|
|
303
|
+
if args.output_cog:
|
|
304
|
+
processor.create_cog()
|
|
305
|
+
|
|
306
|
+
if args.output_parquet:
|
|
307
|
+
if args.h0_index is not None:
|
|
308
|
+
processor.process_h0_region()
|
|
309
|
+
else:
|
|
310
|
+
processor.process_all_h0_regions()
|
|
311
|
+
|
|
312
|
+
elif args.command == "repartition":
|
|
313
|
+
from .vector import repartition_by_h0
|
|
314
|
+
repartition_by_h0(
|
|
315
|
+
chunks_dir=args.chunks_dir,
|
|
316
|
+
output_dir=args.output_dir,
|
|
317
|
+
source_parquet=args.source_parquet,
|
|
318
|
+
cleanup=args.cleanup,
|
|
319
|
+
memory_limit=args.memory_limit,
|
|
320
|
+
)
|
|
321
|
+
|
|
322
|
+
elif args.command == "k8s":
|
|
323
|
+
from .k8s import K8sJobManager
|
|
324
|
+
manager = K8sJobManager(namespace=getattr(args, 'namespace', 'biodiversity'))
|
|
325
|
+
if args.chunks:
|
|
326
|
+
job_spec = manager.generate_chunked_job(
|
|
327
|
+
job_name=args.job_name,
|
|
328
|
+
script_path=args.container_command[0],
|
|
329
|
+
num_chunks=args.chunks,
|
|
330
|
+
)
|
|
331
|
+
else:
|
|
332
|
+
job_spec = manager.generate_job_yaml(
|
|
333
|
+
job_name=args.job_name,
|
|
334
|
+
command=args.container_command,
|
|
335
|
+
)
|
|
336
|
+
manager.save_job_yaml(job_spec, args.output)
|
|
337
|
+
|
|
338
|
+
elif args.command == "sync-job":
|
|
339
|
+
from .k8s import generate_sync_job
|
|
340
|
+
generate_sync_job(
|
|
341
|
+
job_name=args.job_name,
|
|
342
|
+
source=args.source,
|
|
343
|
+
destination=args.destination,
|
|
344
|
+
output_file=args.output,
|
|
345
|
+
namespace=args.namespace,
|
|
346
|
+
cpu=args.cpu,
|
|
347
|
+
memory=args.memory,
|
|
348
|
+
dry_run=args.dry_run,
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
elif args.command == "workflow":
|
|
352
|
+
from .k8s import generate_dataset_workflow
|
|
353
|
+
from .vector.h3_tiling import parse_resolution_by_area
|
|
354
|
+
# Parse parent resolutions from comma-separated string
|
|
355
|
+
parent_res = [int(x.strip()) for x in args.parent_resolutions.split(',') if x.strip()]
|
|
356
|
+
if args.resolution_by_area and args.h3_resolution is not None:
|
|
357
|
+
raise ValueError("--resolution-by-area and --h3-resolution are mutually exclusive")
|
|
358
|
+
# Validate the spec early so workflow generation fails fast on a bad bin.
|
|
359
|
+
if args.resolution_by_area:
|
|
360
|
+
parse_resolution_by_area(args.resolution_by_area)
|
|
361
|
+
generate_dataset_workflow(
|
|
362
|
+
dataset_name=args.dataset,
|
|
363
|
+
source_urls=args.source_urls,
|
|
364
|
+
bucket=args.bucket,
|
|
365
|
+
output_dir=args.output_dir,
|
|
366
|
+
namespace=args.namespace,
|
|
367
|
+
h3_resolution=args.h3_resolution,
|
|
368
|
+
resolution_by_area=args.resolution_by_area,
|
|
369
|
+
parent_resolutions=parent_res,
|
|
370
|
+
id_column=args.id_column,
|
|
371
|
+
layer=args.layer,
|
|
372
|
+
hex_memory=args.hex_memory,
|
|
373
|
+
max_parallelism=args.max_parallelism,
|
|
374
|
+
max_completions=args.max_completions,
|
|
375
|
+
intermediate_chunk_size=args.intermediate_chunk_size,
|
|
376
|
+
row_group_size=args.row_group_size,
|
|
377
|
+
backend=args.backend,
|
|
378
|
+
hex_storage=args.hex_storage,
|
|
379
|
+
repartition_storage=args.repartition_storage,
|
|
380
|
+
repartition_memory=args.repartition_memory,
|
|
381
|
+
profile=args.profile,
|
|
382
|
+
s3_endpoint=args.s3_endpoint,
|
|
383
|
+
s3_public_endpoint=args.s3_public_endpoint,
|
|
384
|
+
s3_secret_name=args.s3_secret_name,
|
|
385
|
+
rclone_secret_name=args.rclone_secret_name,
|
|
386
|
+
rclone_remote=args.rclone_remote,
|
|
387
|
+
priority_class=args.priority_class,
|
|
388
|
+
node_affinity=args.node_affinity,
|
|
389
|
+
)
|
|
390
|
+
|
|
391
|
+
elif args.command == "raster-workflow":
|
|
392
|
+
from .k8s import generate_raster_workflow
|
|
393
|
+
# Parse parent resolutions
|
|
394
|
+
parent_res = [int(x.strip()) for x in args.parent_resolutions.split(',') if x.strip()]
|
|
395
|
+
# Parse optional mosaic parameters
|
|
396
|
+
target_extent = None
|
|
397
|
+
if getattr(args, 'target_extent', None):
|
|
398
|
+
parts = [float(x) for x in args.target_extent.split(',')]
|
|
399
|
+
target_extent = tuple(parts)
|
|
400
|
+
generate_raster_workflow(
|
|
401
|
+
dataset_name=args.dataset,
|
|
402
|
+
source_urls=args.source_urls,
|
|
403
|
+
bucket=args.bucket,
|
|
404
|
+
output_dir=args.output_dir,
|
|
405
|
+
namespace=args.namespace,
|
|
406
|
+
h3_resolution=args.h3_resolution,
|
|
407
|
+
parent_resolutions=parent_res,
|
|
408
|
+
value_column=args.value_column,
|
|
409
|
+
nodata_value=args.nodata,
|
|
410
|
+
hex_resampling=args.hex_resampling,
|
|
411
|
+
hex_memory=args.hex_memory,
|
|
412
|
+
max_parallelism=args.max_parallelism,
|
|
413
|
+
hex_storage=args.hex_storage,
|
|
414
|
+
cog_storage=args.cog_storage,
|
|
415
|
+
target_extent=target_extent,
|
|
416
|
+
target_resolution=getattr(args, 'target_resolution', None),
|
|
417
|
+
band=getattr(args, 'band', None),
|
|
418
|
+
output_cog_name=getattr(args, 'output_cog_name', None),
|
|
419
|
+
backend=args.backend,
|
|
420
|
+
profile=args.profile,
|
|
421
|
+
s3_endpoint=args.s3_endpoint,
|
|
422
|
+
s3_public_endpoint=args.s3_public_endpoint,
|
|
423
|
+
s3_secret_name=args.s3_secret_name,
|
|
424
|
+
rclone_secret_name=args.rclone_secret_name,
|
|
425
|
+
rclone_remote=args.rclone_remote,
|
|
426
|
+
priority_class=args.priority_class,
|
|
427
|
+
node_affinity=args.node_affinity,
|
|
428
|
+
)
|
|
429
|
+
|
|
430
|
+
elif args.command == "storage":
|
|
431
|
+
if args.storage_command == "cors":
|
|
432
|
+
from .storage import configure_bucket_cors
|
|
433
|
+
configure_bucket_cors(
|
|
434
|
+
bucket_name=args.bucket,
|
|
435
|
+
endpoint_url=args.endpoint,
|
|
436
|
+
)
|
|
437
|
+
elif args.storage_command == "sync":
|
|
438
|
+
from .storage import RcloneSync
|
|
439
|
+
syncer = RcloneSync(dry_run=args.dry_run)
|
|
440
|
+
syncer.sync(args.source, args.destination)
|
|
441
|
+
elif args.storage_command == "setup-bucket":
|
|
442
|
+
from .storage import setup_public_bucket
|
|
443
|
+
from .storage.setup_bucket import verify_bucket_config
|
|
444
|
+
import json
|
|
445
|
+
|
|
446
|
+
success = setup_public_bucket(
|
|
447
|
+
bucket_name=args.bucket,
|
|
448
|
+
remote=args.remote,
|
|
449
|
+
endpoint=args.endpoint,
|
|
450
|
+
set_cors=not args.no_cors,
|
|
451
|
+
verbose=True
|
|
452
|
+
)
|
|
453
|
+
|
|
454
|
+
if success and args.verify:
|
|
455
|
+
print("\nVerifying configuration...")
|
|
456
|
+
results = verify_bucket_config(args.bucket, args.endpoint)
|
|
457
|
+
print(json.dumps(results, indent=2))
|
|
458
|
+
|
|
459
|
+
sys.exit(0 if success else 1)
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
if __name__ == "__main__":
|
|
463
|
+
main()
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""Post-build invariants for H3-indexed (hex) datasets.
|
|
2
|
+
|
|
3
|
+
These are schema-only assertions run after the hex write step (vector
|
|
4
|
+
repartition and raster hex) to fail fast on a corrupt build rather than
|
|
5
|
+
silently shipping a dataset that breaks downstream consumers.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from typing import Callable, List, Tuple
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def assert_h3_columns_unsigned(
|
|
12
|
+
fetch: Callable[[str], List[Tuple]],
|
|
13
|
+
hex_glob: str,
|
|
14
|
+
) -> None:
|
|
15
|
+
"""Assert every physical H3 index column ``h{N>=1}`` reads back as ``UBIGINT``.
|
|
16
|
+
|
|
17
|
+
Native-cell columns (``h3_polygon_wkt_to_cells`` / ``h3_latlng_to_cell``)
|
|
18
|
+
are always ``UBIGINT``, but ``h3_cell_to_parent`` changed its return type
|
|
19
|
+
from signed ``BIGINT`` to ``UBIGINT`` in a newer h3 community-extension
|
|
20
|
+
release. Because both the pipeline and the MCP do *unpinned*
|
|
21
|
+
``INSTALL h3 FROM community``, a dataset's parent-column sign would
|
|
22
|
+
otherwise be a fossil of the extension version on its build date. We keep
|
|
23
|
+
tracking the latest h3, so this is an assertion (fail the build) rather
|
|
24
|
+
than a version pin — it also catches a bespoke ingest that casts the native
|
|
25
|
+
cell column (e.g. a ``BIGINT`` native ``h6``).
|
|
26
|
+
|
|
27
|
+
``h0`` is exempt: it is the Hive partition key, so DuckDB infers its type
|
|
28
|
+
from the directory string and always reads it back as signed ``BIGINT``
|
|
29
|
+
regardless of the physical type. The convention is therefore: physical
|
|
30
|
+
h-cols ``UBIGINT``; ``h0`` the signed hive key.
|
|
31
|
+
|
|
32
|
+
The check is schema-only (``DESCRIBE``, no data scan) and is evaluated
|
|
33
|
+
through the same path the consumer reads (the hive glob), so it sees the
|
|
34
|
+
types consumers actually get.
|
|
35
|
+
|
|
36
|
+
Args:
|
|
37
|
+
fetch: Callable that runs a SQL string and returns the result rows as a
|
|
38
|
+
list of tuples. Pass ``lambda sql: con.raw_sql(sql).fetchall()`` for
|
|
39
|
+
an ibis DuckDB connection, or ``lambda sql: con.execute(sql).fetchall()``
|
|
40
|
+
for a raw ``duckdb`` connection.
|
|
41
|
+
hex_glob: A ``read_parquet``-compatible path/glob for the written hex
|
|
42
|
+
output (e.g. ``s3://bucket/foo/hex/h0=*/data_0.parquet``).
|
|
43
|
+
|
|
44
|
+
Raises:
|
|
45
|
+
RuntimeError: if any ``h{N>=1}`` column is not ``UBIGINT``. See issue #102.
|
|
46
|
+
"""
|
|
47
|
+
sql = (
|
|
48
|
+
"SELECT column_name, column_type FROM "
|
|
49
|
+
f"(DESCRIBE SELECT * FROM read_parquet('{hex_glob}')) "
|
|
50
|
+
"WHERE column_name SIMILAR TO 'h[1-9][0-9]*' "
|
|
51
|
+
"AND column_type <> 'UBIGINT'"
|
|
52
|
+
)
|
|
53
|
+
offenders = fetch(sql)
|
|
54
|
+
if offenders:
|
|
55
|
+
cols = ", ".join(f"{name} ({typ})" for name, typ in offenders)
|
|
56
|
+
raise RuntimeError(
|
|
57
|
+
"H3 index columns must be UBIGINT after the hex build "
|
|
58
|
+
"(h0 is exempt as the signed hive key), but found non-UBIGINT "
|
|
59
|
+
f"columns: {cols}. This usually means the h3 community extension "
|
|
60
|
+
"emitted signed BIGINT parents (an older h3_cell_to_parent) or an "
|
|
61
|
+
"ingest cast the native cell column. Rebuild with a current h3 "
|
|
62
|
+
"extension or recast the offending columns to UBIGINT. See issue #102."
|
|
63
|
+
)
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Kubernetes job generation and management utilities."""
|
|
2
|
+
|
|
3
|
+
from .jobs import K8sJobManager, generate_job_yaml, submit_job
|
|
4
|
+
from .workflows import (
|
|
5
|
+
generate_dataset_workflow,
|
|
6
|
+
generate_raster_workflow,
|
|
7
|
+
generate_sync_job,
|
|
8
|
+
ClusterConfig,
|
|
9
|
+
load_profile,
|
|
10
|
+
cluster_config_from_args,
|
|
11
|
+
)
|
|
12
|
+
from .armada import (
|
|
13
|
+
k8s_job_to_armada,
|
|
14
|
+
k8s_indexed_job_to_armada,
|
|
15
|
+
convert_workflow_to_armada,
|
|
16
|
+
save_armada_yaml,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"K8sJobManager", "generate_job_yaml", "submit_job",
|
|
21
|
+
"generate_dataset_workflow", "generate_raster_workflow", "generate_sync_job",
|
|
22
|
+
"ClusterConfig", "load_profile", "cluster_config_from_args",
|
|
23
|
+
"k8s_job_to_armada", "k8s_indexed_job_to_armada",
|
|
24
|
+
"convert_workflow_to_armada", "save_armada_yaml",
|
|
25
|
+
]
|