usdata 0.7.0__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {usdata-0.7.0 → usdata-0.8.0}/PKG-INFO +53 -9
- {usdata-0.7.0 → usdata-0.8.0}/README.md +47 -8
- {usdata-0.7.0 → usdata-0.8.0}/pyproject.toml +13 -1
- {usdata-0.7.0 → usdata-0.8.0}/pyproject.toml.orig +10 -1
- usdata-0.8.0/src/usdata/_netcdf.py +36 -0
- usdata-0.8.0/src/usdata/_radar.py +73 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/data/registry.yaml +23 -19
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/fetch.py +4 -2
- usdata-0.8.0/src/usdata/providers/noaa/goes.py +126 -0
- usdata-0.8.0/src/usdata/providers/noaa/storm_events.py +134 -0
- usdata-0.8.0/src/usdata/readers.py +141 -0
- usdata-0.7.0/src/usdata/readers.py +0 -103
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/__init__.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/_files.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/_progress.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/cache.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/cli/__init__.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/cli/app.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/cli/progress.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/data/nexrad_sites.csv +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/data/places.csv +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/data/places.sources.json +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/manifest.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/models.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/protocols/__init__.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/protocols/erddap.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/protocols/http.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/protocols/s3.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/provenance.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/__init__.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/base.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/noaa/__init__.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/noaa/ghcnd.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/noaa/gsom.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/noaa/nexrad.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/noaa/sites.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/usgs/__init__.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/usgs/daily.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/pull.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/py.typed +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/query.py +0 -0
- {usdata-0.7.0 → usdata-0.8.0}/src/usdata/registry.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: usdata
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
|
|
5
5
|
Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
|
|
6
6
|
Author: Jake Van Slyke
|
|
@@ -14,11 +14,16 @@ Requires-Dist: httpx>=0.28.1
|
|
|
14
14
|
Requires-Dist: pydantic>=2.7
|
|
15
15
|
Requires-Dist: pyyaml>=6.0
|
|
16
16
|
Requires-Dist: typer>=0.12
|
|
17
|
+
Requires-Dist: xarray>=2025.1 ; extra == 'netcdf'
|
|
18
|
+
Requires-Dist: h5netcdf[h5py]>=1.8.1 ; extra == 'netcdf'
|
|
17
19
|
Requires-Dist: pandas>=3.0 ; extra == 'pandas'
|
|
20
|
+
Requires-Dist: xradar>=0.12.0 ; extra == 'radar'
|
|
18
21
|
Requires-Python: >=3.11
|
|
19
22
|
Project-URL: Homepage, https://github.com/jakeryderv/usdata
|
|
20
23
|
Project-URL: Repository, https://github.com/jakeryderv/usdata
|
|
24
|
+
Provides-Extra: netcdf
|
|
21
25
|
Provides-Extra: pandas
|
|
26
|
+
Provides-Extra: radar
|
|
22
27
|
Description-Content-Type: text/markdown
|
|
23
28
|
|
|
24
29
|
# usdata
|
|
@@ -26,10 +31,11 @@ Description-Content-Type: text/markdown
|
|
|
26
31
|
Unified Python SDK and CLI for discovering, fetching, and tracking the
|
|
27
32
|
provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
28
33
|
|
|
29
|
-
> Status: pre-alpha. v0.
|
|
30
|
-
> NEXRAD Level II,
|
|
31
|
-
>
|
|
32
|
-
>
|
|
34
|
+
> Status: pre-alpha. v0.8 supports GHCN-Daily, GSOM monthly summaries,
|
|
35
|
+
> NEXRAD Level II, GOES ABI CONUS imagery, Storm Events annual archives,
|
|
36
|
+
> USGS daily values, and CoastWatch SST subsets with provenance. It includes
|
|
37
|
+
> optional CSV, radar, and NetCDF4 readers, six executed notebooks, Census
|
|
38
|
+
> state/county lookup, and terminal download progress. Other datasets are planned.
|
|
33
39
|
> See [docs/roadmap.md](docs/roadmap.md).
|
|
34
40
|
|
|
35
41
|
## Providers
|
|
@@ -37,7 +43,7 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
|
37
43
|
<!-- registry:start -->
|
|
38
44
|
| Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
|
|
39
45
|
|---|---:|---:|---:|---|---|
|
|
40
|
-
| [NOAA](docs/providers/noaa.md) |
|
|
46
|
+
| [NOAA](docs/providers/noaa.md) | 6 | 0 | 23 | — | `ghcn-daily`, `gsom`, `storm-events`, `nexrad-level2`, `goes-abi`, `coastwatch-sst`, +23 planned |
|
|
41
47
|
| [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
|
|
42
48
|
| [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
|
|
43
49
|
| [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
@@ -94,6 +100,17 @@ usdata pull dataset.yaml # resolve, fetch, write dataset.lock.json
|
|
|
94
100
|
usdata verify dataset.yaml # exit 1 if any cached input drifted
|
|
95
101
|
```
|
|
96
102
|
|
|
103
|
+
Storm Events bulk access is available since v0.8. Dates select complete
|
|
104
|
+
annual details archives; filter rows locally after opening the gzip CSV. For example:
|
|
105
|
+
|
|
106
|
+
```sh
|
|
107
|
+
usdata fetch noaa:storm-events --start 2024-05-01 --end 2024-05-31 --dry-run
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
This lists the entire 2024 archive. Location and variable filters are rejected;
|
|
111
|
+
see the [executed Storm Events notebook](examples/storm-events/example.ipynb)
|
|
112
|
+
for local filtering and reporting limitations.
|
|
113
|
+
|
|
97
114
|
Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
|
|
98
115
|
`USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
|
|
99
116
|
recording source URL, retrieval time, checksum, size, and license.
|
|
@@ -143,6 +160,15 @@ size. Adapters that assemble files from metadata requests show asset-level progr
|
|
|
143
160
|
Use `--no-progress` to disable it. Progress is automatically disabled when either
|
|
144
161
|
stdout or stderr is redirected; existing output lines and exit codes are unchanged.
|
|
145
162
|
|
|
163
|
+
For single-channel GOES CONUS imagery (available since v0.8):
|
|
164
|
+
|
|
165
|
+
```sh
|
|
166
|
+
usdata fetch noaa:goes-abi --start 2024-05-06T12:01:18.1Z --end 2024-05-06T12:01:18.1Z -p satellite=18 -p channel=6
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
The download is a whole NetCDF scene. See [GOES access notes](docs/providers/noaa.md#goes-abi-conus-imagery)
|
|
170
|
+
for supported selectors and scan-start time semantics.
|
|
171
|
+
|
|
146
172
|
## Opening CSV data
|
|
147
173
|
|
|
148
174
|
`FetchedAsset.open()` is available since v0.6 with the optional pandas
|
|
@@ -151,16 +177,32 @@ preserves identifier strings, and keeps CoastWatch units as metadata.
|
|
|
151
177
|
See the [reader reference](docs/reference/readers.md)
|
|
152
178
|
and [fetch → open → analyze example](examples/sst-analysis/README.md).
|
|
153
179
|
|
|
180
|
+
The [examples directory](examples/README.md) contains executed Jupyter notebooks
|
|
181
|
+
with saved data previews, small plots, and source provenance. Start with weather
|
|
182
|
+
and streamflow for manifest workflows, SST for gridded CSV reading, or monthly
|
|
183
|
+
climate for GSOM observations.
|
|
184
|
+
|
|
185
|
+
NetCDF4 scene opening is available since v0.8 with `usdata[netcdf]`.
|
|
186
|
+
See the executed [GOES infrared notebook](examples/goes-imagery/example.ipynb).
|
|
187
|
+
|
|
154
188
|
## Development
|
|
155
189
|
|
|
156
190
|
Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
|
|
157
191
|
|
|
192
|
+
`just setup` uses the tested Python 3.14.7 pin in `.python-version`. Older Linux
|
|
193
|
+
uv Python 3.14 builds can crash during NumPy array operations; see
|
|
194
|
+
[the upstream fix](https://github.com/astral-sh/python-build-standalone/issues/991).
|
|
195
|
+
|
|
158
196
|
```sh
|
|
159
197
|
git clone https://github.com/jakeryderv/usdata && cd usdata
|
|
160
198
|
just setup # install toolchain and dependencies
|
|
161
199
|
just test # unit tests
|
|
162
200
|
just check # format, lint, typecheck, offline tests, generated docs, release notices
|
|
163
201
|
just check-pandas # install the CSV extra and run the same checks
|
|
202
|
+
just check-radar # install the radar extra and run the same checks
|
|
203
|
+
just check-netcdf # install the NetCDF4 extra and run the same checks
|
|
204
|
+
just notebooks # launch the optional Jupyter examples environment
|
|
205
|
+
just run-notebooks # execute notebooks live in fresh kernels and temporary caches
|
|
164
206
|
just build # build wheel and sdist
|
|
165
207
|
just smoke # exercise core and pandas wheel installations outside the checkout
|
|
166
208
|
just run search radar
|
|
@@ -168,9 +210,11 @@ just run search radar
|
|
|
168
210
|
|
|
169
211
|
Unit tests mechanically block network connections. Integration tests that hit
|
|
170
212
|
live services run with `just test-integration`. CI checks Python 3.11 and 3.14 on
|
|
171
|
-
Linux
|
|
172
|
-
profiles on Linux, macOS, and Windows. The full unit and
|
|
173
|
-
|
|
213
|
+
Linux with core-only, pandas, radar, and NetCDF dependency profiles. Installed-wheel
|
|
214
|
+
checks cover all four profiles on Linux, macOS, and Windows. The full unit and
|
|
215
|
+
live-service suites run on Linux. `just setup` restores a core-only development
|
|
216
|
+
environment; the `check-pandas`, `check-radar`, and `check-netcdf` commands install
|
|
217
|
+
their respective extras.
|
|
174
218
|
|
|
175
219
|
Releases: `just release minor` opens a version-bump PR; merging it publishes
|
|
176
220
|
to PyPI and creates the tag and GitHub release. See
|
|
@@ -3,10 +3,11 @@
|
|
|
3
3
|
Unified Python SDK and CLI for discovering, fetching, and tracking the
|
|
4
4
|
provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
5
5
|
|
|
6
|
-
> Status: pre-alpha. v0.
|
|
7
|
-
> NEXRAD Level II,
|
|
8
|
-
>
|
|
9
|
-
>
|
|
6
|
+
> Status: pre-alpha. v0.8 supports GHCN-Daily, GSOM monthly summaries,
|
|
7
|
+
> NEXRAD Level II, GOES ABI CONUS imagery, Storm Events annual archives,
|
|
8
|
+
> USGS daily values, and CoastWatch SST subsets with provenance. It includes
|
|
9
|
+
> optional CSV, radar, and NetCDF4 readers, six executed notebooks, Census
|
|
10
|
+
> state/county lookup, and terminal download progress. Other datasets are planned.
|
|
10
11
|
> See [docs/roadmap.md](docs/roadmap.md).
|
|
11
12
|
|
|
12
13
|
## Providers
|
|
@@ -14,7 +15,7 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
|
14
15
|
<!-- registry:start -->
|
|
15
16
|
| Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
|
|
16
17
|
|---|---:|---:|---:|---|---|
|
|
17
|
-
| [NOAA](docs/providers/noaa.md) |
|
|
18
|
+
| [NOAA](docs/providers/noaa.md) | 6 | 0 | 23 | — | `ghcn-daily`, `gsom`, `storm-events`, `nexrad-level2`, `goes-abi`, `coastwatch-sst`, +23 planned |
|
|
18
19
|
| [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
|
|
19
20
|
| [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
|
|
20
21
|
| [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
@@ -71,6 +72,17 @@ usdata pull dataset.yaml # resolve, fetch, write dataset.lock.json
|
|
|
71
72
|
usdata verify dataset.yaml # exit 1 if any cached input drifted
|
|
72
73
|
```
|
|
73
74
|
|
|
75
|
+
Storm Events bulk access is available since v0.8. Dates select complete
|
|
76
|
+
annual details archives; filter rows locally after opening the gzip CSV. For example:
|
|
77
|
+
|
|
78
|
+
```sh
|
|
79
|
+
usdata fetch noaa:storm-events --start 2024-05-01 --end 2024-05-31 --dry-run
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
This lists the entire 2024 archive. Location and variable filters are rejected;
|
|
83
|
+
see the [executed Storm Events notebook](examples/storm-events/example.ipynb)
|
|
84
|
+
for local filtering and reporting limitations.
|
|
85
|
+
|
|
74
86
|
Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
|
|
75
87
|
`USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
|
|
76
88
|
recording source URL, retrieval time, checksum, size, and license.
|
|
@@ -120,6 +132,15 @@ size. Adapters that assemble files from metadata requests show asset-level progr
|
|
|
120
132
|
Use `--no-progress` to disable it. Progress is automatically disabled when either
|
|
121
133
|
stdout or stderr is redirected; existing output lines and exit codes are unchanged.
|
|
122
134
|
|
|
135
|
+
For single-channel GOES CONUS imagery (available since v0.8):
|
|
136
|
+
|
|
137
|
+
```sh
|
|
138
|
+
usdata fetch noaa:goes-abi --start 2024-05-06T12:01:18.1Z --end 2024-05-06T12:01:18.1Z -p satellite=18 -p channel=6
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
The download is a whole NetCDF scene. See [GOES access notes](docs/providers/noaa.md#goes-abi-conus-imagery)
|
|
142
|
+
for supported selectors and scan-start time semantics.
|
|
143
|
+
|
|
123
144
|
## Opening CSV data
|
|
124
145
|
|
|
125
146
|
`FetchedAsset.open()` is available since v0.6 with the optional pandas
|
|
@@ -128,16 +149,32 @@ preserves identifier strings, and keeps CoastWatch units as metadata.
|
|
|
128
149
|
See the [reader reference](docs/reference/readers.md)
|
|
129
150
|
and [fetch → open → analyze example](examples/sst-analysis/README.md).
|
|
130
151
|
|
|
152
|
+
The [examples directory](examples/README.md) contains executed Jupyter notebooks
|
|
153
|
+
with saved data previews, small plots, and source provenance. Start with weather
|
|
154
|
+
and streamflow for manifest workflows, SST for gridded CSV reading, or monthly
|
|
155
|
+
climate for GSOM observations.
|
|
156
|
+
|
|
157
|
+
NetCDF4 scene opening is available since v0.8 with `usdata[netcdf]`.
|
|
158
|
+
See the executed [GOES infrared notebook](examples/goes-imagery/example.ipynb).
|
|
159
|
+
|
|
131
160
|
## Development
|
|
132
161
|
|
|
133
162
|
Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
|
|
134
163
|
|
|
164
|
+
`just setup` uses the tested Python 3.14.7 pin in `.python-version`. Older Linux
|
|
165
|
+
uv Python 3.14 builds can crash during NumPy array operations; see
|
|
166
|
+
[the upstream fix](https://github.com/astral-sh/python-build-standalone/issues/991).
|
|
167
|
+
|
|
135
168
|
```sh
|
|
136
169
|
git clone https://github.com/jakeryderv/usdata && cd usdata
|
|
137
170
|
just setup # install toolchain and dependencies
|
|
138
171
|
just test # unit tests
|
|
139
172
|
just check # format, lint, typecheck, offline tests, generated docs, release notices
|
|
140
173
|
just check-pandas # install the CSV extra and run the same checks
|
|
174
|
+
just check-radar # install the radar extra and run the same checks
|
|
175
|
+
just check-netcdf # install the NetCDF4 extra and run the same checks
|
|
176
|
+
just notebooks # launch the optional Jupyter examples environment
|
|
177
|
+
just run-notebooks # execute notebooks live in fresh kernels and temporary caches
|
|
141
178
|
just build # build wheel and sdist
|
|
142
179
|
just smoke # exercise core and pandas wheel installations outside the checkout
|
|
143
180
|
just run search radar
|
|
@@ -145,9 +182,11 @@ just run search radar
|
|
|
145
182
|
|
|
146
183
|
Unit tests mechanically block network connections. Integration tests that hit
|
|
147
184
|
live services run with `just test-integration`. CI checks Python 3.11 and 3.14 on
|
|
148
|
-
Linux
|
|
149
|
-
profiles on Linux, macOS, and Windows. The full unit and
|
|
150
|
-
|
|
185
|
+
Linux with core-only, pandas, radar, and NetCDF dependency profiles. Installed-wheel
|
|
186
|
+
checks cover all four profiles on Linux, macOS, and Windows. The full unit and
|
|
187
|
+
live-service suites run on Linux. `just setup` restores a core-only development
|
|
188
|
+
environment; the `check-pandas`, `check-radar`, and `check-netcdf` commands install
|
|
189
|
+
their respective extras.
|
|
151
190
|
|
|
152
191
|
Releases: `just release minor` opens a version-bump PR; merging it publishes
|
|
153
192
|
to PyPI and creates the tag and GitHub release. See
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "usdata"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.8.0"
|
|
4
4
|
description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -32,6 +32,11 @@ email = "jakervanslyke@gmail.com"
|
|
|
32
32
|
|
|
33
33
|
[project.optional-dependencies]
|
|
34
34
|
pandas = ["pandas>=3.0"]
|
|
35
|
+
radar = ["xradar>=0.12.0"]
|
|
36
|
+
netcdf = [
|
|
37
|
+
"xarray>=2025.1",
|
|
38
|
+
"h5netcdf[h5py]>=1.8.1",
|
|
39
|
+
]
|
|
35
40
|
|
|
36
41
|
[project.urls]
|
|
37
42
|
Homepage = "https://github.com/jakeryderv/usdata"
|
|
@@ -48,6 +53,13 @@ dev = [
|
|
|
48
53
|
"respx>=0.23.1",
|
|
49
54
|
"ruff>=0.6",
|
|
50
55
|
]
|
|
56
|
+
examples = [
|
|
57
|
+
"ipykernel>=6.29",
|
|
58
|
+
"jupyterlab>=4.3",
|
|
59
|
+
"matplotlib>=3.9",
|
|
60
|
+
"nbclient>=0.10",
|
|
61
|
+
"pandas>=3.0",
|
|
62
|
+
]
|
|
51
63
|
|
|
52
64
|
[build-system]
|
|
53
65
|
requires = ["uv_build>=0.12.5,<0.13.0"]
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "usdata"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.8.0"
|
|
4
4
|
description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -24,6 +24,8 @@ dependencies = [
|
|
|
24
24
|
|
|
25
25
|
[project.optional-dependencies]
|
|
26
26
|
pandas = ["pandas>=3.0"]
|
|
27
|
+
radar = ["xradar>=0.12.0"]
|
|
28
|
+
netcdf = ["xarray>=2025.1", "h5netcdf[h5py]>=1.8.1"]
|
|
27
29
|
|
|
28
30
|
[project.urls]
|
|
29
31
|
Homepage = "https://github.com/jakeryderv/usdata"
|
|
@@ -40,6 +42,13 @@ dev = [
|
|
|
40
42
|
"respx>=0.23.1",
|
|
41
43
|
"ruff>=0.6",
|
|
42
44
|
]
|
|
45
|
+
examples = [
|
|
46
|
+
"ipykernel>=6.29",
|
|
47
|
+
"jupyterlab>=4.3",
|
|
48
|
+
"matplotlib>=3.9",
|
|
49
|
+
"nbclient>=0.10",
|
|
50
|
+
"pandas>=3.0",
|
|
51
|
+
]
|
|
43
52
|
|
|
44
53
|
[build-system]
|
|
45
54
|
requires = ["uv_build>=0.12.5,<0.13.0"]
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Local, eagerly loaded NetCDF4 reading behind the netcdf extra."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from importlib import import_module
|
|
6
|
+
from typing import TYPE_CHECKING, Any
|
|
7
|
+
|
|
8
|
+
from usdata.readers import MissingReaderDependency
|
|
9
|
+
|
|
10
|
+
if TYPE_CHECKING:
|
|
11
|
+
from usdata.fetch import FetchedAsset
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def open_netcdf(fetched: FetchedAsset) -> Any:
|
|
15
|
+
"""Load a NetCDF4 root Dataset, then close every source file handle."""
|
|
16
|
+
try:
|
|
17
|
+
xarray = import_module("xarray")
|
|
18
|
+
import_module("h5netcdf")
|
|
19
|
+
import_module("h5py")
|
|
20
|
+
except ModuleNotFoundError as error:
|
|
21
|
+
if error.name not in {"xarray", "h5netcdf", "h5py"}:
|
|
22
|
+
raise
|
|
23
|
+
raise MissingReaderDependency(
|
|
24
|
+
'NetCDF4 reading requires xarray and h5netcdf; install: pip install "usdata[netcdf]"'
|
|
25
|
+
) from error
|
|
26
|
+
# A local file object and fixed engine prevent interpretation as an OPeNDAP URL.
|
|
27
|
+
with (
|
|
28
|
+
fetched.path.open("rb") as stream,
|
|
29
|
+
xarray.open_dataset(stream, engine="h5netcdf", chunks=None) as dataset,
|
|
30
|
+
):
|
|
31
|
+
dataset.load()
|
|
32
|
+
dataset.attrs["usdata"] = {
|
|
33
|
+
"asset_id": fetched.asset.id,
|
|
34
|
+
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
35
|
+
}
|
|
36
|
+
return dataset
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Local NEXRAD Level II decoding behind the radar extra."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import bz2
|
|
6
|
+
import gzip
|
|
7
|
+
from importlib import import_module
|
|
8
|
+
from typing import TYPE_CHECKING, Any
|
|
9
|
+
|
|
10
|
+
from usdata.readers import MissingReaderDependency
|
|
11
|
+
|
|
12
|
+
# NOAA RDA/RPG ICD 2620002Y, Table XVII-I notes 21 and 30.
|
|
13
|
+
MOMENT_FLAG_COUNTS = {
|
|
14
|
+
"DBZH": 2,
|
|
15
|
+
"VRADH": 2,
|
|
16
|
+
"WRADH": 2,
|
|
17
|
+
"ZDR": 2,
|
|
18
|
+
"PHIDP": 2,
|
|
19
|
+
"RHOHV": 2,
|
|
20
|
+
"CCORH": 8,
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
if TYPE_CHECKING:
|
|
24
|
+
from usdata.fetch import FetchedAsset
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def open_nexrad(fetched: FetchedAsset) -> Any:
|
|
28
|
+
"""Decode a local volume into a fully loaded xarray DataTree."""
|
|
29
|
+
try:
|
|
30
|
+
xradar = import_module("xradar")
|
|
31
|
+
except ModuleNotFoundError as error:
|
|
32
|
+
if error.name != "xradar":
|
|
33
|
+
raise
|
|
34
|
+
raise MissingReaderDependency(
|
|
35
|
+
'NEXRAD reading requires xradar; install it with: pip install "usdata[radar]" '
|
|
36
|
+
'(or uv add "usdata[radar]")'
|
|
37
|
+
) from error
|
|
38
|
+
|
|
39
|
+
# Bytes prevent remote URL interpretation and work across the backend's
|
|
40
|
+
# repeated sweep reads. Compressed source files stay unchanged in the cache.
|
|
41
|
+
content = fetched.path.read_bytes()
|
|
42
|
+
if content.startswith(b"\x1f\x8b"):
|
|
43
|
+
content = gzip.decompress(content)
|
|
44
|
+
elif content.startswith(b"BZh"):
|
|
45
|
+
content = bz2.decompress(content)
|
|
46
|
+
radar = xradar.io.open_nexradlevel2_datatree(content, incomplete_sweep="pad")
|
|
47
|
+
try:
|
|
48
|
+
radar.load()
|
|
49
|
+
finally:
|
|
50
|
+
radar.close()
|
|
51
|
+
for node in radar.subtree:
|
|
52
|
+
for name, variable in node.ds.variables.items():
|
|
53
|
+
# The backend records the entire input byte string as `source`.
|
|
54
|
+
# Provenance below is the durable reference, not that decoder buffer.
|
|
55
|
+
variable.encoding.pop("source", None)
|
|
56
|
+
if name in MOMENT_FLAG_COUNTS and "range" in variable.dims:
|
|
57
|
+
scale = variable.encoding.get("scale_factor")
|
|
58
|
+
offset = variable.encoding.get("add_offset")
|
|
59
|
+
if scale is not None and offset is not None:
|
|
60
|
+
# xradar 0.12 does not supply _FillValue for NEXRAD flags.
|
|
61
|
+
# Compare using each moment's native scale, not fixed units.
|
|
62
|
+
data = node[name]
|
|
63
|
+
valid = data.notnull()
|
|
64
|
+
for code in range(MOMENT_FLAG_COUNTS[name]):
|
|
65
|
+
valid = valid & (data != offset + code * scale)
|
|
66
|
+
masked = data.where(valid)
|
|
67
|
+
masked.encoding = data.encoding.copy()
|
|
68
|
+
node[name] = masked
|
|
69
|
+
radar.attrs["usdata"] = {
|
|
70
|
+
"asset_id": fetched.asset.id,
|
|
71
|
+
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
72
|
+
}
|
|
73
|
+
return radar
|
|
@@ -105,21 +105,22 @@ datasets:
|
|
|
105
105
|
|
|
106
106
|
- id: noaa:goes-abi
|
|
107
107
|
provider: noaa
|
|
108
|
-
status:
|
|
108
|
+
status: available
|
|
109
109
|
domain: weather-satellites
|
|
110
|
-
|
|
111
|
-
title: GOES-R ABI
|
|
112
|
-
description: >-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
keywords: [satellite, imagery, goes, abi, clouds,
|
|
110
|
+
since: "0.8"
|
|
111
|
+
title: GOES-R ABI CONUS Cloud and Moisture Imagery
|
|
112
|
+
description: >-
|
|
113
|
+
Single-channel CONUS Cloud and Moisture Imagery (ABI-L2-CMIPC) from
|
|
114
|
+
GOES-16, 17, 18, and 19 in anonymous NOAA S3 buckets. Select an explicit
|
|
115
|
+
satellite, channel, and scan-start interval; each asset is a complete
|
|
116
|
+
NetCDF scene with no geographic or variable subsetting.
|
|
117
|
+
keywords: [satellite, imagery, goes, abi, clouds, infrared, reflectance, netcdf, conus]
|
|
118
118
|
protocol: s3
|
|
119
119
|
homepage: https://registry.opendata.aws/noaa-goes/
|
|
120
120
|
license: US Government Work (public domain)
|
|
121
|
-
temporal_extent: { start: "2017-
|
|
122
|
-
capabilities: { spatial_subset: false, temporal_subset: true, variable_subset:
|
|
121
|
+
temporal_extent: { start: "2017-02-28T00:00:00Z" }
|
|
122
|
+
capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: false }
|
|
123
|
+
adapter: usdata.providers.noaa.goes:GoesAbi
|
|
123
124
|
|
|
124
125
|
- id: noaa:goes-glm
|
|
125
126
|
provider: noaa
|
|
@@ -139,21 +140,24 @@ datasets:
|
|
|
139
140
|
|
|
140
141
|
- id: noaa:storm-events
|
|
141
142
|
provider: noaa
|
|
142
|
-
status:
|
|
143
|
+
status: available
|
|
143
144
|
domain: severe-weather
|
|
144
|
-
|
|
145
|
+
since: "0.8"
|
|
145
146
|
title: Storm Events Database
|
|
146
147
|
description: >-
|
|
147
|
-
NCEI's
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
148
|
+
NCEI's significant-weather event details since 1950, with locations,
|
|
149
|
+
impacts, and narratives. Anonymous whole-year gzipped CSV archives;
|
|
150
|
+
select the latest creation-date revision for each requested year.
|
|
151
|
+
No server-side row, location, or variable subsetting. Historical event
|
|
152
|
+
coverage and reporting practices vary; fatalities and locations tables
|
|
153
|
+
are separate products not included by this adapter.
|
|
151
154
|
keywords: [storms, tornado, hail, wind, flood, damage, severe weather, events]
|
|
152
155
|
protocol: http
|
|
153
|
-
homepage: https://www.
|
|
156
|
+
homepage: https://www.ncei.noaa.gov/access/storm-events-database/
|
|
154
157
|
license: US Government Work (public domain)
|
|
155
158
|
temporal_extent: { start: "1950-01-01T00:00:00Z" }
|
|
156
|
-
capabilities: { spatial_subset: false, temporal_subset:
|
|
159
|
+
capabilities: { spatial_subset: false, temporal_subset: false, variable_subset: false }
|
|
160
|
+
adapter: usdata.providers.noaa.storm_events:StormEvents
|
|
157
161
|
|
|
158
162
|
- id: noaa:hurdat2
|
|
159
163
|
provider: noaa
|
|
@@ -35,10 +35,12 @@ class FetchedAsset(BaseModel):
|
|
|
35
35
|
usecols: list[str] | None = None,
|
|
36
36
|
nrows: int | None = None,
|
|
37
37
|
) -> Any:
|
|
38
|
-
"""Open
|
|
38
|
+
"""Open local data with an optional ``pandas``, ``radar``, or ``netcdf`` reader.
|
|
39
39
|
|
|
40
40
|
ERDDAP units are kept in ``frame.attrs["units"]`` and source provenance
|
|
41
|
-
in ``frame.attrs["usdata"]``.
|
|
41
|
+
in ``frame.attrs["usdata"]``. NEXRAD returns a xarray DataTree with provenance
|
|
42
|
+
in ``radar.attrs["usdata"]``. NetCDF4 returns a loaded xarray Dataset with
|
|
43
|
+
matching provenance in its attributes. See ``usdata.readers.open_asset`` for options.
|
|
42
44
|
Cached files and provenance sidecars are never changed.
|
|
43
45
|
"""
|
|
44
46
|
from usdata.readers import open_asset
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""GOES ABI CONUS Cloud and Moisture Imagery from anonymous NOAA S3 buckets.
|
|
2
|
+
|
|
3
|
+
Require ``satellite`` (16, 17, 18, or 19), ``channel`` (1--16 or C01--C16), and
|
|
4
|
+
both timestamps. The optional ``product`` must be ``ABI-L2-CMIPC``. Select whole
|
|
5
|
+
single-channel NetCDF files by inclusive scan-start time, never by scan overlap.
|
|
6
|
+
Geographic and variable subsetting are not available for these archived files.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from datetime import UTC, datetime, timedelta
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
import httpx
|
|
16
|
+
|
|
17
|
+
from usdata.models import Asset, Dataset, Protocol, Query, TimeRange
|
|
18
|
+
from usdata.protocols import http, s3
|
|
19
|
+
from usdata.providers.base import Provider, QueryError
|
|
20
|
+
|
|
21
|
+
PRODUCT = "ABI-L2-CMIPC"
|
|
22
|
+
PUBLIC_START = datetime(2017, 2, 28, tzinfo=UTC)
|
|
23
|
+
KEY_RE = re.compile(
|
|
24
|
+
r"OR_ABI-L2-CMIPC-M[346]C(?P<channel>0[1-9]|1[0-6])_G(?P<satellite>1[6-9])"
|
|
25
|
+
r"_s(?P<start>\d{14})_e(?P<end>\d{14})_c\d{14}\.nc"
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _timestamp(raw: str) -> datetime:
|
|
30
|
+
stamp = datetime.strptime(raw, "%Y%j%H%M%S%f").replace(tzinfo=UTC)
|
|
31
|
+
# strptime accepts day 366 in a non-leap year by spilling into the next year.
|
|
32
|
+
if stamp.strftime("%Y%j%H%M%S") + str(stamp.microsecond // 100000) != raw:
|
|
33
|
+
raise ValueError("invalid ABI timestamp")
|
|
34
|
+
return stamp
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _number(raw: object, label: str, low: int, high: int) -> int:
|
|
38
|
+
if isinstance(raw, bool) or not isinstance(raw, (str, int)):
|
|
39
|
+
raise QueryError(f"{label} must be an integer from {low} to {high}")
|
|
40
|
+
text = str(raw).strip()
|
|
41
|
+
if label == "channel" and text.startswith("C"):
|
|
42
|
+
text = text[1:]
|
|
43
|
+
if not text.isascii() or not text.isdigit() or not low <= int(text) <= high:
|
|
44
|
+
raise QueryError(f"{label} must be an integer from {low} to {high}")
|
|
45
|
+
return int(text)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class GoesAbi(Provider):
|
|
49
|
+
"""Single-channel CONUS ABI imagery; params: satellite, channel, product."""
|
|
50
|
+
|
|
51
|
+
def __init__(self, dataset: Dataset, client: httpx.Client | None = None) -> None:
|
|
52
|
+
super().__init__(dataset)
|
|
53
|
+
self._client = client
|
|
54
|
+
self._owns_client = client is None
|
|
55
|
+
|
|
56
|
+
def close(self) -> None:
|
|
57
|
+
"""Close the owned HTTP client, leaving injected clients to their caller."""
|
|
58
|
+
if self._owns_client and self._client is not None:
|
|
59
|
+
self._client.close()
|
|
60
|
+
self._client = None
|
|
61
|
+
|
|
62
|
+
def _http(self) -> httpx.Client:
|
|
63
|
+
if self._client is None:
|
|
64
|
+
self._client = http.client()
|
|
65
|
+
return self._client
|
|
66
|
+
|
|
67
|
+
def list_assets(self, query: Query) -> list[Asset]:
|
|
68
|
+
"""List complete scenes whose scan starts fall inside the inclusive UTC interval."""
|
|
69
|
+
if unknown := set(query.params) - {"satellite", "channel", "product"}:
|
|
70
|
+
raise QueryError(f"unsupported GOES params: {', '.join(sorted(unknown))}")
|
|
71
|
+
if query.params.get("product", PRODUCT) != PRODUCT:
|
|
72
|
+
raise QueryError(f"only product={PRODUCT} is supported")
|
|
73
|
+
if query.bbox is not None:
|
|
74
|
+
raise QueryError(
|
|
75
|
+
"GOES imagery has no geographic subsetting; omit location/bbox/lat/lon"
|
|
76
|
+
)
|
|
77
|
+
if query.text:
|
|
78
|
+
raise QueryError("GOES imagery has no text filtering; select satellite and channel")
|
|
79
|
+
if query.variables:
|
|
80
|
+
raise QueryError("GOES imagery has no variable subsetting; select a channel instead")
|
|
81
|
+
if query.time is None or query.time.start is None or query.time.end is None:
|
|
82
|
+
raise QueryError(f"{self.dataset.id} requires both start and end times")
|
|
83
|
+
satellite = _number(query.params.get("satellite"), "satellite", 16, 19)
|
|
84
|
+
channel = _number(query.params.get("channel"), "channel", 1, 16)
|
|
85
|
+
start = query.time.start.replace(tzinfo=query.time.start.tzinfo or UTC).astimezone(UTC)
|
|
86
|
+
end = query.time.end.replace(tzinfo=query.time.end.tzinfo or UTC).astimezone(UTC)
|
|
87
|
+
if end < PUBLIC_START:
|
|
88
|
+
raise QueryError("GOES CMIPC public observations begin on 2017-02-28")
|
|
89
|
+
start = max(start, PUBLIC_START)
|
|
90
|
+
hour = start.replace(minute=0, second=0, microsecond=0)
|
|
91
|
+
bucket = f"noaa-goes{satellite}"
|
|
92
|
+
assets: dict[str, Asset] = {}
|
|
93
|
+
while hour <= end:
|
|
94
|
+
prefix = f"{PRODUCT}/{hour:%Y/%j/%H}/"
|
|
95
|
+
for obj in s3.list_objects(bucket, prefix, self._http()):
|
|
96
|
+
if not obj.key.startswith(prefix):
|
|
97
|
+
continue
|
|
98
|
+
name = obj.key.removeprefix(prefix)
|
|
99
|
+
match = KEY_RE.fullmatch(name)
|
|
100
|
+
if (
|
|
101
|
+
match is None
|
|
102
|
+
or int(match["satellite"]) != satellite
|
|
103
|
+
or int(match["channel"]) != channel
|
|
104
|
+
):
|
|
105
|
+
continue
|
|
106
|
+
try:
|
|
107
|
+
scan_start, scan_end = _timestamp(match["start"]), _timestamp(match["end"])
|
|
108
|
+
except ValueError:
|
|
109
|
+
continue
|
|
110
|
+
if scan_end < scan_start or not start <= scan_start <= end:
|
|
111
|
+
continue
|
|
112
|
+
assets[obj.key] = Asset(
|
|
113
|
+
id=name,
|
|
114
|
+
dataset_id=self.dataset.id,
|
|
115
|
+
href=f"s3://{bucket}/{obj.key}",
|
|
116
|
+
protocol=Protocol.S3,
|
|
117
|
+
media_type="application/x-netcdf",
|
|
118
|
+
size=obj.size,
|
|
119
|
+
time=TimeRange(start=scan_start, end=scan_end),
|
|
120
|
+
)
|
|
121
|
+
hour += timedelta(hours=1)
|
|
122
|
+
return sorted(assets.values(), key=lambda asset: asset.id)
|
|
123
|
+
|
|
124
|
+
def fetch(self, asset: Asset, dest: Path) -> Path:
|
|
125
|
+
"""Download the complete archived NetCDF object without modification."""
|
|
126
|
+
return s3.download(asset.href, dest, self._http())
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
"""Annual Storm Events details archives from the NCEI bulk directory.
|
|
2
|
+
|
|
3
|
+
Both dates are required. Every UTC calendar year touched by the interval is
|
|
4
|
+
selected in full; geographic, variable, text, and provider-specific filters are
|
|
5
|
+
unsupported. The most recently created supported details file wins per year.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
from datetime import UTC, datetime
|
|
12
|
+
from html.parser import HTMLParser
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
import httpx
|
|
16
|
+
|
|
17
|
+
from usdata.models import Asset, Dataset, Protocol, Query, TimeRange
|
|
18
|
+
from usdata.protocols import http
|
|
19
|
+
from usdata.providers.base import Provider, QueryError
|
|
20
|
+
|
|
21
|
+
DIRECTORY_URL = "https://www.ncei.noaa.gov/pub/data/swdi/stormevents/csvfiles/"
|
|
22
|
+
DETAILS_NAME = re.compile(r"StormEvents_details-ftp_v1\.0_d(\d{4})_c(\d{8})\.csv\.gz", re.ASCII)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class _Directory(HTMLParser):
|
|
26
|
+
"""Read filenames and exact byte sizes from NCEI's HTML directory table."""
|
|
27
|
+
|
|
28
|
+
def __init__(self) -> None:
|
|
29
|
+
super().__init__()
|
|
30
|
+
self.files: list[tuple[str, int | None]] = []
|
|
31
|
+
self._cells: list[str] = []
|
|
32
|
+
self._name: str | None = None
|
|
33
|
+
self._in_cell = False
|
|
34
|
+
|
|
35
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
36
|
+
if tag == "tr":
|
|
37
|
+
self._cells, self._name = [], None
|
|
38
|
+
elif tag == "td":
|
|
39
|
+
self._cells.append("")
|
|
40
|
+
self._in_cell = True
|
|
41
|
+
elif tag == "a":
|
|
42
|
+
href = dict(attrs).get("href", "") or ""
|
|
43
|
+
# Only literal local filenames; never follow arbitrary links from a listing.
|
|
44
|
+
if DETAILS_NAME.fullmatch(href):
|
|
45
|
+
self._name = href
|
|
46
|
+
|
|
47
|
+
def handle_data(self, data: str) -> None:
|
|
48
|
+
if self._in_cell:
|
|
49
|
+
self._cells[-1] += data
|
|
50
|
+
|
|
51
|
+
def handle_endtag(self, tag: str) -> None:
|
|
52
|
+
if tag == "td":
|
|
53
|
+
self._in_cell = False
|
|
54
|
+
elif tag == "tr" and self._name is not None:
|
|
55
|
+
size = self._cells[2].strip() if len(self._cells) > 2 else ""
|
|
56
|
+
self.files.append(
|
|
57
|
+
(self._name, int(size) if size.isascii() and size.isdigit() else None)
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class StormEvents(Provider):
|
|
62
|
+
"""Resolve whole-year details archives; preserve the original gzip bytes."""
|
|
63
|
+
|
|
64
|
+
def __init__(self, dataset: Dataset, *, client: httpx.Client | None = None) -> None:
|
|
65
|
+
super().__init__(dataset)
|
|
66
|
+
self._client = client
|
|
67
|
+
self._owns_client = client is None
|
|
68
|
+
|
|
69
|
+
def _http(self) -> httpx.Client:
|
|
70
|
+
if self._client is None:
|
|
71
|
+
self._client = http.client()
|
|
72
|
+
return self._client
|
|
73
|
+
|
|
74
|
+
def close(self) -> None:
|
|
75
|
+
"""Close an internally created HTTP client; injected clients stay open."""
|
|
76
|
+
if self._owns_client and self._client is not None:
|
|
77
|
+
self._client.close()
|
|
78
|
+
self._client = None
|
|
79
|
+
|
|
80
|
+
def list_assets(self, query: Query) -> list[Asset]:
|
|
81
|
+
"""Select the latest supported details revision for every requested year."""
|
|
82
|
+
if query.params:
|
|
83
|
+
raise QueryError(f"unsupported Storm Events params: {', '.join(sorted(query.params))}")
|
|
84
|
+
if query.bbox is not None or query.variables or query.text:
|
|
85
|
+
raise QueryError(
|
|
86
|
+
"Storm Events downloads whole annual details files; location/bbox, variables, "
|
|
87
|
+
"and text filters are unsupported. Filter locally after opening the CSV."
|
|
88
|
+
)
|
|
89
|
+
if query.time is None or query.time.start is None or query.time.end is None:
|
|
90
|
+
raise QueryError(f"{self.dataset.id} requires both start and end dates")
|
|
91
|
+
start = query.time.start.replace(tzinfo=query.time.start.tzinfo or UTC).astimezone(UTC)
|
|
92
|
+
end = query.time.end.replace(tzinfo=query.time.end.tzinfo or UTC).astimezone(UTC)
|
|
93
|
+
if start.year < 1950:
|
|
94
|
+
raise QueryError("Storm Events annual details files start in 1950")
|
|
95
|
+
years = range(start.year, end.year + 1)
|
|
96
|
+
listing = _Directory()
|
|
97
|
+
listing.feed(http.get(DIRECTORY_URL, self._http()).text)
|
|
98
|
+
selected: dict[int, tuple[str, int | None]] = {}
|
|
99
|
+
for name, size in listing.files:
|
|
100
|
+
match = DETAILS_NAME.fullmatch(name)
|
|
101
|
+
assert match is not None
|
|
102
|
+
year = int(match[1])
|
|
103
|
+
if year not in years:
|
|
104
|
+
continue
|
|
105
|
+
try:
|
|
106
|
+
datetime.strptime(match[2], "%Y%m%d")
|
|
107
|
+
except ValueError:
|
|
108
|
+
continue
|
|
109
|
+
# Fixed-width YYYYMMDD names sort by creation date. Duplicate rows are harmless.
|
|
110
|
+
if year not in selected or name > selected[year][0]:
|
|
111
|
+
selected[year] = name, size
|
|
112
|
+
if missing := [str(year) for year in years if year not in selected]:
|
|
113
|
+
raise QueryError(
|
|
114
|
+
"no supported Storm Events details file for year(s): " + ", ".join(missing)
|
|
115
|
+
)
|
|
116
|
+
return [
|
|
117
|
+
Asset(
|
|
118
|
+
id=selected[year][0],
|
|
119
|
+
dataset_id=self.dataset.id,
|
|
120
|
+
href=DIRECTORY_URL + selected[year][0],
|
|
121
|
+
protocol=Protocol.HTTP,
|
|
122
|
+
media_type="application/gzip",
|
|
123
|
+
size=selected[year][1],
|
|
124
|
+
time=TimeRange(
|
|
125
|
+
start=datetime(year, 1, 1, tzinfo=UTC),
|
|
126
|
+
end=datetime(year, 12, 31, 23, 59, 59, 999999, tzinfo=UTC),
|
|
127
|
+
),
|
|
128
|
+
)
|
|
129
|
+
for year in years
|
|
130
|
+
]
|
|
131
|
+
|
|
132
|
+
def fetch(self, asset: Asset, dest: Path) -> Path:
|
|
133
|
+
"""Download the pinned annual archive, without decompressing or subsetting."""
|
|
134
|
+
return http.download(asset.href, dest, self._http())
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
"""Optional readers for local fetched files; never fetch or modify cached bytes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import gzip
|
|
7
|
+
import io
|
|
8
|
+
from importlib import import_module
|
|
9
|
+
from typing import TYPE_CHECKING, Any
|
|
10
|
+
|
|
11
|
+
from usdata.models import Protocol
|
|
12
|
+
|
|
13
|
+
if TYPE_CHECKING:
|
|
14
|
+
from usdata.fetch import FetchedAsset
|
|
15
|
+
|
|
16
|
+
NETCDF_MEDIA_TYPES = {"application/x-netcdf", "application/netcdf", "application/x-netcdf4"}
|
|
17
|
+
CSV_MEDIA_TYPES = {"text/csv", "application/csv"}
|
|
18
|
+
GZIP_MEDIA_TYPES = {"application/gzip", "application/x-gzip"}
|
|
19
|
+
IDENTIFIER_COLUMNS = {
|
|
20
|
+
"station",
|
|
21
|
+
"station_id",
|
|
22
|
+
"site_no",
|
|
23
|
+
"monitoring_location_id",
|
|
24
|
+
"parameter_code",
|
|
25
|
+
"statistic_id",
|
|
26
|
+
"event_id",
|
|
27
|
+
"episode_id",
|
|
28
|
+
"state_fips",
|
|
29
|
+
"cz_fips",
|
|
30
|
+
"tor_other_cz_fips",
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class MissingReaderDependency(ImportError):
|
|
35
|
+
"""The optional dependency required to open an asset is not installed."""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class UnsupportedFormat(ValueError):
|
|
39
|
+
"""No reader is implemented for this asset's format."""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def open_asset(
|
|
43
|
+
fetched: FetchedAsset,
|
|
44
|
+
*,
|
|
45
|
+
reader: str | None = None,
|
|
46
|
+
dtype: dict[str, str] | None = None,
|
|
47
|
+
parse_dates: list[str] | None = None,
|
|
48
|
+
usecols: list[str] | None = None,
|
|
49
|
+
nrows: int | None = None,
|
|
50
|
+
) -> Any:
|
|
51
|
+
"""Open local CSV, NetCDF4, or NEXRAD data, retaining units and provenance.
|
|
52
|
+
|
|
53
|
+
Gzip CSVs are decompressed locally without changing cached bytes.
|
|
54
|
+
Infer ``csv`` or ``erddap-csv`` from media type and protocol, or use an
|
|
55
|
+
explicit reader for ambiguous metadata. Identifier columns default to pandas
|
|
56
|
+
strings; explicit dtype entries override those defaults. Dates remain strings
|
|
57
|
+
unless named in parse_dates. No checksum verification or downloading occurs.
|
|
58
|
+
"""
|
|
59
|
+
if reader is None:
|
|
60
|
+
media_type = (fetched.asset.media_type or "").split(";", 1)[0].strip().lower()
|
|
61
|
+
gzip_csv = media_type in GZIP_MEDIA_TYPES and fetched.asset.id.lower().endswith(".csv.gz")
|
|
62
|
+
if fetched.asset.dataset_id == "noaa:nexrad-level2":
|
|
63
|
+
reader = "nexrad-level2"
|
|
64
|
+
elif media_type in NETCDF_MEDIA_TYPES:
|
|
65
|
+
reader = "netcdf"
|
|
66
|
+
elif media_type in CSV_MEDIA_TYPES or gzip_csv:
|
|
67
|
+
reader = "erddap-csv" if fetched.asset.protocol is Protocol.ERDDAP else "csv"
|
|
68
|
+
else:
|
|
69
|
+
raise UnsupportedFormat(
|
|
70
|
+
f"no reader for {fetched.asset.media_type!r}; supported formats are CSV, "
|
|
71
|
+
"ERDDAP CSV, NetCDF4, and NEXRAD Level II. "
|
|
72
|
+
"For a known CSV with ambiguous metadata, "
|
|
73
|
+
"pass reader='csv' "
|
|
74
|
+
"or reader='erddap-csv'; otherwise use fetched.path with a format-specific reader"
|
|
75
|
+
)
|
|
76
|
+
if reader == "netcdf":
|
|
77
|
+
if any(value is not None for value in (dtype, parse_dates, usecols, nrows)):
|
|
78
|
+
raise ValueError(
|
|
79
|
+
"CSV options dtype, parse_dates, usecols and nrows do not apply to NetCDF"
|
|
80
|
+
)
|
|
81
|
+
from usdata._netcdf import open_netcdf
|
|
82
|
+
|
|
83
|
+
return open_netcdf(fetched)
|
|
84
|
+
if reader == "nexrad-level2":
|
|
85
|
+
if any(value is not None for value in (dtype, parse_dates, usecols, nrows)):
|
|
86
|
+
raise ValueError("dtype, parse_dates, usecols, and nrows apply only to CSV readers")
|
|
87
|
+
from usdata._radar import open_nexrad
|
|
88
|
+
|
|
89
|
+
return open_nexrad(fetched)
|
|
90
|
+
if reader not in {"csv", "erddap-csv"}:
|
|
91
|
+
raise UnsupportedFormat(
|
|
92
|
+
f"unsupported reader {reader!r}; use 'csv', 'erddap-csv', 'netcdf', or 'nexrad-level2'"
|
|
93
|
+
)
|
|
94
|
+
try:
|
|
95
|
+
pandas = import_module("pandas")
|
|
96
|
+
except ModuleNotFoundError as error:
|
|
97
|
+
if error.name != "pandas":
|
|
98
|
+
raise
|
|
99
|
+
raise MissingReaderDependency(
|
|
100
|
+
'CSV reading requires pandas; install it with: pip install "usdata[pandas]" '
|
|
101
|
+
'(or uv add "usdata[pandas]")'
|
|
102
|
+
) from error
|
|
103
|
+
|
|
104
|
+
# Pass a file object to pandas: reading a fetched asset is strictly local.
|
|
105
|
+
with fetched.path.open("rb") as raw:
|
|
106
|
+
compressed = raw.read(2) == b"\x1f\x8b"
|
|
107
|
+
raw.seek(0)
|
|
108
|
+
binary = gzip.GzipFile(fileobj=raw) if compressed else raw
|
|
109
|
+
with io.TextIOWrapper(binary, encoding="utf-8-sig", newline="") as stream:
|
|
110
|
+
records = csv.reader(stream)
|
|
111
|
+
columns = next(records, [])
|
|
112
|
+
if (
|
|
113
|
+
not columns
|
|
114
|
+
or any(not column for column in columns)
|
|
115
|
+
or len(set(columns)) != len(columns)
|
|
116
|
+
):
|
|
117
|
+
raise ValueError("CSV must have a non-empty header with unique column names")
|
|
118
|
+
units = {}
|
|
119
|
+
if reader == "erddap-csv":
|
|
120
|
+
values = next(records, [])
|
|
121
|
+
if len(values) != len(columns):
|
|
122
|
+
raise ValueError("ERDDAP CSV must have a units row matching the header")
|
|
123
|
+
units = dict(zip(columns, values, strict=True))
|
|
124
|
+
types = {name: "string" for name in columns if name.casefold() in IDENTIFIER_COLUMNS}
|
|
125
|
+
types.update(dtype or {})
|
|
126
|
+
frame = pandas.read_csv(
|
|
127
|
+
stream,
|
|
128
|
+
header=None,
|
|
129
|
+
names=columns,
|
|
130
|
+
dtype=types,
|
|
131
|
+
parse_dates=parse_dates,
|
|
132
|
+
usecols=usecols,
|
|
133
|
+
nrows=nrows,
|
|
134
|
+
)
|
|
135
|
+
if units:
|
|
136
|
+
frame.attrs["units"] = {name: units[name] for name in frame.columns}
|
|
137
|
+
frame.attrs["usdata"] = {
|
|
138
|
+
"asset_id": fetched.asset.id,
|
|
139
|
+
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
140
|
+
}
|
|
141
|
+
return frame
|
|
@@ -1,103 +0,0 @@
|
|
|
1
|
-
"""Optional readers for local fetched files; never fetch or modify cached bytes."""
|
|
2
|
-
|
|
3
|
-
from __future__ import annotations
|
|
4
|
-
|
|
5
|
-
import csv
|
|
6
|
-
from importlib import import_module
|
|
7
|
-
from typing import TYPE_CHECKING, Any
|
|
8
|
-
|
|
9
|
-
from usdata.models import Protocol
|
|
10
|
-
|
|
11
|
-
if TYPE_CHECKING:
|
|
12
|
-
from usdata.fetch import FetchedAsset
|
|
13
|
-
|
|
14
|
-
CSV_MEDIA_TYPES = {"text/csv", "application/csv"}
|
|
15
|
-
IDENTIFIER_COLUMNS = {
|
|
16
|
-
"station",
|
|
17
|
-
"station_id",
|
|
18
|
-
"site_no",
|
|
19
|
-
"monitoring_location_id",
|
|
20
|
-
"parameter_code",
|
|
21
|
-
"statistic_id",
|
|
22
|
-
}
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
class MissingReaderDependency(ImportError):
|
|
26
|
-
"""The optional dependency required to open an asset is not installed."""
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
class UnsupportedFormat(ValueError):
|
|
30
|
-
"""No reader is implemented for this asset's format."""
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
def open_asset(
|
|
34
|
-
fetched: FetchedAsset,
|
|
35
|
-
*,
|
|
36
|
-
reader: str | None = None,
|
|
37
|
-
dtype: dict[str, str] | None = None,
|
|
38
|
-
parse_dates: list[str] | None = None,
|
|
39
|
-
usecols: list[str] | None = None,
|
|
40
|
-
nrows: int | None = None,
|
|
41
|
-
) -> Any:
|
|
42
|
-
"""Read a local CSV into a pandas DataFrame, retaining units and provenance.
|
|
43
|
-
|
|
44
|
-
Infer ``csv`` or ``erddap-csv`` from media type and protocol, or use an
|
|
45
|
-
explicit reader for ambiguous metadata. Identifier columns default to pandas
|
|
46
|
-
strings; explicit dtype entries override those defaults. Dates remain strings
|
|
47
|
-
unless named in parse_dates. No checksum verification or downloading occurs.
|
|
48
|
-
"""
|
|
49
|
-
if reader is None:
|
|
50
|
-
media_type = (fetched.asset.media_type or "").split(";", 1)[0].strip().lower()
|
|
51
|
-
if media_type not in CSV_MEDIA_TYPES:
|
|
52
|
-
raise UnsupportedFormat(
|
|
53
|
-
f"no reader for {fetched.asset.media_type!r}; supported formats are CSV and "
|
|
54
|
-
"ERDDAP CSV. For a known CSV with ambiguous metadata, pass reader='csv' "
|
|
55
|
-
"or reader='erddap-csv'; otherwise use fetched.path with a format-specific reader"
|
|
56
|
-
)
|
|
57
|
-
reader = "erddap-csv" if fetched.asset.protocol is Protocol.ERDDAP else "csv"
|
|
58
|
-
if reader not in {"csv", "erddap-csv"}:
|
|
59
|
-
raise UnsupportedFormat(f"unsupported reader {reader!r}; use 'csv' or 'erddap-csv'")
|
|
60
|
-
try:
|
|
61
|
-
pandas = import_module("pandas")
|
|
62
|
-
except ModuleNotFoundError as error:
|
|
63
|
-
if error.name != "pandas":
|
|
64
|
-
raise
|
|
65
|
-
raise MissingReaderDependency(
|
|
66
|
-
'CSV reading requires pandas; install it with: pip install "usdata[pandas]" '
|
|
67
|
-
'(or uv add "usdata[pandas]")'
|
|
68
|
-
) from error
|
|
69
|
-
|
|
70
|
-
# Pass a file object to pandas: reading a fetched asset is strictly local.
|
|
71
|
-
with fetched.path.open(encoding="utf-8-sig", newline="") as stream:
|
|
72
|
-
records = csv.reader(stream)
|
|
73
|
-
columns = next(records, [])
|
|
74
|
-
if (
|
|
75
|
-
not columns
|
|
76
|
-
or any(not column for column in columns)
|
|
77
|
-
or len(set(columns)) != len(columns)
|
|
78
|
-
):
|
|
79
|
-
raise ValueError("CSV must have a non-empty header with unique column names")
|
|
80
|
-
units = {}
|
|
81
|
-
if reader == "erddap-csv":
|
|
82
|
-
values = next(records, [])
|
|
83
|
-
if len(values) != len(columns):
|
|
84
|
-
raise ValueError("ERDDAP CSV must have a units row matching the header")
|
|
85
|
-
units = dict(zip(columns, values, strict=True))
|
|
86
|
-
types = {name: "string" for name in columns if name.casefold() in IDENTIFIER_COLUMNS}
|
|
87
|
-
types.update(dtype or {})
|
|
88
|
-
frame = pandas.read_csv(
|
|
89
|
-
stream,
|
|
90
|
-
header=None,
|
|
91
|
-
names=columns,
|
|
92
|
-
dtype=types,
|
|
93
|
-
parse_dates=parse_dates,
|
|
94
|
-
usecols=usecols,
|
|
95
|
-
nrows=nrows,
|
|
96
|
-
)
|
|
97
|
-
if units:
|
|
98
|
-
frame.attrs["units"] = {name: units[name] for name in frame.columns}
|
|
99
|
-
frame.attrs["usdata"] = {
|
|
100
|
-
"asset_id": fetched.asset.id,
|
|
101
|
-
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
102
|
-
}
|
|
103
|
-
return frame
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|