usdata 0.7.0__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. {usdata-0.7.0 → usdata-0.8.0}/PKG-INFO +53 -9
  2. {usdata-0.7.0 → usdata-0.8.0}/README.md +47 -8
  3. {usdata-0.7.0 → usdata-0.8.0}/pyproject.toml +13 -1
  4. {usdata-0.7.0 → usdata-0.8.0}/pyproject.toml.orig +10 -1
  5. usdata-0.8.0/src/usdata/_netcdf.py +36 -0
  6. usdata-0.8.0/src/usdata/_radar.py +73 -0
  7. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/data/registry.yaml +23 -19
  8. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/fetch.py +4 -2
  9. usdata-0.8.0/src/usdata/providers/noaa/goes.py +126 -0
  10. usdata-0.8.0/src/usdata/providers/noaa/storm_events.py +134 -0
  11. usdata-0.8.0/src/usdata/readers.py +141 -0
  12. usdata-0.7.0/src/usdata/readers.py +0 -103
  13. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/__init__.py +0 -0
  14. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/_files.py +0 -0
  15. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/_progress.py +0 -0
  16. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/cache.py +0 -0
  17. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/cli/__init__.py +0 -0
  18. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/cli/app.py +0 -0
  19. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/cli/progress.py +0 -0
  20. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/data/nexrad_sites.csv +0 -0
  21. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/data/places.csv +0 -0
  22. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/data/places.sources.json +0 -0
  23. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/manifest.py +0 -0
  24. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/models.py +0 -0
  25. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/protocols/__init__.py +0 -0
  26. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/protocols/erddap.py +0 -0
  27. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/protocols/http.py +0 -0
  28. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/protocols/s3.py +0 -0
  29. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/provenance.py +0 -0
  30. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/__init__.py +0 -0
  31. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/base.py +0 -0
  32. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/noaa/__init__.py +0 -0
  33. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
  34. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/noaa/ghcnd.py +0 -0
  35. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/noaa/gsom.py +0 -0
  36. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/noaa/nexrad.py +0 -0
  37. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/noaa/sites.py +0 -0
  38. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/usgs/__init__.py +0 -0
  39. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/providers/usgs/daily.py +0 -0
  40. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/pull.py +0 -0
  41. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/py.typed +0 -0
  42. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/query.py +0 -0
  43. {usdata-0.7.0 → usdata-0.8.0}/src/usdata/registry.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: usdata
3
- Version: 0.7.0
3
+ Version: 0.8.0
4
4
  Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
5
5
  Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
6
6
  Author: Jake Van Slyke
@@ -14,11 +14,16 @@ Requires-Dist: httpx>=0.28.1
14
14
  Requires-Dist: pydantic>=2.7
15
15
  Requires-Dist: pyyaml>=6.0
16
16
  Requires-Dist: typer>=0.12
17
+ Requires-Dist: xarray>=2025.1 ; extra == 'netcdf'
18
+ Requires-Dist: h5netcdf[h5py]>=1.8.1 ; extra == 'netcdf'
17
19
  Requires-Dist: pandas>=3.0 ; extra == 'pandas'
20
+ Requires-Dist: xradar>=0.12.0 ; extra == 'radar'
18
21
  Requires-Python: >=3.11
19
22
  Project-URL: Homepage, https://github.com/jakeryderv/usdata
20
23
  Project-URL: Repository, https://github.com/jakeryderv/usdata
24
+ Provides-Extra: netcdf
21
25
  Provides-Extra: pandas
26
+ Provides-Extra: radar
22
27
  Description-Content-Type: text/markdown
23
28
 
24
29
  # usdata
@@ -26,10 +31,11 @@ Description-Content-Type: text/markdown
26
31
  Unified Python SDK and CLI for discovering, fetching, and tracking the
27
32
  provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
28
33
 
29
- > Status: pre-alpha. v0.7 supports GHCN-Daily, GSOM monthly summaries,
30
- > NEXRAD Level II, USGS daily values, and CoastWatch SST subsets with provenance,
31
- > plus Census state/county lookup, optional pandas CSV readers, and terminal
32
- > download progress. Other datasets are planned.
34
+ > Status: pre-alpha. v0.8 supports GHCN-Daily, GSOM monthly summaries,
35
+ > NEXRAD Level II, GOES ABI CONUS imagery, Storm Events annual archives,
36
+ > USGS daily values, and CoastWatch SST subsets with provenance. It includes
37
+ > optional CSV, radar, and NetCDF4 readers, six executed notebooks, Census
38
+ > state/county lookup, and terminal download progress. Other datasets are planned.
33
39
  > See [docs/roadmap.md](docs/roadmap.md).
34
40
 
35
41
  ## Providers
@@ -37,7 +43,7 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
37
43
  <!-- registry:start -->
38
44
  | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
39
45
  |---|---:|---:|---:|---|---|
40
- | [NOAA](docs/providers/noaa.md) | 4 | 0 | 25 | — | `ghcn-daily`, `gsom`, `nexrad-level2`, `coastwatch-sst`, +25 planned |
46
+ | [NOAA](docs/providers/noaa.md) | 6 | 0 | 23 | — | `ghcn-daily`, `gsom`, `storm-events`, `nexrad-level2`, `goes-abi`, `coastwatch-sst`, +23 planned |
41
47
  | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
42
48
  | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
43
49
  | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
@@ -94,6 +100,17 @@ usdata pull dataset.yaml # resolve, fetch, write dataset.lock.json
94
100
  usdata verify dataset.yaml # exit 1 if any cached input drifted
95
101
  ```
96
102
 
103
+ Storm Events bulk access is available since v0.8. Dates select complete
104
+ annual details archives; filter rows locally after opening the gzip CSV. For example:
105
+
106
+ ```sh
107
+ usdata fetch noaa:storm-events --start 2024-05-01 --end 2024-05-31 --dry-run
108
+ ```
109
+
110
+ This lists the entire 2024 archive. Location and variable filters are rejected;
111
+ see the [executed Storm Events notebook](examples/storm-events/example.ipynb)
112
+ for local filtering and reporting limitations.
113
+
97
114
  Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
98
115
  `USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
99
116
  recording source URL, retrieval time, checksum, size, and license.
@@ -143,6 +160,15 @@ size. Adapters that assemble files from metadata requests show asset-level progr
143
160
  Use `--no-progress` to disable it. Progress is automatically disabled when either
144
161
  stdout or stderr is redirected; existing output lines and exit codes are unchanged.
145
162
 
163
+ For single-channel GOES CONUS imagery (available since v0.8):
164
+
165
+ ```sh
166
+ usdata fetch noaa:goes-abi --start 2024-05-06T12:01:18.1Z --end 2024-05-06T12:01:18.1Z -p satellite=18 -p channel=6
167
+ ```
168
+
169
+ The download is a whole NetCDF scene. See [GOES access notes](docs/providers/noaa.md#goes-abi-conus-imagery)
170
+ for supported selectors and scan-start time semantics.
171
+
146
172
  ## Opening CSV data
147
173
 
148
174
  `FetchedAsset.open()` is available since v0.6 with the optional pandas
@@ -151,16 +177,32 @@ preserves identifier strings, and keeps CoastWatch units as metadata.
151
177
  See the [reader reference](docs/reference/readers.md)
152
178
  and [fetch → open → analyze example](examples/sst-analysis/README.md).
153
179
 
180
+ The [examples directory](examples/README.md) contains executed Jupyter notebooks
181
+ with saved data previews, small plots, and source provenance. Start with weather
182
+ and streamflow for manifest workflows, SST for gridded CSV reading, or monthly
183
+ climate for GSOM observations.
184
+
185
+ NetCDF4 scene opening is available since v0.8 with `usdata[netcdf]`.
186
+ See the executed [GOES infrared notebook](examples/goes-imagery/example.ipynb).
187
+
154
188
  ## Development
155
189
 
156
190
  Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
157
191
 
192
+ `just setup` uses the tested Python 3.14.7 pin in `.python-version`. Older Linux
193
+ uv Python 3.14 builds can crash during NumPy array operations; see
194
+ [the upstream fix](https://github.com/astral-sh/python-build-standalone/issues/991).
195
+
158
196
  ```sh
159
197
  git clone https://github.com/jakeryderv/usdata && cd usdata
160
198
  just setup # install toolchain and dependencies
161
199
  just test # unit tests
162
200
  just check # format, lint, typecheck, offline tests, generated docs, release notices
163
201
  just check-pandas # install the CSV extra and run the same checks
202
+ just check-radar # install the radar extra and run the same checks
203
+ just check-netcdf # install the NetCDF4 extra and run the same checks
204
+ just notebooks # launch the optional Jupyter examples environment
205
+ just run-notebooks # execute notebooks live in fresh kernels and temporary caches
164
206
  just build # build wheel and sdist
165
207
  just smoke # exercise core and pandas wheel installations outside the checkout
166
208
  just run search radar
@@ -168,9 +210,11 @@ just run search radar
168
210
 
169
211
  Unit tests mechanically block network connections. Integration tests that hit
170
212
  live services run with `just test-integration`. CI checks Python 3.11 and 3.14 on
171
- Linux, both with and without pandas, and smoke-tests both installed-wheel
172
- profiles on Linux, macOS, and Windows. The full unit and live-service suites currently run on Linux. `just setup` restores
173
- a core-only development environment; `just check-pandas` installs the extra.
213
+ Linux with core-only, pandas, radar, and NetCDF dependency profiles. Installed-wheel
214
+ checks cover all four profiles on Linux, macOS, and Windows. The full unit and
215
+ live-service suites run on Linux. `just setup` restores a core-only development
216
+ environment; the `check-pandas`, `check-radar`, and `check-netcdf` commands install
217
+ their respective extras.
174
218
 
175
219
  Releases: `just release minor` opens a version-bump PR; merging it publishes
176
220
  to PyPI and creates the tag and GitHub release. See
@@ -3,10 +3,11 @@
3
3
  Unified Python SDK and CLI for discovering, fetching, and tracking the
4
4
  provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
5
5
 
6
- > Status: pre-alpha. v0.7 supports GHCN-Daily, GSOM monthly summaries,
7
- > NEXRAD Level II, USGS daily values, and CoastWatch SST subsets with provenance,
8
- > plus Census state/county lookup, optional pandas CSV readers, and terminal
9
- > download progress. Other datasets are planned.
6
+ > Status: pre-alpha. v0.8 supports GHCN-Daily, GSOM monthly summaries,
7
+ > NEXRAD Level II, GOES ABI CONUS imagery, Storm Events annual archives,
8
+ > USGS daily values, and CoastWatch SST subsets with provenance. It includes
9
+ > optional CSV, radar, and NetCDF4 readers, six executed notebooks, Census
10
+ > state/county lookup, and terminal download progress. Other datasets are planned.
10
11
  > See [docs/roadmap.md](docs/roadmap.md).
11
12
 
12
13
  ## Providers
@@ -14,7 +15,7 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
14
15
  <!-- registry:start -->
15
16
  | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
16
17
  |---|---:|---:|---:|---|---|
17
- | [NOAA](docs/providers/noaa.md) | 4 | 0 | 25 | — | `ghcn-daily`, `gsom`, `nexrad-level2`, `coastwatch-sst`, +25 planned |
18
+ | [NOAA](docs/providers/noaa.md) | 6 | 0 | 23 | — | `ghcn-daily`, `gsom`, `storm-events`, `nexrad-level2`, `goes-abi`, `coastwatch-sst`, +23 planned |
18
19
  | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
19
20
  | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
20
21
  | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
@@ -71,6 +72,17 @@ usdata pull dataset.yaml # resolve, fetch, write dataset.lock.json
71
72
  usdata verify dataset.yaml # exit 1 if any cached input drifted
72
73
  ```
73
74
 
75
+ Storm Events bulk access is available since v0.8. Dates select complete
76
+ annual details archives; filter rows locally after opening the gzip CSV. For example:
77
+
78
+ ```sh
79
+ usdata fetch noaa:storm-events --start 2024-05-01 --end 2024-05-31 --dry-run
80
+ ```
81
+
82
+ This lists the entire 2024 archive. Location and variable filters are rejected;
83
+ see the [executed Storm Events notebook](examples/storm-events/example.ipynb)
84
+ for local filtering and reporting limitations.
85
+
74
86
  Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
75
87
  `USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
76
88
  recording source URL, retrieval time, checksum, size, and license.
@@ -120,6 +132,15 @@ size. Adapters that assemble files from metadata requests show asset-level progr
120
132
  Use `--no-progress` to disable it. Progress is automatically disabled when either
121
133
  stdout or stderr is redirected; existing output lines and exit codes are unchanged.
122
134
 
135
+ For single-channel GOES CONUS imagery (available since v0.8):
136
+
137
+ ```sh
138
+ usdata fetch noaa:goes-abi --start 2024-05-06T12:01:18.1Z --end 2024-05-06T12:01:18.1Z -p satellite=18 -p channel=6
139
+ ```
140
+
141
+ The download is a whole NetCDF scene. See [GOES access notes](docs/providers/noaa.md#goes-abi-conus-imagery)
142
+ for supported selectors and scan-start time semantics.
143
+
123
144
  ## Opening CSV data
124
145
 
125
146
  `FetchedAsset.open()` is available since v0.6 with the optional pandas
@@ -128,16 +149,32 @@ preserves identifier strings, and keeps CoastWatch units as metadata.
128
149
  See the [reader reference](docs/reference/readers.md)
129
150
  and [fetch → open → analyze example](examples/sst-analysis/README.md).
130
151
 
152
+ The [examples directory](examples/README.md) contains executed Jupyter notebooks
153
+ with saved data previews, small plots, and source provenance. Start with weather
154
+ and streamflow for manifest workflows, SST for gridded CSV reading, or monthly
155
+ climate for GSOM observations.
156
+
157
+ NetCDF4 scene opening is available since v0.8 with `usdata[netcdf]`.
158
+ See the executed [GOES infrared notebook](examples/goes-imagery/example.ipynb).
159
+
131
160
  ## Development
132
161
 
133
162
  Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
134
163
 
164
+ `just setup` uses the tested Python 3.14.7 pin in `.python-version`. Older Linux
165
+ uv Python 3.14 builds can crash during NumPy array operations; see
166
+ [the upstream fix](https://github.com/astral-sh/python-build-standalone/issues/991).
167
+
135
168
  ```sh
136
169
  git clone https://github.com/jakeryderv/usdata && cd usdata
137
170
  just setup # install toolchain and dependencies
138
171
  just test # unit tests
139
172
  just check # format, lint, typecheck, offline tests, generated docs, release notices
140
173
  just check-pandas # install the CSV extra and run the same checks
174
+ just check-radar # install the radar extra and run the same checks
175
+ just check-netcdf # install the NetCDF4 extra and run the same checks
176
+ just notebooks # launch the optional Jupyter examples environment
177
+ just run-notebooks # execute notebooks live in fresh kernels and temporary caches
141
178
  just build # build wheel and sdist
142
179
  just smoke # exercise core and pandas wheel installations outside the checkout
143
180
  just run search radar
@@ -145,9 +182,11 @@ just run search radar
145
182
 
146
183
  Unit tests mechanically block network connections. Integration tests that hit
147
184
  live services run with `just test-integration`. CI checks Python 3.11 and 3.14 on
148
- Linux, both with and without pandas, and smoke-tests both installed-wheel
149
- profiles on Linux, macOS, and Windows. The full unit and live-service suites currently run on Linux. `just setup` restores
150
- a core-only development environment; `just check-pandas` installs the extra.
185
+ Linux with core-only, pandas, radar, and NetCDF dependency profiles. Installed-wheel
186
+ checks cover all four profiles on Linux, macOS, and Windows. The full unit and
187
+ live-service suites run on Linux. `just setup` restores a core-only development
188
+ environment; the `check-pandas`, `check-radar`, and `check-netcdf` commands install
189
+ their respective extras.
151
190
 
152
191
  Releases: `just release minor` opens a version-bump PR; merging it publishes
153
192
  to PyPI and creates the tag and GitHub release. See
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.7.0"
3
+ version = "0.8.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -32,6 +32,11 @@ email = "jakervanslyke@gmail.com"
32
32
 
33
33
  [project.optional-dependencies]
34
34
  pandas = ["pandas>=3.0"]
35
+ radar = ["xradar>=0.12.0"]
36
+ netcdf = [
37
+ "xarray>=2025.1",
38
+ "h5netcdf[h5py]>=1.8.1",
39
+ ]
35
40
 
36
41
  [project.urls]
37
42
  Homepage = "https://github.com/jakeryderv/usdata"
@@ -48,6 +53,13 @@ dev = [
48
53
  "respx>=0.23.1",
49
54
  "ruff>=0.6",
50
55
  ]
56
+ examples = [
57
+ "ipykernel>=6.29",
58
+ "jupyterlab>=4.3",
59
+ "matplotlib>=3.9",
60
+ "nbclient>=0.10",
61
+ "pandas>=3.0",
62
+ ]
51
63
 
52
64
  [build-system]
53
65
  requires = ["uv_build>=0.12.5,<0.13.0"]
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.7.0"
3
+ version = "0.8.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -24,6 +24,8 @@ dependencies = [
24
24
 
25
25
  [project.optional-dependencies]
26
26
  pandas = ["pandas>=3.0"]
27
+ radar = ["xradar>=0.12.0"]
28
+ netcdf = ["xarray>=2025.1", "h5netcdf[h5py]>=1.8.1"]
27
29
 
28
30
  [project.urls]
29
31
  Homepage = "https://github.com/jakeryderv/usdata"
@@ -40,6 +42,13 @@ dev = [
40
42
  "respx>=0.23.1",
41
43
  "ruff>=0.6",
42
44
  ]
45
+ examples = [
46
+ "ipykernel>=6.29",
47
+ "jupyterlab>=4.3",
48
+ "matplotlib>=3.9",
49
+ "nbclient>=0.10",
50
+ "pandas>=3.0",
51
+ ]
43
52
 
44
53
  [build-system]
45
54
  requires = ["uv_build>=0.12.5,<0.13.0"]
@@ -0,0 +1,36 @@
1
+ """Local, eagerly loaded NetCDF4 reading behind the netcdf extra."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from importlib import import_module
6
+ from typing import TYPE_CHECKING, Any
7
+
8
+ from usdata.readers import MissingReaderDependency
9
+
10
+ if TYPE_CHECKING:
11
+ from usdata.fetch import FetchedAsset
12
+
13
+
14
+ def open_netcdf(fetched: FetchedAsset) -> Any:
15
+ """Load a NetCDF4 root Dataset, then close every source file handle."""
16
+ try:
17
+ xarray = import_module("xarray")
18
+ import_module("h5netcdf")
19
+ import_module("h5py")
20
+ except ModuleNotFoundError as error:
21
+ if error.name not in {"xarray", "h5netcdf", "h5py"}:
22
+ raise
23
+ raise MissingReaderDependency(
24
+ 'NetCDF4 reading requires xarray and h5netcdf; install: pip install "usdata[netcdf]"'
25
+ ) from error
26
+ # A local file object and fixed engine prevent interpretation as an OPeNDAP URL.
27
+ with (
28
+ fetched.path.open("rb") as stream,
29
+ xarray.open_dataset(stream, engine="h5netcdf", chunks=None) as dataset,
30
+ ):
31
+ dataset.load()
32
+ dataset.attrs["usdata"] = {
33
+ "asset_id": fetched.asset.id,
34
+ "provenance": fetched.provenance.model_dump(mode="json"),
35
+ }
36
+ return dataset
@@ -0,0 +1,73 @@
1
+ """Local NEXRAD Level II decoding behind the radar extra."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import bz2
6
+ import gzip
7
+ from importlib import import_module
8
+ from typing import TYPE_CHECKING, Any
9
+
10
+ from usdata.readers import MissingReaderDependency
11
+
12
+ # NOAA RDA/RPG ICD 2620002Y, Table XVII-I notes 21 and 30.
13
+ MOMENT_FLAG_COUNTS = {
14
+ "DBZH": 2,
15
+ "VRADH": 2,
16
+ "WRADH": 2,
17
+ "ZDR": 2,
18
+ "PHIDP": 2,
19
+ "RHOHV": 2,
20
+ "CCORH": 8,
21
+ }
22
+
23
+ if TYPE_CHECKING:
24
+ from usdata.fetch import FetchedAsset
25
+
26
+
27
+ def open_nexrad(fetched: FetchedAsset) -> Any:
28
+ """Decode a local volume into a fully loaded xarray DataTree."""
29
+ try:
30
+ xradar = import_module("xradar")
31
+ except ModuleNotFoundError as error:
32
+ if error.name != "xradar":
33
+ raise
34
+ raise MissingReaderDependency(
35
+ 'NEXRAD reading requires xradar; install it with: pip install "usdata[radar]" '
36
+ '(or uv add "usdata[radar]")'
37
+ ) from error
38
+
39
+ # Bytes prevent remote URL interpretation and work across the backend's
40
+ # repeated sweep reads. Compressed source files stay unchanged in the cache.
41
+ content = fetched.path.read_bytes()
42
+ if content.startswith(b"\x1f\x8b"):
43
+ content = gzip.decompress(content)
44
+ elif content.startswith(b"BZh"):
45
+ content = bz2.decompress(content)
46
+ radar = xradar.io.open_nexradlevel2_datatree(content, incomplete_sweep="pad")
47
+ try:
48
+ radar.load()
49
+ finally:
50
+ radar.close()
51
+ for node in radar.subtree:
52
+ for name, variable in node.ds.variables.items():
53
+ # The backend records the entire input byte string as `source`.
54
+ # Provenance below is the durable reference, not that decoder buffer.
55
+ variable.encoding.pop("source", None)
56
+ if name in MOMENT_FLAG_COUNTS and "range" in variable.dims:
57
+ scale = variable.encoding.get("scale_factor")
58
+ offset = variable.encoding.get("add_offset")
59
+ if scale is not None and offset is not None:
60
+ # xradar 0.12 does not supply _FillValue for NEXRAD flags.
61
+ # Compare using each moment's native scale, not fixed units.
62
+ data = node[name]
63
+ valid = data.notnull()
64
+ for code in range(MOMENT_FLAG_COUNTS[name]):
65
+ valid = valid & (data != offset + code * scale)
66
+ masked = data.where(valid)
67
+ masked.encoding = data.encoding.copy()
68
+ node[name] = masked
69
+ radar.attrs["usdata"] = {
70
+ "asset_id": fetched.asset.id,
71
+ "provenance": fetched.provenance.model_dump(mode="json"),
72
+ }
73
+ return radar
@@ -105,21 +105,22 @@ datasets:
105
105
 
106
106
  - id: noaa:goes-abi
107
107
  provider: noaa
108
- status: planned
108
+ status: available
109
109
  domain: weather-satellites
110
- target: later
111
- title: GOES-R ABI Satellite Imagery
112
- description: >-
113
- Advanced Baseline Imager products from GOES-16, 18, and 19 in the public
114
- noaa-goes16/18/19 S3 buckets, laid out as PRODUCT/YYYY/DDD/HH/ with one
115
- NetCDF per scan (for example ABI-L2-CMIPC). Same anonymous S3 pattern as
116
- NEXRAD, plus product, satellite, and channel selection.
117
- keywords: [satellite, imagery, goes, abi, clouds, fire, radiance, netcdf]
110
+ since: "0.8"
111
+ title: GOES-R ABI CONUS Cloud and Moisture Imagery
112
+ description: >-
113
+ Single-channel CONUS Cloud and Moisture Imagery (ABI-L2-CMIPC) from
114
+ GOES-16, 17, 18, and 19 in anonymous NOAA S3 buckets. Select an explicit
115
+ satellite, channel, and scan-start interval; each asset is a complete
116
+ NetCDF scene with no geographic or variable subsetting.
117
+ keywords: [satellite, imagery, goes, abi, clouds, infrared, reflectance, netcdf, conus]
118
118
  protocol: s3
119
119
  homepage: https://registry.opendata.aws/noaa-goes/
120
120
  license: US Government Work (public domain)
121
- temporal_extent: { start: "2017-01-01T00:00:00Z" }
122
- capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: true }
121
+ temporal_extent: { start: "2017-02-28T00:00:00Z" }
122
+ capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: false }
123
+ adapter: usdata.providers.noaa.goes:GoesAbi
123
124
 
124
125
  - id: noaa:goes-glm
125
126
  provider: noaa
@@ -139,21 +140,24 @@ datasets:
139
140
 
140
141
  - id: noaa:storm-events
141
142
  provider: noaa
142
- status: planned
143
+ status: available
143
144
  domain: severe-weather
144
- target: later
145
+ since: "0.8"
145
146
  title: Storm Events Database
146
147
  description: >-
147
- NCEI's record of significant weather events since 1950 (tornadoes, hail,
148
- wind, floods, and more) with locations, damage, and narratives. Published
149
- as per-year gzipped CSV files (details, fatalities, locations) in a plain
150
- HTTPS directory; the first NCEI bulk-directory dataset.
148
+ NCEI's significant-weather event details since 1950, with locations,
149
+ impacts, and narratives. Anonymous whole-year gzipped CSV archives;
150
+ select the latest creation-date revision for each requested year.
151
+ No server-side row, location, or variable subsetting. Historical event
152
+ coverage and reporting practices vary; fatalities and locations tables
153
+ are separate products not included by this adapter.
151
154
  keywords: [storms, tornado, hail, wind, flood, damage, severe weather, events]
152
155
  protocol: http
153
- homepage: https://www.ncdc.noaa.gov/stormevents/
156
+ homepage: https://www.ncei.noaa.gov/access/storm-events-database/
154
157
  license: US Government Work (public domain)
155
158
  temporal_extent: { start: "1950-01-01T00:00:00Z" }
156
- capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: false }
159
+ capabilities: { spatial_subset: false, temporal_subset: false, variable_subset: false }
160
+ adapter: usdata.providers.noaa.storm_events:StormEvents
157
161
 
158
162
  - id: noaa:hurdat2
159
163
  provider: noaa
@@ -35,10 +35,12 @@ class FetchedAsset(BaseModel):
35
35
  usecols: list[str] | None = None,
36
36
  nrows: int | None = None,
37
37
  ) -> Any:
38
- """Open this local CSV as a DataFrame; requires the ``pandas`` extra.
38
+ """Open local data with an optional ``pandas``, ``radar``, or ``netcdf`` reader.
39
39
 
40
40
  ERDDAP units are kept in ``frame.attrs["units"]`` and source provenance
41
- in ``frame.attrs["usdata"]``. See ``usdata.readers.open_asset`` for options.
41
+ in ``frame.attrs["usdata"]``. NEXRAD returns a xarray DataTree with provenance
42
+ in ``radar.attrs["usdata"]``. NetCDF4 returns a loaded xarray Dataset with
43
+ matching provenance in its attributes. See ``usdata.readers.open_asset`` for options.
42
44
  Cached files and provenance sidecars are never changed.
43
45
  """
44
46
  from usdata.readers import open_asset
@@ -0,0 +1,126 @@
1
+ """GOES ABI CONUS Cloud and Moisture Imagery from anonymous NOAA S3 buckets.
2
+
3
+ Require ``satellite`` (16, 17, 18, or 19), ``channel`` (1--16 or C01--C16), and
4
+ both timestamps. The optional ``product`` must be ``ABI-L2-CMIPC``. Select whole
5
+ single-channel NetCDF files by inclusive scan-start time, never by scan overlap.
6
+ Geographic and variable subsetting are not available for these archived files.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import re
12
+ from datetime import UTC, datetime, timedelta
13
+ from pathlib import Path
14
+
15
+ import httpx
16
+
17
+ from usdata.models import Asset, Dataset, Protocol, Query, TimeRange
18
+ from usdata.protocols import http, s3
19
+ from usdata.providers.base import Provider, QueryError
20
+
21
+ PRODUCT = "ABI-L2-CMIPC"
22
+ PUBLIC_START = datetime(2017, 2, 28, tzinfo=UTC)
23
+ KEY_RE = re.compile(
24
+ r"OR_ABI-L2-CMIPC-M[346]C(?P<channel>0[1-9]|1[0-6])_G(?P<satellite>1[6-9])"
25
+ r"_s(?P<start>\d{14})_e(?P<end>\d{14})_c\d{14}\.nc"
26
+ )
27
+
28
+
29
+ def _timestamp(raw: str) -> datetime:
30
+ stamp = datetime.strptime(raw, "%Y%j%H%M%S%f").replace(tzinfo=UTC)
31
+ # strptime accepts day 366 in a non-leap year by spilling into the next year.
32
+ if stamp.strftime("%Y%j%H%M%S") + str(stamp.microsecond // 100000) != raw:
33
+ raise ValueError("invalid ABI timestamp")
34
+ return stamp
35
+
36
+
37
+ def _number(raw: object, label: str, low: int, high: int) -> int:
38
+ if isinstance(raw, bool) or not isinstance(raw, (str, int)):
39
+ raise QueryError(f"{label} must be an integer from {low} to {high}")
40
+ text = str(raw).strip()
41
+ if label == "channel" and text.startswith("C"):
42
+ text = text[1:]
43
+ if not text.isascii() or not text.isdigit() or not low <= int(text) <= high:
44
+ raise QueryError(f"{label} must be an integer from {low} to {high}")
45
+ return int(text)
46
+
47
+
48
+ class GoesAbi(Provider):
49
+ """Single-channel CONUS ABI imagery; params: satellite, channel, product."""
50
+
51
+ def __init__(self, dataset: Dataset, client: httpx.Client | None = None) -> None:
52
+ super().__init__(dataset)
53
+ self._client = client
54
+ self._owns_client = client is None
55
+
56
+ def close(self) -> None:
57
+ """Close the owned HTTP client, leaving injected clients to their caller."""
58
+ if self._owns_client and self._client is not None:
59
+ self._client.close()
60
+ self._client = None
61
+
62
+ def _http(self) -> httpx.Client:
63
+ if self._client is None:
64
+ self._client = http.client()
65
+ return self._client
66
+
67
+ def list_assets(self, query: Query) -> list[Asset]:
68
+ """List complete scenes whose scan starts fall inside the inclusive UTC interval."""
69
+ if unknown := set(query.params) - {"satellite", "channel", "product"}:
70
+ raise QueryError(f"unsupported GOES params: {', '.join(sorted(unknown))}")
71
+ if query.params.get("product", PRODUCT) != PRODUCT:
72
+ raise QueryError(f"only product={PRODUCT} is supported")
73
+ if query.bbox is not None:
74
+ raise QueryError(
75
+ "GOES imagery has no geographic subsetting; omit location/bbox/lat/lon"
76
+ )
77
+ if query.text:
78
+ raise QueryError("GOES imagery has no text filtering; select satellite and channel")
79
+ if query.variables:
80
+ raise QueryError("GOES imagery has no variable subsetting; select a channel instead")
81
+ if query.time is None or query.time.start is None or query.time.end is None:
82
+ raise QueryError(f"{self.dataset.id} requires both start and end times")
83
+ satellite = _number(query.params.get("satellite"), "satellite", 16, 19)
84
+ channel = _number(query.params.get("channel"), "channel", 1, 16)
85
+ start = query.time.start.replace(tzinfo=query.time.start.tzinfo or UTC).astimezone(UTC)
86
+ end = query.time.end.replace(tzinfo=query.time.end.tzinfo or UTC).astimezone(UTC)
87
+ if end < PUBLIC_START:
88
+ raise QueryError("GOES CMIPC public observations begin on 2017-02-28")
89
+ start = max(start, PUBLIC_START)
90
+ hour = start.replace(minute=0, second=0, microsecond=0)
91
+ bucket = f"noaa-goes{satellite}"
92
+ assets: dict[str, Asset] = {}
93
+ while hour <= end:
94
+ prefix = f"{PRODUCT}/{hour:%Y/%j/%H}/"
95
+ for obj in s3.list_objects(bucket, prefix, self._http()):
96
+ if not obj.key.startswith(prefix):
97
+ continue
98
+ name = obj.key.removeprefix(prefix)
99
+ match = KEY_RE.fullmatch(name)
100
+ if (
101
+ match is None
102
+ or int(match["satellite"]) != satellite
103
+ or int(match["channel"]) != channel
104
+ ):
105
+ continue
106
+ try:
107
+ scan_start, scan_end = _timestamp(match["start"]), _timestamp(match["end"])
108
+ except ValueError:
109
+ continue
110
+ if scan_end < scan_start or not start <= scan_start <= end:
111
+ continue
112
+ assets[obj.key] = Asset(
113
+ id=name,
114
+ dataset_id=self.dataset.id,
115
+ href=f"s3://{bucket}/{obj.key}",
116
+ protocol=Protocol.S3,
117
+ media_type="application/x-netcdf",
118
+ size=obj.size,
119
+ time=TimeRange(start=scan_start, end=scan_end),
120
+ )
121
+ hour += timedelta(hours=1)
122
+ return sorted(assets.values(), key=lambda asset: asset.id)
123
+
124
+ def fetch(self, asset: Asset, dest: Path) -> Path:
125
+ """Download the complete archived NetCDF object without modification."""
126
+ return s3.download(asset.href, dest, self._http())
@@ -0,0 +1,134 @@
1
+ """Annual Storm Events details archives from the NCEI bulk directory.
2
+
3
+ Both dates are required. Every UTC calendar year touched by the interval is
4
+ selected in full; geographic, variable, text, and provider-specific filters are
5
+ unsupported. The most recently created supported details file wins per year.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import re
11
+ from datetime import UTC, datetime
12
+ from html.parser import HTMLParser
13
+ from pathlib import Path
14
+
15
+ import httpx
16
+
17
+ from usdata.models import Asset, Dataset, Protocol, Query, TimeRange
18
+ from usdata.protocols import http
19
+ from usdata.providers.base import Provider, QueryError
20
+
21
+ DIRECTORY_URL = "https://www.ncei.noaa.gov/pub/data/swdi/stormevents/csvfiles/"
22
+ DETAILS_NAME = re.compile(r"StormEvents_details-ftp_v1\.0_d(\d{4})_c(\d{8})\.csv\.gz", re.ASCII)
23
+
24
+
25
+ class _Directory(HTMLParser):
26
+ """Read filenames and exact byte sizes from NCEI's HTML directory table."""
27
+
28
+ def __init__(self) -> None:
29
+ super().__init__()
30
+ self.files: list[tuple[str, int | None]] = []
31
+ self._cells: list[str] = []
32
+ self._name: str | None = None
33
+ self._in_cell = False
34
+
35
+ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
36
+ if tag == "tr":
37
+ self._cells, self._name = [], None
38
+ elif tag == "td":
39
+ self._cells.append("")
40
+ self._in_cell = True
41
+ elif tag == "a":
42
+ href = dict(attrs).get("href", "") or ""
43
+ # Only literal local filenames; never follow arbitrary links from a listing.
44
+ if DETAILS_NAME.fullmatch(href):
45
+ self._name = href
46
+
47
+ def handle_data(self, data: str) -> None:
48
+ if self._in_cell:
49
+ self._cells[-1] += data
50
+
51
+ def handle_endtag(self, tag: str) -> None:
52
+ if tag == "td":
53
+ self._in_cell = False
54
+ elif tag == "tr" and self._name is not None:
55
+ size = self._cells[2].strip() if len(self._cells) > 2 else ""
56
+ self.files.append(
57
+ (self._name, int(size) if size.isascii() and size.isdigit() else None)
58
+ )
59
+
60
+
61
+ class StormEvents(Provider):
62
+ """Resolve whole-year details archives; preserve the original gzip bytes."""
63
+
64
+ def __init__(self, dataset: Dataset, *, client: httpx.Client | None = None) -> None:
65
+ super().__init__(dataset)
66
+ self._client = client
67
+ self._owns_client = client is None
68
+
69
+ def _http(self) -> httpx.Client:
70
+ if self._client is None:
71
+ self._client = http.client()
72
+ return self._client
73
+
74
+ def close(self) -> None:
75
+ """Close an internally created HTTP client; injected clients stay open."""
76
+ if self._owns_client and self._client is not None:
77
+ self._client.close()
78
+ self._client = None
79
+
80
+ def list_assets(self, query: Query) -> list[Asset]:
81
+ """Select the latest supported details revision for every requested year."""
82
+ if query.params:
83
+ raise QueryError(f"unsupported Storm Events params: {', '.join(sorted(query.params))}")
84
+ if query.bbox is not None or query.variables or query.text:
85
+ raise QueryError(
86
+ "Storm Events downloads whole annual details files; location/bbox, variables, "
87
+ "and text filters are unsupported. Filter locally after opening the CSV."
88
+ )
89
+ if query.time is None or query.time.start is None or query.time.end is None:
90
+ raise QueryError(f"{self.dataset.id} requires both start and end dates")
91
+ start = query.time.start.replace(tzinfo=query.time.start.tzinfo or UTC).astimezone(UTC)
92
+ end = query.time.end.replace(tzinfo=query.time.end.tzinfo or UTC).astimezone(UTC)
93
+ if start.year < 1950:
94
+ raise QueryError("Storm Events annual details files start in 1950")
95
+ years = range(start.year, end.year + 1)
96
+ listing = _Directory()
97
+ listing.feed(http.get(DIRECTORY_URL, self._http()).text)
98
+ selected: dict[int, tuple[str, int | None]] = {}
99
+ for name, size in listing.files:
100
+ match = DETAILS_NAME.fullmatch(name)
101
+ assert match is not None
102
+ year = int(match[1])
103
+ if year not in years:
104
+ continue
105
+ try:
106
+ datetime.strptime(match[2], "%Y%m%d")
107
+ except ValueError:
108
+ continue
109
+ # Fixed-width YYYYMMDD names sort by creation date. Duplicate rows are harmless.
110
+ if year not in selected or name > selected[year][0]:
111
+ selected[year] = name, size
112
+ if missing := [str(year) for year in years if year not in selected]:
113
+ raise QueryError(
114
+ "no supported Storm Events details file for year(s): " + ", ".join(missing)
115
+ )
116
+ return [
117
+ Asset(
118
+ id=selected[year][0],
119
+ dataset_id=self.dataset.id,
120
+ href=DIRECTORY_URL + selected[year][0],
121
+ protocol=Protocol.HTTP,
122
+ media_type="application/gzip",
123
+ size=selected[year][1],
124
+ time=TimeRange(
125
+ start=datetime(year, 1, 1, tzinfo=UTC),
126
+ end=datetime(year, 12, 31, 23, 59, 59, 999999, tzinfo=UTC),
127
+ ),
128
+ )
129
+ for year in years
130
+ ]
131
+
132
+ def fetch(self, asset: Asset, dest: Path) -> Path:
133
+ """Download the pinned annual archive, without decompressing or subsetting."""
134
+ return http.download(asset.href, dest, self._http())
@@ -0,0 +1,141 @@
1
+ """Optional readers for local fetched files; never fetch or modify cached bytes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import csv
6
+ import gzip
7
+ import io
8
+ from importlib import import_module
9
+ from typing import TYPE_CHECKING, Any
10
+
11
+ from usdata.models import Protocol
12
+
13
+ if TYPE_CHECKING:
14
+ from usdata.fetch import FetchedAsset
15
+
16
+ NETCDF_MEDIA_TYPES = {"application/x-netcdf", "application/netcdf", "application/x-netcdf4"}
17
+ CSV_MEDIA_TYPES = {"text/csv", "application/csv"}
18
+ GZIP_MEDIA_TYPES = {"application/gzip", "application/x-gzip"}
19
+ IDENTIFIER_COLUMNS = {
20
+ "station",
21
+ "station_id",
22
+ "site_no",
23
+ "monitoring_location_id",
24
+ "parameter_code",
25
+ "statistic_id",
26
+ "event_id",
27
+ "episode_id",
28
+ "state_fips",
29
+ "cz_fips",
30
+ "tor_other_cz_fips",
31
+ }
32
+
33
+
34
+ class MissingReaderDependency(ImportError):
35
+ """The optional dependency required to open an asset is not installed."""
36
+
37
+
38
+ class UnsupportedFormat(ValueError):
39
+ """No reader is implemented for this asset's format."""
40
+
41
+
42
+ def open_asset(
43
+ fetched: FetchedAsset,
44
+ *,
45
+ reader: str | None = None,
46
+ dtype: dict[str, str] | None = None,
47
+ parse_dates: list[str] | None = None,
48
+ usecols: list[str] | None = None,
49
+ nrows: int | None = None,
50
+ ) -> Any:
51
+ """Open local CSV, NetCDF4, or NEXRAD data, retaining units and provenance.
52
+
53
+ Gzip CSVs are decompressed locally without changing cached bytes.
54
+ Infer ``csv`` or ``erddap-csv`` from media type and protocol, or use an
55
+ explicit reader for ambiguous metadata. Identifier columns default to pandas
56
+ strings; explicit dtype entries override those defaults. Dates remain strings
57
+ unless named in parse_dates. No checksum verification or downloading occurs.
58
+ """
59
+ if reader is None:
60
+ media_type = (fetched.asset.media_type or "").split(";", 1)[0].strip().lower()
61
+ gzip_csv = media_type in GZIP_MEDIA_TYPES and fetched.asset.id.lower().endswith(".csv.gz")
62
+ if fetched.asset.dataset_id == "noaa:nexrad-level2":
63
+ reader = "nexrad-level2"
64
+ elif media_type in NETCDF_MEDIA_TYPES:
65
+ reader = "netcdf"
66
+ elif media_type in CSV_MEDIA_TYPES or gzip_csv:
67
+ reader = "erddap-csv" if fetched.asset.protocol is Protocol.ERDDAP else "csv"
68
+ else:
69
+ raise UnsupportedFormat(
70
+ f"no reader for {fetched.asset.media_type!r}; supported formats are CSV, "
71
+ "ERDDAP CSV, NetCDF4, and NEXRAD Level II. "
72
+ "For a known CSV with ambiguous metadata, "
73
+ "pass reader='csv' "
74
+ "or reader='erddap-csv'; otherwise use fetched.path with a format-specific reader"
75
+ )
76
+ if reader == "netcdf":
77
+ if any(value is not None for value in (dtype, parse_dates, usecols, nrows)):
78
+ raise ValueError(
79
+ "CSV options dtype, parse_dates, usecols and nrows do not apply to NetCDF"
80
+ )
81
+ from usdata._netcdf import open_netcdf
82
+
83
+ return open_netcdf(fetched)
84
+ if reader == "nexrad-level2":
85
+ if any(value is not None for value in (dtype, parse_dates, usecols, nrows)):
86
+ raise ValueError("dtype, parse_dates, usecols, and nrows apply only to CSV readers")
87
+ from usdata._radar import open_nexrad
88
+
89
+ return open_nexrad(fetched)
90
+ if reader not in {"csv", "erddap-csv"}:
91
+ raise UnsupportedFormat(
92
+ f"unsupported reader {reader!r}; use 'csv', 'erddap-csv', 'netcdf', or 'nexrad-level2'"
93
+ )
94
+ try:
95
+ pandas = import_module("pandas")
96
+ except ModuleNotFoundError as error:
97
+ if error.name != "pandas":
98
+ raise
99
+ raise MissingReaderDependency(
100
+ 'CSV reading requires pandas; install it with: pip install "usdata[pandas]" '
101
+ '(or uv add "usdata[pandas]")'
102
+ ) from error
103
+
104
+ # Pass a file object to pandas: reading a fetched asset is strictly local.
105
+ with fetched.path.open("rb") as raw:
106
+ compressed = raw.read(2) == b"\x1f\x8b"
107
+ raw.seek(0)
108
+ binary = gzip.GzipFile(fileobj=raw) if compressed else raw
109
+ with io.TextIOWrapper(binary, encoding="utf-8-sig", newline="") as stream:
110
+ records = csv.reader(stream)
111
+ columns = next(records, [])
112
+ if (
113
+ not columns
114
+ or any(not column for column in columns)
115
+ or len(set(columns)) != len(columns)
116
+ ):
117
+ raise ValueError("CSV must have a non-empty header with unique column names")
118
+ units = {}
119
+ if reader == "erddap-csv":
120
+ values = next(records, [])
121
+ if len(values) != len(columns):
122
+ raise ValueError("ERDDAP CSV must have a units row matching the header")
123
+ units = dict(zip(columns, values, strict=True))
124
+ types = {name: "string" for name in columns if name.casefold() in IDENTIFIER_COLUMNS}
125
+ types.update(dtype or {})
126
+ frame = pandas.read_csv(
127
+ stream,
128
+ header=None,
129
+ names=columns,
130
+ dtype=types,
131
+ parse_dates=parse_dates,
132
+ usecols=usecols,
133
+ nrows=nrows,
134
+ )
135
+ if units:
136
+ frame.attrs["units"] = {name: units[name] for name in frame.columns}
137
+ frame.attrs["usdata"] = {
138
+ "asset_id": fetched.asset.id,
139
+ "provenance": fetched.provenance.model_dump(mode="json"),
140
+ }
141
+ return frame
@@ -1,103 +0,0 @@
1
- """Optional readers for local fetched files; never fetch or modify cached bytes."""
2
-
3
- from __future__ import annotations
4
-
5
- import csv
6
- from importlib import import_module
7
- from typing import TYPE_CHECKING, Any
8
-
9
- from usdata.models import Protocol
10
-
11
- if TYPE_CHECKING:
12
- from usdata.fetch import FetchedAsset
13
-
14
- CSV_MEDIA_TYPES = {"text/csv", "application/csv"}
15
- IDENTIFIER_COLUMNS = {
16
- "station",
17
- "station_id",
18
- "site_no",
19
- "monitoring_location_id",
20
- "parameter_code",
21
- "statistic_id",
22
- }
23
-
24
-
25
- class MissingReaderDependency(ImportError):
26
- """The optional dependency required to open an asset is not installed."""
27
-
28
-
29
- class UnsupportedFormat(ValueError):
30
- """No reader is implemented for this asset's format."""
31
-
32
-
33
- def open_asset(
34
- fetched: FetchedAsset,
35
- *,
36
- reader: str | None = None,
37
- dtype: dict[str, str] | None = None,
38
- parse_dates: list[str] | None = None,
39
- usecols: list[str] | None = None,
40
- nrows: int | None = None,
41
- ) -> Any:
42
- """Read a local CSV into a pandas DataFrame, retaining units and provenance.
43
-
44
- Infer ``csv`` or ``erddap-csv`` from media type and protocol, or use an
45
- explicit reader for ambiguous metadata. Identifier columns default to pandas
46
- strings; explicit dtype entries override those defaults. Dates remain strings
47
- unless named in parse_dates. No checksum verification or downloading occurs.
48
- """
49
- if reader is None:
50
- media_type = (fetched.asset.media_type or "").split(";", 1)[0].strip().lower()
51
- if media_type not in CSV_MEDIA_TYPES:
52
- raise UnsupportedFormat(
53
- f"no reader for {fetched.asset.media_type!r}; supported formats are CSV and "
54
- "ERDDAP CSV. For a known CSV with ambiguous metadata, pass reader='csv' "
55
- "or reader='erddap-csv'; otherwise use fetched.path with a format-specific reader"
56
- )
57
- reader = "erddap-csv" if fetched.asset.protocol is Protocol.ERDDAP else "csv"
58
- if reader not in {"csv", "erddap-csv"}:
59
- raise UnsupportedFormat(f"unsupported reader {reader!r}; use 'csv' or 'erddap-csv'")
60
- try:
61
- pandas = import_module("pandas")
62
- except ModuleNotFoundError as error:
63
- if error.name != "pandas":
64
- raise
65
- raise MissingReaderDependency(
66
- 'CSV reading requires pandas; install it with: pip install "usdata[pandas]" '
67
- '(or uv add "usdata[pandas]")'
68
- ) from error
69
-
70
- # Pass a file object to pandas: reading a fetched asset is strictly local.
71
- with fetched.path.open(encoding="utf-8-sig", newline="") as stream:
72
- records = csv.reader(stream)
73
- columns = next(records, [])
74
- if (
75
- not columns
76
- or any(not column for column in columns)
77
- or len(set(columns)) != len(columns)
78
- ):
79
- raise ValueError("CSV must have a non-empty header with unique column names")
80
- units = {}
81
- if reader == "erddap-csv":
82
- values = next(records, [])
83
- if len(values) != len(columns):
84
- raise ValueError("ERDDAP CSV must have a units row matching the header")
85
- units = dict(zip(columns, values, strict=True))
86
- types = {name: "string" for name in columns if name.casefold() in IDENTIFIER_COLUMNS}
87
- types.update(dtype or {})
88
- frame = pandas.read_csv(
89
- stream,
90
- header=None,
91
- names=columns,
92
- dtype=types,
93
- parse_dates=parse_dates,
94
- usecols=usecols,
95
- nrows=nrows,
96
- )
97
- if units:
98
- frame.attrs["units"] = {name: units[name] for name in frame.columns}
99
- frame.attrs["usdata"] = {
100
- "asset_id": fetched.asset.id,
101
- "provenance": fetched.provenance.model_dump(mode="json"),
102
- }
103
- return frame
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes