usdata 0.7.0__tar.gz → 0.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. usdata-0.9.0/PKG-INFO +115 -0
  2. usdata-0.9.0/README.md +87 -0
  3. {usdata-0.7.0 → usdata-0.9.0}/pyproject.toml +32 -4
  4. {usdata-0.7.0 → usdata-0.9.0}/pyproject.toml.orig +27 -4
  5. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/__init__.py +4 -1
  6. usdata-0.9.0/src/usdata/_netcdf.py +36 -0
  7. usdata-0.9.0/src/usdata/_radar.py +123 -0
  8. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/data/registry.yaml +23 -19
  9. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/fetch.py +13 -3
  10. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/models.py +24 -2
  11. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/protocols/s3.py +15 -2
  12. usdata-0.9.0/src/usdata/providers/_http.py +27 -0
  13. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/noaa/coastwatch.py +4 -19
  14. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/noaa/ghcnd.py +4 -19
  15. usdata-0.9.0/src/usdata/providers/noaa/goes.py +109 -0
  16. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/noaa/nexrad.py +5 -22
  17. usdata-0.9.0/src/usdata/providers/noaa/storm_events.py +117 -0
  18. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/usgs/daily.py +4 -19
  19. usdata-0.9.0/src/usdata/readers.py +156 -0
  20. usdata-0.9.0/src/usdata/selection.py +71 -0
  21. usdata-0.7.0/PKG-INFO +0 -186
  22. usdata-0.7.0/README.md +0 -163
  23. usdata-0.7.0/src/usdata/readers.py +0 -103
  24. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/_files.py +0 -0
  25. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/_progress.py +0 -0
  26. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/cache.py +0 -0
  27. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/cli/__init__.py +0 -0
  28. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/cli/app.py +0 -0
  29. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/cli/progress.py +0 -0
  30. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/data/nexrad_sites.csv +0 -0
  31. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/data/places.csv +0 -0
  32. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/data/places.sources.json +0 -0
  33. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/manifest.py +0 -0
  34. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/protocols/__init__.py +0 -0
  35. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/protocols/erddap.py +0 -0
  36. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/protocols/http.py +0 -0
  37. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/provenance.py +0 -0
  38. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/__init__.py +0 -0
  39. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/base.py +0 -0
  40. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/noaa/__init__.py +0 -0
  41. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/noaa/gsom.py +0 -0
  42. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/noaa/sites.py +0 -0
  43. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/usgs/__init__.py +0 -0
  44. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/pull.py +0 -0
  45. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/py.typed +0 -0
  46. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/query.py +0 -0
  47. {usdata-0.7.0 → usdata-0.9.0}/src/usdata/registry.py +0 -0
usdata-0.9.0/PKG-INFO ADDED
@@ -0,0 +1,115 @@
1
+ Metadata-Version: 2.4
2
+ Name: usdata
3
+ Version: 0.9.0
4
+ Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
5
+ Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
6
+ Author: Jake Van Slyke
7
+ Author-email: Jake Van Slyke <jakervanslyke@gmail.com>
8
+ License-Expression: Apache-2.0
9
+ Classifier: Development Status :: 2 - Pre-Alpha
10
+ Classifier: Intended Audience :: Science/Research
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Topic :: Scientific/Engineering
13
+ Requires-Dist: httpx>=0.28.1
14
+ Requires-Dist: pydantic>=2.7
15
+ Requires-Dist: pyyaml>=6.0
16
+ Requires-Dist: typer>=0.18
17
+ Requires-Dist: xarray>=2025.1 ; extra == 'netcdf'
18
+ Requires-Dist: h5netcdf[h5py]>=1.8.1 ; extra == 'netcdf'
19
+ Requires-Dist: pandas>=3.0 ; extra == 'pandas'
20
+ Requires-Dist: xradar>=0.12.0 ; extra == 'radar'
21
+ Requires-Python: >=3.11
22
+ Project-URL: Homepage, https://github.com/jakeryderv/usdata
23
+ Project-URL: Repository, https://github.com/jakeryderv/usdata
24
+ Provides-Extra: netcdf
25
+ Provides-Extra: pandas
26
+ Provides-Extra: radar
27
+ Description-Content-Type: text/markdown
28
+
29
+ # usdata
30
+
31
+ Discover U.S. scientific datasets, fetch their files, and keep a reproducible
32
+ record of where every input came from. Use the same Python SDK or CLI across
33
+ supported NOAA and USGS datasets.
34
+
35
+ **Pre-alpha.** These docs describe the current source checkout. Features marked
36
+ **Unreleased** require a source installation; consult the
37
+ [changelog](CHANGELOG.md) for published versions. Other providers are planned.
38
+
39
+ ## Start here
40
+
41
+ ```sh
42
+ pip install usdata
43
+ usdata search precipitation --location Oklahoma
44
+ usdata info noaa:ghcn-daily
45
+ ```
46
+
47
+ Search uses a curated registry. Fetching contacts the upstream service; readers
48
+ open the resulting local files. Provenance and manifests connect those steps.
49
+
50
+ - [Quick start and documentation](docs/index.md)
51
+ - [Fetch and analyze data](docs/guides/fetch-and-analyze.md)
52
+ - [Runnable examples with saved outputs](examples/README.md)
53
+ - [Readers](docs/reference/readers.md) and [reproducible manifests](docs/reference/manifests.md)
54
+
55
+ ## Providers
56
+
57
+ <!-- registry:start -->
58
+ | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
59
+ |---|---:|---:|---:|---|---|
60
+ | [NOAA](docs/providers/noaa.md) | 6 | 0 | 23 | — | `ghcn-daily`, `gsom`, `storm-events`, `nexrad-level2`, `goes-abi`, `coastwatch-sst`, +23 planned |
61
+ | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
62
+ | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
63
+ | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
64
+ | [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
65
+ | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
66
+ | [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
67
+
68
+ Available datasets are in `code`, stubs in _italics_; planned ones are counted. Available means implemented in this source checkout; consult the [releases](https://github.com/jakeryderv/usdata/releases) for published support. Provider pages link access notes to the generated dataset catalog; [the roadmap](docs/roadmap.md) explains future priorities.
69
+ <!-- registry:end -->
70
+
71
+ ## Development
72
+
73
+ Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
74
+
75
+ `just setup` uses the tested Python 3.14.7 pin in `.python-version`. Older Linux
76
+ uv Python 3.14 builds can crash during NumPy array operations; see
77
+ [the upstream fix](https://github.com/astral-sh/python-build-standalone/issues/991).
78
+
79
+ ```sh
80
+ git clone https://github.com/jakeryderv/usdata && cd usdata
81
+ just setup # install toolchain and dependencies
82
+ just test # all offline tests
83
+ just check # format, lint, typecheck, offline tests, generated docs, release notices
84
+ just check-pandas # install the CSV extra and run the same checks
85
+ just check-radar # install the radar extra and run the same checks
86
+ just check-netcdf # install the NetCDF4 extra and run the same checks
87
+ just notebooks # launch the optional Jupyter examples environment
88
+ just run-notebooks # execute notebooks live in fresh kernels and temporary caches
89
+ just docs-serve # build and preview the documentation locally, with reload
90
+ just check-docs # validate generated content and build the site strictly
91
+ just build # build wheel and sdist
92
+ just smoke # exercise core and pandas wheel installations outside the checkout
93
+ just run search radar
94
+ ```
95
+
96
+ Offline tests mechanically block network connections. Tests that hit
97
+ live services run with `just test-live`; see [testing levels and organization](docs/testing.md). CI checks Python 3.11 and 3.14 on
98
+ Linux with core-only, pandas, radar, and NetCDF dependency profiles. Installed-wheel
99
+ checks cover all four profiles on Linux, macOS, and Windows. The full offline and
100
+ live-service suites run on Linux. `just setup` restores a core-only development
101
+ environment; the `check-pandas`, `check-radar`, and `check-netcdf` commands install
102
+ their respective extras.
103
+
104
+ Releases: `just release minor` opens a version-bump PR; merging it publishes
105
+ to PyPI and creates the tag and GitHub release. See
106
+ [docs/versioning.md](docs/versioning.md).
107
+
108
+ See [provider access notes](docs/providers/README.md),
109
+ [docs/architecture.md](docs/architecture.md) for how the pieces fit,
110
+ [architecture decisions](docs/adr/README.md) for why, and [CONTRIBUTING.md](CONTRIBUTING.md) to add
111
+ a dataset.
112
+
113
+ ## License
114
+
115
+ [Apache-2.0](LICENSE).
usdata-0.9.0/README.md ADDED
@@ -0,0 +1,87 @@
1
+ # usdata
2
+
3
+ Discover U.S. scientific datasets, fetch their files, and keep a reproducible
4
+ record of where every input came from. Use the same Python SDK or CLI across
5
+ supported NOAA and USGS datasets.
6
+
7
+ **Pre-alpha.** These docs describe the current source checkout. Features marked
8
+ **Unreleased** require a source installation; consult the
9
+ [changelog](CHANGELOG.md) for published versions. Other providers are planned.
10
+
11
+ ## Start here
12
+
13
+ ```sh
14
+ pip install usdata
15
+ usdata search precipitation --location Oklahoma
16
+ usdata info noaa:ghcn-daily
17
+ ```
18
+
19
+ Search uses a curated registry. Fetching contacts the upstream service; readers
20
+ open the resulting local files. Provenance and manifests connect those steps.
21
+
22
+ - [Quick start and documentation](docs/index.md)
23
+ - [Fetch and analyze data](docs/guides/fetch-and-analyze.md)
24
+ - [Runnable examples with saved outputs](examples/README.md)
25
+ - [Readers](docs/reference/readers.md) and [reproducible manifests](docs/reference/manifests.md)
26
+
27
+ ## Providers
28
+
29
+ <!-- registry:start -->
30
+ | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
31
+ |---|---:|---:|---:|---|---|
32
+ | [NOAA](docs/providers/noaa.md) | 6 | 0 | 23 | — | `ghcn-daily`, `gsom`, `storm-events`, `nexrad-level2`, `goes-abi`, `coastwatch-sst`, +23 planned |
33
+ | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
34
+ | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
35
+ | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
36
+ | [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
37
+ | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
38
+ | [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
39
+
40
+ Available datasets are in `code`, stubs in _italics_; planned ones are counted. Available means implemented in this source checkout; consult the [releases](https://github.com/jakeryderv/usdata/releases) for published support. Provider pages link access notes to the generated dataset catalog; [the roadmap](docs/roadmap.md) explains future priorities.
41
+ <!-- registry:end -->
42
+
43
+ ## Development
44
+
45
+ Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
46
+
47
+ `just setup` uses the tested Python 3.14.7 pin in `.python-version`. Older Linux
48
+ uv Python 3.14 builds can crash during NumPy array operations; see
49
+ [the upstream fix](https://github.com/astral-sh/python-build-standalone/issues/991).
50
+
51
+ ```sh
52
+ git clone https://github.com/jakeryderv/usdata && cd usdata
53
+ just setup # install toolchain and dependencies
54
+ just test # all offline tests
55
+ just check # format, lint, typecheck, offline tests, generated docs, release notices
56
+ just check-pandas # install the CSV extra and run the same checks
57
+ just check-radar # install the radar extra and run the same checks
58
+ just check-netcdf # install the NetCDF4 extra and run the same checks
59
+ just notebooks # launch the optional Jupyter examples environment
60
+ just run-notebooks # execute notebooks live in fresh kernels and temporary caches
61
+ just docs-serve # build and preview the documentation locally, with reload
62
+ just check-docs # validate generated content and build the site strictly
63
+ just build # build wheel and sdist
64
+ just smoke # exercise core and pandas wheel installations outside the checkout
65
+ just run search radar
66
+ ```
67
+
68
+ Offline tests mechanically block network connections. Tests that hit
69
+ live services run with `just test-live`; see [testing levels and organization](docs/testing.md). CI checks Python 3.11 and 3.14 on
70
+ Linux with core-only, pandas, radar, and NetCDF dependency profiles. Installed-wheel
71
+ checks cover all four profiles on Linux, macOS, and Windows. The full offline and
72
+ live-service suites run on Linux. `just setup` restores a core-only development
73
+ environment; the `check-pandas`, `check-radar`, and `check-netcdf` commands install
74
+ their respective extras.
75
+
76
+ Releases: `just release minor` opens a version-bump PR; merging it publishes
77
+ to PyPI and creates the tag and GitHub release. See
78
+ [docs/versioning.md](docs/versioning.md).
79
+
80
+ See [provider access notes](docs/providers/README.md),
81
+ [docs/architecture.md](docs/architecture.md) for how the pieces fit,
82
+ [architecture decisions](docs/adr/README.md) for why, and [CONTRIBUTING.md](CONTRIBUTING.md) to add
83
+ a dataset.
84
+
85
+ ## License
86
+
87
+ [Apache-2.0](LICENSE).
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.7.0"
3
+ version = "0.9.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -23,7 +23,7 @@ dependencies = [
23
23
  "httpx>=0.28.1",
24
24
  "pydantic>=2.7",
25
25
  "pyyaml>=6.0",
26
- "typer>=0.12",
26
+ "typer>=0.18",
27
27
  ]
28
28
 
29
29
  [[project.authors]]
@@ -32,6 +32,11 @@ email = "jakervanslyke@gmail.com"
32
32
 
33
33
  [project.optional-dependencies]
34
34
  pandas = ["pandas>=3.0"]
35
+ radar = ["xradar>=0.12.0"]
36
+ netcdf = [
37
+ "xarray>=2025.1",
38
+ "h5netcdf[h5py]>=1.8.1",
39
+ ]
35
40
 
36
41
  [project.urls]
37
42
  Homepage = "https://github.com/jakeryderv/usdata"
@@ -48,6 +53,18 @@ dev = [
48
53
  "respx>=0.23.1",
49
54
  "ruff>=0.6",
50
55
  ]
56
+ docs = [
57
+ "mkdocstrings-python>=2.0.8",
58
+ "nbconvert>=7.17.1",
59
+ "zensical>=0.0.60",
60
+ ]
61
+ examples = [
62
+ "ipykernel>=6.29",
63
+ "jupyterlab>=4.3",
64
+ "matplotlib>=3.9",
65
+ "nbclient>=0.10",
66
+ "pandas>=3.0",
67
+ ]
51
68
 
52
69
  [build-system]
53
70
  requires = ["uv_build>=0.12.5,<0.13.0"]
@@ -96,5 +113,16 @@ venv = ".venv"
96
113
 
97
114
  [tool.pytest.ini_options]
98
115
  testpaths = ["tests"]
99
- markers = ["integration: hits live services; skipped unless --run-integration is passed"]
100
- addopts = "-ra"
116
+ markers = [
117
+ "l0: pure in-memory logic",
118
+ "l1: in-process components with controlled transport",
119
+ "l2: local functional tests with filesystem or real decoders",
120
+ "l3: controlled service deployment (currently unused)",
121
+ "l4: production or upstream-live compatibility",
122
+ "live: hits upstream services; requires --run-live",
123
+ "integration: compatibility alias for live",
124
+ "pandas: optional CSV reader coverage",
125
+ "radar: optional radar reader coverage",
126
+ "netcdf: optional NetCDF reader coverage",
127
+ ]
128
+ addopts = "-ra --strict-markers"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.7.0"
3
+ version = "0.9.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -19,11 +19,13 @@ dependencies = [
19
19
  "httpx>=0.28.1",
20
20
  "pydantic>=2.7",
21
21
  "pyyaml>=6.0",
22
- "typer>=0.12",
22
+ "typer>=0.18",
23
23
  ]
24
24
 
25
25
  [project.optional-dependencies]
26
26
  pandas = ["pandas>=3.0"]
27
+ radar = ["xradar>=0.12.0"]
28
+ netcdf = ["xarray>=2025.1", "h5netcdf[h5py]>=1.8.1"]
27
29
 
28
30
  [project.urls]
29
31
  Homepage = "https://github.com/jakeryderv/usdata"
@@ -40,6 +42,18 @@ dev = [
40
42
  "respx>=0.23.1",
41
43
  "ruff>=0.6",
42
44
  ]
45
+ docs = [
46
+ "mkdocstrings-python>=2.0.8",
47
+ "nbconvert>=7.17.1",
48
+ "zensical>=0.0.60",
49
+ ]
50
+ examples = [
51
+ "ipykernel>=6.29",
52
+ "jupyterlab>=4.3",
53
+ "matplotlib>=3.9",
54
+ "nbclient>=0.10",
55
+ "pandas>=3.0",
56
+ ]
43
57
 
44
58
  [build-system]
45
59
  requires = ["uv_build>=0.12.5,<0.13.0"]
@@ -72,6 +86,15 @@ venv = ".venv"
72
86
  [tool.pytest.ini_options]
73
87
  testpaths = ["tests"]
74
88
  markers = [
75
- "integration: hits live services; skipped unless --run-integration is passed",
89
+ "l0: pure in-memory logic",
90
+ "l1: in-process components with controlled transport",
91
+ "l2: local functional tests with filesystem or real decoders",
92
+ "l3: controlled service deployment (currently unused)",
93
+ "l4: production or upstream-live compatibility",
94
+ "live: hits upstream services; requires --run-live",
95
+ "integration: compatibility alias for live",
96
+ "pandas: optional CSV reader coverage",
97
+ "radar: optional radar reader coverage",
98
+ "netcdf: optional NetCDF reader coverage",
76
99
  ]
77
- addopts = "-ra"
100
+ addopts = "-ra --strict-markers"
@@ -10,10 +10,11 @@ try:
10
10
  except PackageNotFoundError: # running from a source tree without an install
11
11
  __version__ = "0.0.0"
12
12
 
13
- from usdata.models import Asset, BBox, Dataset, Provenance, Query, TimeRange
13
+ from usdata.models import Asset, BBox, Dataset, Provenance, Query, TemporalSelection, TimeRange
14
14
  from usdata.pull import pull, verify
15
15
  from usdata.query import build_query
16
16
  from usdata.registry import DatasetNotFound, Registry, SearchResult, default_registry
17
+ from usdata.selection import select_by_time
17
18
 
18
19
  __all__ = [
19
20
  "Asset",
@@ -24,6 +25,7 @@ __all__ = [
24
25
  "Query",
25
26
  "Registry",
26
27
  "SearchResult",
28
+ "TemporalSelection",
27
29
  "TimeRange",
28
30
  "__version__",
29
31
  "build_query",
@@ -31,6 +33,7 @@ __all__ = [
31
33
  "get",
32
34
  "pull",
33
35
  "search",
36
+ "select_by_time",
34
37
  "verify",
35
38
  ]
36
39
 
@@ -0,0 +1,36 @@
1
+ """Local, eagerly loaded NetCDF4 reading behind the netcdf extra."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from importlib import import_module
6
+ from typing import TYPE_CHECKING, Any
7
+
8
+ from usdata.readers import MissingReaderDependency
9
+
10
+ if TYPE_CHECKING:
11
+ from usdata.fetch import FetchedAsset
12
+
13
+
14
+ def open_netcdf(fetched: FetchedAsset) -> Any:
15
+ """Load a NetCDF4 root Dataset, then close every source file handle."""
16
+ try:
17
+ xarray = import_module("xarray")
18
+ import_module("h5netcdf")
19
+ import_module("h5py")
20
+ except ModuleNotFoundError as error:
21
+ if error.name not in {"xarray", "h5netcdf", "h5py"}:
22
+ raise
23
+ raise MissingReaderDependency(
24
+ 'NetCDF4 reading requires xarray and h5netcdf; install: pip install "usdata[netcdf]"'
25
+ ) from error
26
+ # A local file object and fixed engine prevent interpretation as an OPeNDAP URL.
27
+ with (
28
+ fetched.path.open("rb") as stream,
29
+ xarray.open_dataset(stream, engine="h5netcdf", chunks=None) as dataset,
30
+ ):
31
+ dataset.load()
32
+ dataset.attrs["usdata"] = {
33
+ "asset_id": fetched.asset.id,
34
+ "provenance": fetched.provenance.model_dump(mode="json"),
35
+ }
36
+ return dataset
@@ -0,0 +1,123 @@
1
+ """Local NEXRAD Level II decoding behind the radar extra."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import bz2
6
+ import gzip
7
+ from importlib import import_module
8
+ from typing import TYPE_CHECKING, Any
9
+
10
+ from usdata.readers import MissingReaderDependency, RadarDecodeError
11
+
12
+ # NOAA RDA/RPG ICD 2620002Y, Table XVII-I notes 21 and 30.
13
+ MOMENT_FLAG_COUNTS = {
14
+ "DBZH": 2,
15
+ "VRADH": 2,
16
+ "WRADH": 2,
17
+ "ZDR": 2,
18
+ "PHIDP": 2,
19
+ "RHOHV": 2,
20
+ "CCORH": 8,
21
+ }
22
+
23
+ if TYPE_CHECKING:
24
+ from usdata.fetch import FetchedAsset
25
+
26
+
27
+ def _check_sweeps(content: bytes, sweep: int | list[int] | None) -> None:
28
+ """Reject decoder tables that pair a sweep with another sweep's coordinates.
29
+
30
+ xradar 0.12 can omit an interior sweep without an end marker from its data
31
+ table while keeping its moment metadata. The coordinate list then shifts.
32
+ Compare record identities, not just ray counts: equal-length sweeps can
33
+ otherwise silently acquire another sweep's coordinates and timestamps.
34
+ """
35
+ backend = import_module("xradar.io.backends.nexrad_level2")
36
+ with backend.NEXRADLevel2File(content, loaddata=False) as volume:
37
+ # Parsing completeness also populates the per-sweep data tables.
38
+ _ = volume.incomplete_sweeps
39
+ moments = volume.msg_31_data_header
40
+ coordinates = volume.msg_31_header
41
+ requested = (
42
+ list(range(len(moments)))
43
+ if sweep is None
44
+ else (sweep if isinstance(sweep, list) else [sweep])
45
+ )
46
+ for index in requested:
47
+ if index >= len(moments):
48
+ raise ValueError(f"sweep {index} is outside this volume's {len(moments)} sweeps")
49
+ data = volume.data.get(index)
50
+ rays = coordinates[index] if index < len(coordinates) else []
51
+ # Non-radial messages may occur within a sweep or after its last
52
+ # received ray. Follow the decoder's traversal, excluding those
53
+ # records, instead of requiring record_end to be a radial message.
54
+ expected_records = []
55
+ if data is not None:
56
+ intermediate = {record["record_number"] for record in data["intermediate_records"]}
57
+ expected_records = [
58
+ record
59
+ for record in range(data["record_number"], data["record_end"] + 1)
60
+ if record not in intermediate
61
+ ]
62
+ if (
63
+ data is None
64
+ or not rays
65
+ or moments[index]["record_number"] != data["record_number"]
66
+ or [ray["record_number"] for ray in rays] != expected_records
67
+ ):
68
+ raise RadarDecodeError(
69
+ f"cannot safely decode sweep {index}: NEXRAD moment and coordinate records "
70
+ "do not agree; select an unaffected sweep explicitly with open(sweep=...) "
71
+ "or use another decoder. No sweeps were silently dropped."
72
+ )
73
+
74
+
75
+ def open_nexrad(fetched: FetchedAsset, *, sweep: int | list[int] | None = None) -> Any:
76
+ """Decode a local volume into a fully loaded xarray DataTree."""
77
+ try:
78
+ xradar = import_module("xradar")
79
+ except ModuleNotFoundError as error:
80
+ if error.name != "xradar":
81
+ raise
82
+ raise MissingReaderDependency(
83
+ 'NEXRAD reading requires xradar; install it with: pip install "usdata[radar]" '
84
+ '(or uv add "usdata[radar]")'
85
+ ) from error
86
+
87
+ # Bytes prevent remote URL interpretation and work across the backend's
88
+ # repeated sweep reads. Compressed source files stay unchanged in the cache.
89
+ content = fetched.path.read_bytes()
90
+ if content.startswith(b"\x1f\x8b"):
91
+ content = gzip.decompress(content)
92
+ elif content.startswith(b"BZh"):
93
+ content = bz2.decompress(content)
94
+ _check_sweeps(content, sweep)
95
+ radar = xradar.io.open_nexradlevel2_datatree(content, sweep=sweep, incomplete_sweep="pad")
96
+ try:
97
+ radar.load()
98
+ finally:
99
+ radar.close()
100
+ for node in radar.subtree:
101
+ for name, variable in node.ds.variables.items():
102
+ # The backend records the entire input byte string as `source`.
103
+ # Provenance below is the durable reference, not that decoder buffer.
104
+ variable.encoding.pop("source", None)
105
+ if name in MOMENT_FLAG_COUNTS and "range" in variable.dims:
106
+ scale = variable.encoding.get("scale_factor")
107
+ offset = variable.encoding.get("add_offset")
108
+ if scale is not None and offset is not None:
109
+ # xradar 0.12 does not supply _FillValue for NEXRAD flags.
110
+ # Compare using each moment's native scale, not fixed units.
111
+ data = node[name]
112
+ valid = data.notnull()
113
+ for code in range(MOMENT_FLAG_COUNTS[name]):
114
+ valid = valid & (data != offset + code * scale)
115
+ masked = data.where(valid)
116
+ masked.encoding = data.encoding.copy()
117
+ node[name] = masked
118
+ radar.attrs["usdata"] = {
119
+ "asset_id": fetched.asset.id,
120
+ "provenance": fetched.provenance.model_dump(mode="json"),
121
+ "sweeps": [name.lstrip("/") for name in radar.groups if name.startswith("/sweep_")],
122
+ }
123
+ return radar
@@ -105,21 +105,22 @@ datasets:
105
105
 
106
106
  - id: noaa:goes-abi
107
107
  provider: noaa
108
- status: planned
108
+ status: available
109
109
  domain: weather-satellites
110
- target: later
111
- title: GOES-R ABI Satellite Imagery
112
- description: >-
113
- Advanced Baseline Imager products from GOES-16, 18, and 19 in the public
114
- noaa-goes16/18/19 S3 buckets, laid out as PRODUCT/YYYY/DDD/HH/ with one
115
- NetCDF per scan (for example ABI-L2-CMIPC). Same anonymous S3 pattern as
116
- NEXRAD, plus product, satellite, and channel selection.
117
- keywords: [satellite, imagery, goes, abi, clouds, fire, radiance, netcdf]
110
+ since: "0.8"
111
+ title: GOES-R ABI CONUS Cloud and Moisture Imagery
112
+ description: >-
113
+ Single-channel CONUS Cloud and Moisture Imagery (ABI-L2-CMIPC) from
114
+ GOES-16, 17, 18, and 19 in anonymous NOAA S3 buckets. Select an explicit
115
+ satellite, channel, and scan-start interval; each asset is a complete
116
+ NetCDF scene with no geographic or variable subsetting.
117
+ keywords: [satellite, imagery, goes, abi, clouds, infrared, reflectance, netcdf, conus]
118
118
  protocol: s3
119
119
  homepage: https://registry.opendata.aws/noaa-goes/
120
120
  license: US Government Work (public domain)
121
- temporal_extent: { start: "2017-01-01T00:00:00Z" }
122
- capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: true }
121
+ temporal_extent: { start: "2017-02-28T00:00:00Z" }
122
+ capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: false }
123
+ adapter: usdata.providers.noaa.goes:GoesAbi
123
124
 
124
125
  - id: noaa:goes-glm
125
126
  provider: noaa
@@ -139,21 +140,24 @@ datasets:
139
140
 
140
141
  - id: noaa:storm-events
141
142
  provider: noaa
142
- status: planned
143
+ status: available
143
144
  domain: severe-weather
144
- target: later
145
+ since: "0.8"
145
146
  title: Storm Events Database
146
147
  description: >-
147
- NCEI's record of significant weather events since 1950 (tornadoes, hail,
148
- wind, floods, and more) with locations, damage, and narratives. Published
149
- as per-year gzipped CSV files (details, fatalities, locations) in a plain
150
- HTTPS directory; the first NCEI bulk-directory dataset.
148
+ NCEI's significant-weather event details since 1950, with locations,
149
+ impacts, and narratives. Anonymous whole-year gzipped CSV archives;
150
+ select the latest creation-date revision for each requested year.
151
+ No server-side row, location, or variable subsetting. Historical event
152
+ coverage and reporting practices vary; fatalities and locations tables
153
+ are separate products not included by this adapter.
151
154
  keywords: [storms, tornado, hail, wind, flood, damage, severe weather, events]
152
155
  protocol: http
153
- homepage: https://www.ncdc.noaa.gov/stormevents/
156
+ homepage: https://www.ncei.noaa.gov/access/storm-events-database/
154
157
  license: US Government Work (public domain)
155
158
  temporal_extent: { start: "1950-01-01T00:00:00Z" }
156
- capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: false }
159
+ capabilities: { spatial_subset: false, temporal_subset: false, variable_subset: false }
160
+ adapter: usdata.providers.noaa.storm_events:StormEvents
157
161
 
158
162
  - id: noaa:hurdat2
159
163
  provider: noaa
@@ -34,17 +34,27 @@ class FetchedAsset(BaseModel):
34
34
  parse_dates: list[str] | None = None,
35
35
  usecols: list[str] | None = None,
36
36
  nrows: int | None = None,
37
+ sweep: int | list[int] | None = None,
37
38
  ) -> Any:
38
- """Open this local CSV as a DataFrame; requires the ``pandas`` extra.
39
+ """Open local data with an optional ``pandas``, ``radar``, or ``netcdf`` reader.
39
40
 
40
41
  ERDDAP units are kept in ``frame.attrs["units"]`` and source provenance
41
- in ``frame.attrs["usdata"]``. See ``usdata.readers.open_asset`` for options.
42
+ in ``frame.attrs["usdata"]``. NEXRAD returns a xarray DataTree with provenance
43
+ in ``radar.attrs["usdata"]``. NetCDF4 returns a loaded xarray Dataset with
44
+ matching provenance in its attributes. See ``usdata.readers.open_asset`` for options.
45
+ Use ``sweep=0`` or ``sweep=[0, 2]`` to load selected zero-based radar sweeps.
42
46
  Cached files and provenance sidecars are never changed.
43
47
  """
44
48
  from usdata.readers import open_asset
45
49
 
46
50
  return open_asset(
47
- self, reader=reader, dtype=dtype, parse_dates=parse_dates, usecols=usecols, nrows=nrows
51
+ self,
52
+ reader=reader,
53
+ dtype=dtype,
54
+ parse_dates=parse_dates,
55
+ usecols=usecols,
56
+ nrows=nrows,
57
+ sweep=sweep,
48
58
  )
49
59
 
50
60
 
@@ -9,9 +9,9 @@ from __future__ import annotations
9
9
 
10
10
  import math
11
11
  import re
12
- from datetime import datetime
12
+ from datetime import datetime, timedelta
13
13
  from enum import StrEnum
14
- from typing import Any
14
+ from typing import Any, Literal
15
15
 
16
16
  from pydantic import BaseModel, ConfigDict, Field, model_validator
17
17
 
@@ -79,6 +79,12 @@ class BBox(BaseModel):
79
79
  @classmethod
80
80
  def from_point(cls, lat: float, lon: float, radius_km: float = 0.0) -> BBox:
81
81
  """Box around a point. Uses a flat-earth approximation, fine for small radii."""
82
+ if not math.isfinite(lat) or not -90 <= lat <= 90:
83
+ raise ValueError("lat must be finite and between -90 and 90")
84
+ if not math.isfinite(lon) or not -180 <= lon <= 180:
85
+ raise ValueError("lon must be finite and between -180 and 180")
86
+ if not math.isfinite(radius_km) or radius_km < 0:
87
+ raise ValueError("radius_km must be finite and nonnegative")
82
88
  dlat = radius_km / 111.0
83
89
  dlon = radius_km / (111.0 * max(math.cos(math.radians(lat)), 1e-6))
84
90
  return cls(
@@ -223,6 +229,22 @@ class Asset(BaseModel):
223
229
  bbox: BBox | None = None
224
230
 
225
231
 
232
+ class TemporalSelection(BaseModel):
233
+ """A start-time selection and its explicit policy; not source provenance.
234
+
235
+ No match has ``asset=None``, ``offset_seconds=None``, and zero eligible
236
+ candidates. Counts refer to the supplied candidates, not a remote catalog.
237
+ """
238
+
239
+ target: datetime
240
+ tolerance: timedelta
241
+ direction: Literal["nearest", "at_or_before"]
242
+ asset: Asset | None
243
+ offset_seconds: float | None
244
+ candidate_count: int = Field(ge=0)
245
+ eligible_count: int = Field(ge=0)
246
+
247
+
226
248
  class Provenance(BaseModel):
227
249
  """Everything needed to say where a local file came from and re-fetch it."""
228
250