usdata 0.7.0__tar.gz → 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- usdata-0.9.0/PKG-INFO +115 -0
- usdata-0.9.0/README.md +87 -0
- {usdata-0.7.0 → usdata-0.9.0}/pyproject.toml +32 -4
- {usdata-0.7.0 → usdata-0.9.0}/pyproject.toml.orig +27 -4
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/__init__.py +4 -1
- usdata-0.9.0/src/usdata/_netcdf.py +36 -0
- usdata-0.9.0/src/usdata/_radar.py +123 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/data/registry.yaml +23 -19
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/fetch.py +13 -3
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/models.py +24 -2
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/protocols/s3.py +15 -2
- usdata-0.9.0/src/usdata/providers/_http.py +27 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/noaa/coastwatch.py +4 -19
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/noaa/ghcnd.py +4 -19
- usdata-0.9.0/src/usdata/providers/noaa/goes.py +109 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/noaa/nexrad.py +5 -22
- usdata-0.9.0/src/usdata/providers/noaa/storm_events.py +117 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/usgs/daily.py +4 -19
- usdata-0.9.0/src/usdata/readers.py +156 -0
- usdata-0.9.0/src/usdata/selection.py +71 -0
- usdata-0.7.0/PKG-INFO +0 -186
- usdata-0.7.0/README.md +0 -163
- usdata-0.7.0/src/usdata/readers.py +0 -103
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/_files.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/_progress.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/cache.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/cli/__init__.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/cli/app.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/cli/progress.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/data/nexrad_sites.csv +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/data/places.csv +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/data/places.sources.json +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/manifest.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/protocols/__init__.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/protocols/erddap.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/protocols/http.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/provenance.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/__init__.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/base.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/noaa/__init__.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/noaa/gsom.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/noaa/sites.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/providers/usgs/__init__.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/pull.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/py.typed +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/query.py +0 -0
- {usdata-0.7.0 → usdata-0.9.0}/src/usdata/registry.py +0 -0
usdata-0.9.0/PKG-INFO
ADDED
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: usdata
|
|
3
|
+
Version: 0.9.0
|
|
4
|
+
Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
|
|
5
|
+
Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
|
|
6
|
+
Author: Jake Van Slyke
|
|
7
|
+
Author-email: Jake Van Slyke <jakervanslyke@gmail.com>
|
|
8
|
+
License-Expression: Apache-2.0
|
|
9
|
+
Classifier: Development Status :: 2 - Pre-Alpha
|
|
10
|
+
Classifier: Intended Audience :: Science/Research
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering
|
|
13
|
+
Requires-Dist: httpx>=0.28.1
|
|
14
|
+
Requires-Dist: pydantic>=2.7
|
|
15
|
+
Requires-Dist: pyyaml>=6.0
|
|
16
|
+
Requires-Dist: typer>=0.18
|
|
17
|
+
Requires-Dist: xarray>=2025.1 ; extra == 'netcdf'
|
|
18
|
+
Requires-Dist: h5netcdf[h5py]>=1.8.1 ; extra == 'netcdf'
|
|
19
|
+
Requires-Dist: pandas>=3.0 ; extra == 'pandas'
|
|
20
|
+
Requires-Dist: xradar>=0.12.0 ; extra == 'radar'
|
|
21
|
+
Requires-Python: >=3.11
|
|
22
|
+
Project-URL: Homepage, https://github.com/jakeryderv/usdata
|
|
23
|
+
Project-URL: Repository, https://github.com/jakeryderv/usdata
|
|
24
|
+
Provides-Extra: netcdf
|
|
25
|
+
Provides-Extra: pandas
|
|
26
|
+
Provides-Extra: radar
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# usdata
|
|
30
|
+
|
|
31
|
+
Discover U.S. scientific datasets, fetch their files, and keep a reproducible
|
|
32
|
+
record of where every input came from. Use the same Python SDK or CLI across
|
|
33
|
+
supported NOAA and USGS datasets.
|
|
34
|
+
|
|
35
|
+
**Pre-alpha.** These docs describe the current source checkout. Features marked
|
|
36
|
+
**Unreleased** require a source installation; consult the
|
|
37
|
+
[changelog](CHANGELOG.md) for published versions. Other providers are planned.
|
|
38
|
+
|
|
39
|
+
## Start here
|
|
40
|
+
|
|
41
|
+
```sh
|
|
42
|
+
pip install usdata
|
|
43
|
+
usdata search precipitation --location Oklahoma
|
|
44
|
+
usdata info noaa:ghcn-daily
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Search uses a curated registry. Fetching contacts the upstream service; readers
|
|
48
|
+
open the resulting local files. Provenance and manifests connect those steps.
|
|
49
|
+
|
|
50
|
+
- [Quick start and documentation](docs/index.md)
|
|
51
|
+
- [Fetch and analyze data](docs/guides/fetch-and-analyze.md)
|
|
52
|
+
- [Runnable examples with saved outputs](examples/README.md)
|
|
53
|
+
- [Readers](docs/reference/readers.md) and [reproducible manifests](docs/reference/manifests.md)
|
|
54
|
+
|
|
55
|
+
## Providers
|
|
56
|
+
|
|
57
|
+
<!-- registry:start -->
|
|
58
|
+
| Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
|
|
59
|
+
|---|---:|---:|---:|---|---|
|
|
60
|
+
| [NOAA](docs/providers/noaa.md) | 6 | 0 | 23 | — | `ghcn-daily`, `gsom`, `storm-events`, `nexrad-level2`, `goes-abi`, `coastwatch-sst`, +23 planned |
|
|
61
|
+
| [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
|
|
62
|
+
| [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
|
|
63
|
+
| [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
64
|
+
| [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
|
|
65
|
+
| [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
66
|
+
| [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
|
|
67
|
+
|
|
68
|
+
Available datasets are in `code`, stubs in _italics_; planned ones are counted. Available means implemented in this source checkout; consult the [releases](https://github.com/jakeryderv/usdata/releases) for published support. Provider pages link access notes to the generated dataset catalog; [the roadmap](docs/roadmap.md) explains future priorities.
|
|
69
|
+
<!-- registry:end -->
|
|
70
|
+
|
|
71
|
+
## Development
|
|
72
|
+
|
|
73
|
+
Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
|
|
74
|
+
|
|
75
|
+
`just setup` uses the tested Python 3.14.7 pin in `.python-version`. Older Linux
|
|
76
|
+
uv Python 3.14 builds can crash during NumPy array operations; see
|
|
77
|
+
[the upstream fix](https://github.com/astral-sh/python-build-standalone/issues/991).
|
|
78
|
+
|
|
79
|
+
```sh
|
|
80
|
+
git clone https://github.com/jakeryderv/usdata && cd usdata
|
|
81
|
+
just setup # install toolchain and dependencies
|
|
82
|
+
just test # all offline tests
|
|
83
|
+
just check # format, lint, typecheck, offline tests, generated docs, release notices
|
|
84
|
+
just check-pandas # install the CSV extra and run the same checks
|
|
85
|
+
just check-radar # install the radar extra and run the same checks
|
|
86
|
+
just check-netcdf # install the NetCDF4 extra and run the same checks
|
|
87
|
+
just notebooks # launch the optional Jupyter examples environment
|
|
88
|
+
just run-notebooks # execute notebooks live in fresh kernels and temporary caches
|
|
89
|
+
just docs-serve # build and preview the documentation locally, with reload
|
|
90
|
+
just check-docs # validate generated content and build the site strictly
|
|
91
|
+
just build # build wheel and sdist
|
|
92
|
+
just smoke # exercise core and pandas wheel installations outside the checkout
|
|
93
|
+
just run search radar
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Offline tests mechanically block network connections. Tests that hit
|
|
97
|
+
live services run with `just test-live`; see [testing levels and organization](docs/testing.md). CI checks Python 3.11 and 3.14 on
|
|
98
|
+
Linux with core-only, pandas, radar, and NetCDF dependency profiles. Installed-wheel
|
|
99
|
+
checks cover all four profiles on Linux, macOS, and Windows. The full offline and
|
|
100
|
+
live-service suites run on Linux. `just setup` restores a core-only development
|
|
101
|
+
environment; the `check-pandas`, `check-radar`, and `check-netcdf` commands install
|
|
102
|
+
their respective extras.
|
|
103
|
+
|
|
104
|
+
Releases: `just release minor` opens a version-bump PR; merging it publishes
|
|
105
|
+
to PyPI and creates the tag and GitHub release. See
|
|
106
|
+
[docs/versioning.md](docs/versioning.md).
|
|
107
|
+
|
|
108
|
+
See [provider access notes](docs/providers/README.md),
|
|
109
|
+
[docs/architecture.md](docs/architecture.md) for how the pieces fit,
|
|
110
|
+
[architecture decisions](docs/adr/README.md) for why, and [CONTRIBUTING.md](CONTRIBUTING.md) to add
|
|
111
|
+
a dataset.
|
|
112
|
+
|
|
113
|
+
## License
|
|
114
|
+
|
|
115
|
+
[Apache-2.0](LICENSE).
|
usdata-0.9.0/README.md
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
# usdata
|
|
2
|
+
|
|
3
|
+
Discover U.S. scientific datasets, fetch their files, and keep a reproducible
|
|
4
|
+
record of where every input came from. Use the same Python SDK or CLI across
|
|
5
|
+
supported NOAA and USGS datasets.
|
|
6
|
+
|
|
7
|
+
**Pre-alpha.** These docs describe the current source checkout. Features marked
|
|
8
|
+
**Unreleased** require a source installation; consult the
|
|
9
|
+
[changelog](CHANGELOG.md) for published versions. Other providers are planned.
|
|
10
|
+
|
|
11
|
+
## Start here
|
|
12
|
+
|
|
13
|
+
```sh
|
|
14
|
+
pip install usdata
|
|
15
|
+
usdata search precipitation --location Oklahoma
|
|
16
|
+
usdata info noaa:ghcn-daily
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
Search uses a curated registry. Fetching contacts the upstream service; readers
|
|
20
|
+
open the resulting local files. Provenance and manifests connect those steps.
|
|
21
|
+
|
|
22
|
+
- [Quick start and documentation](docs/index.md)
|
|
23
|
+
- [Fetch and analyze data](docs/guides/fetch-and-analyze.md)
|
|
24
|
+
- [Runnable examples with saved outputs](examples/README.md)
|
|
25
|
+
- [Readers](docs/reference/readers.md) and [reproducible manifests](docs/reference/manifests.md)
|
|
26
|
+
|
|
27
|
+
## Providers
|
|
28
|
+
|
|
29
|
+
<!-- registry:start -->
|
|
30
|
+
| Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
|
|
31
|
+
|---|---:|---:|---:|---|---|
|
|
32
|
+
| [NOAA](docs/providers/noaa.md) | 6 | 0 | 23 | — | `ghcn-daily`, `gsom`, `storm-events`, `nexrad-level2`, `goes-abi`, `coastwatch-sst`, +23 planned |
|
|
33
|
+
| [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
|
|
34
|
+
| [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
|
|
35
|
+
| [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
36
|
+
| [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
|
|
37
|
+
| [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
38
|
+
| [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
|
|
39
|
+
|
|
40
|
+
Available datasets are in `code`, stubs in _italics_; planned ones are counted. Available means implemented in this source checkout; consult the [releases](https://github.com/jakeryderv/usdata/releases) for published support. Provider pages link access notes to the generated dataset catalog; [the roadmap](docs/roadmap.md) explains future priorities.
|
|
41
|
+
<!-- registry:end -->
|
|
42
|
+
|
|
43
|
+
## Development
|
|
44
|
+
|
|
45
|
+
Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
|
|
46
|
+
|
|
47
|
+
`just setup` uses the tested Python 3.14.7 pin in `.python-version`. Older Linux
|
|
48
|
+
uv Python 3.14 builds can crash during NumPy array operations; see
|
|
49
|
+
[the upstream fix](https://github.com/astral-sh/python-build-standalone/issues/991).
|
|
50
|
+
|
|
51
|
+
```sh
|
|
52
|
+
git clone https://github.com/jakeryderv/usdata && cd usdata
|
|
53
|
+
just setup # install toolchain and dependencies
|
|
54
|
+
just test # all offline tests
|
|
55
|
+
just check # format, lint, typecheck, offline tests, generated docs, release notices
|
|
56
|
+
just check-pandas # install the CSV extra and run the same checks
|
|
57
|
+
just check-radar # install the radar extra and run the same checks
|
|
58
|
+
just check-netcdf # install the NetCDF4 extra and run the same checks
|
|
59
|
+
just notebooks # launch the optional Jupyter examples environment
|
|
60
|
+
just run-notebooks # execute notebooks live in fresh kernels and temporary caches
|
|
61
|
+
just docs-serve # build and preview the documentation locally, with reload
|
|
62
|
+
just check-docs # validate generated content and build the site strictly
|
|
63
|
+
just build # build wheel and sdist
|
|
64
|
+
just smoke # exercise core and pandas wheel installations outside the checkout
|
|
65
|
+
just run search radar
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Offline tests mechanically block network connections. Tests that hit
|
|
69
|
+
live services run with `just test-live`; see [testing levels and organization](docs/testing.md). CI checks Python 3.11 and 3.14 on
|
|
70
|
+
Linux with core-only, pandas, radar, and NetCDF dependency profiles. Installed-wheel
|
|
71
|
+
checks cover all four profiles on Linux, macOS, and Windows. The full offline and
|
|
72
|
+
live-service suites run on Linux. `just setup` restores a core-only development
|
|
73
|
+
environment; the `check-pandas`, `check-radar`, and `check-netcdf` commands install
|
|
74
|
+
their respective extras.
|
|
75
|
+
|
|
76
|
+
Releases: `just release minor` opens a version-bump PR; merging it publishes
|
|
77
|
+
to PyPI and creates the tag and GitHub release. See
|
|
78
|
+
[docs/versioning.md](docs/versioning.md).
|
|
79
|
+
|
|
80
|
+
See [provider access notes](docs/providers/README.md),
|
|
81
|
+
[docs/architecture.md](docs/architecture.md) for how the pieces fit,
|
|
82
|
+
[architecture decisions](docs/adr/README.md) for why, and [CONTRIBUTING.md](CONTRIBUTING.md) to add
|
|
83
|
+
a dataset.
|
|
84
|
+
|
|
85
|
+
## License
|
|
86
|
+
|
|
87
|
+
[Apache-2.0](LICENSE).
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "usdata"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.9.0"
|
|
4
4
|
description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -23,7 +23,7 @@ dependencies = [
|
|
|
23
23
|
"httpx>=0.28.1",
|
|
24
24
|
"pydantic>=2.7",
|
|
25
25
|
"pyyaml>=6.0",
|
|
26
|
-
"typer>=0.
|
|
26
|
+
"typer>=0.18",
|
|
27
27
|
]
|
|
28
28
|
|
|
29
29
|
[[project.authors]]
|
|
@@ -32,6 +32,11 @@ email = "jakervanslyke@gmail.com"
|
|
|
32
32
|
|
|
33
33
|
[project.optional-dependencies]
|
|
34
34
|
pandas = ["pandas>=3.0"]
|
|
35
|
+
radar = ["xradar>=0.12.0"]
|
|
36
|
+
netcdf = [
|
|
37
|
+
"xarray>=2025.1",
|
|
38
|
+
"h5netcdf[h5py]>=1.8.1",
|
|
39
|
+
]
|
|
35
40
|
|
|
36
41
|
[project.urls]
|
|
37
42
|
Homepage = "https://github.com/jakeryderv/usdata"
|
|
@@ -48,6 +53,18 @@ dev = [
|
|
|
48
53
|
"respx>=0.23.1",
|
|
49
54
|
"ruff>=0.6",
|
|
50
55
|
]
|
|
56
|
+
docs = [
|
|
57
|
+
"mkdocstrings-python>=2.0.8",
|
|
58
|
+
"nbconvert>=7.17.1",
|
|
59
|
+
"zensical>=0.0.60",
|
|
60
|
+
]
|
|
61
|
+
examples = [
|
|
62
|
+
"ipykernel>=6.29",
|
|
63
|
+
"jupyterlab>=4.3",
|
|
64
|
+
"matplotlib>=3.9",
|
|
65
|
+
"nbclient>=0.10",
|
|
66
|
+
"pandas>=3.0",
|
|
67
|
+
]
|
|
51
68
|
|
|
52
69
|
[build-system]
|
|
53
70
|
requires = ["uv_build>=0.12.5,<0.13.0"]
|
|
@@ -96,5 +113,16 @@ venv = ".venv"
|
|
|
96
113
|
|
|
97
114
|
[tool.pytest.ini_options]
|
|
98
115
|
testpaths = ["tests"]
|
|
99
|
-
markers = [
|
|
100
|
-
|
|
116
|
+
markers = [
|
|
117
|
+
"l0: pure in-memory logic",
|
|
118
|
+
"l1: in-process components with controlled transport",
|
|
119
|
+
"l2: local functional tests with filesystem or real decoders",
|
|
120
|
+
"l3: controlled service deployment (currently unused)",
|
|
121
|
+
"l4: production or upstream-live compatibility",
|
|
122
|
+
"live: hits upstream services; requires --run-live",
|
|
123
|
+
"integration: compatibility alias for live",
|
|
124
|
+
"pandas: optional CSV reader coverage",
|
|
125
|
+
"radar: optional radar reader coverage",
|
|
126
|
+
"netcdf: optional NetCDF reader coverage",
|
|
127
|
+
]
|
|
128
|
+
addopts = "-ra --strict-markers"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "usdata"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.9.0"
|
|
4
4
|
description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -19,11 +19,13 @@ dependencies = [
|
|
|
19
19
|
"httpx>=0.28.1",
|
|
20
20
|
"pydantic>=2.7",
|
|
21
21
|
"pyyaml>=6.0",
|
|
22
|
-
"typer>=0.
|
|
22
|
+
"typer>=0.18",
|
|
23
23
|
]
|
|
24
24
|
|
|
25
25
|
[project.optional-dependencies]
|
|
26
26
|
pandas = ["pandas>=3.0"]
|
|
27
|
+
radar = ["xradar>=0.12.0"]
|
|
28
|
+
netcdf = ["xarray>=2025.1", "h5netcdf[h5py]>=1.8.1"]
|
|
27
29
|
|
|
28
30
|
[project.urls]
|
|
29
31
|
Homepage = "https://github.com/jakeryderv/usdata"
|
|
@@ -40,6 +42,18 @@ dev = [
|
|
|
40
42
|
"respx>=0.23.1",
|
|
41
43
|
"ruff>=0.6",
|
|
42
44
|
]
|
|
45
|
+
docs = [
|
|
46
|
+
"mkdocstrings-python>=2.0.8",
|
|
47
|
+
"nbconvert>=7.17.1",
|
|
48
|
+
"zensical>=0.0.60",
|
|
49
|
+
]
|
|
50
|
+
examples = [
|
|
51
|
+
"ipykernel>=6.29",
|
|
52
|
+
"jupyterlab>=4.3",
|
|
53
|
+
"matplotlib>=3.9",
|
|
54
|
+
"nbclient>=0.10",
|
|
55
|
+
"pandas>=3.0",
|
|
56
|
+
]
|
|
43
57
|
|
|
44
58
|
[build-system]
|
|
45
59
|
requires = ["uv_build>=0.12.5,<0.13.0"]
|
|
@@ -72,6 +86,15 @@ venv = ".venv"
|
|
|
72
86
|
[tool.pytest.ini_options]
|
|
73
87
|
testpaths = ["tests"]
|
|
74
88
|
markers = [
|
|
75
|
-
"
|
|
89
|
+
"l0: pure in-memory logic",
|
|
90
|
+
"l1: in-process components with controlled transport",
|
|
91
|
+
"l2: local functional tests with filesystem or real decoders",
|
|
92
|
+
"l3: controlled service deployment (currently unused)",
|
|
93
|
+
"l4: production or upstream-live compatibility",
|
|
94
|
+
"live: hits upstream services; requires --run-live",
|
|
95
|
+
"integration: compatibility alias for live",
|
|
96
|
+
"pandas: optional CSV reader coverage",
|
|
97
|
+
"radar: optional radar reader coverage",
|
|
98
|
+
"netcdf: optional NetCDF reader coverage",
|
|
76
99
|
]
|
|
77
|
-
addopts = "-ra"
|
|
100
|
+
addopts = "-ra --strict-markers"
|
|
@@ -10,10 +10,11 @@ try:
|
|
|
10
10
|
except PackageNotFoundError: # running from a source tree without an install
|
|
11
11
|
__version__ = "0.0.0"
|
|
12
12
|
|
|
13
|
-
from usdata.models import Asset, BBox, Dataset, Provenance, Query, TimeRange
|
|
13
|
+
from usdata.models import Asset, BBox, Dataset, Provenance, Query, TemporalSelection, TimeRange
|
|
14
14
|
from usdata.pull import pull, verify
|
|
15
15
|
from usdata.query import build_query
|
|
16
16
|
from usdata.registry import DatasetNotFound, Registry, SearchResult, default_registry
|
|
17
|
+
from usdata.selection import select_by_time
|
|
17
18
|
|
|
18
19
|
__all__ = [
|
|
19
20
|
"Asset",
|
|
@@ -24,6 +25,7 @@ __all__ = [
|
|
|
24
25
|
"Query",
|
|
25
26
|
"Registry",
|
|
26
27
|
"SearchResult",
|
|
28
|
+
"TemporalSelection",
|
|
27
29
|
"TimeRange",
|
|
28
30
|
"__version__",
|
|
29
31
|
"build_query",
|
|
@@ -31,6 +33,7 @@ __all__ = [
|
|
|
31
33
|
"get",
|
|
32
34
|
"pull",
|
|
33
35
|
"search",
|
|
36
|
+
"select_by_time",
|
|
34
37
|
"verify",
|
|
35
38
|
]
|
|
36
39
|
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Local, eagerly loaded NetCDF4 reading behind the netcdf extra."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from importlib import import_module
|
|
6
|
+
from typing import TYPE_CHECKING, Any
|
|
7
|
+
|
|
8
|
+
from usdata.readers import MissingReaderDependency
|
|
9
|
+
|
|
10
|
+
if TYPE_CHECKING:
|
|
11
|
+
from usdata.fetch import FetchedAsset
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def open_netcdf(fetched: FetchedAsset) -> Any:
|
|
15
|
+
"""Load a NetCDF4 root Dataset, then close every source file handle."""
|
|
16
|
+
try:
|
|
17
|
+
xarray = import_module("xarray")
|
|
18
|
+
import_module("h5netcdf")
|
|
19
|
+
import_module("h5py")
|
|
20
|
+
except ModuleNotFoundError as error:
|
|
21
|
+
if error.name not in {"xarray", "h5netcdf", "h5py"}:
|
|
22
|
+
raise
|
|
23
|
+
raise MissingReaderDependency(
|
|
24
|
+
'NetCDF4 reading requires xarray and h5netcdf; install: pip install "usdata[netcdf]"'
|
|
25
|
+
) from error
|
|
26
|
+
# A local file object and fixed engine prevent interpretation as an OPeNDAP URL.
|
|
27
|
+
with (
|
|
28
|
+
fetched.path.open("rb") as stream,
|
|
29
|
+
xarray.open_dataset(stream, engine="h5netcdf", chunks=None) as dataset,
|
|
30
|
+
):
|
|
31
|
+
dataset.load()
|
|
32
|
+
dataset.attrs["usdata"] = {
|
|
33
|
+
"asset_id": fetched.asset.id,
|
|
34
|
+
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
35
|
+
}
|
|
36
|
+
return dataset
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Local NEXRAD Level II decoding behind the radar extra."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import bz2
|
|
6
|
+
import gzip
|
|
7
|
+
from importlib import import_module
|
|
8
|
+
from typing import TYPE_CHECKING, Any
|
|
9
|
+
|
|
10
|
+
from usdata.readers import MissingReaderDependency, RadarDecodeError
|
|
11
|
+
|
|
12
|
+
# NOAA RDA/RPG ICD 2620002Y, Table XVII-I notes 21 and 30.
|
|
13
|
+
MOMENT_FLAG_COUNTS = {
|
|
14
|
+
"DBZH": 2,
|
|
15
|
+
"VRADH": 2,
|
|
16
|
+
"WRADH": 2,
|
|
17
|
+
"ZDR": 2,
|
|
18
|
+
"PHIDP": 2,
|
|
19
|
+
"RHOHV": 2,
|
|
20
|
+
"CCORH": 8,
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
if TYPE_CHECKING:
|
|
24
|
+
from usdata.fetch import FetchedAsset
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _check_sweeps(content: bytes, sweep: int | list[int] | None) -> None:
|
|
28
|
+
"""Reject decoder tables that pair a sweep with another sweep's coordinates.
|
|
29
|
+
|
|
30
|
+
xradar 0.12 can omit an interior sweep without an end marker from its data
|
|
31
|
+
table while keeping its moment metadata. The coordinate list then shifts.
|
|
32
|
+
Compare record identities, not just ray counts: equal-length sweeps can
|
|
33
|
+
otherwise silently acquire another sweep's coordinates and timestamps.
|
|
34
|
+
"""
|
|
35
|
+
backend = import_module("xradar.io.backends.nexrad_level2")
|
|
36
|
+
with backend.NEXRADLevel2File(content, loaddata=False) as volume:
|
|
37
|
+
# Parsing completeness also populates the per-sweep data tables.
|
|
38
|
+
_ = volume.incomplete_sweeps
|
|
39
|
+
moments = volume.msg_31_data_header
|
|
40
|
+
coordinates = volume.msg_31_header
|
|
41
|
+
requested = (
|
|
42
|
+
list(range(len(moments)))
|
|
43
|
+
if sweep is None
|
|
44
|
+
else (sweep if isinstance(sweep, list) else [sweep])
|
|
45
|
+
)
|
|
46
|
+
for index in requested:
|
|
47
|
+
if index >= len(moments):
|
|
48
|
+
raise ValueError(f"sweep {index} is outside this volume's {len(moments)} sweeps")
|
|
49
|
+
data = volume.data.get(index)
|
|
50
|
+
rays = coordinates[index] if index < len(coordinates) else []
|
|
51
|
+
# Non-radial messages may occur within a sweep or after its last
|
|
52
|
+
# received ray. Follow the decoder's traversal, excluding those
|
|
53
|
+
# records, instead of requiring record_end to be a radial message.
|
|
54
|
+
expected_records = []
|
|
55
|
+
if data is not None:
|
|
56
|
+
intermediate = {record["record_number"] for record in data["intermediate_records"]}
|
|
57
|
+
expected_records = [
|
|
58
|
+
record
|
|
59
|
+
for record in range(data["record_number"], data["record_end"] + 1)
|
|
60
|
+
if record not in intermediate
|
|
61
|
+
]
|
|
62
|
+
if (
|
|
63
|
+
data is None
|
|
64
|
+
or not rays
|
|
65
|
+
or moments[index]["record_number"] != data["record_number"]
|
|
66
|
+
or [ray["record_number"] for ray in rays] != expected_records
|
|
67
|
+
):
|
|
68
|
+
raise RadarDecodeError(
|
|
69
|
+
f"cannot safely decode sweep {index}: NEXRAD moment and coordinate records "
|
|
70
|
+
"do not agree; select an unaffected sweep explicitly with open(sweep=...) "
|
|
71
|
+
"or use another decoder. No sweeps were silently dropped."
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def open_nexrad(fetched: FetchedAsset, *, sweep: int | list[int] | None = None) -> Any:
|
|
76
|
+
"""Decode a local volume into a fully loaded xarray DataTree."""
|
|
77
|
+
try:
|
|
78
|
+
xradar = import_module("xradar")
|
|
79
|
+
except ModuleNotFoundError as error:
|
|
80
|
+
if error.name != "xradar":
|
|
81
|
+
raise
|
|
82
|
+
raise MissingReaderDependency(
|
|
83
|
+
'NEXRAD reading requires xradar; install it with: pip install "usdata[radar]" '
|
|
84
|
+
'(or uv add "usdata[radar]")'
|
|
85
|
+
) from error
|
|
86
|
+
|
|
87
|
+
# Bytes prevent remote URL interpretation and work across the backend's
|
|
88
|
+
# repeated sweep reads. Compressed source files stay unchanged in the cache.
|
|
89
|
+
content = fetched.path.read_bytes()
|
|
90
|
+
if content.startswith(b"\x1f\x8b"):
|
|
91
|
+
content = gzip.decompress(content)
|
|
92
|
+
elif content.startswith(b"BZh"):
|
|
93
|
+
content = bz2.decompress(content)
|
|
94
|
+
_check_sweeps(content, sweep)
|
|
95
|
+
radar = xradar.io.open_nexradlevel2_datatree(content, sweep=sweep, incomplete_sweep="pad")
|
|
96
|
+
try:
|
|
97
|
+
radar.load()
|
|
98
|
+
finally:
|
|
99
|
+
radar.close()
|
|
100
|
+
for node in radar.subtree:
|
|
101
|
+
for name, variable in node.ds.variables.items():
|
|
102
|
+
# The backend records the entire input byte string as `source`.
|
|
103
|
+
# Provenance below is the durable reference, not that decoder buffer.
|
|
104
|
+
variable.encoding.pop("source", None)
|
|
105
|
+
if name in MOMENT_FLAG_COUNTS and "range" in variable.dims:
|
|
106
|
+
scale = variable.encoding.get("scale_factor")
|
|
107
|
+
offset = variable.encoding.get("add_offset")
|
|
108
|
+
if scale is not None and offset is not None:
|
|
109
|
+
# xradar 0.12 does not supply _FillValue for NEXRAD flags.
|
|
110
|
+
# Compare using each moment's native scale, not fixed units.
|
|
111
|
+
data = node[name]
|
|
112
|
+
valid = data.notnull()
|
|
113
|
+
for code in range(MOMENT_FLAG_COUNTS[name]):
|
|
114
|
+
valid = valid & (data != offset + code * scale)
|
|
115
|
+
masked = data.where(valid)
|
|
116
|
+
masked.encoding = data.encoding.copy()
|
|
117
|
+
node[name] = masked
|
|
118
|
+
radar.attrs["usdata"] = {
|
|
119
|
+
"asset_id": fetched.asset.id,
|
|
120
|
+
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
121
|
+
"sweeps": [name.lstrip("/") for name in radar.groups if name.startswith("/sweep_")],
|
|
122
|
+
}
|
|
123
|
+
return radar
|
|
@@ -105,21 +105,22 @@ datasets:
|
|
|
105
105
|
|
|
106
106
|
- id: noaa:goes-abi
|
|
107
107
|
provider: noaa
|
|
108
|
-
status:
|
|
108
|
+
status: available
|
|
109
109
|
domain: weather-satellites
|
|
110
|
-
|
|
111
|
-
title: GOES-R ABI
|
|
112
|
-
description: >-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
keywords: [satellite, imagery, goes, abi, clouds,
|
|
110
|
+
since: "0.8"
|
|
111
|
+
title: GOES-R ABI CONUS Cloud and Moisture Imagery
|
|
112
|
+
description: >-
|
|
113
|
+
Single-channel CONUS Cloud and Moisture Imagery (ABI-L2-CMIPC) from
|
|
114
|
+
GOES-16, 17, 18, and 19 in anonymous NOAA S3 buckets. Select an explicit
|
|
115
|
+
satellite, channel, and scan-start interval; each asset is a complete
|
|
116
|
+
NetCDF scene with no geographic or variable subsetting.
|
|
117
|
+
keywords: [satellite, imagery, goes, abi, clouds, infrared, reflectance, netcdf, conus]
|
|
118
118
|
protocol: s3
|
|
119
119
|
homepage: https://registry.opendata.aws/noaa-goes/
|
|
120
120
|
license: US Government Work (public domain)
|
|
121
|
-
temporal_extent: { start: "2017-
|
|
122
|
-
capabilities: { spatial_subset: false, temporal_subset: true, variable_subset:
|
|
121
|
+
temporal_extent: { start: "2017-02-28T00:00:00Z" }
|
|
122
|
+
capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: false }
|
|
123
|
+
adapter: usdata.providers.noaa.goes:GoesAbi
|
|
123
124
|
|
|
124
125
|
- id: noaa:goes-glm
|
|
125
126
|
provider: noaa
|
|
@@ -139,21 +140,24 @@ datasets:
|
|
|
139
140
|
|
|
140
141
|
- id: noaa:storm-events
|
|
141
142
|
provider: noaa
|
|
142
|
-
status:
|
|
143
|
+
status: available
|
|
143
144
|
domain: severe-weather
|
|
144
|
-
|
|
145
|
+
since: "0.8"
|
|
145
146
|
title: Storm Events Database
|
|
146
147
|
description: >-
|
|
147
|
-
NCEI's
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
148
|
+
NCEI's significant-weather event details since 1950, with locations,
|
|
149
|
+
impacts, and narratives. Anonymous whole-year gzipped CSV archives;
|
|
150
|
+
select the latest creation-date revision for each requested year.
|
|
151
|
+
No server-side row, location, or variable subsetting. Historical event
|
|
152
|
+
coverage and reporting practices vary; fatalities and locations tables
|
|
153
|
+
are separate products not included by this adapter.
|
|
151
154
|
keywords: [storms, tornado, hail, wind, flood, damage, severe weather, events]
|
|
152
155
|
protocol: http
|
|
153
|
-
homepage: https://www.
|
|
156
|
+
homepage: https://www.ncei.noaa.gov/access/storm-events-database/
|
|
154
157
|
license: US Government Work (public domain)
|
|
155
158
|
temporal_extent: { start: "1950-01-01T00:00:00Z" }
|
|
156
|
-
capabilities: { spatial_subset: false, temporal_subset:
|
|
159
|
+
capabilities: { spatial_subset: false, temporal_subset: false, variable_subset: false }
|
|
160
|
+
adapter: usdata.providers.noaa.storm_events:StormEvents
|
|
157
161
|
|
|
158
162
|
- id: noaa:hurdat2
|
|
159
163
|
provider: noaa
|
|
@@ -34,17 +34,27 @@ class FetchedAsset(BaseModel):
|
|
|
34
34
|
parse_dates: list[str] | None = None,
|
|
35
35
|
usecols: list[str] | None = None,
|
|
36
36
|
nrows: int | None = None,
|
|
37
|
+
sweep: int | list[int] | None = None,
|
|
37
38
|
) -> Any:
|
|
38
|
-
"""Open
|
|
39
|
+
"""Open local data with an optional ``pandas``, ``radar``, or ``netcdf`` reader.
|
|
39
40
|
|
|
40
41
|
ERDDAP units are kept in ``frame.attrs["units"]`` and source provenance
|
|
41
|
-
in ``frame.attrs["usdata"]``.
|
|
42
|
+
in ``frame.attrs["usdata"]``. NEXRAD returns a xarray DataTree with provenance
|
|
43
|
+
in ``radar.attrs["usdata"]``. NetCDF4 returns a loaded xarray Dataset with
|
|
44
|
+
matching provenance in its attributes. See ``usdata.readers.open_asset`` for options.
|
|
45
|
+
Use ``sweep=0`` or ``sweep=[0, 2]`` to load selected zero-based radar sweeps.
|
|
42
46
|
Cached files and provenance sidecars are never changed.
|
|
43
47
|
"""
|
|
44
48
|
from usdata.readers import open_asset
|
|
45
49
|
|
|
46
50
|
return open_asset(
|
|
47
|
-
self,
|
|
51
|
+
self,
|
|
52
|
+
reader=reader,
|
|
53
|
+
dtype=dtype,
|
|
54
|
+
parse_dates=parse_dates,
|
|
55
|
+
usecols=usecols,
|
|
56
|
+
nrows=nrows,
|
|
57
|
+
sweep=sweep,
|
|
48
58
|
)
|
|
49
59
|
|
|
50
60
|
|
|
@@ -9,9 +9,9 @@ from __future__ import annotations
|
|
|
9
9
|
|
|
10
10
|
import math
|
|
11
11
|
import re
|
|
12
|
-
from datetime import datetime
|
|
12
|
+
from datetime import datetime, timedelta
|
|
13
13
|
from enum import StrEnum
|
|
14
|
-
from typing import Any
|
|
14
|
+
from typing import Any, Literal
|
|
15
15
|
|
|
16
16
|
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
|
17
17
|
|
|
@@ -79,6 +79,12 @@ class BBox(BaseModel):
|
|
|
79
79
|
@classmethod
|
|
80
80
|
def from_point(cls, lat: float, lon: float, radius_km: float = 0.0) -> BBox:
|
|
81
81
|
"""Box around a point. Uses a flat-earth approximation, fine for small radii."""
|
|
82
|
+
if not math.isfinite(lat) or not -90 <= lat <= 90:
|
|
83
|
+
raise ValueError("lat must be finite and between -90 and 90")
|
|
84
|
+
if not math.isfinite(lon) or not -180 <= lon <= 180:
|
|
85
|
+
raise ValueError("lon must be finite and between -180 and 180")
|
|
86
|
+
if not math.isfinite(radius_km) or radius_km < 0:
|
|
87
|
+
raise ValueError("radius_km must be finite and nonnegative")
|
|
82
88
|
dlat = radius_km / 111.0
|
|
83
89
|
dlon = radius_km / (111.0 * max(math.cos(math.radians(lat)), 1e-6))
|
|
84
90
|
return cls(
|
|
@@ -223,6 +229,22 @@ class Asset(BaseModel):
|
|
|
223
229
|
bbox: BBox | None = None
|
|
224
230
|
|
|
225
231
|
|
|
232
|
+
class TemporalSelection(BaseModel):
|
|
233
|
+
"""A start-time selection and its explicit policy; not source provenance.
|
|
234
|
+
|
|
235
|
+
No match has ``asset=None``, ``offset_seconds=None``, and zero eligible
|
|
236
|
+
candidates. Counts refer to the supplied candidates, not a remote catalog.
|
|
237
|
+
"""
|
|
238
|
+
|
|
239
|
+
target: datetime
|
|
240
|
+
tolerance: timedelta
|
|
241
|
+
direction: Literal["nearest", "at_or_before"]
|
|
242
|
+
asset: Asset | None
|
|
243
|
+
offset_seconds: float | None
|
|
244
|
+
candidate_count: int = Field(ge=0)
|
|
245
|
+
eligible_count: int = Field(ge=0)
|
|
246
|
+
|
|
247
|
+
|
|
226
248
|
class Provenance(BaseModel):
|
|
227
249
|
"""Everything needed to say where a local file came from and re-fetch it."""
|
|
228
250
|
|