usdata 0.8.0__tar.gz → 0.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. usdata-0.9.0/PKG-INFO +115 -0
  2. usdata-0.9.0/README.md +87 -0
  3. {usdata-0.8.0 → usdata-0.9.0}/pyproject.toml +20 -4
  4. {usdata-0.8.0 → usdata-0.9.0}/pyproject.toml.orig +18 -4
  5. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/__init__.py +4 -1
  6. usdata-0.9.0/src/usdata/_radar.py +123 -0
  7. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/fetch.py +9 -1
  8. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/models.py +24 -2
  9. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/protocols/s3.py +15 -2
  10. usdata-0.9.0/src/usdata/providers/_http.py +27 -0
  11. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/providers/noaa/coastwatch.py +4 -19
  12. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/providers/noaa/ghcnd.py +4 -19
  13. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/providers/noaa/goes.py +5 -22
  14. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/providers/noaa/nexrad.py +5 -22
  15. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/providers/noaa/storm_events.py +4 -21
  16. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/providers/usgs/daily.py +4 -19
  17. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/readers.py +16 -1
  18. usdata-0.9.0/src/usdata/selection.py +71 -0
  19. usdata-0.8.0/PKG-INFO +0 -230
  20. usdata-0.8.0/README.md +0 -202
  21. usdata-0.8.0/src/usdata/_radar.py +0 -73
  22. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/_files.py +0 -0
  23. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/_netcdf.py +0 -0
  24. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/_progress.py +0 -0
  25. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/cache.py +0 -0
  26. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/cli/__init__.py +0 -0
  27. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/cli/app.py +0 -0
  28. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/cli/progress.py +0 -0
  29. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/data/nexrad_sites.csv +0 -0
  30. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/data/places.csv +0 -0
  31. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/data/places.sources.json +0 -0
  32. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/data/registry.yaml +0 -0
  33. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/manifest.py +0 -0
  34. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/protocols/__init__.py +0 -0
  35. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/protocols/erddap.py +0 -0
  36. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/protocols/http.py +0 -0
  37. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/provenance.py +0 -0
  38. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/providers/__init__.py +0 -0
  39. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/providers/base.py +0 -0
  40. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/providers/noaa/__init__.py +0 -0
  41. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/providers/noaa/gsom.py +0 -0
  42. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/providers/noaa/sites.py +0 -0
  43. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/providers/usgs/__init__.py +0 -0
  44. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/pull.py +0 -0
  45. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/py.typed +0 -0
  46. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/query.py +0 -0
  47. {usdata-0.8.0 → usdata-0.9.0}/src/usdata/registry.py +0 -0
usdata-0.9.0/PKG-INFO ADDED
@@ -0,0 +1,115 @@
1
+ Metadata-Version: 2.4
2
+ Name: usdata
3
+ Version: 0.9.0
4
+ Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
5
+ Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
6
+ Author: Jake Van Slyke
7
+ Author-email: Jake Van Slyke <jakervanslyke@gmail.com>
8
+ License-Expression: Apache-2.0
9
+ Classifier: Development Status :: 2 - Pre-Alpha
10
+ Classifier: Intended Audience :: Science/Research
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Topic :: Scientific/Engineering
13
+ Requires-Dist: httpx>=0.28.1
14
+ Requires-Dist: pydantic>=2.7
15
+ Requires-Dist: pyyaml>=6.0
16
+ Requires-Dist: typer>=0.18
17
+ Requires-Dist: xarray>=2025.1 ; extra == 'netcdf'
18
+ Requires-Dist: h5netcdf[h5py]>=1.8.1 ; extra == 'netcdf'
19
+ Requires-Dist: pandas>=3.0 ; extra == 'pandas'
20
+ Requires-Dist: xradar>=0.12.0 ; extra == 'radar'
21
+ Requires-Python: >=3.11
22
+ Project-URL: Homepage, https://github.com/jakeryderv/usdata
23
+ Project-URL: Repository, https://github.com/jakeryderv/usdata
24
+ Provides-Extra: netcdf
25
+ Provides-Extra: pandas
26
+ Provides-Extra: radar
27
+ Description-Content-Type: text/markdown
28
+
29
+ # usdata
30
+
31
+ Discover U.S. scientific datasets, fetch their files, and keep a reproducible
32
+ record of where every input came from. Use the same Python SDK or CLI across
33
+ supported NOAA and USGS datasets.
34
+
35
+ **Pre-alpha.** These docs describe the current source checkout. Features marked
36
+ **Unreleased** require a source installation; consult the
37
+ [changelog](CHANGELOG.md) for published versions. Other providers are planned.
38
+
39
+ ## Start here
40
+
41
+ ```sh
42
+ pip install usdata
43
+ usdata search precipitation --location Oklahoma
44
+ usdata info noaa:ghcn-daily
45
+ ```
46
+
47
+ Search uses a curated registry. Fetching contacts the upstream service; readers
48
+ open the resulting local files. Provenance and manifests connect those steps.
49
+
50
+ - [Quick start and documentation](docs/index.md)
51
+ - [Fetch and analyze data](docs/guides/fetch-and-analyze.md)
52
+ - [Runnable examples with saved outputs](examples/README.md)
53
+ - [Readers](docs/reference/readers.md) and [reproducible manifests](docs/reference/manifests.md)
54
+
55
+ ## Providers
56
+
57
+ <!-- registry:start -->
58
+ | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
59
+ |---|---:|---:|---:|---|---|
60
+ | [NOAA](docs/providers/noaa.md) | 6 | 0 | 23 | — | `ghcn-daily`, `gsom`, `storm-events`, `nexrad-level2`, `goes-abi`, `coastwatch-sst`, +23 planned |
61
+ | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
62
+ | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
63
+ | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
64
+ | [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
65
+ | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
66
+ | [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
67
+
68
+ Available datasets are in `code`, stubs in _italics_; planned ones are counted. Available means implemented in this source checkout; consult the [releases](https://github.com/jakeryderv/usdata/releases) for published support. Provider pages link access notes to the generated dataset catalog; [the roadmap](docs/roadmap.md) explains future priorities.
69
+ <!-- registry:end -->
70
+
71
+ ## Development
72
+
73
+ Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
74
+
75
+ `just setup` uses the tested Python 3.14.7 pin in `.python-version`. Older Linux
76
+ uv Python 3.14 builds can crash during NumPy array operations; see
77
+ [the upstream fix](https://github.com/astral-sh/python-build-standalone/issues/991).
78
+
79
+ ```sh
80
+ git clone https://github.com/jakeryderv/usdata && cd usdata
81
+ just setup # install toolchain and dependencies
82
+ just test # all offline tests
83
+ just check # format, lint, typecheck, offline tests, generated docs, release notices
84
+ just check-pandas # install the CSV extra and run the same checks
85
+ just check-radar # install the radar extra and run the same checks
86
+ just check-netcdf # install the NetCDF4 extra and run the same checks
87
+ just notebooks # launch the optional Jupyter examples environment
88
+ just run-notebooks # execute notebooks live in fresh kernels and temporary caches
89
+ just docs-serve # build and preview the documentation locally, with reload
90
+ just check-docs # validate generated content and build the site strictly
91
+ just build # build wheel and sdist
92
+ just smoke # exercise core and pandas wheel installations outside the checkout
93
+ just run search radar
94
+ ```
95
+
96
+ Offline tests mechanically block network connections. Tests that hit
97
+ live services run with `just test-live`; see [testing levels and organization](docs/testing.md). CI checks Python 3.11 and 3.14 on
98
+ Linux with core-only, pandas, radar, and NetCDF dependency profiles. Installed-wheel
99
+ checks cover all four profiles on Linux, macOS, and Windows. The full offline and
100
+ live-service suites run on Linux. `just setup` restores a core-only development
101
+ environment; the `check-pandas`, `check-radar`, and `check-netcdf` commands install
102
+ their respective extras.
103
+
104
+ Releases: `just release minor` opens a version-bump PR; merging it publishes
105
+ to PyPI and creates the tag and GitHub release. See
106
+ [docs/versioning.md](docs/versioning.md).
107
+
108
+ See [provider access notes](docs/providers/README.md),
109
+ [docs/architecture.md](docs/architecture.md) for how the pieces fit,
110
+ [architecture decisions](docs/adr/README.md) for why, and [CONTRIBUTING.md](CONTRIBUTING.md) to add
111
+ a dataset.
112
+
113
+ ## License
114
+
115
+ [Apache-2.0](LICENSE).
usdata-0.9.0/README.md ADDED
@@ -0,0 +1,87 @@
1
+ # usdata
2
+
3
+ Discover U.S. scientific datasets, fetch their files, and keep a reproducible
4
+ record of where every input came from. Use the same Python SDK or CLI across
5
+ supported NOAA and USGS datasets.
6
+
7
+ **Pre-alpha.** These docs describe the current source checkout. Features marked
8
+ **Unreleased** require a source installation; consult the
9
+ [changelog](CHANGELOG.md) for published versions. Other providers are planned.
10
+
11
+ ## Start here
12
+
13
+ ```sh
14
+ pip install usdata
15
+ usdata search precipitation --location Oklahoma
16
+ usdata info noaa:ghcn-daily
17
+ ```
18
+
19
+ Search uses a curated registry. Fetching contacts the upstream service; readers
20
+ open the resulting local files. Provenance and manifests connect those steps.
21
+
22
+ - [Quick start and documentation](docs/index.md)
23
+ - [Fetch and analyze data](docs/guides/fetch-and-analyze.md)
24
+ - [Runnable examples with saved outputs](examples/README.md)
25
+ - [Readers](docs/reference/readers.md) and [reproducible manifests](docs/reference/manifests.md)
26
+
27
+ ## Providers
28
+
29
+ <!-- registry:start -->
30
+ | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
31
+ |---|---:|---:|---:|---|---|
32
+ | [NOAA](docs/providers/noaa.md) | 6 | 0 | 23 | — | `ghcn-daily`, `gsom`, `storm-events`, `nexrad-level2`, `goes-abi`, `coastwatch-sst`, +23 planned |
33
+ | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
34
+ | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
35
+ | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
36
+ | [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
37
+ | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
38
+ | [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
39
+
40
+ Available datasets are in `code`, stubs in _italics_; planned ones are counted. Available means implemented in this source checkout; consult the [releases](https://github.com/jakeryderv/usdata/releases) for published support. Provider pages link access notes to the generated dataset catalog; [the roadmap](docs/roadmap.md) explains future priorities.
41
+ <!-- registry:end -->
42
+
43
+ ## Development
44
+
45
+ Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
46
+
47
+ `just setup` uses the tested Python 3.14.7 pin in `.python-version`. Older Linux
48
+ uv Python 3.14 builds can crash during NumPy array operations; see
49
+ [the upstream fix](https://github.com/astral-sh/python-build-standalone/issues/991).
50
+
51
+ ```sh
52
+ git clone https://github.com/jakeryderv/usdata && cd usdata
53
+ just setup # install toolchain and dependencies
54
+ just test # all offline tests
55
+ just check # format, lint, typecheck, offline tests, generated docs, release notices
56
+ just check-pandas # install the CSV extra and run the same checks
57
+ just check-radar # install the radar extra and run the same checks
58
+ just check-netcdf # install the NetCDF4 extra and run the same checks
59
+ just notebooks # launch the optional Jupyter examples environment
60
+ just run-notebooks # execute notebooks live in fresh kernels and temporary caches
61
+ just docs-serve # build and preview the documentation locally, with reload
62
+ just check-docs # validate generated content and build the site strictly
63
+ just build # build wheel and sdist
64
+ just smoke # exercise core and pandas wheel installations outside the checkout
65
+ just run search radar
66
+ ```
67
+
68
+ Offline tests mechanically block network connections. Tests that hit
69
+ live services run with `just test-live`; see [testing levels and organization](docs/testing.md). CI checks Python 3.11 and 3.14 on
70
+ Linux with core-only, pandas, radar, and NetCDF dependency profiles. Installed-wheel
71
+ checks cover all four profiles on Linux, macOS, and Windows. The full offline and
72
+ live-service suites run on Linux. `just setup` restores a core-only development
73
+ environment; the `check-pandas`, `check-radar`, and `check-netcdf` commands install
74
+ their respective extras.
75
+
76
+ Releases: `just release minor` opens a version-bump PR; merging it publishes
77
+ to PyPI and creates the tag and GitHub release. See
78
+ [docs/versioning.md](docs/versioning.md).
79
+
80
+ See [provider access notes](docs/providers/README.md),
81
+ [docs/architecture.md](docs/architecture.md) for how the pieces fit,
82
+ [architecture decisions](docs/adr/README.md) for why, and [CONTRIBUTING.md](CONTRIBUTING.md) to add
83
+ a dataset.
84
+
85
+ ## License
86
+
87
+ [Apache-2.0](LICENSE).
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.8.0"
3
+ version = "0.9.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -23,7 +23,7 @@ dependencies = [
23
23
  "httpx>=0.28.1",
24
24
  "pydantic>=2.7",
25
25
  "pyyaml>=6.0",
26
- "typer>=0.12",
26
+ "typer>=0.18",
27
27
  ]
28
28
 
29
29
  [[project.authors]]
@@ -53,6 +53,11 @@ dev = [
53
53
  "respx>=0.23.1",
54
54
  "ruff>=0.6",
55
55
  ]
56
+ docs = [
57
+ "mkdocstrings-python>=2.0.8",
58
+ "nbconvert>=7.17.1",
59
+ "zensical>=0.0.60",
60
+ ]
56
61
  examples = [
57
62
  "ipykernel>=6.29",
58
63
  "jupyterlab>=4.3",
@@ -108,5 +113,16 @@ venv = ".venv"
108
113
 
109
114
  [tool.pytest.ini_options]
110
115
  testpaths = ["tests"]
111
- markers = ["integration: hits live services; skipped unless --run-integration is passed"]
112
- addopts = "-ra"
116
+ markers = [
117
+ "l0: pure in-memory logic",
118
+ "l1: in-process components with controlled transport",
119
+ "l2: local functional tests with filesystem or real decoders",
120
+ "l3: controlled service deployment (currently unused)",
121
+ "l4: production or upstream-live compatibility",
122
+ "live: hits upstream services; requires --run-live",
123
+ "integration: compatibility alias for live",
124
+ "pandas: optional CSV reader coverage",
125
+ "radar: optional radar reader coverage",
126
+ "netcdf: optional NetCDF reader coverage",
127
+ ]
128
+ addopts = "-ra --strict-markers"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.8.0"
3
+ version = "0.9.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -19,7 +19,7 @@ dependencies = [
19
19
  "httpx>=0.28.1",
20
20
  "pydantic>=2.7",
21
21
  "pyyaml>=6.0",
22
- "typer>=0.12",
22
+ "typer>=0.18",
23
23
  ]
24
24
 
25
25
  [project.optional-dependencies]
@@ -42,6 +42,11 @@ dev = [
42
42
  "respx>=0.23.1",
43
43
  "ruff>=0.6",
44
44
  ]
45
+ docs = [
46
+ "mkdocstrings-python>=2.0.8",
47
+ "nbconvert>=7.17.1",
48
+ "zensical>=0.0.60",
49
+ ]
45
50
  examples = [
46
51
  "ipykernel>=6.29",
47
52
  "jupyterlab>=4.3",
@@ -81,6 +86,15 @@ venv = ".venv"
81
86
  [tool.pytest.ini_options]
82
87
  testpaths = ["tests"]
83
88
  markers = [
84
- "integration: hits live services; skipped unless --run-integration is passed",
89
+ "l0: pure in-memory logic",
90
+ "l1: in-process components with controlled transport",
91
+ "l2: local functional tests with filesystem or real decoders",
92
+ "l3: controlled service deployment (currently unused)",
93
+ "l4: production or upstream-live compatibility",
94
+ "live: hits upstream services; requires --run-live",
95
+ "integration: compatibility alias for live",
96
+ "pandas: optional CSV reader coverage",
97
+ "radar: optional radar reader coverage",
98
+ "netcdf: optional NetCDF reader coverage",
85
99
  ]
86
- addopts = "-ra"
100
+ addopts = "-ra --strict-markers"
@@ -10,10 +10,11 @@ try:
10
10
  except PackageNotFoundError: # running from a source tree without an install
11
11
  __version__ = "0.0.0"
12
12
 
13
- from usdata.models import Asset, BBox, Dataset, Provenance, Query, TimeRange
13
+ from usdata.models import Asset, BBox, Dataset, Provenance, Query, TemporalSelection, TimeRange
14
14
  from usdata.pull import pull, verify
15
15
  from usdata.query import build_query
16
16
  from usdata.registry import DatasetNotFound, Registry, SearchResult, default_registry
17
+ from usdata.selection import select_by_time
17
18
 
18
19
  __all__ = [
19
20
  "Asset",
@@ -24,6 +25,7 @@ __all__ = [
24
25
  "Query",
25
26
  "Registry",
26
27
  "SearchResult",
28
+ "TemporalSelection",
27
29
  "TimeRange",
28
30
  "__version__",
29
31
  "build_query",
@@ -31,6 +33,7 @@ __all__ = [
31
33
  "get",
32
34
  "pull",
33
35
  "search",
36
+ "select_by_time",
34
37
  "verify",
35
38
  ]
36
39
 
@@ -0,0 +1,123 @@
1
+ """Local NEXRAD Level II decoding behind the radar extra."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import bz2
6
+ import gzip
7
+ from importlib import import_module
8
+ from typing import TYPE_CHECKING, Any
9
+
10
+ from usdata.readers import MissingReaderDependency, RadarDecodeError
11
+
12
+ # NOAA RDA/RPG ICD 2620002Y, Table XVII-I notes 21 and 30.
13
+ MOMENT_FLAG_COUNTS = {
14
+ "DBZH": 2,
15
+ "VRADH": 2,
16
+ "WRADH": 2,
17
+ "ZDR": 2,
18
+ "PHIDP": 2,
19
+ "RHOHV": 2,
20
+ "CCORH": 8,
21
+ }
22
+
23
+ if TYPE_CHECKING:
24
+ from usdata.fetch import FetchedAsset
25
+
26
+
27
+ def _check_sweeps(content: bytes, sweep: int | list[int] | None) -> None:
28
+ """Reject decoder tables that pair a sweep with another sweep's coordinates.
29
+
30
+ xradar 0.12 can omit an interior sweep without an end marker from its data
31
+ table while keeping its moment metadata. The coordinate list then shifts.
32
+ Compare record identities, not just ray counts: equal-length sweeps can
33
+ otherwise silently acquire another sweep's coordinates and timestamps.
34
+ """
35
+ backend = import_module("xradar.io.backends.nexrad_level2")
36
+ with backend.NEXRADLevel2File(content, loaddata=False) as volume:
37
+ # Parsing completeness also populates the per-sweep data tables.
38
+ _ = volume.incomplete_sweeps
39
+ moments = volume.msg_31_data_header
40
+ coordinates = volume.msg_31_header
41
+ requested = (
42
+ list(range(len(moments)))
43
+ if sweep is None
44
+ else (sweep if isinstance(sweep, list) else [sweep])
45
+ )
46
+ for index in requested:
47
+ if index >= len(moments):
48
+ raise ValueError(f"sweep {index} is outside this volume's {len(moments)} sweeps")
49
+ data = volume.data.get(index)
50
+ rays = coordinates[index] if index < len(coordinates) else []
51
+ # Non-radial messages may occur within a sweep or after its last
52
+ # received ray. Follow the decoder's traversal, excluding those
53
+ # records, instead of requiring record_end to be a radial message.
54
+ expected_records = []
55
+ if data is not None:
56
+ intermediate = {record["record_number"] for record in data["intermediate_records"]}
57
+ expected_records = [
58
+ record
59
+ for record in range(data["record_number"], data["record_end"] + 1)
60
+ if record not in intermediate
61
+ ]
62
+ if (
63
+ data is None
64
+ or not rays
65
+ or moments[index]["record_number"] != data["record_number"]
66
+ or [ray["record_number"] for ray in rays] != expected_records
67
+ ):
68
+ raise RadarDecodeError(
69
+ f"cannot safely decode sweep {index}: NEXRAD moment and coordinate records "
70
+ "do not agree; select an unaffected sweep explicitly with open(sweep=...) "
71
+ "or use another decoder. No sweeps were silently dropped."
72
+ )
73
+
74
+
75
+ def open_nexrad(fetched: FetchedAsset, *, sweep: int | list[int] | None = None) -> Any:
76
+ """Decode a local volume into a fully loaded xarray DataTree."""
77
+ try:
78
+ xradar = import_module("xradar")
79
+ except ModuleNotFoundError as error:
80
+ if error.name != "xradar":
81
+ raise
82
+ raise MissingReaderDependency(
83
+ 'NEXRAD reading requires xradar; install it with: pip install "usdata[radar]" '
84
+ '(or uv add "usdata[radar]")'
85
+ ) from error
86
+
87
+ # Bytes prevent remote URL interpretation and work across the backend's
88
+ # repeated sweep reads. Compressed source files stay unchanged in the cache.
89
+ content = fetched.path.read_bytes()
90
+ if content.startswith(b"\x1f\x8b"):
91
+ content = gzip.decompress(content)
92
+ elif content.startswith(b"BZh"):
93
+ content = bz2.decompress(content)
94
+ _check_sweeps(content, sweep)
95
+ radar = xradar.io.open_nexradlevel2_datatree(content, sweep=sweep, incomplete_sweep="pad")
96
+ try:
97
+ radar.load()
98
+ finally:
99
+ radar.close()
100
+ for node in radar.subtree:
101
+ for name, variable in node.ds.variables.items():
102
+ # The backend records the entire input byte string as `source`.
103
+ # Provenance below is the durable reference, not that decoder buffer.
104
+ variable.encoding.pop("source", None)
105
+ if name in MOMENT_FLAG_COUNTS and "range" in variable.dims:
106
+ scale = variable.encoding.get("scale_factor")
107
+ offset = variable.encoding.get("add_offset")
108
+ if scale is not None and offset is not None:
109
+ # xradar 0.12 does not supply _FillValue for NEXRAD flags.
110
+ # Compare using each moment's native scale, not fixed units.
111
+ data = node[name]
112
+ valid = data.notnull()
113
+ for code in range(MOMENT_FLAG_COUNTS[name]):
114
+ valid = valid & (data != offset + code * scale)
115
+ masked = data.where(valid)
116
+ masked.encoding = data.encoding.copy()
117
+ node[name] = masked
118
+ radar.attrs["usdata"] = {
119
+ "asset_id": fetched.asset.id,
120
+ "provenance": fetched.provenance.model_dump(mode="json"),
121
+ "sweeps": [name.lstrip("/") for name in radar.groups if name.startswith("/sweep_")],
122
+ }
123
+ return radar
@@ -34,6 +34,7 @@ class FetchedAsset(BaseModel):
34
34
  parse_dates: list[str] | None = None,
35
35
  usecols: list[str] | None = None,
36
36
  nrows: int | None = None,
37
+ sweep: int | list[int] | None = None,
37
38
  ) -> Any:
38
39
  """Open local data with an optional ``pandas``, ``radar``, or ``netcdf`` reader.
39
40
 
@@ -41,12 +42,19 @@ class FetchedAsset(BaseModel):
41
42
  in ``frame.attrs["usdata"]``. NEXRAD returns a xarray DataTree with provenance
42
43
  in ``radar.attrs["usdata"]``. NetCDF4 returns a loaded xarray Dataset with
43
44
  matching provenance in its attributes. See ``usdata.readers.open_asset`` for options.
45
+ Use ``sweep=0`` or ``sweep=[0, 2]`` to load selected zero-based radar sweeps.
44
46
  Cached files and provenance sidecars are never changed.
45
47
  """
46
48
  from usdata.readers import open_asset
47
49
 
48
50
  return open_asset(
49
- self, reader=reader, dtype=dtype, parse_dates=parse_dates, usecols=usecols, nrows=nrows
51
+ self,
52
+ reader=reader,
53
+ dtype=dtype,
54
+ parse_dates=parse_dates,
55
+ usecols=usecols,
56
+ nrows=nrows,
57
+ sweep=sweep,
50
58
  )
51
59
 
52
60
 
@@ -9,9 +9,9 @@ from __future__ import annotations
9
9
 
10
10
  import math
11
11
  import re
12
- from datetime import datetime
12
+ from datetime import datetime, timedelta
13
13
  from enum import StrEnum
14
- from typing import Any
14
+ from typing import Any, Literal
15
15
 
16
16
  from pydantic import BaseModel, ConfigDict, Field, model_validator
17
17
 
@@ -79,6 +79,12 @@ class BBox(BaseModel):
79
79
  @classmethod
80
80
  def from_point(cls, lat: float, lon: float, radius_km: float = 0.0) -> BBox:
81
81
  """Box around a point. Uses a flat-earth approximation, fine for small radii."""
82
+ if not math.isfinite(lat) or not -90 <= lat <= 90:
83
+ raise ValueError("lat must be finite and between -90 and 90")
84
+ if not math.isfinite(lon) or not -180 <= lon <= 180:
85
+ raise ValueError("lon must be finite and between -180 and 180")
86
+ if not math.isfinite(radius_km) or radius_km < 0:
87
+ raise ValueError("radius_km must be finite and nonnegative")
82
88
  dlat = radius_km / 111.0
83
89
  dlon = radius_km / (111.0 * max(math.cos(math.radians(lat)), 1e-6))
84
90
  return cls(
@@ -223,6 +229,22 @@ class Asset(BaseModel):
223
229
  bbox: BBox | None = None
224
230
 
225
231
 
232
+ class TemporalSelection(BaseModel):
233
+ """A start-time selection and its explicit policy; not source provenance.
234
+
235
+ No match has ``asset=None``, ``offset_seconds=None``, and zero eligible
236
+ candidates. Counts refer to the supplied candidates, not a remote catalog.
237
+ """
238
+
239
+ target: datetime
240
+ tolerance: timedelta
241
+ direction: Literal["nearest", "at_or_before"]
242
+ asset: Asset | None
243
+ offset_seconds: float | None
244
+ candidate_count: int = Field(ge=0)
245
+ eligible_count: int = Field(ge=0)
246
+
247
+
226
248
  class Provenance(BaseModel):
227
249
  """Everything needed to say where a local file came from and re-fetch it."""
228
250
 
@@ -57,11 +57,24 @@ def list_objects(
57
57
  own = client is None
58
58
  client = client or http.client()
59
59
  params: dict[str, str | int] = {"list-type": 2, "prefix": prefix, "max-keys": page_size}
60
+ seen_tokens: set[str] = set()
60
61
  try:
61
62
  while True:
62
63
  resp = http.get(https_url(bucket), client, params=params)
63
64
  resp.raise_for_status()
64
65
  root = ET.fromstring(resp.text)
66
+ truncated = _text(root, "IsTruncated") == "true"
67
+ token = _text(root, "NextContinuationToken")
68
+ if truncated:
69
+ if not token or not token.strip():
70
+ raise httpx.RemoteProtocolError(
71
+ "Truncated S3 listing has no continuation token", request=resp.request
72
+ )
73
+ if token in seen_tokens:
74
+ raise httpx.RemoteProtocolError(
75
+ "S3 listing repeated a continuation token", request=resp.request
76
+ )
77
+ seen_tokens.add(token)
65
78
  for contents in root.iter(NS + "Contents"):
66
79
  key = _text(contents, "Key")
67
80
  if key is None:
@@ -74,9 +87,9 @@ def list_objects(
74
87
  etag=etag.strip('"') if etag else None,
75
88
  last_modified=datetime.fromisoformat(modified) if modified else None,
76
89
  )
77
- token = _text(root, "NextContinuationToken")
78
- if _text(root, "IsTruncated") != "true" or not token:
90
+ if not truncated:
79
91
  return
92
+ assert token is not None
80
93
  params["continuation-token"] = token
81
94
  finally:
82
95
  if own:
@@ -0,0 +1,27 @@
1
+ """Internal lifecycle shared by adapters using an injected or owned HTTP client."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import httpx
6
+
7
+ from usdata.models import Dataset
8
+ from usdata.protocols import http
9
+ from usdata.providers.base import Provider
10
+
11
+
12
+ class _HttpProvider(Provider):
13
+ def __init__(self, dataset: Dataset, client: httpx.Client | None = None) -> None:
14
+ super().__init__(dataset)
15
+ self._client = client
16
+ self._owns_client = client is None
17
+
18
+ def _http(self) -> httpx.Client:
19
+ if self._client is None:
20
+ self._client = http.client()
21
+ return self._client
22
+
23
+ def close(self) -> None:
24
+ """Close owned connections; injected clients remain the caller's responsibility."""
25
+ if self._owns_client and self._client is not None:
26
+ self._client.close()
27
+ self._client = None
@@ -18,9 +18,10 @@ from typing import cast
18
18
 
19
19
  import httpx
20
20
 
21
- from usdata.models import Asset, BBox, Dataset, Protocol, Query, TimeRange
21
+ from usdata.models import Asset, BBox, Protocol, Query, TimeRange
22
22
  from usdata.protocols import erddap, http
23
- from usdata.providers.base import Provider, QueryError
23
+ from usdata.providers._http import _HttpProvider
24
+ from usdata.providers.base import QueryError
24
25
 
25
26
  BASE = "https://coastwatch.noaa.gov/erddap"
26
27
  DATASET = "noaacwBLENDEDsstDNDaily"
@@ -42,25 +43,9 @@ def _spatial_slice(
42
43
  ), length
43
44
 
44
45
 
45
- class CoastwatchSst(Provider):
46
+ class CoastwatchSst(_HttpProvider):
46
47
  """NOAA's 0.05-degree day/night analysis, including units and grid coordinates."""
47
48
 
48
- def __init__(self, dataset: Dataset, client: httpx.Client | None = None) -> None:
49
- super().__init__(dataset)
50
- self._client = client
51
- self._owns_client = client is None
52
-
53
- def _http(self) -> httpx.Client:
54
- if self._client is None:
55
- self._client = http.client()
56
- return self._client
57
-
58
- def close(self) -> None:
59
- """Release owned connections; injected clients remain the caller's responsibility."""
60
- if self._owns_client and self._client is not None:
61
- self._client.close()
62
- self._client = None
63
-
64
49
  def list_assets(self, query: Query) -> list[Asset]:
65
50
  """Resolve a valid grid intersection into one stable, bounded CSV request."""
66
51
  if (
@@ -18,9 +18,10 @@ from typing import Any
18
18
 
19
19
  import httpx
20
20
 
21
- from usdata.models import Asset, Dataset, Protocol, Query, TimeRange
21
+ from usdata.models import Asset, Protocol, Query, TimeRange
22
22
  from usdata.protocols import http
23
- from usdata.providers.base import Provider, QueryError
23
+ from usdata.providers._http import _HttpProvider
24
+ from usdata.providers.base import QueryError
24
25
 
25
26
  SEARCH_URL = "https://www.ncei.noaa.gov/access/services/search/v1/data"
26
27
  DATA_URL = "https://www.ncei.noaa.gov/access/services/data/v1"
@@ -45,27 +46,11 @@ def _stations_param(raw: Any) -> list[str]:
45
46
  return stations
46
47
 
47
48
 
48
- class GhcnDaily(Provider):
49
+ class GhcnDaily(_HttpProvider):
49
50
  """GHCN-Daily adapter. Params: ``stations`` (list or comma string), ``units``."""
50
51
 
51
52
  ncei_dataset = NCEI_DATASET
52
53
 
53
- def __init__(self, dataset: Dataset, client: httpx.Client | None = None) -> None:
54
- super().__init__(dataset)
55
- self._client = client
56
- self._owns_client = client is None
57
-
58
- def close(self) -> None:
59
- """Close an internally created HTTP client; injected clients belong to the caller."""
60
- if self._owns_client and self._client is not None:
61
- self._client.close()
62
- self._client = None
63
-
64
- def _http(self) -> httpx.Client:
65
- if self._client is None:
66
- self._client = http.client()
67
- return self._client
68
-
69
54
  def find_stations(self, query: Query) -> list[str]:
70
55
  """Station ids with data inside the query's bbox and time range."""
71
56
  if query.bbox is None or query.time is None: