usdata 0.6.0__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. {usdata-0.6.0 → usdata-0.8.0}/PKG-INFO +63 -8
  2. {usdata-0.6.0 → usdata-0.8.0}/README.md +57 -7
  3. {usdata-0.6.0 → usdata-0.8.0}/pyproject.toml +13 -1
  4. {usdata-0.6.0 → usdata-0.8.0}/pyproject.toml.orig +10 -1
  5. usdata-0.8.0/src/usdata/_netcdf.py +36 -0
  6. usdata-0.8.0/src/usdata/_progress.py +62 -0
  7. usdata-0.8.0/src/usdata/_radar.py +73 -0
  8. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/cli/app.py +10 -2
  9. usdata-0.8.0/src/usdata/cli/progress.py +89 -0
  10. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/data/registry.yaml +29 -22
  11. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/fetch.py +9 -3
  12. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/protocols/http.py +18 -1
  13. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/providers/noaa/ghcnd.py +25 -4
  14. usdata-0.8.0/src/usdata/providers/noaa/goes.py +126 -0
  15. usdata-0.8.0/src/usdata/providers/noaa/gsom.py +49 -0
  16. usdata-0.8.0/src/usdata/providers/noaa/storm_events.py +134 -0
  17. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/pull.py +9 -1
  18. usdata-0.8.0/src/usdata/readers.py +141 -0
  19. usdata-0.6.0/src/usdata/readers.py +0 -103
  20. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/__init__.py +0 -0
  21. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/_files.py +0 -0
  22. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/cache.py +0 -0
  23. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/cli/__init__.py +0 -0
  24. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/data/nexrad_sites.csv +0 -0
  25. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/data/places.csv +0 -0
  26. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/data/places.sources.json +0 -0
  27. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/manifest.py +0 -0
  28. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/models.py +0 -0
  29. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/protocols/__init__.py +0 -0
  30. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/protocols/erddap.py +0 -0
  31. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/protocols/s3.py +0 -0
  32. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/provenance.py +0 -0
  33. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/providers/__init__.py +0 -0
  34. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/providers/base.py +0 -0
  35. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/providers/noaa/__init__.py +0 -0
  36. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
  37. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/providers/noaa/nexrad.py +0 -0
  38. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/providers/noaa/sites.py +0 -0
  39. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/providers/usgs/__init__.py +0 -0
  40. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/providers/usgs/daily.py +0 -0
  41. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/py.typed +0 -0
  42. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/query.py +0 -0
  43. {usdata-0.6.0 → usdata-0.8.0}/src/usdata/registry.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: usdata
3
- Version: 0.6.0
3
+ Version: 0.8.0
4
4
  Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
5
5
  Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
6
6
  Author: Jake Van Slyke
@@ -14,11 +14,16 @@ Requires-Dist: httpx>=0.28.1
14
14
  Requires-Dist: pydantic>=2.7
15
15
  Requires-Dist: pyyaml>=6.0
16
16
  Requires-Dist: typer>=0.12
17
+ Requires-Dist: xarray>=2025.1 ; extra == 'netcdf'
18
+ Requires-Dist: h5netcdf[h5py]>=1.8.1 ; extra == 'netcdf'
17
19
  Requires-Dist: pandas>=3.0 ; extra == 'pandas'
20
+ Requires-Dist: xradar>=0.12.0 ; extra == 'radar'
18
21
  Requires-Python: >=3.11
19
22
  Project-URL: Homepage, https://github.com/jakeryderv/usdata
20
23
  Project-URL: Repository, https://github.com/jakeryderv/usdata
24
+ Provides-Extra: netcdf
21
25
  Provides-Extra: pandas
26
+ Provides-Extra: radar
22
27
  Description-Content-Type: text/markdown
23
28
 
24
29
  # usdata
@@ -26,9 +31,11 @@ Description-Content-Type: text/markdown
26
31
  Unified Python SDK and CLI for discovering, fetching, and tracking the
27
32
  provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
28
33
 
29
- > Status: pre-alpha. v0.6 supports GHCN-Daily, NEXRAD Level II, USGS daily
30
- > values, and CoastWatch SST subsets with provenance, plus Census state/county
31
- > lookup and optional pandas CSV readers. Other datasets are planned.
34
+ > Status: pre-alpha. v0.8 supports GHCN-Daily, GSOM monthly summaries,
35
+ > NEXRAD Level II, GOES ABI CONUS imagery, Storm Events annual archives,
36
+ > USGS daily values, and CoastWatch SST subsets with provenance. It includes
37
+ > optional CSV, radar, and NetCDF4 readers, six executed notebooks, Census
38
+ > state/county lookup, and terminal download progress. Other datasets are planned.
32
39
  > See [docs/roadmap.md](docs/roadmap.md).
33
40
 
34
41
  ## Providers
@@ -36,7 +43,7 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
36
43
  <!-- registry:start -->
37
44
  | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
38
45
  |---|---:|---:|---:|---|---|
39
- | [NOAA](docs/providers/noaa.md) | 3 | 0 | 26 | — | `ghcn-daily`, `nexrad-level2`, `coastwatch-sst`, +26 planned |
46
+ | [NOAA](docs/providers/noaa.md) | 6 | 0 | 23 | — | `ghcn-daily`, `gsom`, `storm-events`, `nexrad-level2`, `goes-abi`, `coastwatch-sst`, +23 planned |
40
47
  | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
41
48
  | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
42
49
  | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
@@ -93,6 +100,17 @@ usdata pull dataset.yaml # resolve, fetch, write dataset.lock.json
93
100
  usdata verify dataset.yaml # exit 1 if any cached input drifted
94
101
  ```
95
102
 
103
+ Storm Events bulk access is available since v0.8. Dates select complete
104
+ annual details archives; filter rows locally after opening the gzip CSV. For example:
105
+
106
+ ```sh
107
+ usdata fetch noaa:storm-events --start 2024-05-01 --end 2024-05-31 --dry-run
108
+ ```
109
+
110
+ This lists the entire 2024 archive. Location and variable filters are rejected;
111
+ see the [executed Storm Events notebook](examples/storm-events/example.ipynb)
112
+ for local filtering and reporting limitations.
113
+
96
114
  Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
97
115
  `USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
98
116
  recording source URL, retrieval time, checksum, size, and license.
@@ -132,6 +150,25 @@ sources:
132
150
  end: 2024-05-31
133
151
  ```
134
152
 
153
+ Terminal progress is available since v0.7. On a terminal, `fetch` and
154
+ `pull` show progress on stderr: resolved asset counts,
155
+ known bytes and unknown sizes, HTTP download bytes for the current attempt, and
156
+ validated cache hits. `fetch --dry-run` also summarizes known sizes. Asset totals
157
+ include possible cache hits; each manifest source is resolved separately. Bytes
158
+ from a failed HTTP attempt reset on retry; encoded responses have unknown decoded
159
+ size. Adapters that assemble files from metadata requests show asset-level progress.
160
+ Use `--no-progress` to disable it. Progress is automatically disabled when either
161
+ stdout or stderr is redirected; existing output lines and exit codes are unchanged.
162
+
163
+ For single-channel GOES CONUS imagery (available since v0.8):
164
+
165
+ ```sh
166
+ usdata fetch noaa:goes-abi --start 2024-05-06T12:01:18.1Z --end 2024-05-06T12:01:18.1Z -p satellite=18 -p channel=6
167
+ ```
168
+
169
+ The download is a whole NetCDF scene. See [GOES access notes](docs/providers/noaa.md#goes-abi-conus-imagery)
170
+ for supported selectors and scan-start time semantics.
171
+
135
172
  ## Opening CSV data
136
173
 
137
174
  `FetchedAsset.open()` is available since v0.6 with the optional pandas
@@ -140,16 +177,32 @@ preserves identifier strings, and keeps CoastWatch units as metadata.
140
177
  See the [reader reference](docs/reference/readers.md)
141
178
  and [fetch → open → analyze example](examples/sst-analysis/README.md).
142
179
 
180
+ The [examples directory](examples/README.md) contains executed Jupyter notebooks
181
+ with saved data previews, small plots, and source provenance. Start with weather
182
+ and streamflow for manifest workflows, SST for gridded CSV reading, or monthly
183
+ climate for GSOM observations.
184
+
185
+ NetCDF4 scene opening is available since v0.8 with `usdata[netcdf]`.
186
+ See the executed [GOES infrared notebook](examples/goes-imagery/example.ipynb).
187
+
143
188
  ## Development
144
189
 
145
190
  Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
146
191
 
192
+ `just setup` uses the tested Python 3.14.7 pin in `.python-version`. Older Linux
193
+ uv Python 3.14 builds can crash during NumPy array operations; see
194
+ [the upstream fix](https://github.com/astral-sh/python-build-standalone/issues/991).
195
+
147
196
  ```sh
148
197
  git clone https://github.com/jakeryderv/usdata && cd usdata
149
198
  just setup # install toolchain and dependencies
150
199
  just test # unit tests
151
200
  just check # format, lint, typecheck, offline tests, generated docs, release notices
152
201
  just check-pandas # install the CSV extra and run the same checks
202
+ just check-radar # install the radar extra and run the same checks
203
+ just check-netcdf # install the NetCDF4 extra and run the same checks
204
+ just notebooks # launch the optional Jupyter examples environment
205
+ just run-notebooks # execute notebooks live in fresh kernels and temporary caches
153
206
  just build # build wheel and sdist
154
207
  just smoke # exercise core and pandas wheel installations outside the checkout
155
208
  just run search radar
@@ -157,9 +210,11 @@ just run search radar
157
210
 
158
211
  Unit tests mechanically block network connections. Integration tests that hit
159
212
  live services run with `just test-integration`. CI checks Python 3.11 and 3.14 on
160
- Linux, both with and without pandas, and smoke-tests both installed-wheel
161
- profiles on Linux, macOS, and Windows. The full unit and live-service suites currently run on Linux. `just setup` restores
162
- a core-only development environment; `just check-pandas` installs the extra.
213
+ Linux with core-only, pandas, radar, and NetCDF dependency profiles. Installed-wheel
214
+ checks cover all four profiles on Linux, macOS, and Windows. The full unit and
215
+ live-service suites run on Linux. `just setup` restores a core-only development
216
+ environment; the `check-pandas`, `check-radar`, and `check-netcdf` commands install
217
+ their respective extras.
163
218
 
164
219
  Releases: `just release minor` opens a version-bump PR; merging it publishes
165
220
  to PyPI and creates the tag and GitHub release. See
@@ -3,9 +3,11 @@
3
3
  Unified Python SDK and CLI for discovering, fetching, and tracking the
4
4
  provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
5
5
 
6
- > Status: pre-alpha. v0.6 supports GHCN-Daily, NEXRAD Level II, USGS daily
7
- > values, and CoastWatch SST subsets with provenance, plus Census state/county
8
- > lookup and optional pandas CSV readers. Other datasets are planned.
6
+ > Status: pre-alpha. v0.8 supports GHCN-Daily, GSOM monthly summaries,
7
+ > NEXRAD Level II, GOES ABI CONUS imagery, Storm Events annual archives,
8
+ > USGS daily values, and CoastWatch SST subsets with provenance. It includes
9
+ > optional CSV, radar, and NetCDF4 readers, six executed notebooks, Census
10
+ > state/county lookup, and terminal download progress. Other datasets are planned.
9
11
  > See [docs/roadmap.md](docs/roadmap.md).
10
12
 
11
13
  ## Providers
@@ -13,7 +15,7 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
13
15
  <!-- registry:start -->
14
16
  | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
15
17
  |---|---:|---:|---:|---|---|
16
- | [NOAA](docs/providers/noaa.md) | 3 | 0 | 26 | — | `ghcn-daily`, `nexrad-level2`, `coastwatch-sst`, +26 planned |
18
+ | [NOAA](docs/providers/noaa.md) | 6 | 0 | 23 | — | `ghcn-daily`, `gsom`, `storm-events`, `nexrad-level2`, `goes-abi`, `coastwatch-sst`, +23 planned |
17
19
  | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
18
20
  | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
19
21
  | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
@@ -70,6 +72,17 @@ usdata pull dataset.yaml # resolve, fetch, write dataset.lock.json
70
72
  usdata verify dataset.yaml # exit 1 if any cached input drifted
71
73
  ```
72
74
 
75
+ Storm Events bulk access is available since v0.8. Dates select complete
76
+ annual details archives; filter rows locally after opening the gzip CSV. For example:
77
+
78
+ ```sh
79
+ usdata fetch noaa:storm-events --start 2024-05-01 --end 2024-05-31 --dry-run
80
+ ```
81
+
82
+ This lists the entire 2024 archive. Location and variable filters are rejected;
83
+ see the [executed Storm Events notebook](examples/storm-events/example.ipynb)
84
+ for local filtering and reporting limitations.
85
+
73
86
  Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
74
87
  `USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
75
88
  recording source URL, retrieval time, checksum, size, and license.
@@ -109,6 +122,25 @@ sources:
109
122
  end: 2024-05-31
110
123
  ```
111
124
 
125
+ Terminal progress is available since v0.7. On a terminal, `fetch` and
126
+ `pull` show progress on stderr: resolved asset counts,
127
+ known bytes and unknown sizes, HTTP download bytes for the current attempt, and
128
+ validated cache hits. `fetch --dry-run` also summarizes known sizes. Asset totals
129
+ include possible cache hits; each manifest source is resolved separately. Bytes
130
+ from a failed HTTP attempt reset on retry; encoded responses have unknown decoded
131
+ size. Adapters that assemble files from metadata requests show asset-level progress.
132
+ Use `--no-progress` to disable it. Progress is automatically disabled when either
133
+ stdout or stderr is redirected; existing output lines and exit codes are unchanged.
134
+
135
+ For single-channel GOES CONUS imagery (available since v0.8):
136
+
137
+ ```sh
138
+ usdata fetch noaa:goes-abi --start 2024-05-06T12:01:18.1Z --end 2024-05-06T12:01:18.1Z -p satellite=18 -p channel=6
139
+ ```
140
+
141
+ The download is a whole NetCDF scene. See [GOES access notes](docs/providers/noaa.md#goes-abi-conus-imagery)
142
+ for supported selectors and scan-start time semantics.
143
+
112
144
  ## Opening CSV data
113
145
 
114
146
  `FetchedAsset.open()` is available since v0.6 with the optional pandas
@@ -117,16 +149,32 @@ preserves identifier strings, and keeps CoastWatch units as metadata.
117
149
  See the [reader reference](docs/reference/readers.md)
118
150
  and [fetch → open → analyze example](examples/sst-analysis/README.md).
119
151
 
152
+ The [examples directory](examples/README.md) contains executed Jupyter notebooks
153
+ with saved data previews, small plots, and source provenance. Start with weather
154
+ and streamflow for manifest workflows, SST for gridded CSV reading, or monthly
155
+ climate for GSOM observations.
156
+
157
+ NetCDF4 scene opening is available since v0.8 with `usdata[netcdf]`.
158
+ See the executed [GOES infrared notebook](examples/goes-imagery/example.ipynb).
159
+
120
160
  ## Development
121
161
 
122
162
  Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
123
163
 
164
+ `just setup` uses the tested Python 3.14.7 pin in `.python-version`. Older Linux
165
+ uv Python 3.14 builds can crash during NumPy array operations; see
166
+ [the upstream fix](https://github.com/astral-sh/python-build-standalone/issues/991).
167
+
124
168
  ```sh
125
169
  git clone https://github.com/jakeryderv/usdata && cd usdata
126
170
  just setup # install toolchain and dependencies
127
171
  just test # unit tests
128
172
  just check # format, lint, typecheck, offline tests, generated docs, release notices
129
173
  just check-pandas # install the CSV extra and run the same checks
174
+ just check-radar # install the radar extra and run the same checks
175
+ just check-netcdf # install the NetCDF4 extra and run the same checks
176
+ just notebooks # launch the optional Jupyter examples environment
177
+ just run-notebooks # execute notebooks live in fresh kernels and temporary caches
130
178
  just build # build wheel and sdist
131
179
  just smoke # exercise core and pandas wheel installations outside the checkout
132
180
  just run search radar
@@ -134,9 +182,11 @@ just run search radar
134
182
 
135
183
  Unit tests mechanically block network connections. Integration tests that hit
136
184
  live services run with `just test-integration`. CI checks Python 3.11 and 3.14 on
137
- Linux, both with and without pandas, and smoke-tests both installed-wheel
138
- profiles on Linux, macOS, and Windows. The full unit and live-service suites currently run on Linux. `just setup` restores
139
- a core-only development environment; `just check-pandas` installs the extra.
185
+ Linux with core-only, pandas, radar, and NetCDF dependency profiles. Installed-wheel
186
+ checks cover all four profiles on Linux, macOS, and Windows. The full unit and
187
+ live-service suites run on Linux. `just setup` restores a core-only development
188
+ environment; the `check-pandas`, `check-radar`, and `check-netcdf` commands install
189
+ their respective extras.
140
190
 
141
191
  Releases: `just release minor` opens a version-bump PR; merging it publishes
142
192
  to PyPI and creates the tag and GitHub release. See
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.6.0"
3
+ version = "0.8.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -32,6 +32,11 @@ email = "jakervanslyke@gmail.com"
32
32
 
33
33
  [project.optional-dependencies]
34
34
  pandas = ["pandas>=3.0"]
35
+ radar = ["xradar>=0.12.0"]
36
+ netcdf = [
37
+ "xarray>=2025.1",
38
+ "h5netcdf[h5py]>=1.8.1",
39
+ ]
35
40
 
36
41
  [project.urls]
37
42
  Homepage = "https://github.com/jakeryderv/usdata"
@@ -48,6 +53,13 @@ dev = [
48
53
  "respx>=0.23.1",
49
54
  "ruff>=0.6",
50
55
  ]
56
+ examples = [
57
+ "ipykernel>=6.29",
58
+ "jupyterlab>=4.3",
59
+ "matplotlib>=3.9",
60
+ "nbclient>=0.10",
61
+ "pandas>=3.0",
62
+ ]
51
63
 
52
64
  [build-system]
53
65
  requires = ["uv_build>=0.12.5,<0.13.0"]
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.6.0"
3
+ version = "0.8.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -24,6 +24,8 @@ dependencies = [
24
24
 
25
25
  [project.optional-dependencies]
26
26
  pandas = ["pandas>=3.0"]
27
+ radar = ["xradar>=0.12.0"]
28
+ netcdf = ["xarray>=2025.1", "h5netcdf[h5py]>=1.8.1"]
27
29
 
28
30
  [project.urls]
29
31
  Homepage = "https://github.com/jakeryderv/usdata"
@@ -40,6 +42,13 @@ dev = [
40
42
  "respx>=0.23.1",
41
43
  "ruff>=0.6",
42
44
  ]
45
+ examples = [
46
+ "ipykernel>=6.29",
47
+ "jupyterlab>=4.3",
48
+ "matplotlib>=3.9",
49
+ "nbclient>=0.10",
50
+ "pandas>=3.0",
51
+ ]
43
52
 
44
53
  [build-system]
45
54
  requires = ["uv_build>=0.12.5,<0.13.0"]
@@ -0,0 +1,36 @@
1
+ """Local, eagerly loaded NetCDF4 reading behind the netcdf extra."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from importlib import import_module
6
+ from typing import TYPE_CHECKING, Any
7
+
8
+ from usdata.readers import MissingReaderDependency
9
+
10
+ if TYPE_CHECKING:
11
+ from usdata.fetch import FetchedAsset
12
+
13
+
14
+ def open_netcdf(fetched: FetchedAsset) -> Any:
15
+ """Load a NetCDF4 root Dataset, then close every source file handle."""
16
+ try:
17
+ xarray = import_module("xarray")
18
+ import_module("h5netcdf")
19
+ import_module("h5py")
20
+ except ModuleNotFoundError as error:
21
+ if error.name not in {"xarray", "h5netcdf", "h5py"}:
22
+ raise
23
+ raise MissingReaderDependency(
24
+ 'NetCDF4 reading requires xarray and h5netcdf; install: pip install "usdata[netcdf]"'
25
+ ) from error
26
+ # A local file object and fixed engine prevent interpretation as an OPeNDAP URL.
27
+ with (
28
+ fetched.path.open("rb") as stream,
29
+ xarray.open_dataset(stream, engine="h5netcdf", chunks=None) as dataset,
30
+ ):
31
+ dataset.load()
32
+ dataset.attrs["usdata"] = {
33
+ "asset_id": fetched.asset.id,
34
+ "provenance": fetched.provenance.model_dump(mode="json"),
35
+ }
36
+ return dataset
@@ -0,0 +1,62 @@
1
+ """Internal synchronous progress events, scoped to one CLI operation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable, Iterator, Sequence
6
+ from contextlib import contextmanager
7
+ from contextvars import ContextVar
8
+ from dataclasses import dataclass
9
+ from typing import Literal
10
+
11
+
12
+ @dataclass(frozen=True)
13
+ class Batch:
14
+ """A resolved group; sizes describe assets, including possible cache hits."""
15
+
16
+ count: int
17
+ known_bytes: int
18
+ unknown_sizes: int
19
+
20
+
21
+ @dataclass(frozen=True)
22
+ class AssetProgress:
23
+ """Start or validated completion of an asset."""
24
+
25
+ asset_id: str
26
+ state: Literal["start", "cached", "fetched"]
27
+ size: int | None
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class TransferProgress:
32
+ """Bytes written in the current HTTP attempt, reset on every retry."""
33
+
34
+ completed: int
35
+ total: int | None
36
+ attempt: int
37
+
38
+
39
+ Event = Batch | AssetProgress | TransferProgress
40
+ _observer: ContextVar[Callable[[Event], None] | None] = ContextVar("progress", default=None)
41
+
42
+
43
+ def emit(event: Event) -> None:
44
+ """Notify the active observer without importing CLI code."""
45
+ observer = _observer.get()
46
+ if observer is not None:
47
+ observer(event)
48
+
49
+
50
+ def batch(sizes: Sequence[int | None]) -> None:
51
+ """Report known and unknown sizes separately; never guess a total."""
52
+ emit(Batch(len(sizes), sum(size for size in sizes if size is not None), sizes.count(None)))
53
+
54
+
55
+ @contextmanager
56
+ def observe(callback: Callable[[Event], None]) -> Iterator[None]:
57
+ """Observe an operation and restore the previous observer on every exit."""
58
+ token = _observer.set(callback)
59
+ try:
60
+ yield
61
+ finally:
62
+ _observer.reset(token)
@@ -0,0 +1,73 @@
1
+ """Local NEXRAD Level II decoding behind the radar extra."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import bz2
6
+ import gzip
7
+ from importlib import import_module
8
+ from typing import TYPE_CHECKING, Any
9
+
10
+ from usdata.readers import MissingReaderDependency
11
+
12
+ # NOAA RDA/RPG ICD 2620002Y, Table XVII-I notes 21 and 30.
13
+ MOMENT_FLAG_COUNTS = {
14
+ "DBZH": 2,
15
+ "VRADH": 2,
16
+ "WRADH": 2,
17
+ "ZDR": 2,
18
+ "PHIDP": 2,
19
+ "RHOHV": 2,
20
+ "CCORH": 8,
21
+ }
22
+
23
+ if TYPE_CHECKING:
24
+ from usdata.fetch import FetchedAsset
25
+
26
+
27
+ def open_nexrad(fetched: FetchedAsset) -> Any:
28
+ """Decode a local volume into a fully loaded xarray DataTree."""
29
+ try:
30
+ xradar = import_module("xradar")
31
+ except ModuleNotFoundError as error:
32
+ if error.name != "xradar":
33
+ raise
34
+ raise MissingReaderDependency(
35
+ 'NEXRAD reading requires xradar; install it with: pip install "usdata[radar]" '
36
+ '(or uv add "usdata[radar]")'
37
+ ) from error
38
+
39
+ # Bytes prevent remote URL interpretation and work across the backend's
40
+ # repeated sweep reads. Compressed source files stay unchanged in the cache.
41
+ content = fetched.path.read_bytes()
42
+ if content.startswith(b"\x1f\x8b"):
43
+ content = gzip.decompress(content)
44
+ elif content.startswith(b"BZh"):
45
+ content = bz2.decompress(content)
46
+ radar = xradar.io.open_nexradlevel2_datatree(content, incomplete_sweep="pad")
47
+ try:
48
+ radar.load()
49
+ finally:
50
+ radar.close()
51
+ for node in radar.subtree:
52
+ for name, variable in node.ds.variables.items():
53
+ # The backend records the entire input byte string as `source`.
54
+ # Provenance below is the durable reference, not that decoder buffer.
55
+ variable.encoding.pop("source", None)
56
+ if name in MOMENT_FLAG_COUNTS and "range" in variable.dims:
57
+ scale = variable.encoding.get("scale_factor")
58
+ offset = variable.encoding.get("add_offset")
59
+ if scale is not None and offset is not None:
60
+ # xradar 0.12 does not supply _FillValue for NEXRAD flags.
61
+ # Compare using each moment's native scale, not fixed units.
62
+ data = node[name]
63
+ valid = data.notnull()
64
+ for code in range(MOMENT_FLAG_COUNTS[name]):
65
+ valid = valid & (data != offset + code * scale)
66
+ masked = data.where(valid)
67
+ masked.encoding = data.encoding.copy()
68
+ node[name] = masked
69
+ radar.attrs["usdata"] = {
70
+ "asset_id": fetched.asset.id,
71
+ "provenance": fetched.provenance.model_dump(mode="json"),
72
+ }
73
+ return radar
@@ -9,6 +9,8 @@ import httpx
9
9
  import typer
10
10
 
11
11
  from usdata import __version__, build_query, default_registry
12
+ from usdata._progress import batch
13
+ from usdata.cli.progress import progress
12
14
  from usdata.fetch import ChecksumMismatch
13
15
  from usdata.fetch import fetch as fetch_query
14
16
  from usdata.manifest import lockfile_path
@@ -126,6 +128,7 @@ def fetch(
126
128
  ] = None,
127
129
  cache_dir: Annotated[Path | None, typer.Option(help="Override the cache directory.")] = None,
128
130
  force: Annotated[bool, typer.Option(help="Re-download even if cached.")] = False,
131
+ no_progress: Annotated[bool, typer.Option(help="Disable terminal progress.")] = False,
129
132
  dry_run: Annotated[
130
133
  bool, typer.Option(help="List matching assets without downloading.")
131
134
  ] = False,
@@ -165,8 +168,11 @@ def fetch(
165
168
  for a in assets:
166
169
  typer.echo(f"{a.id}\t{a.href}")
167
170
  typer.echo(f"{len(assets)} asset(s) matched", err=True)
171
+ with progress(disabled=no_progress):
172
+ batch([asset.size for asset in assets])
168
173
  return
169
- fetched = fetch_query(ds, query, root=cache_dir, force=force)
174
+ with progress(disabled=no_progress):
175
+ fetched = fetch_query(ds, query, root=cache_dir, force=force)
170
176
  except (DatasetNotFound, UnknownPlace, ValueError) as e:
171
177
  typer.secho(str(e), err=True, fg="red")
172
178
  raise typer.Exit(code=2) from None
@@ -192,10 +198,12 @@ def pull(
192
198
  bool,
193
199
  typer.Option(help="Ignore an existing lockfile: re-resolve every source and rewrite it."),
194
200
  ] = False,
201
+ no_progress: Annotated[bool, typer.Option(help="Disable terminal progress.")] = False,
195
202
  ) -> None:
196
203
  """Fetch every source in a manifest and write (or restore from) its lockfile."""
197
204
  try:
198
- result = pull_manifest(manifest, root=cache_dir, force=force)
205
+ with progress(disabled=no_progress):
206
+ result = pull_manifest(manifest, root=cache_dir, force=force)
199
207
  except EmptySource as e:
200
208
  typer.secho(str(e), err=True, fg="yellow")
201
209
  raise typer.Exit(code=1) from None
@@ -0,0 +1,89 @@
1
+ """Terminal-only progress rendering; normal CLI output remains machine readable."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import shutil
6
+ import sys
7
+ from collections.abc import Iterator
8
+ from contextlib import contextmanager
9
+ from time import monotonic
10
+ from typing import TextIO
11
+
12
+ from usdata._progress import AssetProgress, Batch, Event, observe
13
+
14
+
15
+ def _interactive() -> bool:
16
+ return sys.stdout.isatty() and sys.stderr.isatty()
17
+
18
+
19
+ class _Display:
20
+ def __init__(self, stream: TextIO) -> None:
21
+ self.stream = stream
22
+ self.width = 0
23
+ self.updated = 0.0
24
+ self.asset = ""
25
+ self.count = 0
26
+ self.done = 0
27
+ self.cached = 0
28
+
29
+ def clear(self) -> None:
30
+ if self.width:
31
+ self.stream.write("\r" + " " * self.width + "\r")
32
+ self.stream.flush()
33
+ self.width = 0
34
+
35
+ def line(self, text: str, *, final: bool = False) -> None:
36
+ self.clear()
37
+ # IDs come from providers: keep control characters out of the terminal.
38
+ text = "".join(c if c.isprintable() else "?" for c in text)
39
+ if final:
40
+ self.stream.write(text + "\n")
41
+ else:
42
+ text = text[: max(1, shutil.get_terminal_size().columns - 1)]
43
+ self.stream.write(text)
44
+ self.width = len(text)
45
+ self.stream.flush()
46
+ self.updated = monotonic()
47
+
48
+ def __call__(self, event: Event) -> None:
49
+ if isinstance(event, Batch):
50
+ self.count, self.done, self.cached = event.count, 0, 0
51
+ sizes = f"{event.known_bytes:,} known bytes"
52
+ if event.unknown_sizes:
53
+ sizes += f"; {event.unknown_sizes} size(s) unknown"
54
+ self.line(f"{event.count} asset(s) resolved; {sizes} (before cache checks)", final=True)
55
+ elif isinstance(event, AssetProgress):
56
+ self.asset = event.asset_id
57
+ if event.state == "start":
58
+ size = f"{event.size:,} bytes" if event.size is not None else "size unknown"
59
+ self.line(f"[{self.done}/{self.count}] checking cache ({size}) | {self.asset}")
60
+ else:
61
+ self.done += 1
62
+ self.cached += event.state == "cached"
63
+ self.line(
64
+ f"[{self.done}/{self.count}] {event.state} "
65
+ f"({event.size:,} bytes; {self.cached} cached) | {self.asset}",
66
+ final=self.done == self.count,
67
+ )
68
+ else:
69
+ if event.completed and monotonic() - self.updated < 0.1:
70
+ return
71
+ size = f"{event.total:,}" if event.total is not None else "unknown"
72
+ self.line(
73
+ f"[{self.done}/{self.count}] {event.completed:,}/{size} bytes "
74
+ f"(attempt {event.attempt}) | {self.asset}"
75
+ )
76
+
77
+
78
+ @contextmanager
79
+ def progress(*, disabled: bool = False) -> Iterator[None]:
80
+ """Render progress on stderr only when both output streams are terminals."""
81
+ if disabled or not _interactive():
82
+ yield
83
+ return
84
+ display = _Display(sys.stderr)
85
+ with observe(display):
86
+ try:
87
+ yield
88
+ finally:
89
+ display.clear()