usdata 0.26.0__tar.gz → 0.28.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {usdata-0.26.0 → usdata-0.28.0}/PKG-INFO +11 -10
- {usdata-0.26.0 → usdata-0.28.0}/README.md +9 -8
- {usdata-0.26.0 → usdata-0.28.0}/pyproject.toml +2 -2
- {usdata-0.26.0 → usdata-0.28.0}/pyproject.toml.orig +2 -2
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/_aqs.py +2 -6
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/_fetch.py +70 -26
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/_grib.py +50 -26
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/_hurdat2.py +2 -5
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/_netcdf.py +2 -5
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/_radar.py +5 -5
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/data/registry.yaml +58 -50
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/models.py +8 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/coops.py +2 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/gfs.py +1 -1
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/ghcnd.py +2 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/hrrr.py +1 -1
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/nbm.py +1 -1
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/normals.py +1 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/rap.py +1 -1
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/pull.py +40 -20
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/readers.py +199 -134
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/__init__.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/__main__.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/_files.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/_progress.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/cache.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/cache_ops.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/cite.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/cli/__init__.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/cli/app.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/cli/cache.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/cli/cite.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/cli/doctor.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/cli/inspect.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/cli/progress.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/data/nexrad_sites.csv +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/data/places.csv +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/data/places.sources.json +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/doctor.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/inspect.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/manifest.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/mirror.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/protocols/__init__.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/protocols/erddap.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/protocols/http.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/protocols/listing.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/protocols/s3.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/provenance.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/__init__.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/base.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/credentials.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/epa/__init__.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/epa/aqs.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/fema/__init__.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/fema/declarations.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/http.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/__init__.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/glm.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/goes.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/grib_index.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/gsom.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/gsoy.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/hurdat2.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/ibtracs.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/lcd.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/mrms.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/nexrad.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/nexrad_level3.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/nws_vtec.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/sites.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/spc.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/noaa/storm_events.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/params.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/usgs/__init__.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/usgs/daily.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/providers/usgs/earthquakes.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/py.typed +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/query.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/registry.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/selection.py +0 -0
- {usdata-0.26.0 → usdata-0.28.0}/src/usdata/testing.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: usdata
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.28.0
|
|
4
4
|
Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
|
|
5
5
|
Keywords: noaa,usgs,open-data,scientific-data,provenance,reproducible-research,weather,climate,meteorology
|
|
6
6
|
Author: Jake Van Slyke
|
|
@@ -34,7 +34,7 @@ Requires-Python: >=3.11
|
|
|
34
34
|
Project-URL: Homepage, https://usdata.dev/
|
|
35
35
|
Project-URL: Documentation, https://docs.usdata.dev/
|
|
36
36
|
Project-URL: Repository, https://github.com/jakeryderv/usdata
|
|
37
|
-
Project-URL: Examples, https://usdata.dev/
|
|
37
|
+
Project-URL: Examples, https://usdata.dev/studies/
|
|
38
38
|
Project-URL: Changelog, https://github.com/jakeryderv/usdata/blob/main/CHANGELOG.md
|
|
39
39
|
Project-URL: Issues, https://github.com/jakeryderv/usdata/issues
|
|
40
40
|
Provides-Extra: grib
|
|
@@ -55,12 +55,13 @@ Description-Content-Type: text/markdown
|
|
|
55
55
|
[](https://usdata.dev/)
|
|
56
56
|
|
|
57
57
|
Reproducible acquisition of U.S. public scientific data. One Python SDK and
|
|
58
|
-
CLI discovers curated
|
|
59
|
-
record of every input: a manifest names them,
|
|
60
|
-
and the record of what was fetched is what a
|
|
61
|
-
stays in pandas and xarray; usdata only
|
|
62
|
-
one agency publishes, that agency's own
|
|
63
|
-
|
|
58
|
+
CLI discovers curated datasets from agencies such as NOAA, USGS, and EPA,
|
|
59
|
+
fetches their files, and keeps a record of every input: a manifest names them,
|
|
60
|
+
a lockfile pins them by checksum, and the record of what was fetched is what a
|
|
61
|
+
methods section cites. Analysis stays in pandas and xarray; usdata only
|
|
62
|
+
acquires. If you need every product one agency publishes, that agency's own
|
|
63
|
+
library is the better tool; usdata is for pinning inputs across sources and
|
|
64
|
+
proving later that they have not changed.
|
|
64
65
|
|
|
65
66
|
```sh
|
|
66
67
|
pip install "usdata[pandas]"
|
|
@@ -108,9 +109,9 @@ breaking changes.
|
|
|
108
109
|
|
|
109
110
|
| | |
|
|
110
111
|
| --- | --- |
|
|
111
|
-
| [usdata.dev](https://usdata.dev/) | What it is for: the [dataset browser](https://usdata.dev/datasets/) and [worked examples](https://usdata.dev/
|
|
112
|
+
| [usdata.dev](https://usdata.dev/) | What it is for: the [dataset browser](https://usdata.dev/datasets/) and [worked examples](https://usdata.dev/studies/) with saved results |
|
|
112
113
|
| [docs.usdata.dev](https://docs.usdata.dev/) | How to use it: [install](https://docs.usdata.dev/install/), [getting started](https://docs.usdata.dev/getting-started/), guides, dataset notes, and reference |
|
|
113
|
-
| [Severe-weather case study](https://usdata.dev/
|
|
114
|
+
| [Severe-weather case study](https://usdata.dev/studies/severe-weather-case-study/) | One tornado, six sources, one manifest and lockfile, ending in a citation |
|
|
114
115
|
|
|
115
116
|
Twenty-seven datasets are available today and nineteen more are planned, grouped
|
|
116
117
|
by agency and product family in the [catalog](docs/providers/README.md).
|
|
@@ -10,12 +10,13 @@
|
|
|
10
10
|
[](https://usdata.dev/)
|
|
11
11
|
|
|
12
12
|
Reproducible acquisition of U.S. public scientific data. One Python SDK and
|
|
13
|
-
CLI discovers curated
|
|
14
|
-
record of every input: a manifest names them,
|
|
15
|
-
and the record of what was fetched is what a
|
|
16
|
-
stays in pandas and xarray; usdata only
|
|
17
|
-
one agency publishes, that agency's own
|
|
18
|
-
|
|
13
|
+
CLI discovers curated datasets from agencies such as NOAA, USGS, and EPA,
|
|
14
|
+
fetches their files, and keeps a record of every input: a manifest names them,
|
|
15
|
+
a lockfile pins them by checksum, and the record of what was fetched is what a
|
|
16
|
+
methods section cites. Analysis stays in pandas and xarray; usdata only
|
|
17
|
+
acquires. If you need every product one agency publishes, that agency's own
|
|
18
|
+
library is the better tool; usdata is for pinning inputs across sources and
|
|
19
|
+
proving later that they have not changed.
|
|
19
20
|
|
|
20
21
|
```sh
|
|
21
22
|
pip install "usdata[pandas]"
|
|
@@ -63,9 +64,9 @@ breaking changes.
|
|
|
63
64
|
|
|
64
65
|
| | |
|
|
65
66
|
| --- | --- |
|
|
66
|
-
| [usdata.dev](https://usdata.dev/) | What it is for: the [dataset browser](https://usdata.dev/datasets/) and [worked examples](https://usdata.dev/
|
|
67
|
+
| [usdata.dev](https://usdata.dev/) | What it is for: the [dataset browser](https://usdata.dev/datasets/) and [worked examples](https://usdata.dev/studies/) with saved results |
|
|
67
68
|
| [docs.usdata.dev](https://docs.usdata.dev/) | How to use it: [install](https://docs.usdata.dev/install/), [getting started](https://docs.usdata.dev/getting-started/), guides, dataset notes, and reference |
|
|
68
|
-
| [Severe-weather case study](https://usdata.dev/
|
|
69
|
+
| [Severe-weather case study](https://usdata.dev/studies/severe-weather-case-study/) | One tornado, six sources, one manifest and lockfile, ending in a citation |
|
|
69
70
|
|
|
70
71
|
Twenty-seven datasets are available today and nineteen more are planned, grouped
|
|
71
72
|
by agency and product family in the [catalog](docs/providers/README.md).
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "usdata"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.28.0"
|
|
4
4
|
description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -59,7 +59,7 @@ grib = [
|
|
|
59
59
|
Homepage = "https://usdata.dev/"
|
|
60
60
|
Documentation = "https://docs.usdata.dev/"
|
|
61
61
|
Repository = "https://github.com/jakeryderv/usdata"
|
|
62
|
-
Examples = "https://usdata.dev/
|
|
62
|
+
Examples = "https://usdata.dev/studies/"
|
|
63
63
|
Changelog = "https://github.com/jakeryderv/usdata/blob/main/CHANGELOG.md"
|
|
64
64
|
Issues = "https://github.com/jakeryderv/usdata/issues"
|
|
65
65
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "usdata"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.28.0"
|
|
4
4
|
description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -56,7 +56,7 @@ grib = [
|
|
|
56
56
|
Homepage = "https://usdata.dev/"
|
|
57
57
|
Documentation = "https://docs.usdata.dev/"
|
|
58
58
|
Repository = "https://github.com/jakeryderv/usdata"
|
|
59
|
-
Examples = "https://usdata.dev/
|
|
59
|
+
Examples = "https://usdata.dev/studies/"
|
|
60
60
|
Changelog = "https://github.com/jakeryderv/usdata/blob/main/CHANGELOG.md"
|
|
61
61
|
Issues = "https://github.com/jakeryderv/usdata/issues"
|
|
62
62
|
|
|
@@ -13,7 +13,7 @@ import json
|
|
|
13
13
|
from importlib import import_module
|
|
14
14
|
from typing import TYPE_CHECKING, Any
|
|
15
15
|
|
|
16
|
-
from usdata.readers import MissingReaderDependency
|
|
16
|
+
from usdata.readers import MissingReaderDependency, source_attrs
|
|
17
17
|
|
|
18
18
|
if TYPE_CHECKING:
|
|
19
19
|
from usdata._fetch import FetchedAsset
|
|
@@ -43,9 +43,5 @@ def open_aqs(fetched: FetchedAsset) -> Any:
|
|
|
43
43
|
for column in DATE_COLUMNS:
|
|
44
44
|
if column in frame:
|
|
45
45
|
frame[column] = pandas.to_datetime(frame[column], format="%Y-%m-%d")
|
|
46
|
-
frame.attrs["usdata"] = {
|
|
47
|
-
"asset_id": fetched.asset.id,
|
|
48
|
-
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
49
|
-
"header": body.get("Header", []),
|
|
50
|
-
}
|
|
46
|
+
frame.attrs["usdata"] = {**source_attrs(fetched), "header": body.get("Header", [])}
|
|
51
47
|
return frame
|
|
@@ -17,6 +17,10 @@ from usdata.models import Asset, Dataset, Provenance, Query
|
|
|
17
17
|
from usdata.providers import Provider, load_adapter
|
|
18
18
|
|
|
19
19
|
if TYPE_CHECKING:
|
|
20
|
+
# Each is an optional extra: without it installed the result is simply untyped.
|
|
21
|
+
import pandas as pd # pyright: ignore[reportMissingImports]
|
|
22
|
+
import xarray as xr # pyright: ignore[reportMissingImports]
|
|
23
|
+
|
|
20
24
|
from usdata.inspect import Summary
|
|
21
25
|
|
|
22
26
|
|
|
@@ -32,46 +36,74 @@ class FetchedAsset(BaseModel):
|
|
|
32
36
|
provenance: Provenance
|
|
33
37
|
from_cache: bool
|
|
34
38
|
|
|
35
|
-
def open(
|
|
39
|
+
def open(self) -> Any:
|
|
40
|
+
"""Open local data with the reader its format implies, with that reader's defaults.
|
|
41
|
+
|
|
42
|
+
CSV and ERDDAP CSV, HURDAT2 and AQS daily JSON return a pandas DataFrame;
|
|
43
|
+
NetCDF4 and GRIB2 a loaded xarray Dataset; NEXRAD Level II an xarray
|
|
44
|
+
DataTree. Provenance is kept in the result's ``attrs["usdata"]``. A file
|
|
45
|
+
that needs options, or whose metadata leaves its format ambiguous, is
|
|
46
|
+
opened with the method for its format instead: ``open_csv``,
|
|
47
|
+
``open_nexrad``, ``open_grib2``, or ``open_netcdf``. Cached files and
|
|
48
|
+
provenance sidecars are never changed. See ``usdata.readers.open_asset``.
|
|
49
|
+
"""
|
|
50
|
+
from usdata.readers import open_asset
|
|
51
|
+
|
|
52
|
+
return open_asset(self)
|
|
53
|
+
|
|
54
|
+
def open_csv(
|
|
36
55
|
self,
|
|
37
56
|
*,
|
|
38
|
-
reader: str | None = None,
|
|
39
57
|
dtype: dict[str, str] | None = None,
|
|
40
58
|
parse_dates: list[str] | None = None,
|
|
41
59
|
usecols: list[str] | None = None,
|
|
42
60
|
nrows: int | None = None,
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
ERDDAP units are kept in ``frame.attrs["units"]`` and source provenance
|
|
50
|
-
in ``frame.attrs["usdata"]``. NEXRAD returns a xarray DataTree with provenance
|
|
51
|
-
in ``radar.attrs["usdata"]``. NetCDF4 and GRIB2 return a loaded xarray Dataset
|
|
52
|
-
with matching provenance in its attributes, and with units and long names
|
|
53
|
-
the file leaves unstated filled from the registry entry's variables. See
|
|
54
|
-
``usdata.readers.open_asset`` for options. Use ``sweep=0`` or ``sweep=[0, 2]``
|
|
55
|
-
to load selected zero-based radar sweeps, and
|
|
56
|
-
``select={"shortName": "cape", "typeOfLevel": "surface"}``
|
|
57
|
-
to choose GRIB2 messages; ``strict=True`` raises instead of warning when a
|
|
58
|
-
GRIB2 select value matches none of the selected messages. Cached files and
|
|
59
|
-
provenance sidecars are never changed.
|
|
61
|
+
units_row: bool | None = None,
|
|
62
|
+
) -> pd.DataFrame:
|
|
63
|
+
"""Open a CSV as a pandas DataFrame; see ``usdata.readers.open_csv``.
|
|
64
|
+
|
|
65
|
+
``units_row`` says whether a units row follows the header, as in ERDDAP
|
|
66
|
+
CSV, and is inferred when ``None``; the units go to ``frame.attrs["units"]``.
|
|
60
67
|
"""
|
|
61
|
-
from usdata.readers import
|
|
68
|
+
from usdata.readers import open_csv
|
|
62
69
|
|
|
63
|
-
return
|
|
70
|
+
return open_csv(
|
|
64
71
|
self,
|
|
65
|
-
reader=reader,
|
|
66
72
|
dtype=dtype,
|
|
67
73
|
parse_dates=parse_dates,
|
|
68
74
|
usecols=usecols,
|
|
69
75
|
nrows=nrows,
|
|
70
|
-
|
|
71
|
-
select=select,
|
|
72
|
-
strict=strict,
|
|
76
|
+
units_row=units_row,
|
|
73
77
|
)
|
|
74
78
|
|
|
79
|
+
def open_nexrad(self, *, sweep: int | list[int] | None = None) -> xr.DataTree:
|
|
80
|
+
"""Open a NEXRAD Level II volume as an xarray DataTree; see ``usdata.readers.open_nexrad``.
|
|
81
|
+
|
|
82
|
+
Use ``sweep=0`` or ``sweep=[0, 2]`` to load selected zero-based sweeps.
|
|
83
|
+
"""
|
|
84
|
+
from usdata.readers import open_nexrad
|
|
85
|
+
|
|
86
|
+
return open_nexrad(self, sweep=sweep)
|
|
87
|
+
|
|
88
|
+
def open_grib2(
|
|
89
|
+
self, *, select: Mapping[str, Any] | None = None, strict: bool = False
|
|
90
|
+
) -> xr.Dataset:
|
|
91
|
+
"""Open GRIB2 messages as one xarray Dataset; see ``usdata.readers.open_grib2``.
|
|
92
|
+
|
|
93
|
+
``select={"shortName": "cape", "typeOfLevel": "surface"}`` chooses
|
|
94
|
+
messages; ``strict=True`` raises instead of warning when a select value
|
|
95
|
+
matches none of the selected messages.
|
|
96
|
+
"""
|
|
97
|
+
from usdata.readers import open_grib2
|
|
98
|
+
|
|
99
|
+
return open_grib2(self, select=select, strict=strict)
|
|
100
|
+
|
|
101
|
+
def open_netcdf(self) -> xr.Dataset:
|
|
102
|
+
"""Open a NetCDF4 file as an xarray Dataset; see ``usdata.readers.open_netcdf``."""
|
|
103
|
+
from usdata.readers import open_netcdf
|
|
104
|
+
|
|
105
|
+
return open_netcdf(self)
|
|
106
|
+
|
|
75
107
|
def inspect(self) -> Summary:
|
|
76
108
|
"""Summarize this file: provenance, format, and what that format holds.
|
|
77
109
|
|
|
@@ -108,12 +140,18 @@ def _fetch_asset(
|
|
|
108
140
|
root: Path | None = None,
|
|
109
141
|
force: bool = False,
|
|
110
142
|
pinned: Provenance | None = None,
|
|
143
|
+
staging: Path | None = None,
|
|
111
144
|
) -> FetchedAsset:
|
|
112
145
|
"""Fetch one asset through the cache, passing any pinned record to the adapter.
|
|
113
146
|
|
|
114
147
|
``pinned`` is the provenance a lockfile holds for this asset. The adapter sees
|
|
115
148
|
it through ``prepare_fetch``, so an asset that pins byte ranges is reproduced
|
|
116
149
|
from the record rather than by resolving its query again.
|
|
150
|
+
|
|
151
|
+
``staging`` is a root a download is written under instead of the cache. The
|
|
152
|
+
cache at ``root`` is still checked, and a hit is returned from there; a
|
|
153
|
+
miss comes back with its ``path`` under ``staging``, for the caller to move
|
|
154
|
+
home once it can pin it (ADR 0031).
|
|
117
155
|
"""
|
|
118
156
|
if asset.dataset_id != dataset.id:
|
|
119
157
|
raise ValueError(f"asset dataset {asset.dataset_id!r} does not match {dataset.id!r}")
|
|
@@ -138,6 +176,8 @@ def _fetch_asset(
|
|
|
138
176
|
):
|
|
139
177
|
_progress.emit(_progress.AssetProgress(asset.id, "cached", prov.size))
|
|
140
178
|
return FetchedAsset(asset=asset, path=path, provenance=prov, from_cache=True)
|
|
179
|
+
if staging is not None:
|
|
180
|
+
path = asset_path(asset, staging)
|
|
141
181
|
with staged_path(path) as tmp:
|
|
142
182
|
partial = adapter.prepare_fetch(asset, pinned)
|
|
143
183
|
if partial is None:
|
|
@@ -174,15 +214,19 @@ def _fetch_with(
|
|
|
174
214
|
*,
|
|
175
215
|
root: Path | None = None,
|
|
176
216
|
force: bool = False,
|
|
217
|
+
staging: Path | None = None,
|
|
177
218
|
) -> list[FetchedAsset]:
|
|
178
219
|
"""Run the loop on an adapter the caller opened, so one adapter can serve many queries.
|
|
179
220
|
|
|
180
221
|
The listing is put in the order every result promises, by ``asset.time.start``
|
|
181
222
|
then id, so no adapter has to sort and no caller has to sort defensively.
|
|
223
|
+
``staging`` is passed to each fetch; see ``_fetch_asset``.
|
|
182
224
|
"""
|
|
183
225
|
assets = ordered(adapter.list_assets(query))
|
|
184
226
|
_progress.batch([asset.size for asset in assets])
|
|
185
|
-
return [
|
|
227
|
+
return [
|
|
228
|
+
_fetch_asset(dataset, a, adapter, root=root, force=force, staging=staging) for a in assets
|
|
229
|
+
]
|
|
186
230
|
|
|
187
231
|
|
|
188
232
|
def ordered(assets: list[Asset]) -> list[Asset]:
|
|
@@ -9,7 +9,8 @@ repeat, so the same select always yields the same names. Values arrive from
|
|
|
9
9
|
ecCodes as float64, are masked to NaN where the message's bitmap marks them
|
|
10
10
|
missing, and are stored as float32; the float64 array is released before the
|
|
11
11
|
Dataset is returned. Rows are ordered north to south and columns west to east
|
|
12
|
-
regardless of the message's scanning mode
|
|
12
|
+
regardless of the message's scanning mode, including grids whose adjacent rows
|
|
13
|
+
scan in opposite directions. See ADR 0022.
|
|
13
14
|
"""
|
|
14
15
|
|
|
15
16
|
from __future__ import annotations
|
|
@@ -21,12 +22,16 @@ from collections.abc import Iterator, Mapping, Sequence
|
|
|
21
22
|
from contextlib import contextmanager
|
|
22
23
|
from datetime import UTC, datetime
|
|
23
24
|
from importlib import import_module
|
|
24
|
-
from inspect import currentframe
|
|
25
25
|
from pathlib import Path
|
|
26
26
|
from typing import TYPE_CHECKING, Any, NamedTuple
|
|
27
27
|
|
|
28
28
|
from usdata.inspect import GribMessage
|
|
29
|
-
from usdata.readers import
|
|
29
|
+
from usdata.readers import (
|
|
30
|
+
MissingReaderDependency,
|
|
31
|
+
caller_stacklevel,
|
|
32
|
+
fill_registry_attrs,
|
|
33
|
+
source_attrs,
|
|
34
|
+
)
|
|
30
35
|
|
|
31
36
|
if TYPE_CHECKING:
|
|
32
37
|
from usdata._fetch import FetchedAsset
|
|
@@ -39,6 +44,11 @@ LIBRARY_HINT = (
|
|
|
39
44
|
)
|
|
40
45
|
MRMS_NAME = re.compile(r"^MRMS_(?P<product>.+?)_\d{2}\.\d{2}_\d{8}-\d{6}\.grib2(?:\.gz)?$")
|
|
41
46
|
INVENTORY_KEYS = ("shortName", "name", "typeOfLevel", "level", "step", "units")
|
|
47
|
+
SYMBOL_POWER = re.compile(r"(?<=[A-Za-z])\*\*")
|
|
48
|
+
"""A power written after a unit symbol, as ecCodes writes ``kg**-1``."""
|
|
49
|
+
NUMERIC_POWER = re.compile(r"(?<![A-Za-z])\*\*")
|
|
50
|
+
"""A power written after anything else, such as the ``10**-3`` of a scale factor."""
|
|
51
|
+
|
|
42
52
|
VARIABLE_KEYS = (
|
|
43
53
|
"name",
|
|
44
54
|
"units",
|
|
@@ -200,18 +210,6 @@ def _unmatched_report(
|
|
|
200
210
|
return " ".join(parts)
|
|
201
211
|
|
|
202
212
|
|
|
203
|
-
def _caller_stacklevel() -> int:
|
|
204
|
-
"""Stack level of the first frame outside usdata, so a warning points at the caller."""
|
|
205
|
-
package = Path(__file__).resolve().parent
|
|
206
|
-
frame = currentframe()
|
|
207
|
-
frame = frame.f_back if frame is not None else None
|
|
208
|
-
level = 1
|
|
209
|
-
while frame is not None and Path(frame.f_code.co_filename).resolve().is_relative_to(package):
|
|
210
|
-
level += 1
|
|
211
|
-
frame = frame.f_back
|
|
212
|
-
return level
|
|
213
|
-
|
|
214
|
-
|
|
215
213
|
def _shape(eccodes: Any, h: int) -> tuple[int, int] | None:
|
|
216
214
|
"""A message's grid rows and columns under either key pair, or None when neither is set."""
|
|
217
215
|
rows, cols = _get(eccodes, h, "Nj", int), _get(eccodes, h, "Ni", int)
|
|
@@ -241,6 +239,20 @@ def inventory(path: Path) -> list[GribMessage]:
|
|
|
241
239
|
return messages
|
|
242
240
|
|
|
243
241
|
|
|
242
|
+
def udunits(units: str) -> str:
|
|
243
|
+
"""EcCodes ``units`` in the UDUNITS notation CF metadata uses, or unchanged.
|
|
244
|
+
|
|
245
|
+
ecCodes writes powers as ``**``: ``J kg**-1``, ``m**2 s**-2``. UDUNITS and
|
|
246
|
+
CF write them as a trailing signed integer, ``J kg-1`` and ``m2 s-2``,
|
|
247
|
+
which is what xarray-based tools expect. Only the notation changes. A power
|
|
248
|
+
of a number, such as ``10**-3``, has no such spelling, so a string holding
|
|
249
|
+
one is returned as it is rather than half rewritten.
|
|
250
|
+
"""
|
|
251
|
+
if NUMERIC_POWER.search(units):
|
|
252
|
+
return units
|
|
253
|
+
return SYMBOL_POWER.sub("", units)
|
|
254
|
+
|
|
255
|
+
|
|
244
256
|
def _available(path: Path) -> str:
|
|
245
257
|
"""Every message as a (shortName, typeOfLevel, level) triple, for a reader error to list."""
|
|
246
258
|
return ", ".join(
|
|
@@ -279,6 +291,8 @@ class _Grid:
|
|
|
279
291
|
rows, cols = shape
|
|
280
292
|
self.flip_rows = bool(_get(eccodes, h, "jScansPositively", int))
|
|
281
293
|
self.flip_cols = bool(_get(eccodes, h, "iScansNegatively", int))
|
|
294
|
+
# Adjacent rows scan in opposite directions (NBM's CONUS grid does this).
|
|
295
|
+
self.alternating = bool(_get(eccodes, h, "alternativeRowScanning", int))
|
|
282
296
|
if _get(eccodes, h, "jPointsAreConsecutive", int):
|
|
283
297
|
raise ValueError("grids with consecutive j points are not supported")
|
|
284
298
|
self.regular = self.grid_type == "regular_ll"
|
|
@@ -305,15 +319,25 @@ class _Grid:
|
|
|
305
319
|
value = _get(eccodes, h, key)
|
|
306
320
|
if value is not None:
|
|
307
321
|
self.attrs[key] = value
|
|
308
|
-
self.key = (self.grid_type, self.shape, self.flip_rows, self.flip_cols)
|
|
322
|
+
self.key = (self.grid_type, self.shape, self.flip_rows, self.flip_cols, self.alternating)
|
|
309
323
|
|
|
310
324
|
def _orient(self, numpy: Any, axis: Any, *, rows: bool) -> Any:
|
|
311
325
|
return axis[::-1].copy() if (self.flip_rows if rows else self.flip_cols) else axis
|
|
312
326
|
|
|
313
|
-
def reshape(self, numpy: Any, flat: Any) -> Any:
|
|
327
|
+
def reshape(self, numpy: Any, flat: Any, *, stored: bool = False) -> Any:
|
|
328
|
+
"""Order a flat array north to south and west to east.
|
|
329
|
+
|
|
330
|
+
``stored`` marks data values, which ecCodes returns in the order the
|
|
331
|
+
message stores them: with alternative row scanning every second row
|
|
332
|
+
runs the other way and is reversed here. The latitudes and longitudes
|
|
333
|
+
ecCodes computes do not alternate, so coordinates skip that step.
|
|
334
|
+
"""
|
|
314
335
|
if flat.size != self.shape[0] * self.shape[1]:
|
|
315
336
|
raise ValueError(f"message has {flat.size} values for a {self.shape} grid")
|
|
316
337
|
grid = flat.reshape(self.shape)
|
|
338
|
+
if stored and self.alternating:
|
|
339
|
+
grid = grid.copy()
|
|
340
|
+
grid[1::2] = grid[1::2, ::-1]
|
|
317
341
|
if self.flip_rows:
|
|
318
342
|
grid = grid[::-1, :]
|
|
319
343
|
if self.flip_cols:
|
|
@@ -423,11 +447,15 @@ def open_grib2(
|
|
|
423
447
|
missing = _get(eccodes, h, "missingValue", float)
|
|
424
448
|
if missing is not None:
|
|
425
449
|
values[values == missing] = numpy.nan
|
|
426
|
-
data = grid.reshape(numpy, values).astype(numpy.float32)
|
|
450
|
+
data = grid.reshape(numpy, values, stored=True).astype(numpy.float32)
|
|
427
451
|
del values
|
|
428
452
|
attrs = {
|
|
429
453
|
key: value for key in VARIABLE_KEYS if (value := _get(eccodes, h, key)) is not None
|
|
430
454
|
}
|
|
455
|
+
if isinstance(units := attrs.get("units"), str) and (plain := udunits(units)) != units:
|
|
456
|
+
# The file's own spelling stays beside the rewritten one, as cfgrib keeps it.
|
|
457
|
+
attrs["GRIB_units"] = units
|
|
458
|
+
attrs["units"] = plain
|
|
431
459
|
for label, date_key, time_key in (
|
|
432
460
|
("reference_time", "dataDate", "dataTime"),
|
|
433
461
|
("valid_time", "validityDate", "validityTime"),
|
|
@@ -454,8 +482,8 @@ def open_grib2(
|
|
|
454
482
|
selected.append(_Field(short=short, data=data, attrs=attrs, message=message))
|
|
455
483
|
if count > 1 and needs_select:
|
|
456
484
|
raise ValueError(
|
|
457
|
-
f"{count} messages; pass select={{...}} with ecCodes keys to
|
|
458
|
-
f"for example select={{'shortName': ..., 'typeOfLevel': ...}}. "
|
|
485
|
+
f"{count} messages; pass select={{...}} to open_grib2 with ecCodes keys to "
|
|
486
|
+
f"choose, for example select={{'shortName': ..., 'typeOfLevel': ...}}. "
|
|
459
487
|
f"Available (shortName, typeOfLevel, level): {_available(fetched.path)}"
|
|
460
488
|
)
|
|
461
489
|
if grid is None or not selected:
|
|
@@ -468,7 +496,7 @@ def open_grib2(
|
|
|
468
496
|
if report := _unmatched_report(options, matched, present):
|
|
469
497
|
if strict:
|
|
470
498
|
raise ValueError(report)
|
|
471
|
-
warnings.warn(report, UserWarning, stacklevel=
|
|
499
|
+
warnings.warn(report, UserWarning, stacklevel=caller_stacklevel())
|
|
472
500
|
variables: dict[str, Any] = {}
|
|
473
501
|
messages: dict[str, dict[str, Any]] = {}
|
|
474
502
|
names = variable_names(
|
|
@@ -484,10 +512,6 @@ def open_grib2(
|
|
|
484
512
|
dataset.latitude.attrs["units"] = "degrees_north"
|
|
485
513
|
dataset.longitude.attrs["units"] = "degrees_east"
|
|
486
514
|
dataset.attrs.update(grid.attrs)
|
|
487
|
-
dataset.attrs["usdata"] = {
|
|
488
|
-
"asset_id": fetched.asset.id,
|
|
489
|
-
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
490
|
-
"messages": messages,
|
|
491
|
-
}
|
|
515
|
+
dataset.attrs["usdata"] = {**source_attrs(fetched), "messages": messages}
|
|
492
516
|
fill_registry_attrs(fetched, dataset)
|
|
493
517
|
return dataset
|
|
@@ -15,7 +15,7 @@ from datetime import UTC, datetime
|
|
|
15
15
|
from importlib import import_module
|
|
16
16
|
from typing import TYPE_CHECKING, Any
|
|
17
17
|
|
|
18
|
-
from usdata.readers import Hurdat2FormatError, MissingReaderDependency
|
|
18
|
+
from usdata.readers import Hurdat2FormatError, MissingReaderDependency, source_attrs
|
|
19
19
|
|
|
20
20
|
if TYPE_CHECKING:
|
|
21
21
|
from usdata._fetch import FetchedAsset
|
|
@@ -174,8 +174,5 @@ def open_hurdat2(fetched: FetchedAsset) -> Any:
|
|
|
174
174
|
data["time"] = pandas.to_datetime(columns["time"], utc=True)
|
|
175
175
|
data.update({name: pandas.array(columns[name], dtype="float64") for name in NUMERIC_COLUMNS})
|
|
176
176
|
frame = pandas.DataFrame(data, columns=COLUMNS)
|
|
177
|
-
frame.attrs["usdata"] =
|
|
178
|
-
"asset_id": fetched.asset.id,
|
|
179
|
-
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
180
|
-
}
|
|
177
|
+
frame.attrs["usdata"] = source_attrs(fetched)
|
|
181
178
|
return frame
|
|
@@ -7,7 +7,7 @@ from pathlib import Path
|
|
|
7
7
|
from typing import TYPE_CHECKING, Any
|
|
8
8
|
|
|
9
9
|
from usdata.inspect import NetcdfVariable
|
|
10
|
-
from usdata.readers import MissingReaderDependency, fill_registry_attrs
|
|
10
|
+
from usdata.readers import MissingReaderDependency, fill_registry_attrs, source_attrs
|
|
11
11
|
|
|
12
12
|
if TYPE_CHECKING:
|
|
13
13
|
from usdata._fetch import FetchedAsset
|
|
@@ -42,10 +42,7 @@ def open_netcdf(fetched: FetchedAsset) -> Any:
|
|
|
42
42
|
xarray.open_dataset(stream, engine="h5netcdf", chunks=None) as dataset,
|
|
43
43
|
):
|
|
44
44
|
dataset.load()
|
|
45
|
-
dataset.attrs["usdata"] =
|
|
46
|
-
"asset_id": fetched.asset.id,
|
|
47
|
-
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
48
|
-
}
|
|
45
|
+
dataset.attrs["usdata"] = source_attrs(fetched)
|
|
49
46
|
fill_registry_attrs(fetched, dataset)
|
|
50
47
|
return dataset
|
|
51
48
|
|
|
@@ -7,7 +7,7 @@ import gzip
|
|
|
7
7
|
from importlib import import_module
|
|
8
8
|
from typing import TYPE_CHECKING, Any
|
|
9
9
|
|
|
10
|
-
from usdata.readers import MissingReaderDependency, RadarDecodeError
|
|
10
|
+
from usdata.readers import MissingReaderDependency, RadarDecodeError, source_attrs
|
|
11
11
|
|
|
12
12
|
# NOAA RDA/RPG ICD 2620002Y, Table XVII-I notes 21 and 30.
|
|
13
13
|
MOMENT_FLAG_COUNTS = {
|
|
@@ -67,8 +67,9 @@ def _check_sweeps(content: bytes, sweep: int | list[int] | None) -> None:
|
|
|
67
67
|
):
|
|
68
68
|
raise RadarDecodeError(
|
|
69
69
|
f"cannot safely decode sweep {index}: NEXRAD moment and coordinate records "
|
|
70
|
-
"do not agree; select an unaffected sweep explicitly with
|
|
71
|
-
"or use another decoder.
|
|
70
|
+
"do not agree; select an unaffected sweep explicitly with "
|
|
71
|
+
"open_nexrad(sweep=...) or use another decoder. "
|
|
72
|
+
"No sweeps were silently dropped."
|
|
72
73
|
)
|
|
73
74
|
|
|
74
75
|
|
|
@@ -116,8 +117,7 @@ def open_nexrad(fetched: FetchedAsset, *, sweep: int | list[int] | None = None)
|
|
|
116
117
|
masked.encoding = data.encoding.copy()
|
|
117
118
|
node[name] = masked
|
|
118
119
|
radar.attrs["usdata"] = {
|
|
119
|
-
|
|
120
|
-
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
120
|
+
**source_attrs(fetched),
|
|
121
121
|
"sweeps": [name.lstrip("/") for name in radar.groups if name.startswith("/sweep_")],
|
|
122
122
|
}
|
|
123
123
|
return radar
|