usdata 0.27.0__tar.gz → 0.28.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {usdata-0.27.0 → usdata-0.28.0}/PKG-INFO +1 -1
- {usdata-0.27.0 → usdata-0.28.0}/pyproject.toml +1 -1
- {usdata-0.27.0 → usdata-0.28.0}/pyproject.toml.orig +1 -1
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_aqs.py +2 -6
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_grib.py +31 -20
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_hurdat2.py +2 -5
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_netcdf.py +2 -5
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_radar.py +2 -3
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/models.py +8 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/coops.py +2 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/ghcnd.py +2 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/normals.py +1 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/readers.py +49 -9
- {usdata-0.27.0 → usdata-0.28.0}/README.md +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/__init__.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/__main__.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_fetch.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_files.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_progress.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cache.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cache_ops.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cite.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/__init__.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/app.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/cache.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/cite.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/doctor.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/inspect.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/progress.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/data/nexrad_sites.csv +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/data/places.csv +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/data/places.sources.json +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/data/registry.yaml +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/doctor.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/inspect.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/manifest.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/mirror.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/protocols/__init__.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/protocols/erddap.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/protocols/http.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/protocols/listing.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/protocols/s3.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/provenance.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/__init__.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/base.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/credentials.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/epa/__init__.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/epa/aqs.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/fema/__init__.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/fema/declarations.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/http.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/__init__.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/gfs.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/glm.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/goes.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/grib_index.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/gsom.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/gsoy.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/hrrr.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/hurdat2.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/ibtracs.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/lcd.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/mrms.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/nbm.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/nexrad.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/nexrad_level3.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/nws_vtec.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/rap.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/sites.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/spc.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/storm_events.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/params.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/usgs/__init__.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/usgs/daily.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/usgs/earthquakes.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/pull.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/py.typed +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/query.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/registry.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/selection.py +0 -0
- {usdata-0.27.0 → usdata-0.28.0}/src/usdata/testing.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: usdata
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.28.0
|
|
4
4
|
Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
|
|
5
5
|
Keywords: noaa,usgs,open-data,scientific-data,provenance,reproducible-research,weather,climate,meteorology
|
|
6
6
|
Author: Jake Van Slyke
|
|
@@ -13,7 +13,7 @@ import json
|
|
|
13
13
|
from importlib import import_module
|
|
14
14
|
from typing import TYPE_CHECKING, Any
|
|
15
15
|
|
|
16
|
-
from usdata.readers import MissingReaderDependency
|
|
16
|
+
from usdata.readers import MissingReaderDependency, source_attrs
|
|
17
17
|
|
|
18
18
|
if TYPE_CHECKING:
|
|
19
19
|
from usdata._fetch import FetchedAsset
|
|
@@ -43,9 +43,5 @@ def open_aqs(fetched: FetchedAsset) -> Any:
|
|
|
43
43
|
for column in DATE_COLUMNS:
|
|
44
44
|
if column in frame:
|
|
45
45
|
frame[column] = pandas.to_datetime(frame[column], format="%Y-%m-%d")
|
|
46
|
-
frame.attrs["usdata"] = {
|
|
47
|
-
"asset_id": fetched.asset.id,
|
|
48
|
-
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
49
|
-
"header": body.get("Header", []),
|
|
50
|
-
}
|
|
46
|
+
frame.attrs["usdata"] = {**source_attrs(fetched), "header": body.get("Header", [])}
|
|
51
47
|
return frame
|
|
@@ -22,12 +22,16 @@ from collections.abc import Iterator, Mapping, Sequence
|
|
|
22
22
|
from contextlib import contextmanager
|
|
23
23
|
from datetime import UTC, datetime
|
|
24
24
|
from importlib import import_module
|
|
25
|
-
from inspect import currentframe
|
|
26
25
|
from pathlib import Path
|
|
27
26
|
from typing import TYPE_CHECKING, Any, NamedTuple
|
|
28
27
|
|
|
29
28
|
from usdata.inspect import GribMessage
|
|
30
|
-
from usdata.readers import
|
|
29
|
+
from usdata.readers import (
|
|
30
|
+
MissingReaderDependency,
|
|
31
|
+
caller_stacklevel,
|
|
32
|
+
fill_registry_attrs,
|
|
33
|
+
source_attrs,
|
|
34
|
+
)
|
|
31
35
|
|
|
32
36
|
if TYPE_CHECKING:
|
|
33
37
|
from usdata._fetch import FetchedAsset
|
|
@@ -40,6 +44,11 @@ LIBRARY_HINT = (
|
|
|
40
44
|
)
|
|
41
45
|
MRMS_NAME = re.compile(r"^MRMS_(?P<product>.+?)_\d{2}\.\d{2}_\d{8}-\d{6}\.grib2(?:\.gz)?$")
|
|
42
46
|
INVENTORY_KEYS = ("shortName", "name", "typeOfLevel", "level", "step", "units")
|
|
47
|
+
SYMBOL_POWER = re.compile(r"(?<=[A-Za-z])\*\*")
|
|
48
|
+
"""A power written after a unit symbol, as ecCodes writes ``kg**-1``."""
|
|
49
|
+
NUMERIC_POWER = re.compile(r"(?<![A-Za-z])\*\*")
|
|
50
|
+
"""A power written after anything else, such as the ``10**-3`` of a scale factor."""
|
|
51
|
+
|
|
43
52
|
VARIABLE_KEYS = (
|
|
44
53
|
"name",
|
|
45
54
|
"units",
|
|
@@ -201,18 +210,6 @@ def _unmatched_report(
|
|
|
201
210
|
return " ".join(parts)
|
|
202
211
|
|
|
203
212
|
|
|
204
|
-
def _caller_stacklevel() -> int:
|
|
205
|
-
"""Stack level of the first frame outside usdata, so a warning points at the caller."""
|
|
206
|
-
package = Path(__file__).resolve().parent
|
|
207
|
-
frame = currentframe()
|
|
208
|
-
frame = frame.f_back if frame is not None else None
|
|
209
|
-
level = 1
|
|
210
|
-
while frame is not None and Path(frame.f_code.co_filename).resolve().is_relative_to(package):
|
|
211
|
-
level += 1
|
|
212
|
-
frame = frame.f_back
|
|
213
|
-
return level
|
|
214
|
-
|
|
215
|
-
|
|
216
213
|
def _shape(eccodes: Any, h: int) -> tuple[int, int] | None:
|
|
217
214
|
"""A message's grid rows and columns under either key pair, or None when neither is set."""
|
|
218
215
|
rows, cols = _get(eccodes, h, "Nj", int), _get(eccodes, h, "Ni", int)
|
|
@@ -242,6 +239,20 @@ def inventory(path: Path) -> list[GribMessage]:
|
|
|
242
239
|
return messages
|
|
243
240
|
|
|
244
241
|
|
|
242
|
+
def udunits(units: str) -> str:
|
|
243
|
+
"""EcCodes ``units`` in the UDUNITS notation CF metadata uses, or unchanged.
|
|
244
|
+
|
|
245
|
+
ecCodes writes powers as ``**``: ``J kg**-1``, ``m**2 s**-2``. UDUNITS and
|
|
246
|
+
CF write them as a trailing signed integer, ``J kg-1`` and ``m2 s-2``,
|
|
247
|
+
which is what xarray-based tools expect. Only the notation changes. A power
|
|
248
|
+
of a number, such as ``10**-3``, has no such spelling, so a string holding
|
|
249
|
+
one is returned as it is rather than half rewritten.
|
|
250
|
+
"""
|
|
251
|
+
if NUMERIC_POWER.search(units):
|
|
252
|
+
return units
|
|
253
|
+
return SYMBOL_POWER.sub("", units)
|
|
254
|
+
|
|
255
|
+
|
|
245
256
|
def _available(path: Path) -> str:
|
|
246
257
|
"""Every message as a (shortName, typeOfLevel, level) triple, for a reader error to list."""
|
|
247
258
|
return ", ".join(
|
|
@@ -441,6 +452,10 @@ def open_grib2(
|
|
|
441
452
|
attrs = {
|
|
442
453
|
key: value for key in VARIABLE_KEYS if (value := _get(eccodes, h, key)) is not None
|
|
443
454
|
}
|
|
455
|
+
if isinstance(units := attrs.get("units"), str) and (plain := udunits(units)) != units:
|
|
456
|
+
# The file's own spelling stays beside the rewritten one, as cfgrib keeps it.
|
|
457
|
+
attrs["GRIB_units"] = units
|
|
458
|
+
attrs["units"] = plain
|
|
444
459
|
for label, date_key, time_key in (
|
|
445
460
|
("reference_time", "dataDate", "dataTime"),
|
|
446
461
|
("valid_time", "validityDate", "validityTime"),
|
|
@@ -481,7 +496,7 @@ def open_grib2(
|
|
|
481
496
|
if report := _unmatched_report(options, matched, present):
|
|
482
497
|
if strict:
|
|
483
498
|
raise ValueError(report)
|
|
484
|
-
warnings.warn(report, UserWarning, stacklevel=
|
|
499
|
+
warnings.warn(report, UserWarning, stacklevel=caller_stacklevel())
|
|
485
500
|
variables: dict[str, Any] = {}
|
|
486
501
|
messages: dict[str, dict[str, Any]] = {}
|
|
487
502
|
names = variable_names(
|
|
@@ -497,10 +512,6 @@ def open_grib2(
|
|
|
497
512
|
dataset.latitude.attrs["units"] = "degrees_north"
|
|
498
513
|
dataset.longitude.attrs["units"] = "degrees_east"
|
|
499
514
|
dataset.attrs.update(grid.attrs)
|
|
500
|
-
dataset.attrs["usdata"] = {
|
|
501
|
-
"asset_id": fetched.asset.id,
|
|
502
|
-
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
503
|
-
"messages": messages,
|
|
504
|
-
}
|
|
515
|
+
dataset.attrs["usdata"] = {**source_attrs(fetched), "messages": messages}
|
|
505
516
|
fill_registry_attrs(fetched, dataset)
|
|
506
517
|
return dataset
|
|
@@ -15,7 +15,7 @@ from datetime import UTC, datetime
|
|
|
15
15
|
from importlib import import_module
|
|
16
16
|
from typing import TYPE_CHECKING, Any
|
|
17
17
|
|
|
18
|
-
from usdata.readers import Hurdat2FormatError, MissingReaderDependency
|
|
18
|
+
from usdata.readers import Hurdat2FormatError, MissingReaderDependency, source_attrs
|
|
19
19
|
|
|
20
20
|
if TYPE_CHECKING:
|
|
21
21
|
from usdata._fetch import FetchedAsset
|
|
@@ -174,8 +174,5 @@ def open_hurdat2(fetched: FetchedAsset) -> Any:
|
|
|
174
174
|
data["time"] = pandas.to_datetime(columns["time"], utc=True)
|
|
175
175
|
data.update({name: pandas.array(columns[name], dtype="float64") for name in NUMERIC_COLUMNS})
|
|
176
176
|
frame = pandas.DataFrame(data, columns=COLUMNS)
|
|
177
|
-
frame.attrs["usdata"] =
|
|
178
|
-
"asset_id": fetched.asset.id,
|
|
179
|
-
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
180
|
-
}
|
|
177
|
+
frame.attrs["usdata"] = source_attrs(fetched)
|
|
181
178
|
return frame
|
|
@@ -7,7 +7,7 @@ from pathlib import Path
|
|
|
7
7
|
from typing import TYPE_CHECKING, Any
|
|
8
8
|
|
|
9
9
|
from usdata.inspect import NetcdfVariable
|
|
10
|
-
from usdata.readers import MissingReaderDependency, fill_registry_attrs
|
|
10
|
+
from usdata.readers import MissingReaderDependency, fill_registry_attrs, source_attrs
|
|
11
11
|
|
|
12
12
|
if TYPE_CHECKING:
|
|
13
13
|
from usdata._fetch import FetchedAsset
|
|
@@ -42,10 +42,7 @@ def open_netcdf(fetched: FetchedAsset) -> Any:
|
|
|
42
42
|
xarray.open_dataset(stream, engine="h5netcdf", chunks=None) as dataset,
|
|
43
43
|
):
|
|
44
44
|
dataset.load()
|
|
45
|
-
dataset.attrs["usdata"] =
|
|
46
|
-
"asset_id": fetched.asset.id,
|
|
47
|
-
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
48
|
-
}
|
|
45
|
+
dataset.attrs["usdata"] = source_attrs(fetched)
|
|
49
46
|
fill_registry_attrs(fetched, dataset)
|
|
50
47
|
return dataset
|
|
51
48
|
|
|
@@ -7,7 +7,7 @@ import gzip
|
|
|
7
7
|
from importlib import import_module
|
|
8
8
|
from typing import TYPE_CHECKING, Any
|
|
9
9
|
|
|
10
|
-
from usdata.readers import MissingReaderDependency, RadarDecodeError
|
|
10
|
+
from usdata.readers import MissingReaderDependency, RadarDecodeError, source_attrs
|
|
11
11
|
|
|
12
12
|
# NOAA RDA/RPG ICD 2620002Y, Table XVII-I notes 21 and 30.
|
|
13
13
|
MOMENT_FLAG_COUNTS = {
|
|
@@ -117,8 +117,7 @@ def open_nexrad(fetched: FetchedAsset, *, sweep: int | list[int] | None = None)
|
|
|
117
117
|
masked.encoding = data.encoding.copy()
|
|
118
118
|
node[name] = masked
|
|
119
119
|
radar.attrs["usdata"] = {
|
|
120
|
-
|
|
121
|
-
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
120
|
+
**source_attrs(fetched),
|
|
122
121
|
"sweeps": [name.lstrip("/") for name in radar.groups if name.startswith("/sweep_")],
|
|
123
122
|
}
|
|
124
123
|
return radar
|
|
@@ -500,6 +500,14 @@ class Asset(BaseModel):
|
|
|
500
500
|
checksum: str | None = Field(default=None, description="'<algo>:<hex>', e.g. 'sha256:ab12...'")
|
|
501
501
|
time: TimeRange | None = None
|
|
502
502
|
bbox: BBox | None = None
|
|
503
|
+
properties: dict[str, str] = Field(
|
|
504
|
+
default_factory=dict,
|
|
505
|
+
description=(
|
|
506
|
+
"Facts of the request the bytes do not state, such as the unit system, datum, "
|
|
507
|
+
"or station, keyed in lowercase snake case; readers copy them into "
|
|
508
|
+
"attrs['usdata']['properties'] (ADR 0043)"
|
|
509
|
+
),
|
|
510
|
+
)
|
|
503
511
|
|
|
504
512
|
|
|
505
513
|
class ByteRange(BaseModel):
|
|
@@ -227,6 +227,8 @@ class _CoopsStation(HttpProvider):
|
|
|
227
227
|
protocol=Protocol.HTTP,
|
|
228
228
|
media_type="text/csv",
|
|
229
229
|
time=TimeRange(start=start, end=end),
|
|
230
|
+
# The CSV states none of these, so they travel with the asset (ADR 0043).
|
|
231
|
+
properties={"station": station, **params, "time_zone": "gmt"},
|
|
230
232
|
)
|
|
231
233
|
|
|
232
234
|
def _validate(self, path: Path, asset: Asset) -> None:
|
|
@@ -7,8 +7,10 @@ import gzip
|
|
|
7
7
|
import io
|
|
8
8
|
import os
|
|
9
9
|
import re
|
|
10
|
+
import warnings
|
|
10
11
|
from collections.abc import Mapping
|
|
11
12
|
from importlib import import_module
|
|
13
|
+
from inspect import currentframe
|
|
12
14
|
from pathlib import Path
|
|
13
15
|
from typing import TYPE_CHECKING, Any
|
|
14
16
|
|
|
@@ -226,6 +228,32 @@ def fill_registry_attrs(fetched: FetchedAsset, data: Any) -> None:
|
|
|
226
228
|
data.attrs["usdata"]["registry_attrs"] = filled
|
|
227
229
|
|
|
228
230
|
|
|
231
|
+
def caller_stacklevel() -> int:
|
|
232
|
+
"""Stack level of the first frame outside usdata, so a warning points at the caller."""
|
|
233
|
+
package = Path(__file__).resolve().parent
|
|
234
|
+
frame = currentframe()
|
|
235
|
+
frame = frame.f_back if frame is not None else None
|
|
236
|
+
level = 1
|
|
237
|
+
while frame is not None and Path(frame.f_code.co_filename).resolve().is_relative_to(package):
|
|
238
|
+
level += 1
|
|
239
|
+
frame = frame.f_back
|
|
240
|
+
return level
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def source_attrs(fetched: FetchedAsset) -> dict[str, Any]:
|
|
244
|
+
"""The ``attrs["usdata"]`` every reader starts from: the asset, its request facts, its record.
|
|
245
|
+
|
|
246
|
+
``properties`` are the facts of the request the bytes do not state, such as
|
|
247
|
+
a unit system or datum, always a mapping and empty when the asset records
|
|
248
|
+
none (ADR 0043).
|
|
249
|
+
"""
|
|
250
|
+
return {
|
|
251
|
+
"asset_id": fetched.asset.id,
|
|
252
|
+
"properties": dict(fetched.asset.properties),
|
|
253
|
+
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
|
|
229
257
|
def _local_timestamps(pandas: Any, values: Any) -> Any:
|
|
230
258
|
"""Storm Events local timestamps as tz-naive datetimes, unparsable strings as NaT.
|
|
231
259
|
|
|
@@ -259,15 +287,29 @@ def derive_storm_events_utc(pandas: Any, frame: Any) -> None:
|
|
|
259
287
|
timestamp does not yield an instant gets ``NaT``, and each derived column,
|
|
260
288
|
its source, the rule, that row count, and the labels that gave no offset
|
|
261
289
|
with how many rows carry each are listed under
|
|
262
|
-
``frame.attrs["usdata"]["derived"]``.
|
|
263
|
-
|
|
290
|
+
``frame.attrs["usdata"]["derived"]``.
|
|
291
|
+
|
|
292
|
+
Each UTC column needs only its own local column and ``CZ_TIMEZONE``, so a
|
|
293
|
+
``usecols`` that keeps ``BEGIN_DATE_TIME`` but not ``END_DATE_TIME`` still
|
|
294
|
+
gets ``BEGIN_UTC``. A frame that keeps a local column without
|
|
295
|
+
``CZ_TIMEZONE`` cannot be converted, and says so with a ``UserWarning``
|
|
296
|
+
rather than coming back without the column it would otherwise have.
|
|
264
297
|
|
|
265
298
|
Args:
|
|
266
299
|
pandas: The imported pandas module.
|
|
267
300
|
frame: The DataFrame read from a Storm Events details CSV.
|
|
268
301
|
"""
|
|
269
|
-
|
|
270
|
-
if not
|
|
302
|
+
sources = [source for source in STORM_EVENTS_UTC_COLUMNS if source in frame.columns]
|
|
303
|
+
if not sources:
|
|
304
|
+
return
|
|
305
|
+
if STORM_EVENTS_TIMEZONE_COLUMN not in frame.columns:
|
|
306
|
+
derivable = ", ".join(STORM_EVENTS_UTC_COLUMNS[source] for source in sources)
|
|
307
|
+
warnings.warn(
|
|
308
|
+
f"{', '.join(sources)} read without {STORM_EVENTS_TIMEZONE_COLUMN}, so "
|
|
309
|
+
f"{derivable} cannot be derived; add {STORM_EVENTS_TIMEZONE_COLUMN} to usecols",
|
|
310
|
+
UserWarning,
|
|
311
|
+
stacklevel=caller_stacklevel(),
|
|
312
|
+
)
|
|
271
313
|
return
|
|
272
314
|
labels = frame[STORM_EVENTS_TIMEZONE_COLUMN].astype("string").str.strip()
|
|
273
315
|
stated = pandas.to_numeric(
|
|
@@ -279,7 +321,8 @@ def derive_storm_events_utc(pandas: Any, frame: Any) -> None:
|
|
|
279
321
|
without_offset = labels[hours.isna() & labels.notna()].value_counts()
|
|
280
322
|
unconverted = {str(label): int(count) for label, count in sorted(without_offset.items())}
|
|
281
323
|
derived = []
|
|
282
|
-
for source
|
|
324
|
+
for source in sources:
|
|
325
|
+
column = STORM_EVENTS_UTC_COLUMNS[source]
|
|
283
326
|
local = _local_timestamps(pandas, frame[source])
|
|
284
327
|
frame[column] = (local - offsets).dt.tz_localize("UTC")
|
|
285
328
|
derived.append(
|
|
@@ -520,10 +563,7 @@ def open_csv(
|
|
|
520
563
|
)
|
|
521
564
|
if units:
|
|
522
565
|
frame.attrs["units"] = {name: units[name] for name in frame.columns}
|
|
523
|
-
frame.attrs["usdata"] =
|
|
524
|
-
"asset_id": fetched.asset.id,
|
|
525
|
-
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
526
|
-
}
|
|
566
|
+
frame.attrs["usdata"] = source_attrs(fetched)
|
|
527
567
|
if fetched.asset.dataset_id == STORM_EVENTS_DATASET:
|
|
528
568
|
derive_storm_events_utc(pandas, frame)
|
|
529
569
|
return frame
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|