usdata 0.27.0__tar.gz → 0.28.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. {usdata-0.27.0 → usdata-0.28.0}/PKG-INFO +1 -1
  2. {usdata-0.27.0 → usdata-0.28.0}/pyproject.toml +1 -1
  3. {usdata-0.27.0 → usdata-0.28.0}/pyproject.toml.orig +1 -1
  4. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_aqs.py +2 -6
  5. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_grib.py +31 -20
  6. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_hurdat2.py +2 -5
  7. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_netcdf.py +2 -5
  8. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_radar.py +2 -3
  9. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/models.py +8 -0
  10. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/coops.py +2 -0
  11. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/ghcnd.py +2 -0
  12. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/normals.py +1 -0
  13. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/readers.py +49 -9
  14. {usdata-0.27.0 → usdata-0.28.0}/README.md +0 -0
  15. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/__init__.py +0 -0
  16. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/__main__.py +0 -0
  17. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_fetch.py +0 -0
  18. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_files.py +0 -0
  19. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/_progress.py +0 -0
  20. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cache.py +0 -0
  21. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cache_ops.py +0 -0
  22. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cite.py +0 -0
  23. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/__init__.py +0 -0
  24. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/app.py +0 -0
  25. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/cache.py +0 -0
  26. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/cite.py +0 -0
  27. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/doctor.py +0 -0
  28. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/inspect.py +0 -0
  29. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/cli/progress.py +0 -0
  30. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/data/nexrad_sites.csv +0 -0
  31. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/data/places.csv +0 -0
  32. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/data/places.sources.json +0 -0
  33. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/data/registry.yaml +0 -0
  34. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/doctor.py +0 -0
  35. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/inspect.py +0 -0
  36. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/manifest.py +0 -0
  37. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/mirror.py +0 -0
  38. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/protocols/__init__.py +0 -0
  39. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/protocols/erddap.py +0 -0
  40. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/protocols/http.py +0 -0
  41. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/protocols/listing.py +0 -0
  42. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/protocols/s3.py +0 -0
  43. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/provenance.py +0 -0
  44. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/__init__.py +0 -0
  45. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/base.py +0 -0
  46. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/credentials.py +0 -0
  47. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/epa/__init__.py +0 -0
  48. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/epa/aqs.py +0 -0
  49. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/fema/__init__.py +0 -0
  50. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/fema/declarations.py +0 -0
  51. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/http.py +0 -0
  52. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/__init__.py +0 -0
  53. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
  54. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/gfs.py +0 -0
  55. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/glm.py +0 -0
  56. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/goes.py +0 -0
  57. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/grib_index.py +0 -0
  58. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/gsom.py +0 -0
  59. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/gsoy.py +0 -0
  60. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/hrrr.py +0 -0
  61. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/hurdat2.py +0 -0
  62. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/ibtracs.py +0 -0
  63. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/lcd.py +0 -0
  64. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/mrms.py +0 -0
  65. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/nbm.py +0 -0
  66. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/nexrad.py +0 -0
  67. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/nexrad_level3.py +0 -0
  68. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/nws_vtec.py +0 -0
  69. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/rap.py +0 -0
  70. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/sites.py +0 -0
  71. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/spc.py +0 -0
  72. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/noaa/storm_events.py +0 -0
  73. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/params.py +0 -0
  74. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/usgs/__init__.py +0 -0
  75. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/usgs/daily.py +0 -0
  76. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/providers/usgs/earthquakes.py +0 -0
  77. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/pull.py +0 -0
  78. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/py.typed +0 -0
  79. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/query.py +0 -0
  80. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/registry.py +0 -0
  81. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/selection.py +0 -0
  82. {usdata-0.27.0 → usdata-0.28.0}/src/usdata/testing.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: usdata
3
- Version: 0.27.0
3
+ Version: 0.28.0
4
4
  Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
5
5
  Keywords: noaa,usgs,open-data,scientific-data,provenance,reproducible-research,weather,climate,meteorology
6
6
  Author: Jake Van Slyke
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.27.0"
3
+ version = "0.28.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.27.0"
3
+ version = "0.28.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -13,7 +13,7 @@ import json
13
13
  from importlib import import_module
14
14
  from typing import TYPE_CHECKING, Any
15
15
 
16
- from usdata.readers import MissingReaderDependency
16
+ from usdata.readers import MissingReaderDependency, source_attrs
17
17
 
18
18
  if TYPE_CHECKING:
19
19
  from usdata._fetch import FetchedAsset
@@ -43,9 +43,5 @@ def open_aqs(fetched: FetchedAsset) -> Any:
43
43
  for column in DATE_COLUMNS:
44
44
  if column in frame:
45
45
  frame[column] = pandas.to_datetime(frame[column], format="%Y-%m-%d")
46
- frame.attrs["usdata"] = {
47
- "asset_id": fetched.asset.id,
48
- "provenance": fetched.provenance.model_dump(mode="json"),
49
- "header": body.get("Header", []),
50
- }
46
+ frame.attrs["usdata"] = {**source_attrs(fetched), "header": body.get("Header", [])}
51
47
  return frame
@@ -22,12 +22,16 @@ from collections.abc import Iterator, Mapping, Sequence
22
22
  from contextlib import contextmanager
23
23
  from datetime import UTC, datetime
24
24
  from importlib import import_module
25
- from inspect import currentframe
26
25
  from pathlib import Path
27
26
  from typing import TYPE_CHECKING, Any, NamedTuple
28
27
 
29
28
  from usdata.inspect import GribMessage
30
- from usdata.readers import MissingReaderDependency, fill_registry_attrs
29
+ from usdata.readers import (
30
+ MissingReaderDependency,
31
+ caller_stacklevel,
32
+ fill_registry_attrs,
33
+ source_attrs,
34
+ )
31
35
 
32
36
  if TYPE_CHECKING:
33
37
  from usdata._fetch import FetchedAsset
@@ -40,6 +44,11 @@ LIBRARY_HINT = (
40
44
  )
41
45
  MRMS_NAME = re.compile(r"^MRMS_(?P<product>.+?)_\d{2}\.\d{2}_\d{8}-\d{6}\.grib2(?:\.gz)?$")
42
46
  INVENTORY_KEYS = ("shortName", "name", "typeOfLevel", "level", "step", "units")
47
+ SYMBOL_POWER = re.compile(r"(?<=[A-Za-z])\*\*")
48
+ """A power written after a unit symbol, as ecCodes writes ``kg**-1``."""
49
+ NUMERIC_POWER = re.compile(r"(?<![A-Za-z])\*\*")
50
+ """A power written after anything else, such as the ``10**-3`` of a scale factor."""
51
+
43
52
  VARIABLE_KEYS = (
44
53
  "name",
45
54
  "units",
@@ -201,18 +210,6 @@ def _unmatched_report(
201
210
  return " ".join(parts)
202
211
 
203
212
 
204
- def _caller_stacklevel() -> int:
205
- """Stack level of the first frame outside usdata, so a warning points at the caller."""
206
- package = Path(__file__).resolve().parent
207
- frame = currentframe()
208
- frame = frame.f_back if frame is not None else None
209
- level = 1
210
- while frame is not None and Path(frame.f_code.co_filename).resolve().is_relative_to(package):
211
- level += 1
212
- frame = frame.f_back
213
- return level
214
-
215
-
216
213
  def _shape(eccodes: Any, h: int) -> tuple[int, int] | None:
217
214
  """A message's grid rows and columns under either key pair, or None when neither is set."""
218
215
  rows, cols = _get(eccodes, h, "Nj", int), _get(eccodes, h, "Ni", int)
@@ -242,6 +239,20 @@ def inventory(path: Path) -> list[GribMessage]:
242
239
  return messages
243
240
 
244
241
 
242
+ def udunits(units: str) -> str:
243
+ """EcCodes ``units`` in the UDUNITS notation CF metadata uses, or unchanged.
244
+
245
+ ecCodes writes powers as ``**``: ``J kg**-1``, ``m**2 s**-2``. UDUNITS and
246
+ CF write them as a trailing signed integer, ``J kg-1`` and ``m2 s-2``,
247
+ which is what xarray-based tools expect. Only the notation changes. A power
248
+ of a number, such as ``10**-3``, has no such spelling, so a string holding
249
+ one is returned as it is rather than half rewritten.
250
+ """
251
+ if NUMERIC_POWER.search(units):
252
+ return units
253
+ return SYMBOL_POWER.sub("", units)
254
+
255
+
245
256
  def _available(path: Path) -> str:
246
257
  """Every message as a (shortName, typeOfLevel, level) triple, for a reader error to list."""
247
258
  return ", ".join(
@@ -441,6 +452,10 @@ def open_grib2(
441
452
  attrs = {
442
453
  key: value for key in VARIABLE_KEYS if (value := _get(eccodes, h, key)) is not None
443
454
  }
455
+ if isinstance(units := attrs.get("units"), str) and (plain := udunits(units)) != units:
456
+ # The file's own spelling stays beside the rewritten one, as cfgrib keeps it.
457
+ attrs["GRIB_units"] = units
458
+ attrs["units"] = plain
444
459
  for label, date_key, time_key in (
445
460
  ("reference_time", "dataDate", "dataTime"),
446
461
  ("valid_time", "validityDate", "validityTime"),
@@ -481,7 +496,7 @@ def open_grib2(
481
496
  if report := _unmatched_report(options, matched, present):
482
497
  if strict:
483
498
  raise ValueError(report)
484
- warnings.warn(report, UserWarning, stacklevel=_caller_stacklevel())
499
+ warnings.warn(report, UserWarning, stacklevel=caller_stacklevel())
485
500
  variables: dict[str, Any] = {}
486
501
  messages: dict[str, dict[str, Any]] = {}
487
502
  names = variable_names(
@@ -497,10 +512,6 @@ def open_grib2(
497
512
  dataset.latitude.attrs["units"] = "degrees_north"
498
513
  dataset.longitude.attrs["units"] = "degrees_east"
499
514
  dataset.attrs.update(grid.attrs)
500
- dataset.attrs["usdata"] = {
501
- "asset_id": fetched.asset.id,
502
- "provenance": fetched.provenance.model_dump(mode="json"),
503
- "messages": messages,
504
- }
515
+ dataset.attrs["usdata"] = {**source_attrs(fetched), "messages": messages}
505
516
  fill_registry_attrs(fetched, dataset)
506
517
  return dataset
@@ -15,7 +15,7 @@ from datetime import UTC, datetime
15
15
  from importlib import import_module
16
16
  from typing import TYPE_CHECKING, Any
17
17
 
18
- from usdata.readers import Hurdat2FormatError, MissingReaderDependency
18
+ from usdata.readers import Hurdat2FormatError, MissingReaderDependency, source_attrs
19
19
 
20
20
  if TYPE_CHECKING:
21
21
  from usdata._fetch import FetchedAsset
@@ -174,8 +174,5 @@ def open_hurdat2(fetched: FetchedAsset) -> Any:
174
174
  data["time"] = pandas.to_datetime(columns["time"], utc=True)
175
175
  data.update({name: pandas.array(columns[name], dtype="float64") for name in NUMERIC_COLUMNS})
176
176
  frame = pandas.DataFrame(data, columns=COLUMNS)
177
- frame.attrs["usdata"] = {
178
- "asset_id": fetched.asset.id,
179
- "provenance": fetched.provenance.model_dump(mode="json"),
180
- }
177
+ frame.attrs["usdata"] = source_attrs(fetched)
181
178
  return frame
@@ -7,7 +7,7 @@ from pathlib import Path
7
7
  from typing import TYPE_CHECKING, Any
8
8
 
9
9
  from usdata.inspect import NetcdfVariable
10
- from usdata.readers import MissingReaderDependency, fill_registry_attrs
10
+ from usdata.readers import MissingReaderDependency, fill_registry_attrs, source_attrs
11
11
 
12
12
  if TYPE_CHECKING:
13
13
  from usdata._fetch import FetchedAsset
@@ -42,10 +42,7 @@ def open_netcdf(fetched: FetchedAsset) -> Any:
42
42
  xarray.open_dataset(stream, engine="h5netcdf", chunks=None) as dataset,
43
43
  ):
44
44
  dataset.load()
45
- dataset.attrs["usdata"] = {
46
- "asset_id": fetched.asset.id,
47
- "provenance": fetched.provenance.model_dump(mode="json"),
48
- }
45
+ dataset.attrs["usdata"] = source_attrs(fetched)
49
46
  fill_registry_attrs(fetched, dataset)
50
47
  return dataset
51
48
 
@@ -7,7 +7,7 @@ import gzip
7
7
  from importlib import import_module
8
8
  from typing import TYPE_CHECKING, Any
9
9
 
10
- from usdata.readers import MissingReaderDependency, RadarDecodeError
10
+ from usdata.readers import MissingReaderDependency, RadarDecodeError, source_attrs
11
11
 
12
12
  # NOAA RDA/RPG ICD 2620002Y, Table XVII-I notes 21 and 30.
13
13
  MOMENT_FLAG_COUNTS = {
@@ -117,8 +117,7 @@ def open_nexrad(fetched: FetchedAsset, *, sweep: int | list[int] | None = None)
117
117
  masked.encoding = data.encoding.copy()
118
118
  node[name] = masked
119
119
  radar.attrs["usdata"] = {
120
- "asset_id": fetched.asset.id,
121
- "provenance": fetched.provenance.model_dump(mode="json"),
120
+ **source_attrs(fetched),
122
121
  "sweeps": [name.lstrip("/") for name in radar.groups if name.startswith("/sweep_")],
123
122
  }
124
123
  return radar
@@ -500,6 +500,14 @@ class Asset(BaseModel):
500
500
  checksum: str | None = Field(default=None, description="'<algo>:<hex>', e.g. 'sha256:ab12...'")
501
501
  time: TimeRange | None = None
502
502
  bbox: BBox | None = None
503
+ properties: dict[str, str] = Field(
504
+ default_factory=dict,
505
+ description=(
506
+ "Facts of the request the bytes do not state, such as the unit system, datum, "
507
+ "or station, keyed in lowercase snake case; readers copy them into "
508
+ "attrs['usdata']['properties'] (ADR 0043)"
509
+ ),
510
+ )
503
511
 
504
512
 
505
513
  class ByteRange(BaseModel):
@@ -227,6 +227,8 @@ class _CoopsStation(HttpProvider):
227
227
  protocol=Protocol.HTTP,
228
228
  media_type="text/csv",
229
229
  time=TimeRange(start=start, end=end),
230
+ # The CSV states none of these, so they travel with the asset (ADR 0043).
231
+ properties={"station": station, **params, "time_zone": "gmt"},
230
232
  )
231
233
 
232
234
  def _validate(self, path: Path, asset: Asset) -> None:
@@ -163,6 +163,8 @@ class GhcnDaily(HttpProvider):
163
163
  media_type="text/csv",
164
164
  time=window,
165
165
  bbox=query.bbox,
166
+ # The CSV has no units row; its stations are in a column.
167
+ properties={"units": params.units},
166
168
  )
167
169
  )
168
170
  return assets
@@ -117,6 +117,7 @@ class ClimateNormals(GhcnDaily):
117
117
  media_type="text/csv",
118
118
  time=NORMALS_PERIOD,
119
119
  bbox=query.bbox,
120
+ properties={"units": params.units},
120
121
  )
121
122
  )
122
123
  return assets
@@ -7,8 +7,10 @@ import gzip
7
7
  import io
8
8
  import os
9
9
  import re
10
+ import warnings
10
11
  from collections.abc import Mapping
11
12
  from importlib import import_module
13
+ from inspect import currentframe
12
14
  from pathlib import Path
13
15
  from typing import TYPE_CHECKING, Any
14
16
 
@@ -226,6 +228,32 @@ def fill_registry_attrs(fetched: FetchedAsset, data: Any) -> None:
226
228
  data.attrs["usdata"]["registry_attrs"] = filled
227
229
 
228
230
 
231
+ def caller_stacklevel() -> int:
232
+ """Stack level of the first frame outside usdata, so a warning points at the caller."""
233
+ package = Path(__file__).resolve().parent
234
+ frame = currentframe()
235
+ frame = frame.f_back if frame is not None else None
236
+ level = 1
237
+ while frame is not None and Path(frame.f_code.co_filename).resolve().is_relative_to(package):
238
+ level += 1
239
+ frame = frame.f_back
240
+ return level
241
+
242
+
243
+ def source_attrs(fetched: FetchedAsset) -> dict[str, Any]:
244
+ """The ``attrs["usdata"]`` every reader starts from: the asset, its request facts, its record.
245
+
246
+ ``properties`` are the facts of the request the bytes do not state, such as
247
+ a unit system or datum, always a mapping and empty when the asset records
248
+ none (ADR 0043).
249
+ """
250
+ return {
251
+ "asset_id": fetched.asset.id,
252
+ "properties": dict(fetched.asset.properties),
253
+ "provenance": fetched.provenance.model_dump(mode="json"),
254
+ }
255
+
256
+
229
257
  def _local_timestamps(pandas: Any, values: Any) -> Any:
230
258
  """Storm Events local timestamps as tz-naive datetimes, unparsable strings as NaT.
231
259
 
@@ -259,15 +287,29 @@ def derive_storm_events_utc(pandas: Any, frame: Any) -> None:
259
287
  timestamp does not yield an instant gets ``NaT``, and each derived column,
260
288
  its source, the rule, that row count, and the labels that gave no offset
261
289
  with how many rows carry each are listed under
262
- ``frame.attrs["usdata"]["derived"]``. A frame missing any of the three
263
- source columns is left alone.
290
+ ``frame.attrs["usdata"]["derived"]``.
291
+
292
+ Each UTC column needs only its own local column and ``CZ_TIMEZONE``, so a
293
+ ``usecols`` that keeps ``BEGIN_DATE_TIME`` but not ``END_DATE_TIME`` still
294
+ gets ``BEGIN_UTC``. A frame that keeps a local column without
295
+ ``CZ_TIMEZONE`` cannot be converted, and says so with a ``UserWarning``
296
+ rather than coming back without the column it would otherwise have.
264
297
 
265
298
  Args:
266
299
  pandas: The imported pandas module.
267
300
  frame: The DataFrame read from a Storm Events details CSV.
268
301
  """
269
- required = {*STORM_EVENTS_UTC_COLUMNS, STORM_EVENTS_TIMEZONE_COLUMN}
270
- if not required.issubset(frame.columns):
302
+ sources = [source for source in STORM_EVENTS_UTC_COLUMNS if source in frame.columns]
303
+ if not sources:
304
+ return
305
+ if STORM_EVENTS_TIMEZONE_COLUMN not in frame.columns:
306
+ derivable = ", ".join(STORM_EVENTS_UTC_COLUMNS[source] for source in sources)
307
+ warnings.warn(
308
+ f"{', '.join(sources)} read without {STORM_EVENTS_TIMEZONE_COLUMN}, so "
309
+ f"{derivable} cannot be derived; add {STORM_EVENTS_TIMEZONE_COLUMN} to usecols",
310
+ UserWarning,
311
+ stacklevel=caller_stacklevel(),
312
+ )
271
313
  return
272
314
  labels = frame[STORM_EVENTS_TIMEZONE_COLUMN].astype("string").str.strip()
273
315
  stated = pandas.to_numeric(
@@ -279,7 +321,8 @@ def derive_storm_events_utc(pandas: Any, frame: Any) -> None:
279
321
  without_offset = labels[hours.isna() & labels.notna()].value_counts()
280
322
  unconverted = {str(label): int(count) for label, count in sorted(without_offset.items())}
281
323
  derived = []
282
- for source, column in STORM_EVENTS_UTC_COLUMNS.items():
324
+ for source in sources:
325
+ column = STORM_EVENTS_UTC_COLUMNS[source]
283
326
  local = _local_timestamps(pandas, frame[source])
284
327
  frame[column] = (local - offsets).dt.tz_localize("UTC")
285
328
  derived.append(
@@ -520,10 +563,7 @@ def open_csv(
520
563
  )
521
564
  if units:
522
565
  frame.attrs["units"] = {name: units[name] for name in frame.columns}
523
- frame.attrs["usdata"] = {
524
- "asset_id": fetched.asset.id,
525
- "provenance": fetched.provenance.model_dump(mode="json"),
526
- }
566
+ frame.attrs["usdata"] = source_attrs(fetched)
527
567
  if fetched.asset.dataset_id == STORM_EVENTS_DATASET:
528
568
  derive_storm_events_utc(pandas, frame)
529
569
  return frame
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes