usdata 0.5.0__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {usdata-0.5.0 → usdata-0.6.0}/PKG-INFO +22 -10
  2. {usdata-0.5.0 → usdata-0.6.0}/README.md +19 -9
  3. {usdata-0.5.0 → usdata-0.6.0}/pyproject.toml +4 -1
  4. {usdata-0.5.0 → usdata-0.6.0}/pyproject.toml.orig +4 -1
  5. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/data/registry.yaml +5 -5
  6. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/fetch.py +22 -0
  7. usdata-0.6.0/src/usdata/readers.py +103 -0
  8. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/__init__.py +0 -0
  9. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/_files.py +0 -0
  10. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/cache.py +0 -0
  11. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/cli/__init__.py +0 -0
  12. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/cli/app.py +0 -0
  13. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/data/nexrad_sites.csv +0 -0
  14. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/data/places.csv +0 -0
  15. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/data/places.sources.json +0 -0
  16. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/manifest.py +0 -0
  17. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/models.py +0 -0
  18. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/protocols/__init__.py +0 -0
  19. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/protocols/erddap.py +0 -0
  20. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/protocols/http.py +0 -0
  21. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/protocols/s3.py +0 -0
  22. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/provenance.py +0 -0
  23. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/providers/__init__.py +0 -0
  24. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/providers/base.py +0 -0
  25. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/providers/noaa/__init__.py +0 -0
  26. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
  27. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/providers/noaa/ghcnd.py +0 -0
  28. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/providers/noaa/nexrad.py +0 -0
  29. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/providers/noaa/sites.py +0 -0
  30. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/providers/usgs/__init__.py +0 -0
  31. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/providers/usgs/daily.py +0 -0
  32. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/pull.py +0 -0
  33. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/py.typed +0 -0
  34. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/query.py +0 -0
  35. {usdata-0.5.0 → usdata-0.6.0}/src/usdata/registry.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: usdata
3
- Version: 0.5.0
3
+ Version: 0.6.0
4
4
  Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
5
5
  Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
6
6
  Author: Jake Van Slyke
@@ -14,9 +14,11 @@ Requires-Dist: httpx>=0.28.1
14
14
  Requires-Dist: pydantic>=2.7
15
15
  Requires-Dist: pyyaml>=6.0
16
16
  Requires-Dist: typer>=0.12
17
+ Requires-Dist: pandas>=3.0 ; extra == 'pandas'
17
18
  Requires-Python: >=3.11
18
19
  Project-URL: Homepage, https://github.com/jakeryderv/usdata
19
20
  Project-URL: Repository, https://github.com/jakeryderv/usdata
21
+ Provides-Extra: pandas
20
22
  Description-Content-Type: text/markdown
21
23
 
22
24
  # usdata
@@ -24,22 +26,22 @@ Description-Content-Type: text/markdown
24
26
  Unified Python SDK and CLI for discovering, fetching, and tracking the
25
27
  provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
26
28
 
27
- > Status: pre-alpha. v0.5 supports GHCN-Daily, NEXRAD Level II, USGS daily
29
+ > Status: pre-alpha. v0.6 supports GHCN-Daily, NEXRAD Level II, USGS daily
28
30
  > values, and CoastWatch SST subsets with provenance, plus Census state/county
29
- > lookup. Other datasets are planned.
31
+ > lookup and optional pandas CSV readers. Other datasets are planned.
30
32
  > See [docs/roadmap.md](docs/roadmap.md).
31
33
 
32
34
  ## Providers
33
35
 
34
36
  <!-- registry:start -->
35
- | Provider | Available | Stub | Planned | Next up (0.6) | Datasets |
37
+ | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
36
38
  |---|---:|---:|---:|---|---|
37
- | [NOAA](docs/providers/noaa.md) | 3 | 0 | 26 | `gfs`, `hrrr`, `oisst`, `etopo` | `ghcn-daily`, `nexrad-level2`, `coastwatch-sst`, +26 planned |
39
+ | [NOAA](docs/providers/noaa.md) | 3 | 0 | 26 | — | `ghcn-daily`, `nexrad-level2`, `coastwatch-sst`, +26 planned |
38
40
  | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
39
41
  | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
40
42
  | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
41
43
  | [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
42
- | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | `gpm-imerg` | +1 planned |
44
+ | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
43
45
  | [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
44
46
 
45
47
  Available datasets are in `code`, stubs in _italics_; planned ones are counted. Available means implemented in this source checkout; consult the [releases](https://github.com/jakeryderv/usdata/releases) for published support. Each provider page has access notes and full dataset details; [docs/roadmap.md](docs/roadmap.md) lists datasets by target version.
@@ -130,6 +132,14 @@ sources:
130
132
  end: 2024-05-31
131
133
  ```
132
134
 
135
+ ## Opening CSV data
136
+
137
+ `FetchedAsset.open()` is available since v0.6 with the optional pandas
138
+ extra (`pip install "usdata[pandas]"`). It reads cached CSV into a DataFrame,
139
+ preserves identifier strings, and keeps CoastWatch units as metadata.
140
+ See the [reader reference](docs/reference/readers.md)
141
+ and [fetch → open → analyze example](examples/sst-analysis/README.md).
142
+
133
143
  ## Development
134
144
 
135
145
  Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
@@ -138,16 +148,18 @@ Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
138
148
  git clone https://github.com/jakeryderv/usdata && cd usdata
139
149
  just setup # install toolchain and dependencies
140
150
  just test # unit tests
141
- just check # format, lint, typecheck, offline tests, generated docs
151
+ just check # format, lint, typecheck, offline tests, generated docs, release notices
152
+ just check-pandas # install the CSV extra and run the same checks
142
153
  just build # build wheel and sdist
143
- just smoke # install and exercise the built wheel outside the checkout
154
+ just smoke # exercise core and pandas wheel installations outside the checkout
144
155
  just run search radar
145
156
  ```
146
157
 
147
158
  Unit tests mechanically block network connections. Integration tests that hit
148
159
  live services run with `just test-integration`. CI checks Python 3.11 and 3.14 on
149
- Linux and smoke-tests the installed wheel on Linux, macOS, and Windows. The
150
- full unit and live-service suites currently run on Linux.
160
+ Linux, both with and without pandas, and smoke-tests both installed-wheel
161
+ profiles on Linux, macOS, and Windows. The full unit and live-service suites currently run on Linux. `just setup` restores
162
+ a core-only development environment; `just check-pandas` installs the extra.
151
163
 
152
164
  Releases: `just release minor` opens a version-bump PR; merging it publishes
153
165
  to PyPI and creates the tag and GitHub release. See
@@ -3,22 +3,22 @@
3
3
  Unified Python SDK and CLI for discovering, fetching, and tracking the
4
4
  provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
5
5
 
6
- > Status: pre-alpha. v0.5 supports GHCN-Daily, NEXRAD Level II, USGS daily
6
+ > Status: pre-alpha. v0.6 supports GHCN-Daily, NEXRAD Level II, USGS daily
7
7
  > values, and CoastWatch SST subsets with provenance, plus Census state/county
8
- > lookup. Other datasets are planned.
8
+ > lookup and optional pandas CSV readers. Other datasets are planned.
9
9
  > See [docs/roadmap.md](docs/roadmap.md).
10
10
 
11
11
  ## Providers
12
12
 
13
13
  <!-- registry:start -->
14
- | Provider | Available | Stub | Planned | Next up (0.6) | Datasets |
14
+ | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
15
15
  |---|---:|---:|---:|---|---|
16
- | [NOAA](docs/providers/noaa.md) | 3 | 0 | 26 | `gfs`, `hrrr`, `oisst`, `etopo` | `ghcn-daily`, `nexrad-level2`, `coastwatch-sst`, +26 planned |
16
+ | [NOAA](docs/providers/noaa.md) | 3 | 0 | 26 | — | `ghcn-daily`, `nexrad-level2`, `coastwatch-sst`, +26 planned |
17
17
  | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
18
18
  | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
19
19
  | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
20
20
  | [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
21
- | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | `gpm-imerg` | +1 planned |
21
+ | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
22
22
  | [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
23
23
 
24
24
  Available datasets are in `code`, stubs in _italics_; planned ones are counted. Available means implemented in this source checkout; consult the [releases](https://github.com/jakeryderv/usdata/releases) for published support. Each provider page has access notes and full dataset details; [docs/roadmap.md](docs/roadmap.md) lists datasets by target version.
@@ -109,6 +109,14 @@ sources:
109
109
  end: 2024-05-31
110
110
  ```
111
111
 
112
+ ## Opening CSV data
113
+
114
+ `FetchedAsset.open()` is available since v0.6 with the optional pandas
115
+ extra (`pip install "usdata[pandas]"`). It reads cached CSV into a DataFrame,
116
+ preserves identifier strings, and keeps CoastWatch units as metadata.
117
+ See the [reader reference](docs/reference/readers.md)
118
+ and [fetch → open → analyze example](examples/sst-analysis/README.md).
119
+
112
120
  ## Development
113
121
 
114
122
  Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
@@ -117,16 +125,18 @@ Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
117
125
  git clone https://github.com/jakeryderv/usdata && cd usdata
118
126
  just setup # install toolchain and dependencies
119
127
  just test # unit tests
120
- just check # format, lint, typecheck, offline tests, generated docs
128
+ just check # format, lint, typecheck, offline tests, generated docs, release notices
129
+ just check-pandas # install the CSV extra and run the same checks
121
130
  just build # build wheel and sdist
122
- just smoke # install and exercise the built wheel outside the checkout
131
+ just smoke # exercise core and pandas wheel installations outside the checkout
123
132
  just run search radar
124
133
  ```
125
134
 
126
135
  Unit tests mechanically block network connections. Integration tests that hit
127
136
  live services run with `just test-integration`. CI checks Python 3.11 and 3.14 on
128
- Linux and smoke-tests the installed wheel on Linux, macOS, and Windows. The
129
- full unit and live-service suites currently run on Linux.
137
+ Linux, both with and without pandas, and smoke-tests both installed-wheel
138
+ profiles on Linux, macOS, and Windows. The full unit and live-service suites currently run on Linux. `just setup` restores
139
+ a core-only development environment; `just check-pandas` installs the extra.
130
140
 
131
141
  Releases: `just release minor` opens a version-bump PR; merging it publishes
132
142
  to PyPI and creates the tag and GitHub release. See
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.5.0"
3
+ version = "0.6.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -30,6 +30,9 @@ dependencies = [
30
30
  name = "Jake Van Slyke"
31
31
  email = "jakervanslyke@gmail.com"
32
32
 
33
+ [project.optional-dependencies]
34
+ pandas = ["pandas>=3.0"]
35
+
33
36
  [project.urls]
34
37
  Homepage = "https://github.com/jakeryderv/usdata"
35
38
  Repository = "https://github.com/jakeryderv/usdata"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.5.0"
3
+ version = "0.6.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -22,6 +22,9 @@ dependencies = [
22
22
  "typer>=0.12",
23
23
  ]
24
24
 
25
+ [project.optional-dependencies]
26
+ pandas = ["pandas>=3.0"]
27
+
25
28
  [project.urls]
26
29
  Homepage = "https://github.com/jakeryderv/usdata"
27
30
  Repository = "https://github.com/jakeryderv/usdata"
@@ -293,7 +293,7 @@ datasets:
293
293
  provider: noaa
294
294
  status: planned
295
295
  domain: weather-models
296
- target: "0.6"
296
+ target: later
297
297
  title: HRRR Forecast Model Output
298
298
  description: >-
299
299
  High-Resolution Rapid Refresh 3 km hourly forecasts in GRIB2 from the
@@ -310,7 +310,7 @@ datasets:
310
310
  provider: noaa
311
311
  status: planned
312
312
  domain: weather-models
313
- target: "0.6"
313
+ target: later
314
314
  title: GFS Forecast Model Output
315
315
  description: >-
316
316
  Global Forecast System output in GRIB2 from the public noaa-gfs-bdp-pds
@@ -342,7 +342,7 @@ datasets:
342
342
  provider: noaa
343
343
  status: planned
344
344
  domain: ocean-physics
345
- target: "0.6"
345
+ target: later
346
346
  title: OISST Daily Sea Surface Temperature
347
347
  description: >-
348
348
  Optimum Interpolation SST v2.1: daily global 0.25 degree analysis since
@@ -375,7 +375,7 @@ datasets:
375
375
  provider: noaa
376
376
  status: planned
377
377
  domain: bathymetry
378
- target: "0.6"
378
+ target: later
379
379
  title: ETOPO 2022 Global Relief
380
380
  description: >-
381
381
  Global topography and bathymetry at 15, 30, and 60 arc-seconds as
@@ -591,7 +591,7 @@ datasets:
591
591
  provider: nasa
592
592
  status: planned
593
593
  domain: weather-satellites
594
- target: "0.6"
594
+ target: later
595
595
  title: GPM IMERG Precipitation
596
596
  description: >-
597
597
  Global half-hourly and daily merged satellite precipitation estimates.
@@ -3,6 +3,7 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  from pathlib import Path
6
+ from typing import Any
6
7
 
7
8
  from pydantic import BaseModel
8
9
 
@@ -25,6 +26,27 @@ class FetchedAsset(BaseModel):
25
26
  provenance: Provenance
26
27
  from_cache: bool
27
28
 
29
+ def open(
30
+ self,
31
+ *,
32
+ reader: str | None = None,
33
+ dtype: dict[str, str] | None = None,
34
+ parse_dates: list[str] | None = None,
35
+ usecols: list[str] | None = None,
36
+ nrows: int | None = None,
37
+ ) -> Any:
38
+ """Open this local CSV as a DataFrame; requires the ``pandas`` extra.
39
+
40
+ ERDDAP units are kept in ``frame.attrs["units"]`` and source provenance
41
+ in ``frame.attrs["usdata"]``. See ``usdata.readers.open_asset`` for options.
42
+ Cached files and provenance sidecars are never changed.
43
+ """
44
+ from usdata.readers import open_asset
45
+
46
+ return open_asset(
47
+ self, reader=reader, dtype=dtype, parse_dates=parse_dates, usecols=usecols, nrows=nrows
48
+ )
49
+
28
50
 
29
51
  def _fetch_asset(
30
52
  dataset: Dataset,
@@ -0,0 +1,103 @@
1
+ """Optional readers for local fetched files; never fetch or modify cached bytes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import csv
6
+ from importlib import import_module
7
+ from typing import TYPE_CHECKING, Any
8
+
9
+ from usdata.models import Protocol
10
+
11
+ if TYPE_CHECKING:
12
+ from usdata.fetch import FetchedAsset
13
+
14
+ CSV_MEDIA_TYPES = {"text/csv", "application/csv"}
15
+ IDENTIFIER_COLUMNS = {
16
+ "station",
17
+ "station_id",
18
+ "site_no",
19
+ "monitoring_location_id",
20
+ "parameter_code",
21
+ "statistic_id",
22
+ }
23
+
24
+
25
+ class MissingReaderDependency(ImportError):
26
+ """The optional dependency required to open an asset is not installed."""
27
+
28
+
29
+ class UnsupportedFormat(ValueError):
30
+ """No reader is implemented for this asset's format."""
31
+
32
+
33
+ def open_asset(
34
+ fetched: FetchedAsset,
35
+ *,
36
+ reader: str | None = None,
37
+ dtype: dict[str, str] | None = None,
38
+ parse_dates: list[str] | None = None,
39
+ usecols: list[str] | None = None,
40
+ nrows: int | None = None,
41
+ ) -> Any:
42
+ """Read a local CSV into a pandas DataFrame, retaining units and provenance.
43
+
44
+ Infer ``csv`` or ``erddap-csv`` from media type and protocol, or use an
45
+ explicit reader for ambiguous metadata. Identifier columns default to pandas
46
+ strings; explicit dtype entries override those defaults. Dates remain strings
47
+ unless named in parse_dates. No checksum verification or downloading occurs.
48
+ """
49
+ if reader is None:
50
+ media_type = (fetched.asset.media_type or "").split(";", 1)[0].strip().lower()
51
+ if media_type not in CSV_MEDIA_TYPES:
52
+ raise UnsupportedFormat(
53
+ f"no reader for {fetched.asset.media_type!r}; supported formats are CSV and "
54
+ "ERDDAP CSV. For a known CSV with ambiguous metadata, pass reader='csv' "
55
+ "or reader='erddap-csv'; otherwise use fetched.path with a format-specific reader"
56
+ )
57
+ reader = "erddap-csv" if fetched.asset.protocol is Protocol.ERDDAP else "csv"
58
+ if reader not in {"csv", "erddap-csv"}:
59
+ raise UnsupportedFormat(f"unsupported reader {reader!r}; use 'csv' or 'erddap-csv'")
60
+ try:
61
+ pandas = import_module("pandas")
62
+ except ModuleNotFoundError as error:
63
+ if error.name != "pandas":
64
+ raise
65
+ raise MissingReaderDependency(
66
+ 'CSV reading requires pandas; install it with: pip install "usdata[pandas]" '
67
+ '(or uv add "usdata[pandas]")'
68
+ ) from error
69
+
70
+ # Pass a file object to pandas: reading a fetched asset is strictly local.
71
+ with fetched.path.open(encoding="utf-8-sig", newline="") as stream:
72
+ records = csv.reader(stream)
73
+ columns = next(records, [])
74
+ if (
75
+ not columns
76
+ or any(not column for column in columns)
77
+ or len(set(columns)) != len(columns)
78
+ ):
79
+ raise ValueError("CSV must have a non-empty header with unique column names")
80
+ units = {}
81
+ if reader == "erddap-csv":
82
+ values = next(records, [])
83
+ if len(values) != len(columns):
84
+ raise ValueError("ERDDAP CSV must have a units row matching the header")
85
+ units = dict(zip(columns, values, strict=True))
86
+ types = {name: "string" for name in columns if name.casefold() in IDENTIFIER_COLUMNS}
87
+ types.update(dtype or {})
88
+ frame = pandas.read_csv(
89
+ stream,
90
+ header=None,
91
+ names=columns,
92
+ dtype=types,
93
+ parse_dates=parse_dates,
94
+ usecols=usecols,
95
+ nrows=nrows,
96
+ )
97
+ if units:
98
+ frame.attrs["units"] = {name: units[name] for name in frame.columns}
99
+ frame.attrs["usdata"] = {
100
+ "asset_id": fetched.asset.id,
101
+ "provenance": fetched.provenance.model_dump(mode="json"),
102
+ }
103
+ return frame
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes