usdata 0.5.0__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. {usdata-0.5.0 → usdata-0.7.0}/PKG-INFO +34 -11
  2. {usdata-0.5.0 → usdata-0.7.0}/README.md +31 -10
  3. {usdata-0.5.0 → usdata-0.7.0}/pyproject.toml +4 -1
  4. {usdata-0.5.0 → usdata-0.7.0}/pyproject.toml.orig +4 -1
  5. usdata-0.7.0/src/usdata/_progress.py +62 -0
  6. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/cli/app.py +10 -2
  7. usdata-0.7.0/src/usdata/cli/progress.py +89 -0
  8. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/data/registry.yaml +11 -8
  9. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/fetch.py +27 -1
  10. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/protocols/http.py +18 -1
  11. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/noaa/ghcnd.py +25 -4
  12. usdata-0.7.0/src/usdata/providers/noaa/gsom.py +49 -0
  13. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/pull.py +9 -1
  14. usdata-0.7.0/src/usdata/readers.py +103 -0
  15. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/__init__.py +0 -0
  16. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/_files.py +0 -0
  17. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/cache.py +0 -0
  18. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/cli/__init__.py +0 -0
  19. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/data/nexrad_sites.csv +0 -0
  20. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/data/places.csv +0 -0
  21. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/data/places.sources.json +0 -0
  22. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/manifest.py +0 -0
  23. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/models.py +0 -0
  24. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/protocols/__init__.py +0 -0
  25. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/protocols/erddap.py +0 -0
  26. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/protocols/s3.py +0 -0
  27. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/provenance.py +0 -0
  28. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/__init__.py +0 -0
  29. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/base.py +0 -0
  30. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/noaa/__init__.py +0 -0
  31. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
  32. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/noaa/nexrad.py +0 -0
  33. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/noaa/sites.py +0 -0
  34. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/usgs/__init__.py +0 -0
  35. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/usgs/daily.py +0 -0
  36. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/py.typed +0 -0
  37. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/query.py +0 -0
  38. {usdata-0.5.0 → usdata-0.7.0}/src/usdata/registry.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: usdata
3
- Version: 0.5.0
3
+ Version: 0.7.0
4
4
  Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
5
5
  Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
6
6
  Author: Jake Van Slyke
@@ -14,9 +14,11 @@ Requires-Dist: httpx>=0.28.1
14
14
  Requires-Dist: pydantic>=2.7
15
15
  Requires-Dist: pyyaml>=6.0
16
16
  Requires-Dist: typer>=0.12
17
+ Requires-Dist: pandas>=3.0 ; extra == 'pandas'
17
18
  Requires-Python: >=3.11
18
19
  Project-URL: Homepage, https://github.com/jakeryderv/usdata
19
20
  Project-URL: Repository, https://github.com/jakeryderv/usdata
21
+ Provides-Extra: pandas
20
22
  Description-Content-Type: text/markdown
21
23
 
22
24
  # usdata
@@ -24,22 +26,23 @@ Description-Content-Type: text/markdown
24
26
  Unified Python SDK and CLI for discovering, fetching, and tracking the
25
27
  provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
26
28
 
27
- > Status: pre-alpha. v0.5 supports GHCN-Daily, NEXRAD Level II, USGS daily
28
- > values, and CoastWatch SST subsets with provenance, plus Census state/county
29
- > lookup. Other datasets are planned.
29
+ > Status: pre-alpha. v0.7 supports GHCN-Daily, GSOM monthly summaries,
30
+ > NEXRAD Level II, USGS daily values, and CoastWatch SST subsets with provenance,
31
+ > plus Census state/county lookup, optional pandas CSV readers, and terminal
32
+ > download progress. Other datasets are planned.
30
33
  > See [docs/roadmap.md](docs/roadmap.md).
31
34
 
32
35
  ## Providers
33
36
 
34
37
  <!-- registry:start -->
35
- | Provider | Available | Stub | Planned | Next up (0.6) | Datasets |
38
+ | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
36
39
  |---|---:|---:|---:|---|---|
37
- | [NOAA](docs/providers/noaa.md) | 3 | 0 | 26 | `gfs`, `hrrr`, `oisst`, `etopo` | `ghcn-daily`, `nexrad-level2`, `coastwatch-sst`, +26 planned |
40
+ | [NOAA](docs/providers/noaa.md) | 4 | 0 | 25 | — | `ghcn-daily`, `gsom`, `nexrad-level2`, `coastwatch-sst`, +25 planned |
38
41
  | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
39
42
  | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
40
43
  | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
41
44
  | [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
42
- | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | `gpm-imerg` | +1 planned |
45
+ | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
43
46
  | [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
44
47
 
45
48
  Available datasets are in `code`, stubs in _italics_; planned ones are counted. Available means implemented in this source checkout; consult the [releases](https://github.com/jakeryderv/usdata/releases) for published support. Each provider page has access notes and full dataset details; [docs/roadmap.md](docs/roadmap.md) lists datasets by target version.
@@ -130,6 +133,24 @@ sources:
130
133
  end: 2024-05-31
131
134
  ```
132
135
 
136
+ Terminal progress is available since v0.7. On a terminal, `fetch` and
137
+ `pull` show progress on stderr: resolved asset counts,
138
+ known bytes and unknown sizes, HTTP download bytes for the current attempt, and
139
+ validated cache hits. `fetch --dry-run` also summarizes known sizes. Asset totals
140
+ include possible cache hits; each manifest source is resolved separately. Bytes
141
+ from a failed HTTP attempt reset on retry; encoded responses have unknown decoded
142
+ size. Adapters that assemble files from metadata requests show asset-level progress.
143
+ Use `--no-progress` to disable it. Progress is automatically disabled when either
144
+ stdout or stderr is redirected; existing output lines and exit codes are unchanged.
145
+
146
+ ## Opening CSV data
147
+
148
+ `FetchedAsset.open()` is available since v0.6 with the optional pandas
149
+ extra (`pip install "usdata[pandas]"`). It reads cached CSV into a DataFrame,
150
+ preserves identifier strings, and keeps CoastWatch units as metadata.
151
+ See the [reader reference](docs/reference/readers.md)
152
+ and [fetch → open → analyze example](examples/sst-analysis/README.md).
153
+
133
154
  ## Development
134
155
 
135
156
  Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
@@ -138,16 +159,18 @@ Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
138
159
  git clone https://github.com/jakeryderv/usdata && cd usdata
139
160
  just setup # install toolchain and dependencies
140
161
  just test # unit tests
141
- just check # format, lint, typecheck, offline tests, generated docs
162
+ just check # format, lint, typecheck, offline tests, generated docs, release notices
163
+ just check-pandas # install the CSV extra and run the same checks
142
164
  just build # build wheel and sdist
143
- just smoke # install and exercise the built wheel outside the checkout
165
+ just smoke # exercise core and pandas wheel installations outside the checkout
144
166
  just run search radar
145
167
  ```
146
168
 
147
169
  Unit tests mechanically block network connections. Integration tests that hit
148
170
  live services run with `just test-integration`. CI checks Python 3.11 and 3.14 on
149
- Linux and smoke-tests the installed wheel on Linux, macOS, and Windows. The
150
- full unit and live-service suites currently run on Linux.
171
+ Linux, both with and without pandas, and smoke-tests both installed-wheel
172
+ profiles on Linux, macOS, and Windows. The full unit and live-service suites currently run on Linux. `just setup` restores
173
+ a core-only development environment; `just check-pandas` installs the extra.
151
174
 
152
175
  Releases: `just release minor` opens a version-bump PR; merging it publishes
153
176
  to PyPI and creates the tag and GitHub release. See
@@ -3,22 +3,23 @@
3
3
  Unified Python SDK and CLI for discovering, fetching, and tracking the
4
4
  provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
5
5
 
6
- > Status: pre-alpha. v0.5 supports GHCN-Daily, NEXRAD Level II, USGS daily
7
- > values, and CoastWatch SST subsets with provenance, plus Census state/county
8
- > lookup. Other datasets are planned.
6
+ > Status: pre-alpha. v0.7 supports GHCN-Daily, GSOM monthly summaries,
7
+ > NEXRAD Level II, USGS daily values, and CoastWatch SST subsets with provenance,
8
+ > plus Census state/county lookup, optional pandas CSV readers, and terminal
9
+ > download progress. Other datasets are planned.
9
10
  > See [docs/roadmap.md](docs/roadmap.md).
10
11
 
11
12
  ## Providers
12
13
 
13
14
  <!-- registry:start -->
14
- | Provider | Available | Stub | Planned | Next up (0.6) | Datasets |
15
+ | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
15
16
  |---|---:|---:|---:|---|---|
16
- | [NOAA](docs/providers/noaa.md) | 3 | 0 | 26 | `gfs`, `hrrr`, `oisst`, `etopo` | `ghcn-daily`, `nexrad-level2`, `coastwatch-sst`, +26 planned |
17
+ | [NOAA](docs/providers/noaa.md) | 4 | 0 | 25 | — | `ghcn-daily`, `gsom`, `nexrad-level2`, `coastwatch-sst`, +25 planned |
17
18
  | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
18
19
  | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
19
20
  | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
20
21
  | [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
21
- | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | `gpm-imerg` | +1 planned |
22
+ | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
22
23
  | [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
23
24
 
24
25
  Available datasets are in `code`, stubs in _italics_; planned ones are counted. Available means implemented in this source checkout; consult the [releases](https://github.com/jakeryderv/usdata/releases) for published support. Each provider page has access notes and full dataset details; [docs/roadmap.md](docs/roadmap.md) lists datasets by target version.
@@ -109,6 +110,24 @@ sources:
109
110
  end: 2024-05-31
110
111
  ```
111
112
 
113
+ Terminal progress is available since v0.7. On a terminal, `fetch` and
114
+ `pull` show progress on stderr: resolved asset counts,
115
+ known bytes and unknown sizes, HTTP download bytes for the current attempt, and
116
+ validated cache hits. `fetch --dry-run` also summarizes known sizes. Asset totals
117
+ include possible cache hits; each manifest source is resolved separately. Bytes
118
+ from a failed HTTP attempt reset on retry; encoded responses have unknown decoded
119
+ size. Adapters that assemble files from metadata requests show asset-level progress.
120
+ Use `--no-progress` to disable it. Progress is automatically disabled when either
121
+ stdout or stderr is redirected; existing output lines and exit codes are unchanged.
122
+
123
+ ## Opening CSV data
124
+
125
+ `FetchedAsset.open()` is available since v0.6 with the optional pandas
126
+ extra (`pip install "usdata[pandas]"`). It reads cached CSV into a DataFrame,
127
+ preserves identifier strings, and keeps CoastWatch units as metadata.
128
+ See the [reader reference](docs/reference/readers.md)
129
+ and [fetch → open → analyze example](examples/sst-analysis/README.md).
130
+
112
131
  ## Development
113
132
 
114
133
  Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
@@ -117,16 +136,18 @@ Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
117
136
  git clone https://github.com/jakeryderv/usdata && cd usdata
118
137
  just setup # install toolchain and dependencies
119
138
  just test # unit tests
120
- just check # format, lint, typecheck, offline tests, generated docs
139
+ just check # format, lint, typecheck, offline tests, generated docs, release notices
140
+ just check-pandas # install the CSV extra and run the same checks
121
141
  just build # build wheel and sdist
122
- just smoke # install and exercise the built wheel outside the checkout
142
+ just smoke # exercise core and pandas wheel installations outside the checkout
123
143
  just run search radar
124
144
  ```
125
145
 
126
146
  Unit tests mechanically block network connections. Integration tests that hit
127
147
  live services run with `just test-integration`. CI checks Python 3.11 and 3.14 on
128
- Linux and smoke-tests the installed wheel on Linux, macOS, and Windows. The
129
- full unit and live-service suites currently run on Linux.
148
+ Linux, both with and without pandas, and smoke-tests both installed-wheel
149
+ profiles on Linux, macOS, and Windows. The full unit and live-service suites currently run on Linux. `just setup` restores
150
+ a core-only development environment; `just check-pandas` installs the extra.
130
151
 
131
152
  Releases: `just release minor` opens a version-bump PR; merging it publishes
132
153
  to PyPI and creates the tag and GitHub release. See
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.5.0"
3
+ version = "0.7.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -30,6 +30,9 @@ dependencies = [
30
30
  name = "Jake Van Slyke"
31
31
  email = "jakervanslyke@gmail.com"
32
32
 
33
+ [project.optional-dependencies]
34
+ pandas = ["pandas>=3.0"]
35
+
33
36
  [project.urls]
34
37
  Homepage = "https://github.com/jakeryderv/usdata"
35
38
  Repository = "https://github.com/jakeryderv/usdata"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.5.0"
3
+ version = "0.7.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -22,6 +22,9 @@ dependencies = [
22
22
  "typer>=0.12",
23
23
  ]
24
24
 
25
+ [project.optional-dependencies]
26
+ pandas = ["pandas>=3.0"]
27
+
25
28
  [project.urls]
26
29
  Homepage = "https://github.com/jakeryderv/usdata"
27
30
  Repository = "https://github.com/jakeryderv/usdata"
@@ -0,0 +1,62 @@
1
+ """Internal synchronous progress events, scoped to one CLI operation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable, Iterator, Sequence
6
+ from contextlib import contextmanager
7
+ from contextvars import ContextVar
8
+ from dataclasses import dataclass
9
+ from typing import Literal
10
+
11
+
12
+ @dataclass(frozen=True)
13
+ class Batch:
14
+ """A resolved group; sizes describe assets, including possible cache hits."""
15
+
16
+ count: int
17
+ known_bytes: int
18
+ unknown_sizes: int
19
+
20
+
21
+ @dataclass(frozen=True)
22
+ class AssetProgress:
23
+ """Start or validated completion of an asset."""
24
+
25
+ asset_id: str
26
+ state: Literal["start", "cached", "fetched"]
27
+ size: int | None
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class TransferProgress:
32
+ """Bytes written in the current HTTP attempt, reset on every retry."""
33
+
34
+ completed: int
35
+ total: int | None
36
+ attempt: int
37
+
38
+
39
+ Event = Batch | AssetProgress | TransferProgress
40
+ _observer: ContextVar[Callable[[Event], None] | None] = ContextVar("progress", default=None)
41
+
42
+
43
+ def emit(event: Event) -> None:
44
+ """Notify the active observer without importing CLI code."""
45
+ observer = _observer.get()
46
+ if observer is not None:
47
+ observer(event)
48
+
49
+
50
+ def batch(sizes: Sequence[int | None]) -> None:
51
+ """Report known and unknown sizes separately; never guess a total."""
52
+ emit(Batch(len(sizes), sum(size for size in sizes if size is not None), sizes.count(None)))
53
+
54
+
55
+ @contextmanager
56
+ def observe(callback: Callable[[Event], None]) -> Iterator[None]:
57
+ """Observe an operation and restore the previous observer on every exit."""
58
+ token = _observer.set(callback)
59
+ try:
60
+ yield
61
+ finally:
62
+ _observer.reset(token)
@@ -9,6 +9,8 @@ import httpx
9
9
  import typer
10
10
 
11
11
  from usdata import __version__, build_query, default_registry
12
+ from usdata._progress import batch
13
+ from usdata.cli.progress import progress
12
14
  from usdata.fetch import ChecksumMismatch
13
15
  from usdata.fetch import fetch as fetch_query
14
16
  from usdata.manifest import lockfile_path
@@ -126,6 +128,7 @@ def fetch(
126
128
  ] = None,
127
129
  cache_dir: Annotated[Path | None, typer.Option(help="Override the cache directory.")] = None,
128
130
  force: Annotated[bool, typer.Option(help="Re-download even if cached.")] = False,
131
+ no_progress: Annotated[bool, typer.Option(help="Disable terminal progress.")] = False,
129
132
  dry_run: Annotated[
130
133
  bool, typer.Option(help="List matching assets without downloading.")
131
134
  ] = False,
@@ -165,8 +168,11 @@ def fetch(
165
168
  for a in assets:
166
169
  typer.echo(f"{a.id}\t{a.href}")
167
170
  typer.echo(f"{len(assets)} asset(s) matched", err=True)
171
+ with progress(disabled=no_progress):
172
+ batch([asset.size for asset in assets])
168
173
  return
169
- fetched = fetch_query(ds, query, root=cache_dir, force=force)
174
+ with progress(disabled=no_progress):
175
+ fetched = fetch_query(ds, query, root=cache_dir, force=force)
170
176
  except (DatasetNotFound, UnknownPlace, ValueError) as e:
171
177
  typer.secho(str(e), err=True, fg="red")
172
178
  raise typer.Exit(code=2) from None
@@ -192,10 +198,12 @@ def pull(
192
198
  bool,
193
199
  typer.Option(help="Ignore an existing lockfile: re-resolve every source and rewrite it."),
194
200
  ] = False,
201
+ no_progress: Annotated[bool, typer.Option(help="Disable terminal progress.")] = False,
195
202
  ) -> None:
196
203
  """Fetch every source in a manifest and write (or restore from) its lockfile."""
197
204
  try:
198
- result = pull_manifest(manifest, root=cache_dir, force=force)
205
+ with progress(disabled=no_progress):
206
+ result = pull_manifest(manifest, root=cache_dir, force=force)
199
207
  except EmptySource as e:
200
208
  typer.secho(str(e), err=True, fg="yellow")
201
209
  raise typer.Exit(code=1) from None
@@ -0,0 +1,89 @@
1
+ """Terminal-only progress rendering; normal CLI output remains machine readable."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import shutil
6
+ import sys
7
+ from collections.abc import Iterator
8
+ from contextlib import contextmanager
9
+ from time import monotonic
10
+ from typing import TextIO
11
+
12
+ from usdata._progress import AssetProgress, Batch, Event, observe
13
+
14
+
15
+ def _interactive() -> bool:
16
+ return sys.stdout.isatty() and sys.stderr.isatty()
17
+
18
+
19
+ class _Display:
20
+ def __init__(self, stream: TextIO) -> None:
21
+ self.stream = stream
22
+ self.width = 0
23
+ self.updated = 0.0
24
+ self.asset = ""
25
+ self.count = 0
26
+ self.done = 0
27
+ self.cached = 0
28
+
29
+ def clear(self) -> None:
30
+ if self.width:
31
+ self.stream.write("\r" + " " * self.width + "\r")
32
+ self.stream.flush()
33
+ self.width = 0
34
+
35
+ def line(self, text: str, *, final: bool = False) -> None:
36
+ self.clear()
37
+ # IDs come from providers: keep control characters out of the terminal.
38
+ text = "".join(c if c.isprintable() else "?" for c in text)
39
+ if final:
40
+ self.stream.write(text + "\n")
41
+ else:
42
+ text = text[: max(1, shutil.get_terminal_size().columns - 1)]
43
+ self.stream.write(text)
44
+ self.width = len(text)
45
+ self.stream.flush()
46
+ self.updated = monotonic()
47
+
48
+ def __call__(self, event: Event) -> None:
49
+ if isinstance(event, Batch):
50
+ self.count, self.done, self.cached = event.count, 0, 0
51
+ sizes = f"{event.known_bytes:,} known bytes"
52
+ if event.unknown_sizes:
53
+ sizes += f"; {event.unknown_sizes} size(s) unknown"
54
+ self.line(f"{event.count} asset(s) resolved; {sizes} (before cache checks)", final=True)
55
+ elif isinstance(event, AssetProgress):
56
+ self.asset = event.asset_id
57
+ if event.state == "start":
58
+ size = f"{event.size:,} bytes" if event.size is not None else "size unknown"
59
+ self.line(f"[{self.done}/{self.count}] checking cache ({size}) | {self.asset}")
60
+ else:
61
+ self.done += 1
62
+ self.cached += event.state == "cached"
63
+ self.line(
64
+ f"[{self.done}/{self.count}] {event.state} "
65
+ f"({event.size:,} bytes; {self.cached} cached) | {self.asset}",
66
+ final=self.done == self.count,
67
+ )
68
+ else:
69
+ if event.completed and monotonic() - self.updated < 0.1:
70
+ return
71
+ size = f"{event.total:,}" if event.total is not None else "unknown"
72
+ self.line(
73
+ f"[{self.done}/{self.count}] {event.completed:,}/{size} bytes "
74
+ f"(attempt {event.attempt}) | {self.asset}"
75
+ )
76
+
77
+
78
+ @contextmanager
79
+ def progress(*, disabled: bool = False) -> Iterator[None]:
80
+ """Render progress on stderr only when both output streams are terminals."""
81
+ if disabled or not _interactive():
82
+ yield
83
+ return
84
+ display = _Display(sys.stderr)
85
+ with observe(display):
86
+ try:
87
+ yield
88
+ finally:
89
+ display.clear()
@@ -209,19 +209,22 @@ datasets:
209
209
 
210
210
  - id: noaa:gsom
211
211
  provider: noaa
212
- status: planned
212
+ status: available
213
213
  domain: surface-weather
214
- target: later
214
+ since: "0.7"
215
215
  title: Global Summary of the Month
216
216
  description: >-
217
217
  Monthly station summaries derived from GHCN-Daily (means, extremes,
218
218
  totals) via the NCEI Access Data Service dataset global-summary-of-the-month.
219
- Shares the GHCN-Daily client and station search.
219
+ Selects whole calendar months and explicit stations, or discovers stations
220
+ through the companion search service.
220
221
  keywords: [climate, monthly, stations, temperature, precipitation, gsom, ncei]
221
222
  protocol: http
222
223
  homepage: https://www.ncei.noaa.gov/access/search/data-search/global-summary-of-the-month
223
224
  license: US Government Work (public domain)
224
225
  capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: true }
226
+ spatial_extent: { west: -180.0, south: -90.0, east: 180.0, north: 90.0 }
227
+ adapter: usdata.providers.noaa.gsom:GlobalSummaryMonthly
225
228
 
226
229
  - id: noaa:gsoy
227
230
  provider: noaa
@@ -293,7 +296,7 @@ datasets:
293
296
  provider: noaa
294
297
  status: planned
295
298
  domain: weather-models
296
- target: "0.6"
299
+ target: later
297
300
  title: HRRR Forecast Model Output
298
301
  description: >-
299
302
  High-Resolution Rapid Refresh 3 km hourly forecasts in GRIB2 from the
@@ -310,7 +313,7 @@ datasets:
310
313
  provider: noaa
311
314
  status: planned
312
315
  domain: weather-models
313
- target: "0.6"
316
+ target: later
314
317
  title: GFS Forecast Model Output
315
318
  description: >-
316
319
  Global Forecast System output in GRIB2 from the public noaa-gfs-bdp-pds
@@ -342,7 +345,7 @@ datasets:
342
345
  provider: noaa
343
346
  status: planned
344
347
  domain: ocean-physics
345
- target: "0.6"
348
+ target: later
346
349
  title: OISST Daily Sea Surface Temperature
347
350
  description: >-
348
351
  Optimum Interpolation SST v2.1: daily global 0.25 degree analysis since
@@ -375,7 +378,7 @@ datasets:
375
378
  provider: noaa
376
379
  status: planned
377
380
  domain: bathymetry
378
- target: "0.6"
381
+ target: later
379
382
  title: ETOPO 2022 Global Relief
380
383
  description: >-
381
384
  Global topography and bathymetry at 15, 30, and 60 arc-seconds as
@@ -591,7 +594,7 @@ datasets:
591
594
  provider: nasa
592
595
  status: planned
593
596
  domain: weather-satellites
594
- target: "0.6"
597
+ target: later
595
598
  title: GPM IMERG Precipitation
596
599
  description: >-
597
600
  Global half-hourly and daily merged satellite precipitation estimates.
@@ -3,10 +3,11 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  from pathlib import Path
6
+ from typing import Any
6
7
 
7
8
  from pydantic import BaseModel
8
9
 
9
- from usdata import provenance
10
+ from usdata import _progress, provenance
10
11
  from usdata._files import staged_path
11
12
  from usdata.cache import asset_path, sha256_file
12
13
  from usdata.models import Asset, Dataset, Provenance, Query
@@ -25,6 +26,27 @@ class FetchedAsset(BaseModel):
25
26
  provenance: Provenance
26
27
  from_cache: bool
27
28
 
29
+ def open(
30
+ self,
31
+ *,
32
+ reader: str | None = None,
33
+ dtype: dict[str, str] | None = None,
34
+ parse_dates: list[str] | None = None,
35
+ usecols: list[str] | None = None,
36
+ nrows: int | None = None,
37
+ ) -> Any:
38
+ """Open this local CSV as a DataFrame; requires the ``pandas`` extra.
39
+
40
+ ERDDAP units are kept in ``frame.attrs["units"]`` and source provenance
41
+ in ``frame.attrs["usdata"]``. See ``usdata.readers.open_asset`` for options.
42
+ Cached files and provenance sidecars are never changed.
43
+ """
44
+ from usdata.readers import open_asset
45
+
46
+ return open_asset(
47
+ self, reader=reader, dtype=dtype, parse_dates=parse_dates, usecols=usecols, nrows=nrows
48
+ )
49
+
28
50
 
29
51
  def _fetch_asset(
30
52
  dataset: Dataset,
@@ -36,6 +58,7 @@ def _fetch_asset(
36
58
  ) -> FetchedAsset:
37
59
  if asset.dataset_id != dataset.id:
38
60
  raise ValueError(f"asset dataset {asset.dataset_id!r} does not match {dataset.id!r}")
61
+ _progress.emit(_progress.AssetProgress(asset.id, "start", asset.size))
39
62
  path = asset_path(asset, root)
40
63
  if not force and path.is_file():
41
64
  try:
@@ -51,6 +74,7 @@ def _fetch_asset(
51
74
  and (asset.checksum is None or prov.checksum == asset.checksum)
52
75
  and sha256_file(path) == prov.checksum
53
76
  ):
77
+ _progress.emit(_progress.AssetProgress(asset.id, "cached", prov.size))
54
78
  return FetchedAsset(asset=asset, path=path, provenance=prov, from_cache=True)
55
79
  with staged_path(path) as tmp:
56
80
  adapter.fetch(asset, tmp)
@@ -59,6 +83,7 @@ def _fetch_asset(
59
83
  raise ChecksumMismatch(f"{asset.id}: expected {asset.checksum}, got {prov.checksum}")
60
84
  # A crash between replacements leaves a detectable mismatch, never a trusted partial file.
61
85
  provenance.write(prov, path)
86
+ _progress.emit(_progress.AssetProgress(asset.id, "fetched", prov.size))
62
87
  return FetchedAsset(asset=asset, path=path, provenance=prov, from_cache=False)
63
88
 
64
89
 
@@ -76,4 +101,5 @@ def fetch(
76
101
  """Resolve and fetch a query, sharing one adapter and closing its owned resources."""
77
102
  with load_adapter(dataset) as adapter:
78
103
  assets = adapter.list_assets(query)
104
+ _progress.batch([asset.size for asset in assets])
79
105
  return [_fetch_asset(dataset, a, adapter, root=root, force=force) for a in assets]
@@ -11,7 +11,7 @@ from typing import Any, TypeVar
11
11
 
12
12
  import httpx
13
13
 
14
- from usdata import __version__
14
+ from usdata import __version__, _progress
15
15
  from usdata._files import staged_path
16
16
 
17
17
  USER_AGENT = f"usdata/{__version__} (+https://github.com/jakeryderv/usdata)"
@@ -80,13 +80,30 @@ def download(url: str, dest: Path, http: httpx.Client | None = None) -> Path:
80
80
  """Download atomically, restarting interrupted GETs up to three total attempts."""
81
81
  own = http is None
82
82
  active = http or client()
83
+ attempt = 0
83
84
 
84
85
  def request() -> Path:
86
+ nonlocal attempt
87
+ attempt += 1
88
+ _progress.emit(_progress.TransferProgress(0, None, attempt))
85
89
  with staged_path(dest) as tmp, active.stream("GET", url) as resp:
86
90
  resp.raise_for_status()
91
+ length = resp.headers.get("Content-Length", "")
92
+ # iter_bytes writes decoded bytes; an encoded length is not comparable.
93
+ total = (
94
+ int(length)
95
+ if length.isascii()
96
+ and length.isdigit()
97
+ and resp.headers.get("Content-Encoding", "identity").lower() == "identity"
98
+ else None
99
+ )
100
+ completed = 0
101
+ _progress.emit(_progress.TransferProgress(completed, total, attempt))
87
102
  with tmp.open("wb") as f:
88
103
  for chunk in resp.iter_bytes():
89
104
  f.write(chunk)
105
+ completed += len(chunk)
106
+ _progress.emit(_progress.TransferProgress(completed, total, attempt))
90
107
  return dest
91
108
 
92
109
  try:
@@ -12,6 +12,7 @@ always goes through search first. Stations are chunked so URLs stay short.
12
12
  from __future__ import annotations
13
13
 
14
14
  import hashlib
15
+ import logging
15
16
  from pathlib import Path
16
17
  from typing import Any
17
18
 
@@ -26,6 +27,7 @@ DATA_URL = "https://www.ncei.noaa.gov/access/services/data/v1"
26
27
  NCEI_DATASET = "daily-summaries"
27
28
  SEARCH_PAGE_SIZE = 1000
28
29
  STATIONS_PER_ASSET = 50
30
+ logger = logging.getLogger(__name__)
29
31
 
30
32
 
31
33
  def _date(value: Any) -> str:
@@ -46,6 +48,8 @@ def _stations_param(raw: Any) -> list[str]:
46
48
  class GhcnDaily(Provider):
47
49
  """GHCN-Daily adapter. Params: ``stations`` (list or comma string), ``units``."""
48
50
 
51
+ ncei_dataset = NCEI_DATASET
52
+
49
53
  def __init__(self, dataset: Dataset, client: httpx.Client | None = None) -> None:
50
54
  super().__init__(dataset)
51
55
  self._client = client
@@ -68,7 +72,7 @@ class GhcnDaily(Provider):
68
72
  raise QueryError("station search needs a bounding box and a time range")
69
73
  b = query.bbox
70
74
  params: dict[str, Any] = {
71
- "dataset": NCEI_DATASET,
75
+ "dataset": self.ncei_dataset,
72
76
  "bbox": f"{b.north},{b.west},{b.south},{b.east}",
73
77
  "startDate": _date(query.time.start),
74
78
  "endDate": _date(query.time.end),
@@ -84,12 +88,29 @@ class GhcnDaily(Provider):
84
88
  resp.raise_for_status()
85
89
  body = resp.json()
86
90
  results = body.get("results", [])
91
+ before = len(found)
87
92
  for result in results:
88
93
  for station in result.get("stations", []):
89
94
  sid = station.get("id")
90
95
  if sid and sid not in seen:
91
96
  seen.add(sid)
92
97
  found.append(sid)
98
+ logger.debug(
99
+ "NCEI station search: url=%s status=%s content_type=%s "
100
+ "count=%r totalCount=%r results=%s new_stations=%s station_sample=%r",
101
+ resp.request.url,
102
+ resp.status_code,
103
+ resp.headers.get("content-type"),
104
+ body.get("count"),
105
+ body.get("totalCount"),
106
+ len(results),
107
+ len(found) - before,
108
+ found[before : before + 5],
109
+ )
110
+ if len(found) == before and logger.isEnabledFor(logging.DEBUG):
111
+ logger.debug(
112
+ "NCEI search page yielded no new stations; response_prefix=%r", resp.text[:512]
113
+ )
93
114
  # "count" is the number matching this query; "totalCount" is dataset-wide.
94
115
  params["offset"] += SEARCH_PAGE_SIZE
95
116
  if not results or params["offset"] >= int(body.get("count", 0)):
@@ -101,7 +122,7 @@ class GhcnDaily(Provider):
101
122
  if query.time is None or query.time.start is None or query.time.end is None:
102
123
  raise QueryError(f"{self.dataset.id} requires both start and end dates")
103
124
  if unknown := set(query.params) - {"stations", "units"}:
104
- raise QueryError(f"unsupported GHCN params: {', '.join(sorted(unknown))}")
125
+ raise QueryError(f"unsupported {self.dataset.id} params: {', '.join(sorted(unknown))}")
105
126
  if query.params.get("units", "metric") not in ("metric", "standard"):
106
127
  raise QueryError("units must be metric or standard")
107
128
  if "stations" in query.params:
@@ -118,7 +139,7 @@ class GhcnDaily(Provider):
118
139
  for i in range(0, len(stations), STATIONS_PER_ASSET):
119
140
  chunk = stations[i : i + STATIONS_PER_ASSET]
120
141
  params: dict[str, Any] = {
121
- "dataset": NCEI_DATASET,
142
+ "dataset": self.ncei_dataset,
122
143
  "stations": ",".join(chunk),
123
144
  "startDate": start,
124
145
  "endDate": end,
@@ -132,7 +153,7 @@ class GhcnDaily(Provider):
132
153
  digest = hashlib.sha1(url.encode()).hexdigest()[:12]
133
154
  assets.append(
134
155
  Asset(
135
- id=f"{NCEI_DATASET}_{start}_{end}_{digest}.csv",
156
+ id=f"{self.ncei_dataset}_{start}_{end}_{digest}.csv",
136
157
  dataset_id=self.dataset.id,
137
158
  href=url,
138
159
  protocol=Protocol.HTTP,
@@ -0,0 +1,49 @@
1
+ """Monthly station summaries through the NCEI Access Data Service.
2
+
3
+ Params: ``stations`` (list or comma string) and ``units`` (metric or standard).
4
+ Both dates are required; every UTC calendar month touched by the interval is
5
+ selected in full. Geographic discovery uses the same normalized month bounds.
6
+ """
7
+
8
+ from calendar import monthrange
9
+ from datetime import UTC
10
+
11
+ from usdata.models import Asset, Query, TimeRange
12
+ from usdata.providers.base import QueryError
13
+ from usdata.providers.noaa.ghcnd import GhcnDaily
14
+
15
+
16
+ class GlobalSummaryMonthly(GhcnDaily):
17
+ """GSOM CSV subsets, sharing NCEI station discovery and transport with GHCN."""
18
+
19
+ ncei_dataset = "global-summary-of-the-month"
20
+
21
+ def _monthly_query(self, query: Query) -> Query:
22
+ if query.time is None or query.time.start is None or query.time.end is None:
23
+ raise QueryError(f"{self.dataset.id} requires both start and end dates")
24
+ start = query.time.start.replace(tzinfo=query.time.start.tzinfo or UTC).astimezone(UTC)
25
+ end = query.time.end.replace(tzinfo=query.time.end.tzinfo or UTC).astimezone(UTC)
26
+ return query.model_copy(
27
+ update={
28
+ "time": TimeRange(
29
+ start=start.replace(day=1, hour=0, minute=0, second=0, microsecond=0),
30
+ end=end.replace(
31
+ day=monthrange(end.year, end.month)[1],
32
+ hour=23,
33
+ minute=59,
34
+ second=59,
35
+ microsecond=999999,
36
+ ),
37
+ )
38
+ }
39
+ )
40
+
41
+ def find_stations(self, query: Query) -> list[str]:
42
+ """Find stations overlapping the selected complete calendar months."""
43
+ return super().find_stations(self._monthly_query(query))
44
+
45
+ def list_assets(self, query: Query) -> list[Asset]:
46
+ """Resolve monthly CSVs with stable URLs and complete-month asset bounds."""
47
+ if "stations" in query.params and query.bbox is not None:
48
+ raise QueryError("pass stations or a location/bbox, not both")
49
+ return super().list_assets(self._monthly_query(query))
@@ -16,7 +16,7 @@ from pathlib import Path
16
16
 
17
17
  from pydantic import BaseModel
18
18
 
19
- from usdata import __version__, provenance
19
+ from usdata import __version__, _progress, provenance
20
20
  from usdata.cache import asset_path, sha256_file
21
21
  from usdata.fetch import ChecksumMismatch, FetchedAsset, _fetch_asset, fetch
22
22
  from usdata.manifest import LockedAsset, Lockfile, Manifest, lockfile_path
@@ -111,13 +111,21 @@ def restore(
111
111
  lock = Lockfile.load(lock_path)
112
112
  _check_manifest(manifest_path, lock)
113
113
  fetched: list[FetchedAsset] = []
114
+ _progress.batch([entry.provenance.size for entry in lock.assets])
114
115
  adapters: dict[str, Provider] = {}
115
116
  with ExitStack() as stack:
116
117
  for entry in lock.assets:
117
118
  dataset = reg.get(entry.asset.dataset_id)
118
119
  path = asset_path(entry.asset, root)
120
+ if path.is_file():
121
+ _progress.emit(
122
+ _progress.AssetProgress(entry.asset.id, "start", entry.provenance.size)
123
+ )
119
124
  if path.is_file() and sha256_file(path) == entry.provenance.checksum:
120
125
  provenance.write(entry.provenance, path)
126
+ _progress.emit(
127
+ _progress.AssetProgress(entry.asset.id, "cached", entry.provenance.size)
128
+ )
121
129
  fetched.append(
122
130
  FetchedAsset(
123
131
  asset=entry.asset, path=path, provenance=entry.provenance, from_cache=True
@@ -0,0 +1,103 @@
1
+ """Optional readers for local fetched files; never fetch or modify cached bytes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import csv
6
+ from importlib import import_module
7
+ from typing import TYPE_CHECKING, Any
8
+
9
+ from usdata.models import Protocol
10
+
11
+ if TYPE_CHECKING:
12
+ from usdata.fetch import FetchedAsset
13
+
14
+ CSV_MEDIA_TYPES = {"text/csv", "application/csv"}
15
+ IDENTIFIER_COLUMNS = {
16
+ "station",
17
+ "station_id",
18
+ "site_no",
19
+ "monitoring_location_id",
20
+ "parameter_code",
21
+ "statistic_id",
22
+ }
23
+
24
+
25
+ class MissingReaderDependency(ImportError):
26
+ """The optional dependency required to open an asset is not installed."""
27
+
28
+
29
+ class UnsupportedFormat(ValueError):
30
+ """No reader is implemented for this asset's format."""
31
+
32
+
33
+ def open_asset(
34
+ fetched: FetchedAsset,
35
+ *,
36
+ reader: str | None = None,
37
+ dtype: dict[str, str] | None = None,
38
+ parse_dates: list[str] | None = None,
39
+ usecols: list[str] | None = None,
40
+ nrows: int | None = None,
41
+ ) -> Any:
42
+ """Read a local CSV into a pandas DataFrame, retaining units and provenance.
43
+
44
+ Infer ``csv`` or ``erddap-csv`` from media type and protocol, or use an
45
+ explicit reader for ambiguous metadata. Identifier columns default to pandas
46
+ strings; explicit dtype entries override those defaults. Dates remain strings
47
+ unless named in parse_dates. No checksum verification or downloading occurs.
48
+ """
49
+ if reader is None:
50
+ media_type = (fetched.asset.media_type or "").split(";", 1)[0].strip().lower()
51
+ if media_type not in CSV_MEDIA_TYPES:
52
+ raise UnsupportedFormat(
53
+ f"no reader for {fetched.asset.media_type!r}; supported formats are CSV and "
54
+ "ERDDAP CSV. For a known CSV with ambiguous metadata, pass reader='csv' "
55
+ "or reader='erddap-csv'; otherwise use fetched.path with a format-specific reader"
56
+ )
57
+ reader = "erddap-csv" if fetched.asset.protocol is Protocol.ERDDAP else "csv"
58
+ if reader not in {"csv", "erddap-csv"}:
59
+ raise UnsupportedFormat(f"unsupported reader {reader!r}; use 'csv' or 'erddap-csv'")
60
+ try:
61
+ pandas = import_module("pandas")
62
+ except ModuleNotFoundError as error:
63
+ if error.name != "pandas":
64
+ raise
65
+ raise MissingReaderDependency(
66
+ 'CSV reading requires pandas; install it with: pip install "usdata[pandas]" '
67
+ '(or uv add "usdata[pandas]")'
68
+ ) from error
69
+
70
+ # Pass a file object to pandas: reading a fetched asset is strictly local.
71
+ with fetched.path.open(encoding="utf-8-sig", newline="") as stream:
72
+ records = csv.reader(stream)
73
+ columns = next(records, [])
74
+ if (
75
+ not columns
76
+ or any(not column for column in columns)
77
+ or len(set(columns)) != len(columns)
78
+ ):
79
+ raise ValueError("CSV must have a non-empty header with unique column names")
80
+ units = {}
81
+ if reader == "erddap-csv":
82
+ values = next(records, [])
83
+ if len(values) != len(columns):
84
+ raise ValueError("ERDDAP CSV must have a units row matching the header")
85
+ units = dict(zip(columns, values, strict=True))
86
+ types = {name: "string" for name in columns if name.casefold() in IDENTIFIER_COLUMNS}
87
+ types.update(dtype or {})
88
+ frame = pandas.read_csv(
89
+ stream,
90
+ header=None,
91
+ names=columns,
92
+ dtype=types,
93
+ parse_dates=parse_dates,
94
+ usecols=usecols,
95
+ nrows=nrows,
96
+ )
97
+ if units:
98
+ frame.attrs["units"] = {name: units[name] for name in frame.columns}
99
+ frame.attrs["usdata"] = {
100
+ "asset_id": fetched.asset.id,
101
+ "provenance": fetched.provenance.model_dump(mode="json"),
102
+ }
103
+ return frame
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes