usdata 0.6.0__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. {usdata-0.6.0 → usdata-0.7.0}/PKG-INFO +16 -5
  2. {usdata-0.6.0 → usdata-0.7.0}/README.md +15 -4
  3. {usdata-0.6.0 → usdata-0.7.0}/pyproject.toml +1 -1
  4. {usdata-0.6.0 → usdata-0.7.0}/pyproject.toml.orig +1 -1
  5. usdata-0.7.0/src/usdata/_progress.py +62 -0
  6. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/cli/app.py +10 -2
  7. usdata-0.7.0/src/usdata/cli/progress.py +89 -0
  8. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/data/registry.yaml +6 -3
  9. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/fetch.py +5 -1
  10. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/protocols/http.py +18 -1
  11. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/noaa/ghcnd.py +25 -4
  12. usdata-0.7.0/src/usdata/providers/noaa/gsom.py +49 -0
  13. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/pull.py +9 -1
  14. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/__init__.py +0 -0
  15. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/_files.py +0 -0
  16. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/cache.py +0 -0
  17. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/cli/__init__.py +0 -0
  18. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/data/nexrad_sites.csv +0 -0
  19. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/data/places.csv +0 -0
  20. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/data/places.sources.json +0 -0
  21. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/manifest.py +0 -0
  22. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/models.py +0 -0
  23. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/protocols/__init__.py +0 -0
  24. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/protocols/erddap.py +0 -0
  25. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/protocols/s3.py +0 -0
  26. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/provenance.py +0 -0
  27. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/__init__.py +0 -0
  28. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/base.py +0 -0
  29. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/noaa/__init__.py +0 -0
  30. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
  31. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/noaa/nexrad.py +0 -0
  32. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/noaa/sites.py +0 -0
  33. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/usgs/__init__.py +0 -0
  34. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/usgs/daily.py +0 -0
  35. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/py.typed +0 -0
  36. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/query.py +0 -0
  37. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/readers.py +0 -0
  38. {usdata-0.6.0 → usdata-0.7.0}/src/usdata/registry.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: usdata
3
- Version: 0.6.0
3
+ Version: 0.7.0
4
4
  Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
5
5
  Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
6
6
  Author: Jake Van Slyke
@@ -26,9 +26,10 @@ Description-Content-Type: text/markdown
26
26
  Unified Python SDK and CLI for discovering, fetching, and tracking the
27
27
  provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
28
28
 
29
- > Status: pre-alpha. v0.6 supports GHCN-Daily, NEXRAD Level II, USGS daily
30
- > values, and CoastWatch SST subsets with provenance, plus Census state/county
31
- > lookup and optional pandas CSV readers. Other datasets are planned.
29
+ > Status: pre-alpha. v0.7 supports GHCN-Daily, GSOM monthly summaries,
30
+ > NEXRAD Level II, USGS daily values, and CoastWatch SST subsets with provenance,
31
+ > plus Census state/county lookup, optional pandas CSV readers, and terminal
32
+ > download progress. Other datasets are planned.
32
33
  > See [docs/roadmap.md](docs/roadmap.md).
33
34
 
34
35
  ## Providers
@@ -36,7 +37,7 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
36
37
  <!-- registry:start -->
37
38
  | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
38
39
  |---|---:|---:|---:|---|---|
39
- | [NOAA](docs/providers/noaa.md) | 3 | 0 | 26 | — | `ghcn-daily`, `nexrad-level2`, `coastwatch-sst`, +26 planned |
40
+ | [NOAA](docs/providers/noaa.md) | 4 | 0 | 25 | — | `ghcn-daily`, `gsom`, `nexrad-level2`, `coastwatch-sst`, +25 planned |
40
41
  | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
41
42
  | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
42
43
  | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
@@ -132,6 +133,16 @@ sources:
132
133
  end: 2024-05-31
133
134
  ```
134
135
 
136
+ Terminal progress is available since v0.7. On a terminal, `fetch` and
137
+ `pull` show progress on stderr: resolved asset counts,
138
+ known bytes and unknown sizes, HTTP download bytes for the current attempt, and
139
+ validated cache hits. `fetch --dry-run` also summarizes known sizes. Asset totals
140
+ include possible cache hits; each manifest source is resolved separately. Bytes
141
+ from a failed HTTP attempt reset on retry; encoded responses have unknown decoded
142
+ size. Adapters that assemble files from metadata requests show asset-level progress.
143
+ Use `--no-progress` to disable it. Progress is automatically disabled when either
144
+ stdout or stderr is redirected; existing output lines and exit codes are unchanged.
145
+
135
146
  ## Opening CSV data
136
147
 
137
148
  `FetchedAsset.open()` is available since v0.6 with the optional pandas
@@ -3,9 +3,10 @@
3
3
  Unified Python SDK and CLI for discovering, fetching, and tracking the
4
4
  provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
5
5
 
6
- > Status: pre-alpha. v0.6 supports GHCN-Daily, NEXRAD Level II, USGS daily
7
- > values, and CoastWatch SST subsets with provenance, plus Census state/county
8
- > lookup and optional pandas CSV readers. Other datasets are planned.
6
+ > Status: pre-alpha. v0.7 supports GHCN-Daily, GSOM monthly summaries,
7
+ > NEXRAD Level II, USGS daily values, and CoastWatch SST subsets with provenance,
8
+ > plus Census state/county lookup, optional pandas CSV readers, and terminal
9
+ > download progress. Other datasets are planned.
9
10
  > See [docs/roadmap.md](docs/roadmap.md).
10
11
 
11
12
  ## Providers
@@ -13,7 +14,7 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
13
14
  <!-- registry:start -->
14
15
  | Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
15
16
  |---|---:|---:|---:|---|---|
16
- | [NOAA](docs/providers/noaa.md) | 3 | 0 | 26 | — | `ghcn-daily`, `nexrad-level2`, `coastwatch-sst`, +26 planned |
17
+ | [NOAA](docs/providers/noaa.md) | 4 | 0 | 25 | — | `ghcn-daily`, `gsom`, `nexrad-level2`, `coastwatch-sst`, +25 planned |
17
18
  | [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
18
19
  | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
19
20
  | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
@@ -109,6 +110,16 @@ sources:
109
110
  end: 2024-05-31
110
111
  ```
111
112
 
113
+ Terminal progress is available since v0.7. On a terminal, `fetch` and
114
+ `pull` show progress on stderr: resolved asset counts,
115
+ known bytes and unknown sizes, HTTP download bytes for the current attempt, and
116
+ validated cache hits. `fetch --dry-run` also summarizes known sizes. Asset totals
117
+ include possible cache hits; each manifest source is resolved separately. Bytes
118
+ from a failed HTTP attempt reset on retry; encoded responses have unknown decoded
119
+ size. Adapters that assemble files from metadata requests show asset-level progress.
120
+ Use `--no-progress` to disable it. Progress is automatically disabled when either
121
+ stdout or stderr is redirected; existing output lines and exit codes are unchanged.
122
+
112
123
  ## Opening CSV data
113
124
 
114
125
  `FetchedAsset.open()` is available since v0.6 with the optional pandas
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.6.0"
3
+ version = "0.7.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.6.0"
3
+ version = "0.7.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -0,0 +1,62 @@
1
+ """Internal synchronous progress events, scoped to one CLI operation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable, Iterator, Sequence
6
+ from contextlib import contextmanager
7
+ from contextvars import ContextVar
8
+ from dataclasses import dataclass
9
+ from typing import Literal
10
+
11
+
12
+ @dataclass(frozen=True)
13
+ class Batch:
14
+ """A resolved group; sizes describe assets, including possible cache hits."""
15
+
16
+ count: int
17
+ known_bytes: int
18
+ unknown_sizes: int
19
+
20
+
21
+ @dataclass(frozen=True)
22
+ class AssetProgress:
23
+ """Start or validated completion of an asset."""
24
+
25
+ asset_id: str
26
+ state: Literal["start", "cached", "fetched"]
27
+ size: int | None
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class TransferProgress:
32
+ """Bytes written in the current HTTP attempt, reset on every retry."""
33
+
34
+ completed: int
35
+ total: int | None
36
+ attempt: int
37
+
38
+
39
+ Event = Batch | AssetProgress | TransferProgress
40
+ _observer: ContextVar[Callable[[Event], None] | None] = ContextVar("progress", default=None)
41
+
42
+
43
+ def emit(event: Event) -> None:
44
+ """Notify the active observer without importing CLI code."""
45
+ observer = _observer.get()
46
+ if observer is not None:
47
+ observer(event)
48
+
49
+
50
+ def batch(sizes: Sequence[int | None]) -> None:
51
+ """Report known and unknown sizes separately; never guess a total."""
52
+ emit(Batch(len(sizes), sum(size for size in sizes if size is not None), sizes.count(None)))
53
+
54
+
55
+ @contextmanager
56
+ def observe(callback: Callable[[Event], None]) -> Iterator[None]:
57
+ """Observe an operation and restore the previous observer on every exit."""
58
+ token = _observer.set(callback)
59
+ try:
60
+ yield
61
+ finally:
62
+ _observer.reset(token)
@@ -9,6 +9,8 @@ import httpx
9
9
  import typer
10
10
 
11
11
  from usdata import __version__, build_query, default_registry
12
+ from usdata._progress import batch
13
+ from usdata.cli.progress import progress
12
14
  from usdata.fetch import ChecksumMismatch
13
15
  from usdata.fetch import fetch as fetch_query
14
16
  from usdata.manifest import lockfile_path
@@ -126,6 +128,7 @@ def fetch(
126
128
  ] = None,
127
129
  cache_dir: Annotated[Path | None, typer.Option(help="Override the cache directory.")] = None,
128
130
  force: Annotated[bool, typer.Option(help="Re-download even if cached.")] = False,
131
+ no_progress: Annotated[bool, typer.Option(help="Disable terminal progress.")] = False,
129
132
  dry_run: Annotated[
130
133
  bool, typer.Option(help="List matching assets without downloading.")
131
134
  ] = False,
@@ -165,8 +168,11 @@ def fetch(
165
168
  for a in assets:
166
169
  typer.echo(f"{a.id}\t{a.href}")
167
170
  typer.echo(f"{len(assets)} asset(s) matched", err=True)
171
+ with progress(disabled=no_progress):
172
+ batch([asset.size for asset in assets])
168
173
  return
169
- fetched = fetch_query(ds, query, root=cache_dir, force=force)
174
+ with progress(disabled=no_progress):
175
+ fetched = fetch_query(ds, query, root=cache_dir, force=force)
170
176
  except (DatasetNotFound, UnknownPlace, ValueError) as e:
171
177
  typer.secho(str(e), err=True, fg="red")
172
178
  raise typer.Exit(code=2) from None
@@ -192,10 +198,12 @@ def pull(
192
198
  bool,
193
199
  typer.Option(help="Ignore an existing lockfile: re-resolve every source and rewrite it."),
194
200
  ] = False,
201
+ no_progress: Annotated[bool, typer.Option(help="Disable terminal progress.")] = False,
195
202
  ) -> None:
196
203
  """Fetch every source in a manifest and write (or restore from) its lockfile."""
197
204
  try:
198
- result = pull_manifest(manifest, root=cache_dir, force=force)
205
+ with progress(disabled=no_progress):
206
+ result = pull_manifest(manifest, root=cache_dir, force=force)
199
207
  except EmptySource as e:
200
208
  typer.secho(str(e), err=True, fg="yellow")
201
209
  raise typer.Exit(code=1) from None
@@ -0,0 +1,89 @@
1
+ """Terminal-only progress rendering; normal CLI output remains machine readable."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import shutil
6
+ import sys
7
+ from collections.abc import Iterator
8
+ from contextlib import contextmanager
9
+ from time import monotonic
10
+ from typing import TextIO
11
+
12
+ from usdata._progress import AssetProgress, Batch, Event, observe
13
+
14
+
15
+ def _interactive() -> bool:
16
+ return sys.stdout.isatty() and sys.stderr.isatty()
17
+
18
+
19
+ class _Display:
20
+ def __init__(self, stream: TextIO) -> None:
21
+ self.stream = stream
22
+ self.width = 0
23
+ self.updated = 0.0
24
+ self.asset = ""
25
+ self.count = 0
26
+ self.done = 0
27
+ self.cached = 0
28
+
29
+ def clear(self) -> None:
30
+ if self.width:
31
+ self.stream.write("\r" + " " * self.width + "\r")
32
+ self.stream.flush()
33
+ self.width = 0
34
+
35
+ def line(self, text: str, *, final: bool = False) -> None:
36
+ self.clear()
37
+ # IDs come from providers: keep control characters out of the terminal.
38
+ text = "".join(c if c.isprintable() else "?" for c in text)
39
+ if final:
40
+ self.stream.write(text + "\n")
41
+ else:
42
+ text = text[: max(1, shutil.get_terminal_size().columns - 1)]
43
+ self.stream.write(text)
44
+ self.width = len(text)
45
+ self.stream.flush()
46
+ self.updated = monotonic()
47
+
48
+ def __call__(self, event: Event) -> None:
49
+ if isinstance(event, Batch):
50
+ self.count, self.done, self.cached = event.count, 0, 0
51
+ sizes = f"{event.known_bytes:,} known bytes"
52
+ if event.unknown_sizes:
53
+ sizes += f"; {event.unknown_sizes} size(s) unknown"
54
+ self.line(f"{event.count} asset(s) resolved; {sizes} (before cache checks)", final=True)
55
+ elif isinstance(event, AssetProgress):
56
+ self.asset = event.asset_id
57
+ if event.state == "start":
58
+ size = f"{event.size:,} bytes" if event.size is not None else "size unknown"
59
+ self.line(f"[{self.done}/{self.count}] checking cache ({size}) | {self.asset}")
60
+ else:
61
+ self.done += 1
62
+ self.cached += event.state == "cached"
63
+ self.line(
64
+ f"[{self.done}/{self.count}] {event.state} "
65
+ f"({event.size:,} bytes; {self.cached} cached) | {self.asset}",
66
+ final=self.done == self.count,
67
+ )
68
+ else:
69
+ if event.completed and monotonic() - self.updated < 0.1:
70
+ return
71
+ size = f"{event.total:,}" if event.total is not None else "unknown"
72
+ self.line(
73
+ f"[{self.done}/{self.count}] {event.completed:,}/{size} bytes "
74
+ f"(attempt {event.attempt}) | {self.asset}"
75
+ )
76
+
77
+
78
+ @contextmanager
79
+ def progress(*, disabled: bool = False) -> Iterator[None]:
80
+ """Render progress on stderr only when both output streams are terminals."""
81
+ if disabled or not _interactive():
82
+ yield
83
+ return
84
+ display = _Display(sys.stderr)
85
+ with observe(display):
86
+ try:
87
+ yield
88
+ finally:
89
+ display.clear()
@@ -209,19 +209,22 @@ datasets:
209
209
 
210
210
  - id: noaa:gsom
211
211
  provider: noaa
212
- status: planned
212
+ status: available
213
213
  domain: surface-weather
214
- target: later
214
+ since: "0.7"
215
215
  title: Global Summary of the Month
216
216
  description: >-
217
217
  Monthly station summaries derived from GHCN-Daily (means, extremes,
218
218
  totals) via the NCEI Access Data Service dataset global-summary-of-the-month.
219
- Shares the GHCN-Daily client and station search.
219
+ Selects whole calendar months and explicit stations, or discovers stations
220
+ through the companion search service.
220
221
  keywords: [climate, monthly, stations, temperature, precipitation, gsom, ncei]
221
222
  protocol: http
222
223
  homepage: https://www.ncei.noaa.gov/access/search/data-search/global-summary-of-the-month
223
224
  license: US Government Work (public domain)
224
225
  capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: true }
226
+ spatial_extent: { west: -180.0, south: -90.0, east: 180.0, north: 90.0 }
227
+ adapter: usdata.providers.noaa.gsom:GlobalSummaryMonthly
225
228
 
226
229
  - id: noaa:gsoy
227
230
  provider: noaa
@@ -7,7 +7,7 @@ from typing import Any
7
7
 
8
8
  from pydantic import BaseModel
9
9
 
10
- from usdata import provenance
10
+ from usdata import _progress, provenance
11
11
  from usdata._files import staged_path
12
12
  from usdata.cache import asset_path, sha256_file
13
13
  from usdata.models import Asset, Dataset, Provenance, Query
@@ -58,6 +58,7 @@ def _fetch_asset(
58
58
  ) -> FetchedAsset:
59
59
  if asset.dataset_id != dataset.id:
60
60
  raise ValueError(f"asset dataset {asset.dataset_id!r} does not match {dataset.id!r}")
61
+ _progress.emit(_progress.AssetProgress(asset.id, "start", asset.size))
61
62
  path = asset_path(asset, root)
62
63
  if not force and path.is_file():
63
64
  try:
@@ -73,6 +74,7 @@ def _fetch_asset(
73
74
  and (asset.checksum is None or prov.checksum == asset.checksum)
74
75
  and sha256_file(path) == prov.checksum
75
76
  ):
77
+ _progress.emit(_progress.AssetProgress(asset.id, "cached", prov.size))
76
78
  return FetchedAsset(asset=asset, path=path, provenance=prov, from_cache=True)
77
79
  with staged_path(path) as tmp:
78
80
  adapter.fetch(asset, tmp)
@@ -81,6 +83,7 @@ def _fetch_asset(
81
83
  raise ChecksumMismatch(f"{asset.id}: expected {asset.checksum}, got {prov.checksum}")
82
84
  # A crash between replacements leaves a detectable mismatch, never a trusted partial file.
83
85
  provenance.write(prov, path)
86
+ _progress.emit(_progress.AssetProgress(asset.id, "fetched", prov.size))
84
87
  return FetchedAsset(asset=asset, path=path, provenance=prov, from_cache=False)
85
88
 
86
89
 
@@ -98,4 +101,5 @@ def fetch(
98
101
  """Resolve and fetch a query, sharing one adapter and closing its owned resources."""
99
102
  with load_adapter(dataset) as adapter:
100
103
  assets = adapter.list_assets(query)
104
+ _progress.batch([asset.size for asset in assets])
101
105
  return [_fetch_asset(dataset, a, adapter, root=root, force=force) for a in assets]
@@ -11,7 +11,7 @@ from typing import Any, TypeVar
11
11
 
12
12
  import httpx
13
13
 
14
- from usdata import __version__
14
+ from usdata import __version__, _progress
15
15
  from usdata._files import staged_path
16
16
 
17
17
  USER_AGENT = f"usdata/{__version__} (+https://github.com/jakeryderv/usdata)"
@@ -80,13 +80,30 @@ def download(url: str, dest: Path, http: httpx.Client | None = None) -> Path:
80
80
  """Download atomically, restarting interrupted GETs up to three total attempts."""
81
81
  own = http is None
82
82
  active = http or client()
83
+ attempt = 0
83
84
 
84
85
  def request() -> Path:
86
+ nonlocal attempt
87
+ attempt += 1
88
+ _progress.emit(_progress.TransferProgress(0, None, attempt))
85
89
  with staged_path(dest) as tmp, active.stream("GET", url) as resp:
86
90
  resp.raise_for_status()
91
+ length = resp.headers.get("Content-Length", "")
92
+ # iter_bytes writes decoded bytes; an encoded length is not comparable.
93
+ total = (
94
+ int(length)
95
+ if length.isascii()
96
+ and length.isdigit()
97
+ and resp.headers.get("Content-Encoding", "identity").lower() == "identity"
98
+ else None
99
+ )
100
+ completed = 0
101
+ _progress.emit(_progress.TransferProgress(completed, total, attempt))
87
102
  with tmp.open("wb") as f:
88
103
  for chunk in resp.iter_bytes():
89
104
  f.write(chunk)
105
+ completed += len(chunk)
106
+ _progress.emit(_progress.TransferProgress(completed, total, attempt))
90
107
  return dest
91
108
 
92
109
  try:
@@ -12,6 +12,7 @@ always goes through search first. Stations are chunked so URLs stay short.
12
12
  from __future__ import annotations
13
13
 
14
14
  import hashlib
15
+ import logging
15
16
  from pathlib import Path
16
17
  from typing import Any
17
18
 
@@ -26,6 +27,7 @@ DATA_URL = "https://www.ncei.noaa.gov/access/services/data/v1"
26
27
  NCEI_DATASET = "daily-summaries"
27
28
  SEARCH_PAGE_SIZE = 1000
28
29
  STATIONS_PER_ASSET = 50
30
+ logger = logging.getLogger(__name__)
29
31
 
30
32
 
31
33
  def _date(value: Any) -> str:
@@ -46,6 +48,8 @@ def _stations_param(raw: Any) -> list[str]:
46
48
  class GhcnDaily(Provider):
47
49
  """GHCN-Daily adapter. Params: ``stations`` (list or comma string), ``units``."""
48
50
 
51
+ ncei_dataset = NCEI_DATASET
52
+
49
53
  def __init__(self, dataset: Dataset, client: httpx.Client | None = None) -> None:
50
54
  super().__init__(dataset)
51
55
  self._client = client
@@ -68,7 +72,7 @@ class GhcnDaily(Provider):
68
72
  raise QueryError("station search needs a bounding box and a time range")
69
73
  b = query.bbox
70
74
  params: dict[str, Any] = {
71
- "dataset": NCEI_DATASET,
75
+ "dataset": self.ncei_dataset,
72
76
  "bbox": f"{b.north},{b.west},{b.south},{b.east}",
73
77
  "startDate": _date(query.time.start),
74
78
  "endDate": _date(query.time.end),
@@ -84,12 +88,29 @@ class GhcnDaily(Provider):
84
88
  resp.raise_for_status()
85
89
  body = resp.json()
86
90
  results = body.get("results", [])
91
+ before = len(found)
87
92
  for result in results:
88
93
  for station in result.get("stations", []):
89
94
  sid = station.get("id")
90
95
  if sid and sid not in seen:
91
96
  seen.add(sid)
92
97
  found.append(sid)
98
+ logger.debug(
99
+ "NCEI station search: url=%s status=%s content_type=%s "
100
+ "count=%r totalCount=%r results=%s new_stations=%s station_sample=%r",
101
+ resp.request.url,
102
+ resp.status_code,
103
+ resp.headers.get("content-type"),
104
+ body.get("count"),
105
+ body.get("totalCount"),
106
+ len(results),
107
+ len(found) - before,
108
+ found[before : before + 5],
109
+ )
110
+ if len(found) == before and logger.isEnabledFor(logging.DEBUG):
111
+ logger.debug(
112
+ "NCEI search page yielded no new stations; response_prefix=%r", resp.text[:512]
113
+ )
93
114
  # "count" is the number matching this query; "totalCount" is dataset-wide.
94
115
  params["offset"] += SEARCH_PAGE_SIZE
95
116
  if not results or params["offset"] >= int(body.get("count", 0)):
@@ -101,7 +122,7 @@ class GhcnDaily(Provider):
101
122
  if query.time is None or query.time.start is None or query.time.end is None:
102
123
  raise QueryError(f"{self.dataset.id} requires both start and end dates")
103
124
  if unknown := set(query.params) - {"stations", "units"}:
104
- raise QueryError(f"unsupported GHCN params: {', '.join(sorted(unknown))}")
125
+ raise QueryError(f"unsupported {self.dataset.id} params: {', '.join(sorted(unknown))}")
105
126
  if query.params.get("units", "metric") not in ("metric", "standard"):
106
127
  raise QueryError("units must be metric or standard")
107
128
  if "stations" in query.params:
@@ -118,7 +139,7 @@ class GhcnDaily(Provider):
118
139
  for i in range(0, len(stations), STATIONS_PER_ASSET):
119
140
  chunk = stations[i : i + STATIONS_PER_ASSET]
120
141
  params: dict[str, Any] = {
121
- "dataset": NCEI_DATASET,
142
+ "dataset": self.ncei_dataset,
122
143
  "stations": ",".join(chunk),
123
144
  "startDate": start,
124
145
  "endDate": end,
@@ -132,7 +153,7 @@ class GhcnDaily(Provider):
132
153
  digest = hashlib.sha1(url.encode()).hexdigest()[:12]
133
154
  assets.append(
134
155
  Asset(
135
- id=f"{NCEI_DATASET}_{start}_{end}_{digest}.csv",
156
+ id=f"{self.ncei_dataset}_{start}_{end}_{digest}.csv",
136
157
  dataset_id=self.dataset.id,
137
158
  href=url,
138
159
  protocol=Protocol.HTTP,
@@ -0,0 +1,49 @@
1
+ """Monthly station summaries through the NCEI Access Data Service.
2
+
3
+ Params: ``stations`` (list or comma string) and ``units`` (metric or standard).
4
+ Both dates are required; every UTC calendar month touched by the interval is
5
+ selected in full. Geographic discovery uses the same normalized month bounds.
6
+ """
7
+
8
+ from calendar import monthrange
9
+ from datetime import UTC
10
+
11
+ from usdata.models import Asset, Query, TimeRange
12
+ from usdata.providers.base import QueryError
13
+ from usdata.providers.noaa.ghcnd import GhcnDaily
14
+
15
+
16
+ class GlobalSummaryMonthly(GhcnDaily):
17
+ """GSOM CSV subsets, sharing NCEI station discovery and transport with GHCN."""
18
+
19
+ ncei_dataset = "global-summary-of-the-month"
20
+
21
+ def _monthly_query(self, query: Query) -> Query:
22
+ if query.time is None or query.time.start is None or query.time.end is None:
23
+ raise QueryError(f"{self.dataset.id} requires both start and end dates")
24
+ start = query.time.start.replace(tzinfo=query.time.start.tzinfo or UTC).astimezone(UTC)
25
+ end = query.time.end.replace(tzinfo=query.time.end.tzinfo or UTC).astimezone(UTC)
26
+ return query.model_copy(
27
+ update={
28
+ "time": TimeRange(
29
+ start=start.replace(day=1, hour=0, minute=0, second=0, microsecond=0),
30
+ end=end.replace(
31
+ day=monthrange(end.year, end.month)[1],
32
+ hour=23,
33
+ minute=59,
34
+ second=59,
35
+ microsecond=999999,
36
+ ),
37
+ )
38
+ }
39
+ )
40
+
41
+ def find_stations(self, query: Query) -> list[str]:
42
+ """Find stations overlapping the selected complete calendar months."""
43
+ return super().find_stations(self._monthly_query(query))
44
+
45
+ def list_assets(self, query: Query) -> list[Asset]:
46
+ """Resolve monthly CSVs with stable URLs and complete-month asset bounds."""
47
+ if "stations" in query.params and query.bbox is not None:
48
+ raise QueryError("pass stations or a location/bbox, not both")
49
+ return super().list_assets(self._monthly_query(query))
@@ -16,7 +16,7 @@ from pathlib import Path
16
16
 
17
17
  from pydantic import BaseModel
18
18
 
19
- from usdata import __version__, provenance
19
+ from usdata import __version__, _progress, provenance
20
20
  from usdata.cache import asset_path, sha256_file
21
21
  from usdata.fetch import ChecksumMismatch, FetchedAsset, _fetch_asset, fetch
22
22
  from usdata.manifest import LockedAsset, Lockfile, Manifest, lockfile_path
@@ -111,13 +111,21 @@ def restore(
111
111
  lock = Lockfile.load(lock_path)
112
112
  _check_manifest(manifest_path, lock)
113
113
  fetched: list[FetchedAsset] = []
114
+ _progress.batch([entry.provenance.size for entry in lock.assets])
114
115
  adapters: dict[str, Provider] = {}
115
116
  with ExitStack() as stack:
116
117
  for entry in lock.assets:
117
118
  dataset = reg.get(entry.asset.dataset_id)
118
119
  path = asset_path(entry.asset, root)
120
+ if path.is_file():
121
+ _progress.emit(
122
+ _progress.AssetProgress(entry.asset.id, "start", entry.provenance.size)
123
+ )
119
124
  if path.is_file() and sha256_file(path) == entry.provenance.checksum:
120
125
  provenance.write(entry.provenance, path)
126
+ _progress.emit(
127
+ _progress.AssetProgress(entry.asset.id, "cached", entry.provenance.size)
128
+ )
121
129
  fetched.append(
122
130
  FetchedAsset(
123
131
  asset=entry.asset, path=path, provenance=entry.provenance, from_cache=True
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes