usdata 0.6.0__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {usdata-0.6.0 → usdata-0.7.0}/PKG-INFO +16 -5
- {usdata-0.6.0 → usdata-0.7.0}/README.md +15 -4
- {usdata-0.6.0 → usdata-0.7.0}/pyproject.toml +1 -1
- {usdata-0.6.0 → usdata-0.7.0}/pyproject.toml.orig +1 -1
- usdata-0.7.0/src/usdata/_progress.py +62 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/cli/app.py +10 -2
- usdata-0.7.0/src/usdata/cli/progress.py +89 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/data/registry.yaml +6 -3
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/fetch.py +5 -1
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/protocols/http.py +18 -1
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/noaa/ghcnd.py +25 -4
- usdata-0.7.0/src/usdata/providers/noaa/gsom.py +49 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/pull.py +9 -1
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/__init__.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/_files.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/cache.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/cli/__init__.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/data/nexrad_sites.csv +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/data/places.csv +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/data/places.sources.json +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/manifest.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/models.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/protocols/__init__.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/protocols/erddap.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/protocols/s3.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/provenance.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/__init__.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/base.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/noaa/__init__.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/noaa/nexrad.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/noaa/sites.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/usgs/__init__.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/providers/usgs/daily.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/py.typed +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/query.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/readers.py +0 -0
- {usdata-0.6.0 → usdata-0.7.0}/src/usdata/registry.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: usdata
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
|
|
5
5
|
Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
|
|
6
6
|
Author: Jake Van Slyke
|
|
@@ -26,9 +26,10 @@ Description-Content-Type: text/markdown
|
|
|
26
26
|
Unified Python SDK and CLI for discovering, fetching, and tracking the
|
|
27
27
|
provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
28
28
|
|
|
29
|
-
> Status: pre-alpha. v0.
|
|
30
|
-
> values, and CoastWatch SST subsets with provenance,
|
|
31
|
-
> lookup
|
|
29
|
+
> Status: pre-alpha. v0.7 supports GHCN-Daily, GSOM monthly summaries,
|
|
30
|
+
> NEXRAD Level II, USGS daily values, and CoastWatch SST subsets with provenance,
|
|
31
|
+
> plus Census state/county lookup, optional pandas CSV readers, and terminal
|
|
32
|
+
> download progress. Other datasets are planned.
|
|
32
33
|
> See [docs/roadmap.md](docs/roadmap.md).
|
|
33
34
|
|
|
34
35
|
## Providers
|
|
@@ -36,7 +37,7 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
|
36
37
|
<!-- registry:start -->
|
|
37
38
|
| Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
|
|
38
39
|
|---|---:|---:|---:|---|---|
|
|
39
|
-
| [NOAA](docs/providers/noaa.md) |
|
|
40
|
+
| [NOAA](docs/providers/noaa.md) | 4 | 0 | 25 | — | `ghcn-daily`, `gsom`, `nexrad-level2`, `coastwatch-sst`, +25 planned |
|
|
40
41
|
| [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
|
|
41
42
|
| [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
|
|
42
43
|
| [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
@@ -132,6 +133,16 @@ sources:
|
|
|
132
133
|
end: 2024-05-31
|
|
133
134
|
```
|
|
134
135
|
|
|
136
|
+
Terminal progress is available since v0.7. On a terminal, `fetch` and
|
|
137
|
+
`pull` show progress on stderr: resolved asset counts,
|
|
138
|
+
known bytes and unknown sizes, HTTP download bytes for the current attempt, and
|
|
139
|
+
validated cache hits. `fetch --dry-run` also summarizes known sizes. Asset totals
|
|
140
|
+
include possible cache hits; each manifest source is resolved separately. Bytes
|
|
141
|
+
from a failed HTTP attempt reset on retry; encoded responses have unknown decoded
|
|
142
|
+
size. Adapters that assemble files from metadata requests show asset-level progress.
|
|
143
|
+
Use `--no-progress` to disable it. Progress is automatically disabled when either
|
|
144
|
+
stdout or stderr is redirected; existing output lines and exit codes are unchanged.
|
|
145
|
+
|
|
135
146
|
## Opening CSV data
|
|
136
147
|
|
|
137
148
|
`FetchedAsset.open()` is available since v0.6 with the optional pandas
|
|
@@ -3,9 +3,10 @@
|
|
|
3
3
|
Unified Python SDK and CLI for discovering, fetching, and tracking the
|
|
4
4
|
provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
5
5
|
|
|
6
|
-
> Status: pre-alpha. v0.
|
|
7
|
-
> values, and CoastWatch SST subsets with provenance,
|
|
8
|
-
> lookup
|
|
6
|
+
> Status: pre-alpha. v0.7 supports GHCN-Daily, GSOM monthly summaries,
|
|
7
|
+
> NEXRAD Level II, USGS daily values, and CoastWatch SST subsets with provenance,
|
|
8
|
+
> plus Census state/county lookup, optional pandas CSV readers, and terminal
|
|
9
|
+
> download progress. Other datasets are planned.
|
|
9
10
|
> See [docs/roadmap.md](docs/roadmap.md).
|
|
10
11
|
|
|
11
12
|
## Providers
|
|
@@ -13,7 +14,7 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
|
13
14
|
<!-- registry:start -->
|
|
14
15
|
| Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
|
|
15
16
|
|---|---:|---:|---:|---|---|
|
|
16
|
-
| [NOAA](docs/providers/noaa.md) |
|
|
17
|
+
| [NOAA](docs/providers/noaa.md) | 4 | 0 | 25 | — | `ghcn-daily`, `gsom`, `nexrad-level2`, `coastwatch-sst`, +25 planned |
|
|
17
18
|
| [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
|
|
18
19
|
| [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
|
|
19
20
|
| [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
@@ -109,6 +110,16 @@ sources:
|
|
|
109
110
|
end: 2024-05-31
|
|
110
111
|
```
|
|
111
112
|
|
|
113
|
+
Terminal progress is available since v0.7. On a terminal, `fetch` and
|
|
114
|
+
`pull` show progress on stderr: resolved asset counts,
|
|
115
|
+
known bytes and unknown sizes, HTTP download bytes for the current attempt, and
|
|
116
|
+
validated cache hits. `fetch --dry-run` also summarizes known sizes. Asset totals
|
|
117
|
+
include possible cache hits; each manifest source is resolved separately. Bytes
|
|
118
|
+
from a failed HTTP attempt reset on retry; encoded responses have unknown decoded
|
|
119
|
+
size. Adapters that assemble files from metadata requests show asset-level progress.
|
|
120
|
+
Use `--no-progress` to disable it. Progress is automatically disabled when either
|
|
121
|
+
stdout or stderr is redirected; existing output lines and exit codes are unchanged.
|
|
122
|
+
|
|
112
123
|
## Opening CSV data
|
|
113
124
|
|
|
114
125
|
`FetchedAsset.open()` is available since v0.6 with the optional pandas
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Internal synchronous progress events, scoped to one CLI operation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable, Iterator, Sequence
|
|
6
|
+
from contextlib import contextmanager
|
|
7
|
+
from contextvars import ContextVar
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from typing import Literal
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(frozen=True)
|
|
13
|
+
class Batch:
|
|
14
|
+
"""A resolved group; sizes describe assets, including possible cache hits."""
|
|
15
|
+
|
|
16
|
+
count: int
|
|
17
|
+
known_bytes: int
|
|
18
|
+
unknown_sizes: int
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class AssetProgress:
|
|
23
|
+
"""Start or validated completion of an asset."""
|
|
24
|
+
|
|
25
|
+
asset_id: str
|
|
26
|
+
state: Literal["start", "cached", "fetched"]
|
|
27
|
+
size: int | None
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class TransferProgress:
|
|
32
|
+
"""Bytes written in the current HTTP attempt, reset on every retry."""
|
|
33
|
+
|
|
34
|
+
completed: int
|
|
35
|
+
total: int | None
|
|
36
|
+
attempt: int
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
Event = Batch | AssetProgress | TransferProgress
|
|
40
|
+
_observer: ContextVar[Callable[[Event], None] | None] = ContextVar("progress", default=None)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def emit(event: Event) -> None:
|
|
44
|
+
"""Notify the active observer without importing CLI code."""
|
|
45
|
+
observer = _observer.get()
|
|
46
|
+
if observer is not None:
|
|
47
|
+
observer(event)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def batch(sizes: Sequence[int | None]) -> None:
|
|
51
|
+
"""Report known and unknown sizes separately; never guess a total."""
|
|
52
|
+
emit(Batch(len(sizes), sum(size for size in sizes if size is not None), sizes.count(None)))
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@contextmanager
|
|
56
|
+
def observe(callback: Callable[[Event], None]) -> Iterator[None]:
|
|
57
|
+
"""Observe an operation and restore the previous observer on every exit."""
|
|
58
|
+
token = _observer.set(callback)
|
|
59
|
+
try:
|
|
60
|
+
yield
|
|
61
|
+
finally:
|
|
62
|
+
_observer.reset(token)
|
|
@@ -9,6 +9,8 @@ import httpx
|
|
|
9
9
|
import typer
|
|
10
10
|
|
|
11
11
|
from usdata import __version__, build_query, default_registry
|
|
12
|
+
from usdata._progress import batch
|
|
13
|
+
from usdata.cli.progress import progress
|
|
12
14
|
from usdata.fetch import ChecksumMismatch
|
|
13
15
|
from usdata.fetch import fetch as fetch_query
|
|
14
16
|
from usdata.manifest import lockfile_path
|
|
@@ -126,6 +128,7 @@ def fetch(
|
|
|
126
128
|
] = None,
|
|
127
129
|
cache_dir: Annotated[Path | None, typer.Option(help="Override the cache directory.")] = None,
|
|
128
130
|
force: Annotated[bool, typer.Option(help="Re-download even if cached.")] = False,
|
|
131
|
+
no_progress: Annotated[bool, typer.Option(help="Disable terminal progress.")] = False,
|
|
129
132
|
dry_run: Annotated[
|
|
130
133
|
bool, typer.Option(help="List matching assets without downloading.")
|
|
131
134
|
] = False,
|
|
@@ -165,8 +168,11 @@ def fetch(
|
|
|
165
168
|
for a in assets:
|
|
166
169
|
typer.echo(f"{a.id}\t{a.href}")
|
|
167
170
|
typer.echo(f"{len(assets)} asset(s) matched", err=True)
|
|
171
|
+
with progress(disabled=no_progress):
|
|
172
|
+
batch([asset.size for asset in assets])
|
|
168
173
|
return
|
|
169
|
-
|
|
174
|
+
with progress(disabled=no_progress):
|
|
175
|
+
fetched = fetch_query(ds, query, root=cache_dir, force=force)
|
|
170
176
|
except (DatasetNotFound, UnknownPlace, ValueError) as e:
|
|
171
177
|
typer.secho(str(e), err=True, fg="red")
|
|
172
178
|
raise typer.Exit(code=2) from None
|
|
@@ -192,10 +198,12 @@ def pull(
|
|
|
192
198
|
bool,
|
|
193
199
|
typer.Option(help="Ignore an existing lockfile: re-resolve every source and rewrite it."),
|
|
194
200
|
] = False,
|
|
201
|
+
no_progress: Annotated[bool, typer.Option(help="Disable terminal progress.")] = False,
|
|
195
202
|
) -> None:
|
|
196
203
|
"""Fetch every source in a manifest and write (or restore from) its lockfile."""
|
|
197
204
|
try:
|
|
198
|
-
|
|
205
|
+
with progress(disabled=no_progress):
|
|
206
|
+
result = pull_manifest(manifest, root=cache_dir, force=force)
|
|
199
207
|
except EmptySource as e:
|
|
200
208
|
typer.secho(str(e), err=True, fg="yellow")
|
|
201
209
|
raise typer.Exit(code=1) from None
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""Terminal-only progress rendering; normal CLI output remains machine readable."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import shutil
|
|
6
|
+
import sys
|
|
7
|
+
from collections.abc import Iterator
|
|
8
|
+
from contextlib import contextmanager
|
|
9
|
+
from time import monotonic
|
|
10
|
+
from typing import TextIO
|
|
11
|
+
|
|
12
|
+
from usdata._progress import AssetProgress, Batch, Event, observe
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _interactive() -> bool:
|
|
16
|
+
return sys.stdout.isatty() and sys.stderr.isatty()
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class _Display:
|
|
20
|
+
def __init__(self, stream: TextIO) -> None:
|
|
21
|
+
self.stream = stream
|
|
22
|
+
self.width = 0
|
|
23
|
+
self.updated = 0.0
|
|
24
|
+
self.asset = ""
|
|
25
|
+
self.count = 0
|
|
26
|
+
self.done = 0
|
|
27
|
+
self.cached = 0
|
|
28
|
+
|
|
29
|
+
def clear(self) -> None:
|
|
30
|
+
if self.width:
|
|
31
|
+
self.stream.write("\r" + " " * self.width + "\r")
|
|
32
|
+
self.stream.flush()
|
|
33
|
+
self.width = 0
|
|
34
|
+
|
|
35
|
+
def line(self, text: str, *, final: bool = False) -> None:
|
|
36
|
+
self.clear()
|
|
37
|
+
# IDs come from providers: keep control characters out of the terminal.
|
|
38
|
+
text = "".join(c if c.isprintable() else "?" for c in text)
|
|
39
|
+
if final:
|
|
40
|
+
self.stream.write(text + "\n")
|
|
41
|
+
else:
|
|
42
|
+
text = text[: max(1, shutil.get_terminal_size().columns - 1)]
|
|
43
|
+
self.stream.write(text)
|
|
44
|
+
self.width = len(text)
|
|
45
|
+
self.stream.flush()
|
|
46
|
+
self.updated = monotonic()
|
|
47
|
+
|
|
48
|
+
def __call__(self, event: Event) -> None:
|
|
49
|
+
if isinstance(event, Batch):
|
|
50
|
+
self.count, self.done, self.cached = event.count, 0, 0
|
|
51
|
+
sizes = f"{event.known_bytes:,} known bytes"
|
|
52
|
+
if event.unknown_sizes:
|
|
53
|
+
sizes += f"; {event.unknown_sizes} size(s) unknown"
|
|
54
|
+
self.line(f"{event.count} asset(s) resolved; {sizes} (before cache checks)", final=True)
|
|
55
|
+
elif isinstance(event, AssetProgress):
|
|
56
|
+
self.asset = event.asset_id
|
|
57
|
+
if event.state == "start":
|
|
58
|
+
size = f"{event.size:,} bytes" if event.size is not None else "size unknown"
|
|
59
|
+
self.line(f"[{self.done}/{self.count}] checking cache ({size}) | {self.asset}")
|
|
60
|
+
else:
|
|
61
|
+
self.done += 1
|
|
62
|
+
self.cached += event.state == "cached"
|
|
63
|
+
self.line(
|
|
64
|
+
f"[{self.done}/{self.count}] {event.state} "
|
|
65
|
+
f"({event.size:,} bytes; {self.cached} cached) | {self.asset}",
|
|
66
|
+
final=self.done == self.count,
|
|
67
|
+
)
|
|
68
|
+
else:
|
|
69
|
+
if event.completed and monotonic() - self.updated < 0.1:
|
|
70
|
+
return
|
|
71
|
+
size = f"{event.total:,}" if event.total is not None else "unknown"
|
|
72
|
+
self.line(
|
|
73
|
+
f"[{self.done}/{self.count}] {event.completed:,}/{size} bytes "
|
|
74
|
+
f"(attempt {event.attempt}) | {self.asset}"
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
@contextmanager
|
|
79
|
+
def progress(*, disabled: bool = False) -> Iterator[None]:
|
|
80
|
+
"""Render progress on stderr only when both output streams are terminals."""
|
|
81
|
+
if disabled or not _interactive():
|
|
82
|
+
yield
|
|
83
|
+
return
|
|
84
|
+
display = _Display(sys.stderr)
|
|
85
|
+
with observe(display):
|
|
86
|
+
try:
|
|
87
|
+
yield
|
|
88
|
+
finally:
|
|
89
|
+
display.clear()
|
|
@@ -209,19 +209,22 @@ datasets:
|
|
|
209
209
|
|
|
210
210
|
- id: noaa:gsom
|
|
211
211
|
provider: noaa
|
|
212
|
-
status:
|
|
212
|
+
status: available
|
|
213
213
|
domain: surface-weather
|
|
214
|
-
|
|
214
|
+
since: "0.7"
|
|
215
215
|
title: Global Summary of the Month
|
|
216
216
|
description: >-
|
|
217
217
|
Monthly station summaries derived from GHCN-Daily (means, extremes,
|
|
218
218
|
totals) via the NCEI Access Data Service dataset global-summary-of-the-month.
|
|
219
|
-
|
|
219
|
+
Selects whole calendar months and explicit stations, or discovers stations
|
|
220
|
+
through the companion search service.
|
|
220
221
|
keywords: [climate, monthly, stations, temperature, precipitation, gsom, ncei]
|
|
221
222
|
protocol: http
|
|
222
223
|
homepage: https://www.ncei.noaa.gov/access/search/data-search/global-summary-of-the-month
|
|
223
224
|
license: US Government Work (public domain)
|
|
224
225
|
capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: true }
|
|
226
|
+
spatial_extent: { west: -180.0, south: -90.0, east: 180.0, north: 90.0 }
|
|
227
|
+
adapter: usdata.providers.noaa.gsom:GlobalSummaryMonthly
|
|
225
228
|
|
|
226
229
|
- id: noaa:gsoy
|
|
227
230
|
provider: noaa
|
|
@@ -7,7 +7,7 @@ from typing import Any
|
|
|
7
7
|
|
|
8
8
|
from pydantic import BaseModel
|
|
9
9
|
|
|
10
|
-
from usdata import provenance
|
|
10
|
+
from usdata import _progress, provenance
|
|
11
11
|
from usdata._files import staged_path
|
|
12
12
|
from usdata.cache import asset_path, sha256_file
|
|
13
13
|
from usdata.models import Asset, Dataset, Provenance, Query
|
|
@@ -58,6 +58,7 @@ def _fetch_asset(
|
|
|
58
58
|
) -> FetchedAsset:
|
|
59
59
|
if asset.dataset_id != dataset.id:
|
|
60
60
|
raise ValueError(f"asset dataset {asset.dataset_id!r} does not match {dataset.id!r}")
|
|
61
|
+
_progress.emit(_progress.AssetProgress(asset.id, "start", asset.size))
|
|
61
62
|
path = asset_path(asset, root)
|
|
62
63
|
if not force and path.is_file():
|
|
63
64
|
try:
|
|
@@ -73,6 +74,7 @@ def _fetch_asset(
|
|
|
73
74
|
and (asset.checksum is None or prov.checksum == asset.checksum)
|
|
74
75
|
and sha256_file(path) == prov.checksum
|
|
75
76
|
):
|
|
77
|
+
_progress.emit(_progress.AssetProgress(asset.id, "cached", prov.size))
|
|
76
78
|
return FetchedAsset(asset=asset, path=path, provenance=prov, from_cache=True)
|
|
77
79
|
with staged_path(path) as tmp:
|
|
78
80
|
adapter.fetch(asset, tmp)
|
|
@@ -81,6 +83,7 @@ def _fetch_asset(
|
|
|
81
83
|
raise ChecksumMismatch(f"{asset.id}: expected {asset.checksum}, got {prov.checksum}")
|
|
82
84
|
# A crash between replacements leaves a detectable mismatch, never a trusted partial file.
|
|
83
85
|
provenance.write(prov, path)
|
|
86
|
+
_progress.emit(_progress.AssetProgress(asset.id, "fetched", prov.size))
|
|
84
87
|
return FetchedAsset(asset=asset, path=path, provenance=prov, from_cache=False)
|
|
85
88
|
|
|
86
89
|
|
|
@@ -98,4 +101,5 @@ def fetch(
|
|
|
98
101
|
"""Resolve and fetch a query, sharing one adapter and closing its owned resources."""
|
|
99
102
|
with load_adapter(dataset) as adapter:
|
|
100
103
|
assets = adapter.list_assets(query)
|
|
104
|
+
_progress.batch([asset.size for asset in assets])
|
|
101
105
|
return [_fetch_asset(dataset, a, adapter, root=root, force=force) for a in assets]
|
|
@@ -11,7 +11,7 @@ from typing import Any, TypeVar
|
|
|
11
11
|
|
|
12
12
|
import httpx
|
|
13
13
|
|
|
14
|
-
from usdata import __version__
|
|
14
|
+
from usdata import __version__, _progress
|
|
15
15
|
from usdata._files import staged_path
|
|
16
16
|
|
|
17
17
|
USER_AGENT = f"usdata/{__version__} (+https://github.com/jakeryderv/usdata)"
|
|
@@ -80,13 +80,30 @@ def download(url: str, dest: Path, http: httpx.Client | None = None) -> Path:
|
|
|
80
80
|
"""Download atomically, restarting interrupted GETs up to three total attempts."""
|
|
81
81
|
own = http is None
|
|
82
82
|
active = http or client()
|
|
83
|
+
attempt = 0
|
|
83
84
|
|
|
84
85
|
def request() -> Path:
|
|
86
|
+
nonlocal attempt
|
|
87
|
+
attempt += 1
|
|
88
|
+
_progress.emit(_progress.TransferProgress(0, None, attempt))
|
|
85
89
|
with staged_path(dest) as tmp, active.stream("GET", url) as resp:
|
|
86
90
|
resp.raise_for_status()
|
|
91
|
+
length = resp.headers.get("Content-Length", "")
|
|
92
|
+
# iter_bytes writes decoded bytes; an encoded length is not comparable.
|
|
93
|
+
total = (
|
|
94
|
+
int(length)
|
|
95
|
+
if length.isascii()
|
|
96
|
+
and length.isdigit()
|
|
97
|
+
and resp.headers.get("Content-Encoding", "identity").lower() == "identity"
|
|
98
|
+
else None
|
|
99
|
+
)
|
|
100
|
+
completed = 0
|
|
101
|
+
_progress.emit(_progress.TransferProgress(completed, total, attempt))
|
|
87
102
|
with tmp.open("wb") as f:
|
|
88
103
|
for chunk in resp.iter_bytes():
|
|
89
104
|
f.write(chunk)
|
|
105
|
+
completed += len(chunk)
|
|
106
|
+
_progress.emit(_progress.TransferProgress(completed, total, attempt))
|
|
90
107
|
return dest
|
|
91
108
|
|
|
92
109
|
try:
|
|
@@ -12,6 +12,7 @@ always goes through search first. Stations are chunked so URLs stay short.
|
|
|
12
12
|
from __future__ import annotations
|
|
13
13
|
|
|
14
14
|
import hashlib
|
|
15
|
+
import logging
|
|
15
16
|
from pathlib import Path
|
|
16
17
|
from typing import Any
|
|
17
18
|
|
|
@@ -26,6 +27,7 @@ DATA_URL = "https://www.ncei.noaa.gov/access/services/data/v1"
|
|
|
26
27
|
NCEI_DATASET = "daily-summaries"
|
|
27
28
|
SEARCH_PAGE_SIZE = 1000
|
|
28
29
|
STATIONS_PER_ASSET = 50
|
|
30
|
+
logger = logging.getLogger(__name__)
|
|
29
31
|
|
|
30
32
|
|
|
31
33
|
def _date(value: Any) -> str:
|
|
@@ -46,6 +48,8 @@ def _stations_param(raw: Any) -> list[str]:
|
|
|
46
48
|
class GhcnDaily(Provider):
|
|
47
49
|
"""GHCN-Daily adapter. Params: ``stations`` (list or comma string), ``units``."""
|
|
48
50
|
|
|
51
|
+
ncei_dataset = NCEI_DATASET
|
|
52
|
+
|
|
49
53
|
def __init__(self, dataset: Dataset, client: httpx.Client | None = None) -> None:
|
|
50
54
|
super().__init__(dataset)
|
|
51
55
|
self._client = client
|
|
@@ -68,7 +72,7 @@ class GhcnDaily(Provider):
|
|
|
68
72
|
raise QueryError("station search needs a bounding box and a time range")
|
|
69
73
|
b = query.bbox
|
|
70
74
|
params: dict[str, Any] = {
|
|
71
|
-
"dataset":
|
|
75
|
+
"dataset": self.ncei_dataset,
|
|
72
76
|
"bbox": f"{b.north},{b.west},{b.south},{b.east}",
|
|
73
77
|
"startDate": _date(query.time.start),
|
|
74
78
|
"endDate": _date(query.time.end),
|
|
@@ -84,12 +88,29 @@ class GhcnDaily(Provider):
|
|
|
84
88
|
resp.raise_for_status()
|
|
85
89
|
body = resp.json()
|
|
86
90
|
results = body.get("results", [])
|
|
91
|
+
before = len(found)
|
|
87
92
|
for result in results:
|
|
88
93
|
for station in result.get("stations", []):
|
|
89
94
|
sid = station.get("id")
|
|
90
95
|
if sid and sid not in seen:
|
|
91
96
|
seen.add(sid)
|
|
92
97
|
found.append(sid)
|
|
98
|
+
logger.debug(
|
|
99
|
+
"NCEI station search: url=%s status=%s content_type=%s "
|
|
100
|
+
"count=%r totalCount=%r results=%s new_stations=%s station_sample=%r",
|
|
101
|
+
resp.request.url,
|
|
102
|
+
resp.status_code,
|
|
103
|
+
resp.headers.get("content-type"),
|
|
104
|
+
body.get("count"),
|
|
105
|
+
body.get("totalCount"),
|
|
106
|
+
len(results),
|
|
107
|
+
len(found) - before,
|
|
108
|
+
found[before : before + 5],
|
|
109
|
+
)
|
|
110
|
+
if len(found) == before and logger.isEnabledFor(logging.DEBUG):
|
|
111
|
+
logger.debug(
|
|
112
|
+
"NCEI search page yielded no new stations; response_prefix=%r", resp.text[:512]
|
|
113
|
+
)
|
|
93
114
|
# "count" is the number matching this query; "totalCount" is dataset-wide.
|
|
94
115
|
params["offset"] += SEARCH_PAGE_SIZE
|
|
95
116
|
if not results or params["offset"] >= int(body.get("count", 0)):
|
|
@@ -101,7 +122,7 @@ class GhcnDaily(Provider):
|
|
|
101
122
|
if query.time is None or query.time.start is None or query.time.end is None:
|
|
102
123
|
raise QueryError(f"{self.dataset.id} requires both start and end dates")
|
|
103
124
|
if unknown := set(query.params) - {"stations", "units"}:
|
|
104
|
-
raise QueryError(f"unsupported
|
|
125
|
+
raise QueryError(f"unsupported {self.dataset.id} params: {', '.join(sorted(unknown))}")
|
|
105
126
|
if query.params.get("units", "metric") not in ("metric", "standard"):
|
|
106
127
|
raise QueryError("units must be metric or standard")
|
|
107
128
|
if "stations" in query.params:
|
|
@@ -118,7 +139,7 @@ class GhcnDaily(Provider):
|
|
|
118
139
|
for i in range(0, len(stations), STATIONS_PER_ASSET):
|
|
119
140
|
chunk = stations[i : i + STATIONS_PER_ASSET]
|
|
120
141
|
params: dict[str, Any] = {
|
|
121
|
-
"dataset":
|
|
142
|
+
"dataset": self.ncei_dataset,
|
|
122
143
|
"stations": ",".join(chunk),
|
|
123
144
|
"startDate": start,
|
|
124
145
|
"endDate": end,
|
|
@@ -132,7 +153,7 @@ class GhcnDaily(Provider):
|
|
|
132
153
|
digest = hashlib.sha1(url.encode()).hexdigest()[:12]
|
|
133
154
|
assets.append(
|
|
134
155
|
Asset(
|
|
135
|
-
id=f"{
|
|
156
|
+
id=f"{self.ncei_dataset}_{start}_{end}_{digest}.csv",
|
|
136
157
|
dataset_id=self.dataset.id,
|
|
137
158
|
href=url,
|
|
138
159
|
protocol=Protocol.HTTP,
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""Monthly station summaries through the NCEI Access Data Service.
|
|
2
|
+
|
|
3
|
+
Params: ``stations`` (list or comma string) and ``units`` (metric or standard).
|
|
4
|
+
Both dates are required; every UTC calendar month touched by the interval is
|
|
5
|
+
selected in full. Geographic discovery uses the same normalized month bounds.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from calendar import monthrange
|
|
9
|
+
from datetime import UTC
|
|
10
|
+
|
|
11
|
+
from usdata.models import Asset, Query, TimeRange
|
|
12
|
+
from usdata.providers.base import QueryError
|
|
13
|
+
from usdata.providers.noaa.ghcnd import GhcnDaily
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class GlobalSummaryMonthly(GhcnDaily):
|
|
17
|
+
"""GSOM CSV subsets, sharing NCEI station discovery and transport with GHCN."""
|
|
18
|
+
|
|
19
|
+
ncei_dataset = "global-summary-of-the-month"
|
|
20
|
+
|
|
21
|
+
def _monthly_query(self, query: Query) -> Query:
|
|
22
|
+
if query.time is None or query.time.start is None or query.time.end is None:
|
|
23
|
+
raise QueryError(f"{self.dataset.id} requires both start and end dates")
|
|
24
|
+
start = query.time.start.replace(tzinfo=query.time.start.tzinfo or UTC).astimezone(UTC)
|
|
25
|
+
end = query.time.end.replace(tzinfo=query.time.end.tzinfo or UTC).astimezone(UTC)
|
|
26
|
+
return query.model_copy(
|
|
27
|
+
update={
|
|
28
|
+
"time": TimeRange(
|
|
29
|
+
start=start.replace(day=1, hour=0, minute=0, second=0, microsecond=0),
|
|
30
|
+
end=end.replace(
|
|
31
|
+
day=monthrange(end.year, end.month)[1],
|
|
32
|
+
hour=23,
|
|
33
|
+
minute=59,
|
|
34
|
+
second=59,
|
|
35
|
+
microsecond=999999,
|
|
36
|
+
),
|
|
37
|
+
)
|
|
38
|
+
}
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
def find_stations(self, query: Query) -> list[str]:
|
|
42
|
+
"""Find stations overlapping the selected complete calendar months."""
|
|
43
|
+
return super().find_stations(self._monthly_query(query))
|
|
44
|
+
|
|
45
|
+
def list_assets(self, query: Query) -> list[Asset]:
|
|
46
|
+
"""Resolve monthly CSVs with stable URLs and complete-month asset bounds."""
|
|
47
|
+
if "stations" in query.params and query.bbox is not None:
|
|
48
|
+
raise QueryError("pass stations or a location/bbox, not both")
|
|
49
|
+
return super().list_assets(self._monthly_query(query))
|
|
@@ -16,7 +16,7 @@ from pathlib import Path
|
|
|
16
16
|
|
|
17
17
|
from pydantic import BaseModel
|
|
18
18
|
|
|
19
|
-
from usdata import __version__, provenance
|
|
19
|
+
from usdata import __version__, _progress, provenance
|
|
20
20
|
from usdata.cache import asset_path, sha256_file
|
|
21
21
|
from usdata.fetch import ChecksumMismatch, FetchedAsset, _fetch_asset, fetch
|
|
22
22
|
from usdata.manifest import LockedAsset, Lockfile, Manifest, lockfile_path
|
|
@@ -111,13 +111,21 @@ def restore(
|
|
|
111
111
|
lock = Lockfile.load(lock_path)
|
|
112
112
|
_check_manifest(manifest_path, lock)
|
|
113
113
|
fetched: list[FetchedAsset] = []
|
|
114
|
+
_progress.batch([entry.provenance.size for entry in lock.assets])
|
|
114
115
|
adapters: dict[str, Provider] = {}
|
|
115
116
|
with ExitStack() as stack:
|
|
116
117
|
for entry in lock.assets:
|
|
117
118
|
dataset = reg.get(entry.asset.dataset_id)
|
|
118
119
|
path = asset_path(entry.asset, root)
|
|
120
|
+
if path.is_file():
|
|
121
|
+
_progress.emit(
|
|
122
|
+
_progress.AssetProgress(entry.asset.id, "start", entry.provenance.size)
|
|
123
|
+
)
|
|
119
124
|
if path.is_file() and sha256_file(path) == entry.provenance.checksum:
|
|
120
125
|
provenance.write(entry.provenance, path)
|
|
126
|
+
_progress.emit(
|
|
127
|
+
_progress.AssetProgress(entry.asset.id, "cached", entry.provenance.size)
|
|
128
|
+
)
|
|
121
129
|
fetched.append(
|
|
122
130
|
FetchedAsset(
|
|
123
131
|
asset=entry.asset, path=path, provenance=entry.provenance, from_cache=True
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|