usdata 0.5.0__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {usdata-0.5.0 → usdata-0.7.0}/PKG-INFO +34 -11
- {usdata-0.5.0 → usdata-0.7.0}/README.md +31 -10
- {usdata-0.5.0 → usdata-0.7.0}/pyproject.toml +4 -1
- {usdata-0.5.0 → usdata-0.7.0}/pyproject.toml.orig +4 -1
- usdata-0.7.0/src/usdata/_progress.py +62 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/cli/app.py +10 -2
- usdata-0.7.0/src/usdata/cli/progress.py +89 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/data/registry.yaml +11 -8
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/fetch.py +27 -1
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/protocols/http.py +18 -1
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/noaa/ghcnd.py +25 -4
- usdata-0.7.0/src/usdata/providers/noaa/gsom.py +49 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/pull.py +9 -1
- usdata-0.7.0/src/usdata/readers.py +103 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/__init__.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/_files.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/cache.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/cli/__init__.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/data/nexrad_sites.csv +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/data/places.csv +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/data/places.sources.json +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/manifest.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/models.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/protocols/__init__.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/protocols/erddap.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/protocols/s3.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/provenance.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/__init__.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/base.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/noaa/__init__.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/noaa/nexrad.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/noaa/sites.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/usgs/__init__.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/providers/usgs/daily.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/py.typed +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/query.py +0 -0
- {usdata-0.5.0 → usdata-0.7.0}/src/usdata/registry.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: usdata
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
|
|
5
5
|
Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
|
|
6
6
|
Author: Jake Van Slyke
|
|
@@ -14,9 +14,11 @@ Requires-Dist: httpx>=0.28.1
|
|
|
14
14
|
Requires-Dist: pydantic>=2.7
|
|
15
15
|
Requires-Dist: pyyaml>=6.0
|
|
16
16
|
Requires-Dist: typer>=0.12
|
|
17
|
+
Requires-Dist: pandas>=3.0 ; extra == 'pandas'
|
|
17
18
|
Requires-Python: >=3.11
|
|
18
19
|
Project-URL: Homepage, https://github.com/jakeryderv/usdata
|
|
19
20
|
Project-URL: Repository, https://github.com/jakeryderv/usdata
|
|
21
|
+
Provides-Extra: pandas
|
|
20
22
|
Description-Content-Type: text/markdown
|
|
21
23
|
|
|
22
24
|
# usdata
|
|
@@ -24,22 +26,23 @@ Description-Content-Type: text/markdown
|
|
|
24
26
|
Unified Python SDK and CLI for discovering, fetching, and tracking the
|
|
25
27
|
provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
26
28
|
|
|
27
|
-
> Status: pre-alpha. v0.
|
|
28
|
-
> values, and CoastWatch SST subsets with provenance,
|
|
29
|
-
> lookup
|
|
29
|
+
> Status: pre-alpha. v0.7 supports GHCN-Daily, GSOM monthly summaries,
|
|
30
|
+
> NEXRAD Level II, USGS daily values, and CoastWatch SST subsets with provenance,
|
|
31
|
+
> plus Census state/county lookup, optional pandas CSV readers, and terminal
|
|
32
|
+
> download progress. Other datasets are planned.
|
|
30
33
|
> See [docs/roadmap.md](docs/roadmap.md).
|
|
31
34
|
|
|
32
35
|
## Providers
|
|
33
36
|
|
|
34
37
|
<!-- registry:start -->
|
|
35
|
-
| Provider | Available | Stub | Planned | Next up (
|
|
38
|
+
| Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
|
|
36
39
|
|---|---:|---:|---:|---|---|
|
|
37
|
-
| [NOAA](docs/providers/noaa.md) |
|
|
40
|
+
| [NOAA](docs/providers/noaa.md) | 4 | 0 | 25 | — | `ghcn-daily`, `gsom`, `nexrad-level2`, `coastwatch-sst`, +25 planned |
|
|
38
41
|
| [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
|
|
39
42
|
| [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
|
|
40
43
|
| [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
41
44
|
| [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
|
|
42
|
-
| [NASA](docs/providers/nasa.md) | 0 | 0 | 1 |
|
|
45
|
+
| [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
43
46
|
| [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
|
|
44
47
|
|
|
45
48
|
Available datasets are in `code`, stubs in _italics_; planned ones are counted. Available means implemented in this source checkout; consult the [releases](https://github.com/jakeryderv/usdata/releases) for published support. Each provider page has access notes and full dataset details; [docs/roadmap.md](docs/roadmap.md) lists datasets by target version.
|
|
@@ -130,6 +133,24 @@ sources:
|
|
|
130
133
|
end: 2024-05-31
|
|
131
134
|
```
|
|
132
135
|
|
|
136
|
+
Terminal progress is available since v0.7. On a terminal, `fetch` and
|
|
137
|
+
`pull` show progress on stderr: resolved asset counts,
|
|
138
|
+
known bytes and unknown sizes, HTTP download bytes for the current attempt, and
|
|
139
|
+
validated cache hits. `fetch --dry-run` also summarizes known sizes. Asset totals
|
|
140
|
+
include possible cache hits; each manifest source is resolved separately. Bytes
|
|
141
|
+
from a failed HTTP attempt reset on retry; encoded responses have unknown decoded
|
|
142
|
+
size. Adapters that assemble files from metadata requests show asset-level progress.
|
|
143
|
+
Use `--no-progress` to disable it. Progress is automatically disabled when either
|
|
144
|
+
stdout or stderr is redirected; existing output lines and exit codes are unchanged.
|
|
145
|
+
|
|
146
|
+
## Opening CSV data
|
|
147
|
+
|
|
148
|
+
`FetchedAsset.open()` is available since v0.6 with the optional pandas
|
|
149
|
+
extra (`pip install "usdata[pandas]"`). It reads cached CSV into a DataFrame,
|
|
150
|
+
preserves identifier strings, and keeps CoastWatch units as metadata.
|
|
151
|
+
See the [reader reference](docs/reference/readers.md)
|
|
152
|
+
and [fetch → open → analyze example](examples/sst-analysis/README.md).
|
|
153
|
+
|
|
133
154
|
## Development
|
|
134
155
|
|
|
135
156
|
Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
|
|
@@ -138,16 +159,18 @@ Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
|
|
|
138
159
|
git clone https://github.com/jakeryderv/usdata && cd usdata
|
|
139
160
|
just setup # install toolchain and dependencies
|
|
140
161
|
just test # unit tests
|
|
141
|
-
just check # format, lint, typecheck, offline tests, generated docs
|
|
162
|
+
just check # format, lint, typecheck, offline tests, generated docs, release notices
|
|
163
|
+
just check-pandas # install the CSV extra and run the same checks
|
|
142
164
|
just build # build wheel and sdist
|
|
143
|
-
just smoke #
|
|
165
|
+
just smoke # exercise core and pandas wheel installations outside the checkout
|
|
144
166
|
just run search radar
|
|
145
167
|
```
|
|
146
168
|
|
|
147
169
|
Unit tests mechanically block network connections. Integration tests that hit
|
|
148
170
|
live services run with `just test-integration`. CI checks Python 3.11 and 3.14 on
|
|
149
|
-
Linux and smoke-tests
|
|
150
|
-
full unit and live-service suites currently run on Linux.
|
|
171
|
+
Linux, both with and without pandas, and smoke-tests both installed-wheel
|
|
172
|
+
profiles on Linux, macOS, and Windows. The full unit and live-service suites currently run on Linux. `just setup` restores
|
|
173
|
+
a core-only development environment; `just check-pandas` installs the extra.
|
|
151
174
|
|
|
152
175
|
Releases: `just release minor` opens a version-bump PR; merging it publishes
|
|
153
176
|
to PyPI and creates the tag and GitHub release. See
|
|
@@ -3,22 +3,23 @@
|
|
|
3
3
|
Unified Python SDK and CLI for discovering, fetching, and tracking the
|
|
4
4
|
provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
5
5
|
|
|
6
|
-
> Status: pre-alpha. v0.
|
|
7
|
-
> values, and CoastWatch SST subsets with provenance,
|
|
8
|
-
> lookup
|
|
6
|
+
> Status: pre-alpha. v0.7 supports GHCN-Daily, GSOM monthly summaries,
|
|
7
|
+
> NEXRAD Level II, USGS daily values, and CoastWatch SST subsets with provenance,
|
|
8
|
+
> plus Census state/county lookup, optional pandas CSV readers, and terminal
|
|
9
|
+
> download progress. Other datasets are planned.
|
|
9
10
|
> See [docs/roadmap.md](docs/roadmap.md).
|
|
10
11
|
|
|
11
12
|
## Providers
|
|
12
13
|
|
|
13
14
|
<!-- registry:start -->
|
|
14
|
-
| Provider | Available | Stub | Planned | Next up (
|
|
15
|
+
| Provider | Available | Stub | Planned | Next up (unassigned) | Datasets |
|
|
15
16
|
|---|---:|---:|---:|---|---|
|
|
16
|
-
| [NOAA](docs/providers/noaa.md) |
|
|
17
|
+
| [NOAA](docs/providers/noaa.md) | 4 | 0 | 25 | — | `ghcn-daily`, `gsom`, `nexrad-level2`, `coastwatch-sst`, +25 planned |
|
|
17
18
|
| [USGS](docs/providers/usgs.md) | 1 | 0 | 2 | — | `water-daily`, +2 planned |
|
|
18
19
|
| [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
|
|
19
20
|
| [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
20
21
|
| [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
|
|
21
|
-
| [NASA](docs/providers/nasa.md) | 0 | 0 | 1 |
|
|
22
|
+
| [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
22
23
|
| [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
|
|
23
24
|
|
|
24
25
|
Available datasets are in `code`, stubs in _italics_; planned ones are counted. Available means implemented in this source checkout; consult the [releases](https://github.com/jakeryderv/usdata/releases) for published support. Each provider page has access notes and full dataset details; [docs/roadmap.md](docs/roadmap.md) lists datasets by target version.
|
|
@@ -109,6 +110,24 @@ sources:
|
|
|
109
110
|
end: 2024-05-31
|
|
110
111
|
```
|
|
111
112
|
|
|
113
|
+
Terminal progress is available since v0.7. On a terminal, `fetch` and
|
|
114
|
+
`pull` show progress on stderr: resolved asset counts,
|
|
115
|
+
known bytes and unknown sizes, HTTP download bytes for the current attempt, and
|
|
116
|
+
validated cache hits. `fetch --dry-run` also summarizes known sizes. Asset totals
|
|
117
|
+
include possible cache hits; each manifest source is resolved separately. Bytes
|
|
118
|
+
from a failed HTTP attempt reset on retry; encoded responses have unknown decoded
|
|
119
|
+
size. Adapters that assemble files from metadata requests show asset-level progress.
|
|
120
|
+
Use `--no-progress` to disable it. Progress is automatically disabled when either
|
|
121
|
+
stdout or stderr is redirected; existing output lines and exit codes are unchanged.
|
|
122
|
+
|
|
123
|
+
## Opening CSV data
|
|
124
|
+
|
|
125
|
+
`FetchedAsset.open()` is available since v0.6 with the optional pandas
|
|
126
|
+
extra (`pip install "usdata[pandas]"`). It reads cached CSV into a DataFrame,
|
|
127
|
+
preserves identifier strings, and keeps CoastWatch units as metadata.
|
|
128
|
+
See the [reader reference](docs/reference/readers.md)
|
|
129
|
+
and [fetch → open → analyze example](examples/sst-analysis/README.md).
|
|
130
|
+
|
|
112
131
|
## Development
|
|
113
132
|
|
|
114
133
|
Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
|
|
@@ -117,16 +136,18 @@ Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
|
|
|
117
136
|
git clone https://github.com/jakeryderv/usdata && cd usdata
|
|
118
137
|
just setup # install toolchain and dependencies
|
|
119
138
|
just test # unit tests
|
|
120
|
-
just check # format, lint, typecheck, offline tests, generated docs
|
|
139
|
+
just check # format, lint, typecheck, offline tests, generated docs, release notices
|
|
140
|
+
just check-pandas # install the CSV extra and run the same checks
|
|
121
141
|
just build # build wheel and sdist
|
|
122
|
-
just smoke #
|
|
142
|
+
just smoke # exercise core and pandas wheel installations outside the checkout
|
|
123
143
|
just run search radar
|
|
124
144
|
```
|
|
125
145
|
|
|
126
146
|
Unit tests mechanically block network connections. Integration tests that hit
|
|
127
147
|
live services run with `just test-integration`. CI checks Python 3.11 and 3.14 on
|
|
128
|
-
Linux and smoke-tests
|
|
129
|
-
full unit and live-service suites currently run on Linux.
|
|
148
|
+
Linux, both with and without pandas, and smoke-tests both installed-wheel
|
|
149
|
+
profiles on Linux, macOS, and Windows. The full unit and live-service suites currently run on Linux. `just setup` restores
|
|
150
|
+
a core-only development environment; `just check-pandas` installs the extra.
|
|
130
151
|
|
|
131
152
|
Releases: `just release minor` opens a version-bump PR; merging it publishes
|
|
132
153
|
to PyPI and creates the tag and GitHub release. See
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "usdata"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.7.0"
|
|
4
4
|
description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -30,6 +30,9 @@ dependencies = [
|
|
|
30
30
|
name = "Jake Van Slyke"
|
|
31
31
|
email = "jakervanslyke@gmail.com"
|
|
32
32
|
|
|
33
|
+
[project.optional-dependencies]
|
|
34
|
+
pandas = ["pandas>=3.0"]
|
|
35
|
+
|
|
33
36
|
[project.urls]
|
|
34
37
|
Homepage = "https://github.com/jakeryderv/usdata"
|
|
35
38
|
Repository = "https://github.com/jakeryderv/usdata"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "usdata"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.7.0"
|
|
4
4
|
description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -22,6 +22,9 @@ dependencies = [
|
|
|
22
22
|
"typer>=0.12",
|
|
23
23
|
]
|
|
24
24
|
|
|
25
|
+
[project.optional-dependencies]
|
|
26
|
+
pandas = ["pandas>=3.0"]
|
|
27
|
+
|
|
25
28
|
[project.urls]
|
|
26
29
|
Homepage = "https://github.com/jakeryderv/usdata"
|
|
27
30
|
Repository = "https://github.com/jakeryderv/usdata"
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Internal synchronous progress events, scoped to one CLI operation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable, Iterator, Sequence
|
|
6
|
+
from contextlib import contextmanager
|
|
7
|
+
from contextvars import ContextVar
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from typing import Literal
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(frozen=True)
|
|
13
|
+
class Batch:
|
|
14
|
+
"""A resolved group; sizes describe assets, including possible cache hits."""
|
|
15
|
+
|
|
16
|
+
count: int
|
|
17
|
+
known_bytes: int
|
|
18
|
+
unknown_sizes: int
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class AssetProgress:
|
|
23
|
+
"""Start or validated completion of an asset."""
|
|
24
|
+
|
|
25
|
+
asset_id: str
|
|
26
|
+
state: Literal["start", "cached", "fetched"]
|
|
27
|
+
size: int | None
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class TransferProgress:
|
|
32
|
+
"""Bytes written in the current HTTP attempt, reset on every retry."""
|
|
33
|
+
|
|
34
|
+
completed: int
|
|
35
|
+
total: int | None
|
|
36
|
+
attempt: int
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
Event = Batch | AssetProgress | TransferProgress
|
|
40
|
+
_observer: ContextVar[Callable[[Event], None] | None] = ContextVar("progress", default=None)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def emit(event: Event) -> None:
|
|
44
|
+
"""Notify the active observer without importing CLI code."""
|
|
45
|
+
observer = _observer.get()
|
|
46
|
+
if observer is not None:
|
|
47
|
+
observer(event)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def batch(sizes: Sequence[int | None]) -> None:
|
|
51
|
+
"""Report known and unknown sizes separately; never guess a total."""
|
|
52
|
+
emit(Batch(len(sizes), sum(size for size in sizes if size is not None), sizes.count(None)))
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@contextmanager
|
|
56
|
+
def observe(callback: Callable[[Event], None]) -> Iterator[None]:
|
|
57
|
+
"""Observe an operation and restore the previous observer on every exit."""
|
|
58
|
+
token = _observer.set(callback)
|
|
59
|
+
try:
|
|
60
|
+
yield
|
|
61
|
+
finally:
|
|
62
|
+
_observer.reset(token)
|
|
@@ -9,6 +9,8 @@ import httpx
|
|
|
9
9
|
import typer
|
|
10
10
|
|
|
11
11
|
from usdata import __version__, build_query, default_registry
|
|
12
|
+
from usdata._progress import batch
|
|
13
|
+
from usdata.cli.progress import progress
|
|
12
14
|
from usdata.fetch import ChecksumMismatch
|
|
13
15
|
from usdata.fetch import fetch as fetch_query
|
|
14
16
|
from usdata.manifest import lockfile_path
|
|
@@ -126,6 +128,7 @@ def fetch(
|
|
|
126
128
|
] = None,
|
|
127
129
|
cache_dir: Annotated[Path | None, typer.Option(help="Override the cache directory.")] = None,
|
|
128
130
|
force: Annotated[bool, typer.Option(help="Re-download even if cached.")] = False,
|
|
131
|
+
no_progress: Annotated[bool, typer.Option(help="Disable terminal progress.")] = False,
|
|
129
132
|
dry_run: Annotated[
|
|
130
133
|
bool, typer.Option(help="List matching assets without downloading.")
|
|
131
134
|
] = False,
|
|
@@ -165,8 +168,11 @@ def fetch(
|
|
|
165
168
|
for a in assets:
|
|
166
169
|
typer.echo(f"{a.id}\t{a.href}")
|
|
167
170
|
typer.echo(f"{len(assets)} asset(s) matched", err=True)
|
|
171
|
+
with progress(disabled=no_progress):
|
|
172
|
+
batch([asset.size for asset in assets])
|
|
168
173
|
return
|
|
169
|
-
|
|
174
|
+
with progress(disabled=no_progress):
|
|
175
|
+
fetched = fetch_query(ds, query, root=cache_dir, force=force)
|
|
170
176
|
except (DatasetNotFound, UnknownPlace, ValueError) as e:
|
|
171
177
|
typer.secho(str(e), err=True, fg="red")
|
|
172
178
|
raise typer.Exit(code=2) from None
|
|
@@ -192,10 +198,12 @@ def pull(
|
|
|
192
198
|
bool,
|
|
193
199
|
typer.Option(help="Ignore an existing lockfile: re-resolve every source and rewrite it."),
|
|
194
200
|
] = False,
|
|
201
|
+
no_progress: Annotated[bool, typer.Option(help="Disable terminal progress.")] = False,
|
|
195
202
|
) -> None:
|
|
196
203
|
"""Fetch every source in a manifest and write (or restore from) its lockfile."""
|
|
197
204
|
try:
|
|
198
|
-
|
|
205
|
+
with progress(disabled=no_progress):
|
|
206
|
+
result = pull_manifest(manifest, root=cache_dir, force=force)
|
|
199
207
|
except EmptySource as e:
|
|
200
208
|
typer.secho(str(e), err=True, fg="yellow")
|
|
201
209
|
raise typer.Exit(code=1) from None
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""Terminal-only progress rendering; normal CLI output remains machine readable."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import shutil
|
|
6
|
+
import sys
|
|
7
|
+
from collections.abc import Iterator
|
|
8
|
+
from contextlib import contextmanager
|
|
9
|
+
from time import monotonic
|
|
10
|
+
from typing import TextIO
|
|
11
|
+
|
|
12
|
+
from usdata._progress import AssetProgress, Batch, Event, observe
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _interactive() -> bool:
|
|
16
|
+
return sys.stdout.isatty() and sys.stderr.isatty()
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class _Display:
|
|
20
|
+
def __init__(self, stream: TextIO) -> None:
|
|
21
|
+
self.stream = stream
|
|
22
|
+
self.width = 0
|
|
23
|
+
self.updated = 0.0
|
|
24
|
+
self.asset = ""
|
|
25
|
+
self.count = 0
|
|
26
|
+
self.done = 0
|
|
27
|
+
self.cached = 0
|
|
28
|
+
|
|
29
|
+
def clear(self) -> None:
|
|
30
|
+
if self.width:
|
|
31
|
+
self.stream.write("\r" + " " * self.width + "\r")
|
|
32
|
+
self.stream.flush()
|
|
33
|
+
self.width = 0
|
|
34
|
+
|
|
35
|
+
def line(self, text: str, *, final: bool = False) -> None:
|
|
36
|
+
self.clear()
|
|
37
|
+
# IDs come from providers: keep control characters out of the terminal.
|
|
38
|
+
text = "".join(c if c.isprintable() else "?" for c in text)
|
|
39
|
+
if final:
|
|
40
|
+
self.stream.write(text + "\n")
|
|
41
|
+
else:
|
|
42
|
+
text = text[: max(1, shutil.get_terminal_size().columns - 1)]
|
|
43
|
+
self.stream.write(text)
|
|
44
|
+
self.width = len(text)
|
|
45
|
+
self.stream.flush()
|
|
46
|
+
self.updated = monotonic()
|
|
47
|
+
|
|
48
|
+
def __call__(self, event: Event) -> None:
|
|
49
|
+
if isinstance(event, Batch):
|
|
50
|
+
self.count, self.done, self.cached = event.count, 0, 0
|
|
51
|
+
sizes = f"{event.known_bytes:,} known bytes"
|
|
52
|
+
if event.unknown_sizes:
|
|
53
|
+
sizes += f"; {event.unknown_sizes} size(s) unknown"
|
|
54
|
+
self.line(f"{event.count} asset(s) resolved; {sizes} (before cache checks)", final=True)
|
|
55
|
+
elif isinstance(event, AssetProgress):
|
|
56
|
+
self.asset = event.asset_id
|
|
57
|
+
if event.state == "start":
|
|
58
|
+
size = f"{event.size:,} bytes" if event.size is not None else "size unknown"
|
|
59
|
+
self.line(f"[{self.done}/{self.count}] checking cache ({size}) | {self.asset}")
|
|
60
|
+
else:
|
|
61
|
+
self.done += 1
|
|
62
|
+
self.cached += event.state == "cached"
|
|
63
|
+
self.line(
|
|
64
|
+
f"[{self.done}/{self.count}] {event.state} "
|
|
65
|
+
f"({event.size:,} bytes; {self.cached} cached) | {self.asset}",
|
|
66
|
+
final=self.done == self.count,
|
|
67
|
+
)
|
|
68
|
+
else:
|
|
69
|
+
if event.completed and monotonic() - self.updated < 0.1:
|
|
70
|
+
return
|
|
71
|
+
size = f"{event.total:,}" if event.total is not None else "unknown"
|
|
72
|
+
self.line(
|
|
73
|
+
f"[{self.done}/{self.count}] {event.completed:,}/{size} bytes "
|
|
74
|
+
f"(attempt {event.attempt}) | {self.asset}"
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
@contextmanager
|
|
79
|
+
def progress(*, disabled: bool = False) -> Iterator[None]:
|
|
80
|
+
"""Render progress on stderr only when both output streams are terminals."""
|
|
81
|
+
if disabled or not _interactive():
|
|
82
|
+
yield
|
|
83
|
+
return
|
|
84
|
+
display = _Display(sys.stderr)
|
|
85
|
+
with observe(display):
|
|
86
|
+
try:
|
|
87
|
+
yield
|
|
88
|
+
finally:
|
|
89
|
+
display.clear()
|
|
@@ -209,19 +209,22 @@ datasets:
|
|
|
209
209
|
|
|
210
210
|
- id: noaa:gsom
|
|
211
211
|
provider: noaa
|
|
212
|
-
status:
|
|
212
|
+
status: available
|
|
213
213
|
domain: surface-weather
|
|
214
|
-
|
|
214
|
+
since: "0.7"
|
|
215
215
|
title: Global Summary of the Month
|
|
216
216
|
description: >-
|
|
217
217
|
Monthly station summaries derived from GHCN-Daily (means, extremes,
|
|
218
218
|
totals) via the NCEI Access Data Service dataset global-summary-of-the-month.
|
|
219
|
-
|
|
219
|
+
Selects whole calendar months and explicit stations, or discovers stations
|
|
220
|
+
through the companion search service.
|
|
220
221
|
keywords: [climate, monthly, stations, temperature, precipitation, gsom, ncei]
|
|
221
222
|
protocol: http
|
|
222
223
|
homepage: https://www.ncei.noaa.gov/access/search/data-search/global-summary-of-the-month
|
|
223
224
|
license: US Government Work (public domain)
|
|
224
225
|
capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: true }
|
|
226
|
+
spatial_extent: { west: -180.0, south: -90.0, east: 180.0, north: 90.0 }
|
|
227
|
+
adapter: usdata.providers.noaa.gsom:GlobalSummaryMonthly
|
|
225
228
|
|
|
226
229
|
- id: noaa:gsoy
|
|
227
230
|
provider: noaa
|
|
@@ -293,7 +296,7 @@ datasets:
|
|
|
293
296
|
provider: noaa
|
|
294
297
|
status: planned
|
|
295
298
|
domain: weather-models
|
|
296
|
-
target:
|
|
299
|
+
target: later
|
|
297
300
|
title: HRRR Forecast Model Output
|
|
298
301
|
description: >-
|
|
299
302
|
High-Resolution Rapid Refresh 3 km hourly forecasts in GRIB2 from the
|
|
@@ -310,7 +313,7 @@ datasets:
|
|
|
310
313
|
provider: noaa
|
|
311
314
|
status: planned
|
|
312
315
|
domain: weather-models
|
|
313
|
-
target:
|
|
316
|
+
target: later
|
|
314
317
|
title: GFS Forecast Model Output
|
|
315
318
|
description: >-
|
|
316
319
|
Global Forecast System output in GRIB2 from the public noaa-gfs-bdp-pds
|
|
@@ -342,7 +345,7 @@ datasets:
|
|
|
342
345
|
provider: noaa
|
|
343
346
|
status: planned
|
|
344
347
|
domain: ocean-physics
|
|
345
|
-
target:
|
|
348
|
+
target: later
|
|
346
349
|
title: OISST Daily Sea Surface Temperature
|
|
347
350
|
description: >-
|
|
348
351
|
Optimum Interpolation SST v2.1: daily global 0.25 degree analysis since
|
|
@@ -375,7 +378,7 @@ datasets:
|
|
|
375
378
|
provider: noaa
|
|
376
379
|
status: planned
|
|
377
380
|
domain: bathymetry
|
|
378
|
-
target:
|
|
381
|
+
target: later
|
|
379
382
|
title: ETOPO 2022 Global Relief
|
|
380
383
|
description: >-
|
|
381
384
|
Global topography and bathymetry at 15, 30, and 60 arc-seconds as
|
|
@@ -591,7 +594,7 @@ datasets:
|
|
|
591
594
|
provider: nasa
|
|
592
595
|
status: planned
|
|
593
596
|
domain: weather-satellites
|
|
594
|
-
target:
|
|
597
|
+
target: later
|
|
595
598
|
title: GPM IMERG Precipitation
|
|
596
599
|
description: >-
|
|
597
600
|
Global half-hourly and daily merged satellite precipitation estimates.
|
|
@@ -3,10 +3,11 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
6
7
|
|
|
7
8
|
from pydantic import BaseModel
|
|
8
9
|
|
|
9
|
-
from usdata import provenance
|
|
10
|
+
from usdata import _progress, provenance
|
|
10
11
|
from usdata._files import staged_path
|
|
11
12
|
from usdata.cache import asset_path, sha256_file
|
|
12
13
|
from usdata.models import Asset, Dataset, Provenance, Query
|
|
@@ -25,6 +26,27 @@ class FetchedAsset(BaseModel):
|
|
|
25
26
|
provenance: Provenance
|
|
26
27
|
from_cache: bool
|
|
27
28
|
|
|
29
|
+
def open(
|
|
30
|
+
self,
|
|
31
|
+
*,
|
|
32
|
+
reader: str | None = None,
|
|
33
|
+
dtype: dict[str, str] | None = None,
|
|
34
|
+
parse_dates: list[str] | None = None,
|
|
35
|
+
usecols: list[str] | None = None,
|
|
36
|
+
nrows: int | None = None,
|
|
37
|
+
) -> Any:
|
|
38
|
+
"""Open this local CSV as a DataFrame; requires the ``pandas`` extra.
|
|
39
|
+
|
|
40
|
+
ERDDAP units are kept in ``frame.attrs["units"]`` and source provenance
|
|
41
|
+
in ``frame.attrs["usdata"]``. See ``usdata.readers.open_asset`` for options.
|
|
42
|
+
Cached files and provenance sidecars are never changed.
|
|
43
|
+
"""
|
|
44
|
+
from usdata.readers import open_asset
|
|
45
|
+
|
|
46
|
+
return open_asset(
|
|
47
|
+
self, reader=reader, dtype=dtype, parse_dates=parse_dates, usecols=usecols, nrows=nrows
|
|
48
|
+
)
|
|
49
|
+
|
|
28
50
|
|
|
29
51
|
def _fetch_asset(
|
|
30
52
|
dataset: Dataset,
|
|
@@ -36,6 +58,7 @@ def _fetch_asset(
|
|
|
36
58
|
) -> FetchedAsset:
|
|
37
59
|
if asset.dataset_id != dataset.id:
|
|
38
60
|
raise ValueError(f"asset dataset {asset.dataset_id!r} does not match {dataset.id!r}")
|
|
61
|
+
_progress.emit(_progress.AssetProgress(asset.id, "start", asset.size))
|
|
39
62
|
path = asset_path(asset, root)
|
|
40
63
|
if not force and path.is_file():
|
|
41
64
|
try:
|
|
@@ -51,6 +74,7 @@ def _fetch_asset(
|
|
|
51
74
|
and (asset.checksum is None or prov.checksum == asset.checksum)
|
|
52
75
|
and sha256_file(path) == prov.checksum
|
|
53
76
|
):
|
|
77
|
+
_progress.emit(_progress.AssetProgress(asset.id, "cached", prov.size))
|
|
54
78
|
return FetchedAsset(asset=asset, path=path, provenance=prov, from_cache=True)
|
|
55
79
|
with staged_path(path) as tmp:
|
|
56
80
|
adapter.fetch(asset, tmp)
|
|
@@ -59,6 +83,7 @@ def _fetch_asset(
|
|
|
59
83
|
raise ChecksumMismatch(f"{asset.id}: expected {asset.checksum}, got {prov.checksum}")
|
|
60
84
|
# A crash between replacements leaves a detectable mismatch, never a trusted partial file.
|
|
61
85
|
provenance.write(prov, path)
|
|
86
|
+
_progress.emit(_progress.AssetProgress(asset.id, "fetched", prov.size))
|
|
62
87
|
return FetchedAsset(asset=asset, path=path, provenance=prov, from_cache=False)
|
|
63
88
|
|
|
64
89
|
|
|
@@ -76,4 +101,5 @@ def fetch(
|
|
|
76
101
|
"""Resolve and fetch a query, sharing one adapter and closing its owned resources."""
|
|
77
102
|
with load_adapter(dataset) as adapter:
|
|
78
103
|
assets = adapter.list_assets(query)
|
|
104
|
+
_progress.batch([asset.size for asset in assets])
|
|
79
105
|
return [_fetch_asset(dataset, a, adapter, root=root, force=force) for a in assets]
|
|
@@ -11,7 +11,7 @@ from typing import Any, TypeVar
|
|
|
11
11
|
|
|
12
12
|
import httpx
|
|
13
13
|
|
|
14
|
-
from usdata import __version__
|
|
14
|
+
from usdata import __version__, _progress
|
|
15
15
|
from usdata._files import staged_path
|
|
16
16
|
|
|
17
17
|
USER_AGENT = f"usdata/{__version__} (+https://github.com/jakeryderv/usdata)"
|
|
@@ -80,13 +80,30 @@ def download(url: str, dest: Path, http: httpx.Client | None = None) -> Path:
|
|
|
80
80
|
"""Download atomically, restarting interrupted GETs up to three total attempts."""
|
|
81
81
|
own = http is None
|
|
82
82
|
active = http or client()
|
|
83
|
+
attempt = 0
|
|
83
84
|
|
|
84
85
|
def request() -> Path:
|
|
86
|
+
nonlocal attempt
|
|
87
|
+
attempt += 1
|
|
88
|
+
_progress.emit(_progress.TransferProgress(0, None, attempt))
|
|
85
89
|
with staged_path(dest) as tmp, active.stream("GET", url) as resp:
|
|
86
90
|
resp.raise_for_status()
|
|
91
|
+
length = resp.headers.get("Content-Length", "")
|
|
92
|
+
# iter_bytes writes decoded bytes; an encoded length is not comparable.
|
|
93
|
+
total = (
|
|
94
|
+
int(length)
|
|
95
|
+
if length.isascii()
|
|
96
|
+
and length.isdigit()
|
|
97
|
+
and resp.headers.get("Content-Encoding", "identity").lower() == "identity"
|
|
98
|
+
else None
|
|
99
|
+
)
|
|
100
|
+
completed = 0
|
|
101
|
+
_progress.emit(_progress.TransferProgress(completed, total, attempt))
|
|
87
102
|
with tmp.open("wb") as f:
|
|
88
103
|
for chunk in resp.iter_bytes():
|
|
89
104
|
f.write(chunk)
|
|
105
|
+
completed += len(chunk)
|
|
106
|
+
_progress.emit(_progress.TransferProgress(completed, total, attempt))
|
|
90
107
|
return dest
|
|
91
108
|
|
|
92
109
|
try:
|
|
@@ -12,6 +12,7 @@ always goes through search first. Stations are chunked so URLs stay short.
|
|
|
12
12
|
from __future__ import annotations
|
|
13
13
|
|
|
14
14
|
import hashlib
|
|
15
|
+
import logging
|
|
15
16
|
from pathlib import Path
|
|
16
17
|
from typing import Any
|
|
17
18
|
|
|
@@ -26,6 +27,7 @@ DATA_URL = "https://www.ncei.noaa.gov/access/services/data/v1"
|
|
|
26
27
|
NCEI_DATASET = "daily-summaries"
|
|
27
28
|
SEARCH_PAGE_SIZE = 1000
|
|
28
29
|
STATIONS_PER_ASSET = 50
|
|
30
|
+
logger = logging.getLogger(__name__)
|
|
29
31
|
|
|
30
32
|
|
|
31
33
|
def _date(value: Any) -> str:
|
|
@@ -46,6 +48,8 @@ def _stations_param(raw: Any) -> list[str]:
|
|
|
46
48
|
class GhcnDaily(Provider):
|
|
47
49
|
"""GHCN-Daily adapter. Params: ``stations`` (list or comma string), ``units``."""
|
|
48
50
|
|
|
51
|
+
ncei_dataset = NCEI_DATASET
|
|
52
|
+
|
|
49
53
|
def __init__(self, dataset: Dataset, client: httpx.Client | None = None) -> None:
|
|
50
54
|
super().__init__(dataset)
|
|
51
55
|
self._client = client
|
|
@@ -68,7 +72,7 @@ class GhcnDaily(Provider):
|
|
|
68
72
|
raise QueryError("station search needs a bounding box and a time range")
|
|
69
73
|
b = query.bbox
|
|
70
74
|
params: dict[str, Any] = {
|
|
71
|
-
"dataset":
|
|
75
|
+
"dataset": self.ncei_dataset,
|
|
72
76
|
"bbox": f"{b.north},{b.west},{b.south},{b.east}",
|
|
73
77
|
"startDate": _date(query.time.start),
|
|
74
78
|
"endDate": _date(query.time.end),
|
|
@@ -84,12 +88,29 @@ class GhcnDaily(Provider):
|
|
|
84
88
|
resp.raise_for_status()
|
|
85
89
|
body = resp.json()
|
|
86
90
|
results = body.get("results", [])
|
|
91
|
+
before = len(found)
|
|
87
92
|
for result in results:
|
|
88
93
|
for station in result.get("stations", []):
|
|
89
94
|
sid = station.get("id")
|
|
90
95
|
if sid and sid not in seen:
|
|
91
96
|
seen.add(sid)
|
|
92
97
|
found.append(sid)
|
|
98
|
+
logger.debug(
|
|
99
|
+
"NCEI station search: url=%s status=%s content_type=%s "
|
|
100
|
+
"count=%r totalCount=%r results=%s new_stations=%s station_sample=%r",
|
|
101
|
+
resp.request.url,
|
|
102
|
+
resp.status_code,
|
|
103
|
+
resp.headers.get("content-type"),
|
|
104
|
+
body.get("count"),
|
|
105
|
+
body.get("totalCount"),
|
|
106
|
+
len(results),
|
|
107
|
+
len(found) - before,
|
|
108
|
+
found[before : before + 5],
|
|
109
|
+
)
|
|
110
|
+
if len(found) == before and logger.isEnabledFor(logging.DEBUG):
|
|
111
|
+
logger.debug(
|
|
112
|
+
"NCEI search page yielded no new stations; response_prefix=%r", resp.text[:512]
|
|
113
|
+
)
|
|
93
114
|
# "count" is the number matching this query; "totalCount" is dataset-wide.
|
|
94
115
|
params["offset"] += SEARCH_PAGE_SIZE
|
|
95
116
|
if not results or params["offset"] >= int(body.get("count", 0)):
|
|
@@ -101,7 +122,7 @@ class GhcnDaily(Provider):
|
|
|
101
122
|
if query.time is None or query.time.start is None or query.time.end is None:
|
|
102
123
|
raise QueryError(f"{self.dataset.id} requires both start and end dates")
|
|
103
124
|
if unknown := set(query.params) - {"stations", "units"}:
|
|
104
|
-
raise QueryError(f"unsupported
|
|
125
|
+
raise QueryError(f"unsupported {self.dataset.id} params: {', '.join(sorted(unknown))}")
|
|
105
126
|
if query.params.get("units", "metric") not in ("metric", "standard"):
|
|
106
127
|
raise QueryError("units must be metric or standard")
|
|
107
128
|
if "stations" in query.params:
|
|
@@ -118,7 +139,7 @@ class GhcnDaily(Provider):
|
|
|
118
139
|
for i in range(0, len(stations), STATIONS_PER_ASSET):
|
|
119
140
|
chunk = stations[i : i + STATIONS_PER_ASSET]
|
|
120
141
|
params: dict[str, Any] = {
|
|
121
|
-
"dataset":
|
|
142
|
+
"dataset": self.ncei_dataset,
|
|
122
143
|
"stations": ",".join(chunk),
|
|
123
144
|
"startDate": start,
|
|
124
145
|
"endDate": end,
|
|
@@ -132,7 +153,7 @@ class GhcnDaily(Provider):
|
|
|
132
153
|
digest = hashlib.sha1(url.encode()).hexdigest()[:12]
|
|
133
154
|
assets.append(
|
|
134
155
|
Asset(
|
|
135
|
-
id=f"{
|
|
156
|
+
id=f"{self.ncei_dataset}_{start}_{end}_{digest}.csv",
|
|
136
157
|
dataset_id=self.dataset.id,
|
|
137
158
|
href=url,
|
|
138
159
|
protocol=Protocol.HTTP,
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""Monthly station summaries through the NCEI Access Data Service.
|
|
2
|
+
|
|
3
|
+
Params: ``stations`` (list or comma string) and ``units`` (metric or standard).
|
|
4
|
+
Both dates are required; every UTC calendar month touched by the interval is
|
|
5
|
+
selected in full. Geographic discovery uses the same normalized month bounds.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from calendar import monthrange
|
|
9
|
+
from datetime import UTC
|
|
10
|
+
|
|
11
|
+
from usdata.models import Asset, Query, TimeRange
|
|
12
|
+
from usdata.providers.base import QueryError
|
|
13
|
+
from usdata.providers.noaa.ghcnd import GhcnDaily
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class GlobalSummaryMonthly(GhcnDaily):
|
|
17
|
+
"""GSOM CSV subsets, sharing NCEI station discovery and transport with GHCN."""
|
|
18
|
+
|
|
19
|
+
ncei_dataset = "global-summary-of-the-month"
|
|
20
|
+
|
|
21
|
+
def _monthly_query(self, query: Query) -> Query:
|
|
22
|
+
if query.time is None or query.time.start is None or query.time.end is None:
|
|
23
|
+
raise QueryError(f"{self.dataset.id} requires both start and end dates")
|
|
24
|
+
start = query.time.start.replace(tzinfo=query.time.start.tzinfo or UTC).astimezone(UTC)
|
|
25
|
+
end = query.time.end.replace(tzinfo=query.time.end.tzinfo or UTC).astimezone(UTC)
|
|
26
|
+
return query.model_copy(
|
|
27
|
+
update={
|
|
28
|
+
"time": TimeRange(
|
|
29
|
+
start=start.replace(day=1, hour=0, minute=0, second=0, microsecond=0),
|
|
30
|
+
end=end.replace(
|
|
31
|
+
day=monthrange(end.year, end.month)[1],
|
|
32
|
+
hour=23,
|
|
33
|
+
minute=59,
|
|
34
|
+
second=59,
|
|
35
|
+
microsecond=999999,
|
|
36
|
+
),
|
|
37
|
+
)
|
|
38
|
+
}
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
def find_stations(self, query: Query) -> list[str]:
|
|
42
|
+
"""Find stations overlapping the selected complete calendar months."""
|
|
43
|
+
return super().find_stations(self._monthly_query(query))
|
|
44
|
+
|
|
45
|
+
def list_assets(self, query: Query) -> list[Asset]:
|
|
46
|
+
"""Resolve monthly CSVs with stable URLs and complete-month asset bounds."""
|
|
47
|
+
if "stations" in query.params and query.bbox is not None:
|
|
48
|
+
raise QueryError("pass stations or a location/bbox, not both")
|
|
49
|
+
return super().list_assets(self._monthly_query(query))
|
|
@@ -16,7 +16,7 @@ from pathlib import Path
|
|
|
16
16
|
|
|
17
17
|
from pydantic import BaseModel
|
|
18
18
|
|
|
19
|
-
from usdata import __version__, provenance
|
|
19
|
+
from usdata import __version__, _progress, provenance
|
|
20
20
|
from usdata.cache import asset_path, sha256_file
|
|
21
21
|
from usdata.fetch import ChecksumMismatch, FetchedAsset, _fetch_asset, fetch
|
|
22
22
|
from usdata.manifest import LockedAsset, Lockfile, Manifest, lockfile_path
|
|
@@ -111,13 +111,21 @@ def restore(
|
|
|
111
111
|
lock = Lockfile.load(lock_path)
|
|
112
112
|
_check_manifest(manifest_path, lock)
|
|
113
113
|
fetched: list[FetchedAsset] = []
|
|
114
|
+
_progress.batch([entry.provenance.size for entry in lock.assets])
|
|
114
115
|
adapters: dict[str, Provider] = {}
|
|
115
116
|
with ExitStack() as stack:
|
|
116
117
|
for entry in lock.assets:
|
|
117
118
|
dataset = reg.get(entry.asset.dataset_id)
|
|
118
119
|
path = asset_path(entry.asset, root)
|
|
120
|
+
if path.is_file():
|
|
121
|
+
_progress.emit(
|
|
122
|
+
_progress.AssetProgress(entry.asset.id, "start", entry.provenance.size)
|
|
123
|
+
)
|
|
119
124
|
if path.is_file() and sha256_file(path) == entry.provenance.checksum:
|
|
120
125
|
provenance.write(entry.provenance, path)
|
|
126
|
+
_progress.emit(
|
|
127
|
+
_progress.AssetProgress(entry.asset.id, "cached", entry.provenance.size)
|
|
128
|
+
)
|
|
121
129
|
fetched.append(
|
|
122
130
|
FetchedAsset(
|
|
123
131
|
asset=entry.asset, path=path, provenance=entry.provenance, from_cache=True
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""Optional readers for local fetched files; never fetch or modify cached bytes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
from importlib import import_module
|
|
7
|
+
from typing import TYPE_CHECKING, Any
|
|
8
|
+
|
|
9
|
+
from usdata.models import Protocol
|
|
10
|
+
|
|
11
|
+
if TYPE_CHECKING:
|
|
12
|
+
from usdata.fetch import FetchedAsset
|
|
13
|
+
|
|
14
|
+
CSV_MEDIA_TYPES = {"text/csv", "application/csv"}
|
|
15
|
+
IDENTIFIER_COLUMNS = {
|
|
16
|
+
"station",
|
|
17
|
+
"station_id",
|
|
18
|
+
"site_no",
|
|
19
|
+
"monitoring_location_id",
|
|
20
|
+
"parameter_code",
|
|
21
|
+
"statistic_id",
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class MissingReaderDependency(ImportError):
|
|
26
|
+
"""The optional dependency required to open an asset is not installed."""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class UnsupportedFormat(ValueError):
|
|
30
|
+
"""No reader is implemented for this asset's format."""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def open_asset(
|
|
34
|
+
fetched: FetchedAsset,
|
|
35
|
+
*,
|
|
36
|
+
reader: str | None = None,
|
|
37
|
+
dtype: dict[str, str] | None = None,
|
|
38
|
+
parse_dates: list[str] | None = None,
|
|
39
|
+
usecols: list[str] | None = None,
|
|
40
|
+
nrows: int | None = None,
|
|
41
|
+
) -> Any:
|
|
42
|
+
"""Read a local CSV into a pandas DataFrame, retaining units and provenance.
|
|
43
|
+
|
|
44
|
+
Infer ``csv`` or ``erddap-csv`` from media type and protocol, or use an
|
|
45
|
+
explicit reader for ambiguous metadata. Identifier columns default to pandas
|
|
46
|
+
strings; explicit dtype entries override those defaults. Dates remain strings
|
|
47
|
+
unless named in parse_dates. No checksum verification or downloading occurs.
|
|
48
|
+
"""
|
|
49
|
+
if reader is None:
|
|
50
|
+
media_type = (fetched.asset.media_type or "").split(";", 1)[0].strip().lower()
|
|
51
|
+
if media_type not in CSV_MEDIA_TYPES:
|
|
52
|
+
raise UnsupportedFormat(
|
|
53
|
+
f"no reader for {fetched.asset.media_type!r}; supported formats are CSV and "
|
|
54
|
+
"ERDDAP CSV. For a known CSV with ambiguous metadata, pass reader='csv' "
|
|
55
|
+
"or reader='erddap-csv'; otherwise use fetched.path with a format-specific reader"
|
|
56
|
+
)
|
|
57
|
+
reader = "erddap-csv" if fetched.asset.protocol is Protocol.ERDDAP else "csv"
|
|
58
|
+
if reader not in {"csv", "erddap-csv"}:
|
|
59
|
+
raise UnsupportedFormat(f"unsupported reader {reader!r}; use 'csv' or 'erddap-csv'")
|
|
60
|
+
try:
|
|
61
|
+
pandas = import_module("pandas")
|
|
62
|
+
except ModuleNotFoundError as error:
|
|
63
|
+
if error.name != "pandas":
|
|
64
|
+
raise
|
|
65
|
+
raise MissingReaderDependency(
|
|
66
|
+
'CSV reading requires pandas; install it with: pip install "usdata[pandas]" '
|
|
67
|
+
'(or uv add "usdata[pandas]")'
|
|
68
|
+
) from error
|
|
69
|
+
|
|
70
|
+
# Pass a file object to pandas: reading a fetched asset is strictly local.
|
|
71
|
+
with fetched.path.open(encoding="utf-8-sig", newline="") as stream:
|
|
72
|
+
records = csv.reader(stream)
|
|
73
|
+
columns = next(records, [])
|
|
74
|
+
if (
|
|
75
|
+
not columns
|
|
76
|
+
or any(not column for column in columns)
|
|
77
|
+
or len(set(columns)) != len(columns)
|
|
78
|
+
):
|
|
79
|
+
raise ValueError("CSV must have a non-empty header with unique column names")
|
|
80
|
+
units = {}
|
|
81
|
+
if reader == "erddap-csv":
|
|
82
|
+
values = next(records, [])
|
|
83
|
+
if len(values) != len(columns):
|
|
84
|
+
raise ValueError("ERDDAP CSV must have a units row matching the header")
|
|
85
|
+
units = dict(zip(columns, values, strict=True))
|
|
86
|
+
types = {name: "string" for name in columns if name.casefold() in IDENTIFIER_COLUMNS}
|
|
87
|
+
types.update(dtype or {})
|
|
88
|
+
frame = pandas.read_csv(
|
|
89
|
+
stream,
|
|
90
|
+
header=None,
|
|
91
|
+
names=columns,
|
|
92
|
+
dtype=types,
|
|
93
|
+
parse_dates=parse_dates,
|
|
94
|
+
usecols=usecols,
|
|
95
|
+
nrows=nrows,
|
|
96
|
+
)
|
|
97
|
+
if units:
|
|
98
|
+
frame.attrs["units"] = {name: units[name] for name in frame.columns}
|
|
99
|
+
frame.attrs["usdata"] = {
|
|
100
|
+
"asset_id": fetched.asset.id,
|
|
101
|
+
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
102
|
+
}
|
|
103
|
+
return frame
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|