usdata 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- usdata-0.2.0/PKG-INFO +113 -0
- usdata-0.2.0/README.md +92 -0
- usdata-0.2.0/pyproject.toml +84 -0
- usdata-0.2.0/pyproject.toml.orig +64 -0
- usdata-0.2.0/src/usdata/__init__.py +42 -0
- usdata-0.2.0/src/usdata/cache.py +33 -0
- usdata-0.2.0/src/usdata/cli/__init__.py +3 -0
- usdata-0.2.0/src/usdata/cli/app.py +178 -0
- usdata-0.2.0/src/usdata/data/nexrad_sites.csv +211 -0
- usdata-0.2.0/src/usdata/data/places.yaml +10 -0
- usdata-0.2.0/src/usdata/data/registry.yaml +50 -0
- usdata-0.2.0/src/usdata/fetch.py +50 -0
- usdata-0.2.0/src/usdata/manifest.py +73 -0
- usdata-0.2.0/src/usdata/models.py +169 -0
- usdata-0.2.0/src/usdata/protocols/__init__.py +1 -0
- usdata-0.2.0/src/usdata/protocols/http.py +41 -0
- usdata-0.2.0/src/usdata/protocols/s3.py +85 -0
- usdata-0.2.0/src/usdata/provenance.py +40 -0
- usdata-0.2.0/src/usdata/providers/__init__.py +5 -0
- usdata-0.2.0/src/usdata/providers/base.py +39 -0
- usdata-0.2.0/src/usdata/providers/noaa/__init__.py +1 -0
- usdata-0.2.0/src/usdata/providers/noaa/coastwatch.py +16 -0
- usdata-0.2.0/src/usdata/providers/noaa/ghcnd.py +128 -0
- usdata-0.2.0/src/usdata/providers/noaa/nexrad.py +102 -0
- usdata-0.2.0/src/usdata/providers/noaa/sites.py +65 -0
- usdata-0.2.0/src/usdata/query.py +100 -0
- usdata-0.2.0/src/usdata/registry.py +114 -0
- usdata-0.1.0/PKG-INFO +0 -10
- usdata-0.1.0/README.md +0 -1
- usdata-0.1.0/pyproject.toml +0 -18
- usdata-0.1.0/pyproject.toml.orig +0 -17
- usdata-0.1.0/src/usdata/__init__.py +0 -2
usdata-0.2.0/PKG-INFO
ADDED
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: usdata
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
|
|
5
|
+
Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
|
|
6
|
+
Author: Jake Van Slyke
|
|
7
|
+
Author-email: Jake Van Slyke <jakervanslyke@gmail.com>
|
|
8
|
+
License-Expression: Apache-2.0
|
|
9
|
+
Classifier: Development Status :: 2 - Pre-Alpha
|
|
10
|
+
Classifier: Intended Audience :: Science/Research
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering
|
|
13
|
+
Requires-Dist: httpx>=0.28.1
|
|
14
|
+
Requires-Dist: pydantic>=2.7
|
|
15
|
+
Requires-Dist: pyyaml>=6.0
|
|
16
|
+
Requires-Dist: typer>=0.12
|
|
17
|
+
Requires-Python: >=3.11
|
|
18
|
+
Project-URL: Homepage, https://github.com/jakeryderv/usdata
|
|
19
|
+
Project-URL: Repository, https://github.com/jakeryderv/usdata
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
|
|
22
|
+
# usdata
|
|
23
|
+
|
|
24
|
+
Unified Python SDK and CLI for discovering, fetching, and tracking the
|
|
25
|
+
provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
26
|
+
|
|
27
|
+
> Status: pre-alpha. `noaa:ghcn-daily` and `noaa:nexrad-level2` fetch real
|
|
28
|
+
> data with provenance; other registry entries are stubs.
|
|
29
|
+
> See [docs/roadmap.md](docs/roadmap.md).
|
|
30
|
+
|
|
31
|
+
## Install
|
|
32
|
+
|
|
33
|
+
```sh
|
|
34
|
+
pip install usdata # or: uv add usdata
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Usage
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
from usdata import build_query, get, search
|
|
41
|
+
from usdata.fetch import fetch
|
|
42
|
+
|
|
43
|
+
for r in search("precipitation", location="Oklahoma"):
|
|
44
|
+
print(r.dataset.id, r.dataset.title)
|
|
45
|
+
|
|
46
|
+
ds = get("noaa:ghcn-daily")
|
|
47
|
+
query = build_query(
|
|
48
|
+
lat=35.39,
|
|
49
|
+
lon=-97.60,
|
|
50
|
+
radius_km=15,
|
|
51
|
+
start="2024-05-06",
|
|
52
|
+
end="2024-05-07",
|
|
53
|
+
variables=["PRCP", "TMAX"],
|
|
54
|
+
)
|
|
55
|
+
for item in fetch(ds, query):
|
|
56
|
+
print(item.path, item.provenance.checksum)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
```sh
|
|
60
|
+
usdata search "tornado radar" --state OK
|
|
61
|
+
usdata info noaa:ghcn-daily
|
|
62
|
+
usdata fetch noaa:ghcn-daily --lat 35.39 --lon -97.60 --radius-km 15 \
|
|
63
|
+
--start 2024-05-06 --end 2024-05-07 --vars PRCP,TMAX
|
|
64
|
+
usdata fetch noaa:ghcn-daily -p stations=USW00013967 --start 2024-01-01 --end 2024-12-31
|
|
65
|
+
usdata fetch noaa:nexrad-level2 --lat 35.47 --lon -97.52 \
|
|
66
|
+
--start 2024-05-06T20:00 --end 2024-05-06T23:00 # nearest radar (KTLX)
|
|
67
|
+
usdata fetch noaa:nexrad-level2 -p site=KTLX --start 2024-05-06T20:00 --end 2024-05-06T20:30 --dry-run
|
|
68
|
+
usdata pull dataset.yaml
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
|
|
72
|
+
`USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
|
|
73
|
+
recording source URL, retrieval time, checksum, size, and license.
|
|
74
|
+
|
|
75
|
+
A manifest declares every input a project needs; `pull` fetches them and writes
|
|
76
|
+
a lockfile with checksums and provenance so the inputs can be reproduced:
|
|
77
|
+
|
|
78
|
+
```yaml
|
|
79
|
+
name: tornado-environment
|
|
80
|
+
sources:
|
|
81
|
+
- dataset: noaa:nexrad-level2
|
|
82
|
+
location: oklahoma
|
|
83
|
+
start: 2024-05-06
|
|
84
|
+
end: 2024-05-07
|
|
85
|
+
- dataset: noaa:ghcn-daily
|
|
86
|
+
location: oklahoma
|
|
87
|
+
start: 2024-05-01
|
|
88
|
+
end: 2024-05-31
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
## Development
|
|
92
|
+
|
|
93
|
+
Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
|
|
94
|
+
|
|
95
|
+
```sh
|
|
96
|
+
git clone https://github.com/jakeryderv/usdata && cd usdata
|
|
97
|
+
just setup # install toolchain and dependencies
|
|
98
|
+
just test # unit tests
|
|
99
|
+
just check # format, lint, typecheck, tests (what CI runs)
|
|
100
|
+
just run search radar
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
Integration tests that hit live services run with `just test-integration`.
|
|
104
|
+
|
|
105
|
+
Releases are published to PyPI by tagging: create a GitHub release and the
|
|
106
|
+
`publish.yml` workflow uploads via trusted publishing.
|
|
107
|
+
|
|
108
|
+
See [docs/architecture.md](docs/architecture.md) for how the pieces fit and
|
|
109
|
+
[docs/adr/](docs/adr/) for why.
|
|
110
|
+
|
|
111
|
+
## License
|
|
112
|
+
|
|
113
|
+
Apache-2.0
|
usdata-0.2.0/README.md
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
# usdata
|
|
2
|
+
|
|
3
|
+
Unified Python SDK and CLI for discovering, fetching, and tracking the
|
|
4
|
+
provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
5
|
+
|
|
6
|
+
> Status: pre-alpha. `noaa:ghcn-daily` and `noaa:nexrad-level2` fetch real
|
|
7
|
+
> data with provenance; other registry entries are stubs.
|
|
8
|
+
> See [docs/roadmap.md](docs/roadmap.md).
|
|
9
|
+
|
|
10
|
+
## Install
|
|
11
|
+
|
|
12
|
+
```sh
|
|
13
|
+
pip install usdata # or: uv add usdata
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
## Usage
|
|
17
|
+
|
|
18
|
+
```python
|
|
19
|
+
from usdata import build_query, get, search
|
|
20
|
+
from usdata.fetch import fetch
|
|
21
|
+
|
|
22
|
+
for r in search("precipitation", location="Oklahoma"):
|
|
23
|
+
print(r.dataset.id, r.dataset.title)
|
|
24
|
+
|
|
25
|
+
ds = get("noaa:ghcn-daily")
|
|
26
|
+
query = build_query(
|
|
27
|
+
lat=35.39,
|
|
28
|
+
lon=-97.60,
|
|
29
|
+
radius_km=15,
|
|
30
|
+
start="2024-05-06",
|
|
31
|
+
end="2024-05-07",
|
|
32
|
+
variables=["PRCP", "TMAX"],
|
|
33
|
+
)
|
|
34
|
+
for item in fetch(ds, query):
|
|
35
|
+
print(item.path, item.provenance.checksum)
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
```sh
|
|
39
|
+
usdata search "tornado radar" --state OK
|
|
40
|
+
usdata info noaa:ghcn-daily
|
|
41
|
+
usdata fetch noaa:ghcn-daily --lat 35.39 --lon -97.60 --radius-km 15 \
|
|
42
|
+
--start 2024-05-06 --end 2024-05-07 --vars PRCP,TMAX
|
|
43
|
+
usdata fetch noaa:ghcn-daily -p stations=USW00013967 --start 2024-01-01 --end 2024-12-31
|
|
44
|
+
usdata fetch noaa:nexrad-level2 --lat 35.47 --lon -97.52 \
|
|
45
|
+
--start 2024-05-06T20:00 --end 2024-05-06T23:00 # nearest radar (KTLX)
|
|
46
|
+
usdata fetch noaa:nexrad-level2 -p site=KTLX --start 2024-05-06T20:00 --end 2024-05-06T20:30 --dry-run
|
|
47
|
+
usdata pull dataset.yaml
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
|
|
51
|
+
`USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
|
|
52
|
+
recording source URL, retrieval time, checksum, size, and license.
|
|
53
|
+
|
|
54
|
+
A manifest declares every input a project needs; `pull` fetches them and writes
|
|
55
|
+
a lockfile with checksums and provenance so the inputs can be reproduced:
|
|
56
|
+
|
|
57
|
+
```yaml
|
|
58
|
+
name: tornado-environment
|
|
59
|
+
sources:
|
|
60
|
+
- dataset: noaa:nexrad-level2
|
|
61
|
+
location: oklahoma
|
|
62
|
+
start: 2024-05-06
|
|
63
|
+
end: 2024-05-07
|
|
64
|
+
- dataset: noaa:ghcn-daily
|
|
65
|
+
location: oklahoma
|
|
66
|
+
start: 2024-05-01
|
|
67
|
+
end: 2024-05-31
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
## Development
|
|
71
|
+
|
|
72
|
+
Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
|
|
73
|
+
|
|
74
|
+
```sh
|
|
75
|
+
git clone https://github.com/jakeryderv/usdata && cd usdata
|
|
76
|
+
just setup # install toolchain and dependencies
|
|
77
|
+
just test # unit tests
|
|
78
|
+
just check # format, lint, typecheck, tests (what CI runs)
|
|
79
|
+
just run search radar
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
Integration tests that hit live services run with `just test-integration`.
|
|
83
|
+
|
|
84
|
+
Releases are published to PyPI by tagging: create a GitHub release and the
|
|
85
|
+
`publish.yml` workflow uploads via trusted publishing.
|
|
86
|
+
|
|
87
|
+
See [docs/architecture.md](docs/architecture.md) for how the pieces fit and
|
|
88
|
+
[docs/adr/](docs/adr/) for why.
|
|
89
|
+
|
|
90
|
+
## License
|
|
91
|
+
|
|
92
|
+
Apache-2.0
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "usdata"
|
|
3
|
+
version = "0.2.0"
|
|
4
|
+
description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "Apache-2.0"
|
|
7
|
+
requires-python = ">=3.11"
|
|
8
|
+
keywords = [
|
|
9
|
+
"noaa",
|
|
10
|
+
"usgs",
|
|
11
|
+
"nasa",
|
|
12
|
+
"open-data",
|
|
13
|
+
"scientific-data",
|
|
14
|
+
"provenance",
|
|
15
|
+
]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Development Status :: 2 - Pre-Alpha",
|
|
18
|
+
"Intended Audience :: Science/Research",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Topic :: Scientific/Engineering",
|
|
21
|
+
]
|
|
22
|
+
dependencies = [
|
|
23
|
+
"httpx>=0.28.1",
|
|
24
|
+
"pydantic>=2.7",
|
|
25
|
+
"pyyaml>=6.0",
|
|
26
|
+
"typer>=0.12",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
[[project.authors]]
|
|
30
|
+
name = "Jake Van Slyke"
|
|
31
|
+
email = "jakervanslyke@gmail.com"
|
|
32
|
+
|
|
33
|
+
[project.urls]
|
|
34
|
+
Homepage = "https://github.com/jakeryderv/usdata"
|
|
35
|
+
Repository = "https://github.com/jakeryderv/usdata"
|
|
36
|
+
|
|
37
|
+
[project.scripts]
|
|
38
|
+
usdata = "usdata.cli:app"
|
|
39
|
+
|
|
40
|
+
[dependency-groups]
|
|
41
|
+
dev = [
|
|
42
|
+
"pyright>=1.1.380",
|
|
43
|
+
"pytest>=8.0",
|
|
44
|
+
"respx>=0.23.1",
|
|
45
|
+
"ruff>=0.6",
|
|
46
|
+
]
|
|
47
|
+
|
|
48
|
+
[build-system]
|
|
49
|
+
requires = ["uv_build>=0.12.5,<0.13.0"]
|
|
50
|
+
build-backend = "uv_build"
|
|
51
|
+
|
|
52
|
+
[tool.ruff]
|
|
53
|
+
line-length = 100
|
|
54
|
+
target-version = "py311"
|
|
55
|
+
src = [
|
|
56
|
+
"src",
|
|
57
|
+
"tests",
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
[tool.ruff.lint]
|
|
61
|
+
select = [
|
|
62
|
+
"E",
|
|
63
|
+
"F",
|
|
64
|
+
"I",
|
|
65
|
+
"UP",
|
|
66
|
+
"B",
|
|
67
|
+
"SIM",
|
|
68
|
+
"RUF",
|
|
69
|
+
]
|
|
70
|
+
|
|
71
|
+
[tool.pyright]
|
|
72
|
+
include = [
|
|
73
|
+
"src",
|
|
74
|
+
"tests",
|
|
75
|
+
]
|
|
76
|
+
pythonVersion = "3.11"
|
|
77
|
+
typeCheckingMode = "standard"
|
|
78
|
+
venvPath = "."
|
|
79
|
+
venv = ".venv"
|
|
80
|
+
|
|
81
|
+
[tool.pytest.ini_options]
|
|
82
|
+
testpaths = ["tests"]
|
|
83
|
+
markers = ["integration: hits live services; skipped unless --run-integration is passed"]
|
|
84
|
+
addopts = "-ra"
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "usdata"
|
|
3
|
+
version = "0.2.0"
|
|
4
|
+
description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "Apache-2.0"
|
|
7
|
+
authors = [
|
|
8
|
+
{ name = "Jake Van Slyke", email = "jakervanslyke@gmail.com" }
|
|
9
|
+
]
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
keywords = ["noaa", "usgs", "nasa", "open-data", "scientific-data", "provenance"]
|
|
12
|
+
classifiers = [
|
|
13
|
+
"Development Status :: 2 - Pre-Alpha",
|
|
14
|
+
"Intended Audience :: Science/Research",
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"Topic :: Scientific/Engineering",
|
|
17
|
+
]
|
|
18
|
+
dependencies = [
|
|
19
|
+
"httpx>=0.28.1",
|
|
20
|
+
"pydantic>=2.7",
|
|
21
|
+
"pyyaml>=6.0",
|
|
22
|
+
"typer>=0.12",
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
[project.urls]
|
|
26
|
+
Homepage = "https://github.com/jakeryderv/usdata"
|
|
27
|
+
Repository = "https://github.com/jakeryderv/usdata"
|
|
28
|
+
|
|
29
|
+
[project.scripts]
|
|
30
|
+
usdata = "usdata.cli:app"
|
|
31
|
+
|
|
32
|
+
[dependency-groups]
|
|
33
|
+
dev = [
|
|
34
|
+
"pyright>=1.1.380",
|
|
35
|
+
"pytest>=8.0",
|
|
36
|
+
"respx>=0.23.1",
|
|
37
|
+
"ruff>=0.6",
|
|
38
|
+
]
|
|
39
|
+
|
|
40
|
+
[build-system]
|
|
41
|
+
requires = ["uv_build>=0.12.5,<0.13.0"]
|
|
42
|
+
build-backend = "uv_build"
|
|
43
|
+
|
|
44
|
+
[tool.ruff]
|
|
45
|
+
line-length = 100
|
|
46
|
+
target-version = "py311"
|
|
47
|
+
src = ["src", "tests"]
|
|
48
|
+
|
|
49
|
+
[tool.ruff.lint]
|
|
50
|
+
select = ["E", "F", "I", "UP", "B", "SIM", "RUF"]
|
|
51
|
+
|
|
52
|
+
[tool.pyright]
|
|
53
|
+
include = ["src", "tests"]
|
|
54
|
+
pythonVersion = "3.11"
|
|
55
|
+
typeCheckingMode = "standard"
|
|
56
|
+
venvPath = "."
|
|
57
|
+
venv = ".venv"
|
|
58
|
+
|
|
59
|
+
[tool.pytest.ini_options]
|
|
60
|
+
testpaths = ["tests"]
|
|
61
|
+
markers = [
|
|
62
|
+
"integration: hits live services; skipped unless --run-integration is passed",
|
|
63
|
+
]
|
|
64
|
+
addopts = "-ra"
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""usdata: unified access and provenance for U.S. public scientific data."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
try:
|
|
9
|
+
__version__ = version("usdata")
|
|
10
|
+
except PackageNotFoundError: # running from a source tree without an install
|
|
11
|
+
__version__ = "0.0.0"
|
|
12
|
+
|
|
13
|
+
from usdata.models import Asset, BBox, Dataset, Provenance, Query, TimeRange
|
|
14
|
+
from usdata.query import build_query
|
|
15
|
+
from usdata.registry import DatasetNotFound, Registry, SearchResult, default_registry
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"Asset",
|
|
19
|
+
"BBox",
|
|
20
|
+
"Dataset",
|
|
21
|
+
"DatasetNotFound",
|
|
22
|
+
"Provenance",
|
|
23
|
+
"Query",
|
|
24
|
+
"Registry",
|
|
25
|
+
"SearchResult",
|
|
26
|
+
"TimeRange",
|
|
27
|
+
"__version__",
|
|
28
|
+
"build_query",
|
|
29
|
+
"default_registry",
|
|
30
|
+
"get",
|
|
31
|
+
"search",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def search(text: str | None = None, **kwargs: Any) -> list[SearchResult]:
|
|
36
|
+
"""Search the curated registry. Keyword arguments match ``build_query``."""
|
|
37
|
+
return default_registry().search(build_query(text, **kwargs))
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def get(dataset_id: str) -> Dataset:
|
|
41
|
+
"""Look up a dataset by ``provider:name`` id."""
|
|
42
|
+
return default_registry().get(dataset_id)
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""Local file cache. Layout: ``<cache_dir>/<provider>/<dataset name>/<asset id>``."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import os
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from usdata.models import Asset
|
|
10
|
+
|
|
11
|
+
ENV_VAR = "USDATA_CACHE_DIR"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def cache_dir() -> Path:
|
|
15
|
+
if override := os.environ.get(ENV_VAR):
|
|
16
|
+
return Path(override).expanduser()
|
|
17
|
+
xdg = os.environ.get("XDG_CACHE_HOME")
|
|
18
|
+
base = Path(xdg).expanduser() if xdg else Path.home() / ".cache"
|
|
19
|
+
return base / "usdata"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def asset_path(asset: Asset, root: Path | None = None) -> Path:
|
|
23
|
+
provider, name = asset.dataset_id.split(":", 1)
|
|
24
|
+
safe_id = asset.id.replace("/", "_")
|
|
25
|
+
return (root or cache_dir()) / provider / name / safe_id
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def sha256_file(path: Path, chunk_size: int = 1 << 20) -> str:
|
|
29
|
+
h = hashlib.sha256()
|
|
30
|
+
with path.open("rb") as f:
|
|
31
|
+
while chunk := f.read(chunk_size):
|
|
32
|
+
h.update(chunk)
|
|
33
|
+
return f"sha256:{h.hexdigest()}"
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Annotated
|
|
5
|
+
|
|
6
|
+
import httpx
|
|
7
|
+
import typer
|
|
8
|
+
|
|
9
|
+
from usdata import __version__, build_query, default_registry
|
|
10
|
+
from usdata.fetch import fetch_asset
|
|
11
|
+
from usdata.manifest import Manifest
|
|
12
|
+
from usdata.providers import load_adapter
|
|
13
|
+
from usdata.providers.base import NotImplementedProvider
|
|
14
|
+
from usdata.query import UnknownPlace
|
|
15
|
+
from usdata.registry import DatasetNotFound
|
|
16
|
+
|
|
17
|
+
app = typer.Typer(
|
|
18
|
+
name="usdata",
|
|
19
|
+
help="Discover, fetch, and track provenance of U.S. public scientific data.",
|
|
20
|
+
no_args_is_help=True,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _version_callback(value: bool) -> None:
|
|
25
|
+
if value:
|
|
26
|
+
typer.echo(f"usdata {__version__}")
|
|
27
|
+
raise typer.Exit()
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@app.callback()
|
|
31
|
+
def main(
|
|
32
|
+
version: Annotated[
|
|
33
|
+
bool | None,
|
|
34
|
+
typer.Option("--version", callback=_version_callback, is_eager=True, help="Show version."),
|
|
35
|
+
] = None,
|
|
36
|
+
) -> None:
|
|
37
|
+
pass
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@app.command()
|
|
41
|
+
def search(
|
|
42
|
+
text: Annotated[str | None, typer.Argument(help="Free-text keywords.")] = None,
|
|
43
|
+
provider: Annotated[
|
|
44
|
+
str | None, typer.Option(help="Restrict to one provider, e.g. noaa.")
|
|
45
|
+
] = None,
|
|
46
|
+
state: Annotated[str | None, typer.Option(help="State name or postal code.")] = None,
|
|
47
|
+
start: Annotated[str | None, typer.Option(help="ISO date or datetime.")] = None,
|
|
48
|
+
end: Annotated[str | None, typer.Option(help="ISO date or datetime.")] = None,
|
|
49
|
+
) -> None:
|
|
50
|
+
"""Search the curated dataset registry."""
|
|
51
|
+
try:
|
|
52
|
+
query = build_query(text, provider=provider, location=state, start=start, end=end)
|
|
53
|
+
except UnknownPlace as e:
|
|
54
|
+
typer.secho(f"Unknown place: {e}", err=True, fg="red")
|
|
55
|
+
raise typer.Exit(code=2) from None
|
|
56
|
+
results = default_registry().search(query)
|
|
57
|
+
if not results:
|
|
58
|
+
typer.echo("No datasets matched.")
|
|
59
|
+
raise typer.Exit(code=1)
|
|
60
|
+
width = max(len(r.dataset.id) for r in results)
|
|
61
|
+
for r in results:
|
|
62
|
+
typer.echo(f"{r.dataset.id:<{width}} {r.dataset.title}")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@app.command()
|
|
66
|
+
def info(
|
|
67
|
+
dataset_id: Annotated[str, typer.Argument(help="Dataset id, e.g. noaa:nexrad-level2")],
|
|
68
|
+
) -> None:
|
|
69
|
+
"""Show details for one dataset."""
|
|
70
|
+
try:
|
|
71
|
+
ds = default_registry().get(dataset_id)
|
|
72
|
+
except DatasetNotFound:
|
|
73
|
+
typer.secho(f"Unknown dataset: {dataset_id}", err=True, fg="red")
|
|
74
|
+
raise typer.Exit(code=2) from None
|
|
75
|
+
typer.echo(f"{ds.id}\n {ds.title}\n")
|
|
76
|
+
typer.echo(f" {ds.description.strip()}\n")
|
|
77
|
+
typer.echo(f" provider: {ds.provider}")
|
|
78
|
+
typer.echo(f" protocol: {ds.protocol.value}")
|
|
79
|
+
typer.echo(f" license: {ds.license or 'unknown'}")
|
|
80
|
+
typer.echo(f" homepage: {ds.homepage or '-'}")
|
|
81
|
+
caps = ", ".join(k for k, v in ds.capabilities.model_dump().items() if v) or "none"
|
|
82
|
+
typer.echo(f" subsetting: {caps}")
|
|
83
|
+
if ds.spatial_extent:
|
|
84
|
+
typer.echo(f" extent: {ds.spatial_extent.as_tuple()}")
|
|
85
|
+
if ds.temporal_extent:
|
|
86
|
+
typer.echo(
|
|
87
|
+
f" time: {ds.temporal_extent.start} .. {ds.temporal_extent.end or 'present'}"
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@app.command()
|
|
92
|
+
def fetch(
|
|
93
|
+
dataset_id: Annotated[str, typer.Argument(help="Dataset id, e.g. noaa:ghcn-daily")],
|
|
94
|
+
state: Annotated[str | None, typer.Option(help="State name or postal code.")] = None,
|
|
95
|
+
bbox: Annotated[str | None, typer.Option(help="west,south,east,north in degrees.")] = None,
|
|
96
|
+
lat: Annotated[float | None, typer.Option()] = None,
|
|
97
|
+
lon: Annotated[float | None, typer.Option()] = None,
|
|
98
|
+
radius_km: Annotated[float, typer.Option(help="Radius around --lat/--lon.")] = 50.0,
|
|
99
|
+
start: Annotated[str | None, typer.Option(help="ISO date or datetime.")] = None,
|
|
100
|
+
end: Annotated[str | None, typer.Option(help="ISO date or datetime.")] = None,
|
|
101
|
+
variables: Annotated[
|
|
102
|
+
str | None, typer.Option("--vars", help="Comma-separated variable names.")
|
|
103
|
+
] = None,
|
|
104
|
+
param: Annotated[
|
|
105
|
+
list[str] | None,
|
|
106
|
+
typer.Option("--param", "-p", help="Provider-specific key=value, repeatable."),
|
|
107
|
+
] = None,
|
|
108
|
+
cache_dir: Annotated[Path | None, typer.Option(help="Override the cache directory.")] = None,
|
|
109
|
+
force: Annotated[bool, typer.Option(help="Re-download even if cached.")] = False,
|
|
110
|
+
dry_run: Annotated[
|
|
111
|
+
bool, typer.Option(help="List matching assets without downloading.")
|
|
112
|
+
] = False,
|
|
113
|
+
) -> None:
|
|
114
|
+
"""Resolve a query against one dataset and download the matching assets."""
|
|
115
|
+
params: dict[str, str] = {}
|
|
116
|
+
for item in param or []:
|
|
117
|
+
key, sep, value = item.partition("=")
|
|
118
|
+
if not sep:
|
|
119
|
+
typer.secho(f"--param expects key=value, got {item!r}", err=True, fg="red")
|
|
120
|
+
raise typer.Exit(code=2)
|
|
121
|
+
params[key] = value
|
|
122
|
+
box = None
|
|
123
|
+
if bbox:
|
|
124
|
+
try:
|
|
125
|
+
w, s, e, n = (float(x) for x in bbox.split(","))
|
|
126
|
+
except ValueError:
|
|
127
|
+
typer.secho("--bbox expects west,south,east,north", err=True, fg="red")
|
|
128
|
+
raise typer.Exit(code=2) from None
|
|
129
|
+
box = (w, s, e, n)
|
|
130
|
+
try:
|
|
131
|
+
ds = default_registry().get(dataset_id)
|
|
132
|
+
query = build_query(
|
|
133
|
+
location=state,
|
|
134
|
+
bbox=box,
|
|
135
|
+
lat=lat,
|
|
136
|
+
lon=lon,
|
|
137
|
+
radius_km=radius_km,
|
|
138
|
+
start=start,
|
|
139
|
+
end=end,
|
|
140
|
+
variables=[v.strip() for v in variables.split(",")] if variables else None,
|
|
141
|
+
**params,
|
|
142
|
+
)
|
|
143
|
+
adapter = load_adapter(ds)
|
|
144
|
+
assets = adapter.list_assets(query)
|
|
145
|
+
if dry_run:
|
|
146
|
+
for a in assets:
|
|
147
|
+
typer.echo(f"{a.id}\t{a.href}")
|
|
148
|
+
typer.echo(f"{len(assets)} asset(s) matched", err=True)
|
|
149
|
+
return
|
|
150
|
+
fetched = [fetch_asset(ds, a, root=cache_dir, force=force) for a in assets]
|
|
151
|
+
except (DatasetNotFound, UnknownPlace, ValueError) as e:
|
|
152
|
+
typer.secho(str(e), err=True, fg="red")
|
|
153
|
+
raise typer.Exit(code=2) from None
|
|
154
|
+
except NotImplementedProvider as e:
|
|
155
|
+
typer.secho(str(e), err=True, fg="yellow")
|
|
156
|
+
raise typer.Exit(code=3) from None
|
|
157
|
+
except httpx.HTTPError as e:
|
|
158
|
+
typer.secho(f"request failed: {e}", err=True, fg="red")
|
|
159
|
+
raise typer.Exit(code=4) from None
|
|
160
|
+
if not fetched:
|
|
161
|
+
typer.echo("No assets matched.", err=True)
|
|
162
|
+
raise typer.Exit(code=1)
|
|
163
|
+
for f in fetched:
|
|
164
|
+
tag = "cached" if f.from_cache else "fetched"
|
|
165
|
+
typer.echo(f"{f.path}\t{tag}\t{f.provenance.size} bytes")
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
@app.command()
|
|
169
|
+
def pull(manifest: Annotated[Path, typer.Argument(exists=True, dir_okay=False)]) -> None:
|
|
170
|
+
"""Fetch every source in a manifest and write a lockfile."""
|
|
171
|
+
m = Manifest.load(manifest)
|
|
172
|
+
missing = m.validate_against()
|
|
173
|
+
if missing:
|
|
174
|
+
typer.secho(f"Unknown datasets in manifest: {', '.join(missing)}", err=True, fg="red")
|
|
175
|
+
raise typer.Exit(code=2)
|
|
176
|
+
typer.echo(f"{m.name} v{m.version}: {len(m.sources)} source(s) validated")
|
|
177
|
+
typer.secho("pull is not implemented yet; no data was fetched.", err=True, fg="yellow")
|
|
178
|
+
raise typer.Exit(code=3)
|