usdata 0.1.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
usdata-0.2.1/PKG-INFO ADDED
@@ -0,0 +1,115 @@
1
+ Metadata-Version: 2.4
2
+ Name: usdata
3
+ Version: 0.2.1
4
+ Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
5
+ Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
6
+ Author: Jake Van Slyke
7
+ Author-email: Jake Van Slyke <jakervanslyke@gmail.com>
8
+ License-Expression: Apache-2.0
9
+ Classifier: Development Status :: 2 - Pre-Alpha
10
+ Classifier: Intended Audience :: Science/Research
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Topic :: Scientific/Engineering
13
+ Requires-Dist: httpx>=0.28.1
14
+ Requires-Dist: pydantic>=2.7
15
+ Requires-Dist: pyyaml>=6.0
16
+ Requires-Dist: typer>=0.12
17
+ Requires-Python: >=3.11
18
+ Project-URL: Homepage, https://github.com/jakeryderv/usdata
19
+ Project-URL: Repository, https://github.com/jakeryderv/usdata
20
+ Description-Content-Type: text/markdown
21
+
22
+ # usdata
23
+
24
+ Unified Python SDK and CLI for discovering, fetching, and tracking the
25
+ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
26
+
27
+ > Status: pre-alpha. `noaa:ghcn-daily` and `noaa:nexrad-level2` fetch real
28
+ > data with provenance; other registry entries are stubs.
29
+ > See [docs/roadmap.md](docs/roadmap.md).
30
+
31
+ ## Install
32
+
33
+ ```sh
34
+ pip install usdata # or: uv add usdata
35
+ ```
36
+
37
+ ## Usage
38
+
39
+ ```python
40
+ from usdata import build_query, get, search
41
+ from usdata.fetch import fetch
42
+
43
+ for r in search("precipitation", location="Oklahoma"):
44
+ print(r.dataset.id, r.dataset.title)
45
+
46
+ ds = get("noaa:ghcn-daily")
47
+ query = build_query(
48
+ lat=35.39,
49
+ lon=-97.60,
50
+ radius_km=15,
51
+ start="2024-05-06",
52
+ end="2024-05-07",
53
+ variables=["PRCP", "TMAX"],
54
+ )
55
+ for item in fetch(ds, query):
56
+ print(item.path, item.provenance.checksum)
57
+ ```
58
+
59
+ ```sh
60
+ usdata search "tornado radar" --state OK
61
+ usdata info noaa:ghcn-daily
62
+ usdata fetch noaa:ghcn-daily --lat 35.39 --lon -97.60 --radius-km 15 \
63
+ --start 2024-05-06 --end 2024-05-07 --vars PRCP,TMAX
64
+ usdata fetch noaa:ghcn-daily -p stations=USW00013967 --start 2024-01-01 --end 2024-12-31
65
+ usdata fetch noaa:nexrad-level2 --lat 35.47 --lon -97.52 \
66
+ --start 2024-05-06T20:00 --end 2024-05-06T23:00 # nearest radar (KTLX)
67
+ usdata fetch noaa:nexrad-level2 -p site=KTLX --start 2024-05-06T20:00 --end 2024-05-06T20:30 --dry-run
68
+ usdata pull dataset.yaml
69
+ ```
70
+
71
+ Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
72
+ `USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
73
+ recording source URL, retrieval time, checksum, size, and license.
74
+
75
+ A manifest declares every input a project needs; `pull` fetches them and writes
76
+ a lockfile with checksums and provenance so the inputs can be reproduced:
77
+
78
+ ```yaml
79
+ name: tornado-environment
80
+ sources:
81
+ - dataset: noaa:nexrad-level2
82
+ location: oklahoma
83
+ start: 2024-05-06
84
+ end: 2024-05-07
85
+ - dataset: noaa:ghcn-daily
86
+ location: oklahoma
87
+ start: 2024-05-01
88
+ end: 2024-05-31
89
+ ```
90
+
91
+ ## Development
92
+
93
+ Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
94
+
95
+ ```sh
96
+ git clone https://github.com/jakeryderv/usdata && cd usdata
97
+ just setup # install toolchain and dependencies
98
+ just test # unit tests
99
+ just check # format, lint, typecheck, tests (what CI runs)
100
+ just run search radar
101
+ ```
102
+
103
+ Integration tests that hit live services run with `just test-integration`.
104
+
105
+ Releases: `just release minor` opens a version-bump PR; merging it publishes
106
+ to PyPI and creates the tag and GitHub release. See
107
+ [docs/versioning.md](docs/versioning.md).
108
+
109
+ See [docs/architecture.md](docs/architecture.md) for how the pieces fit,
110
+ [docs/adr/](docs/adr/) for why, and [CONTRIBUTING.md](CONTRIBUTING.md) to add
111
+ a dataset.
112
+
113
+ ## License
114
+
115
+ Apache-2.0
usdata-0.2.1/README.md ADDED
@@ -0,0 +1,94 @@
1
+ # usdata
2
+
3
+ Unified Python SDK and CLI for discovering, fetching, and tracking the
4
+ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
5
+
6
+ > Status: pre-alpha. `noaa:ghcn-daily` and `noaa:nexrad-level2` fetch real
7
+ > data with provenance; other registry entries are stubs.
8
+ > See [docs/roadmap.md](docs/roadmap.md).
9
+
10
+ ## Install
11
+
12
+ ```sh
13
+ pip install usdata # or: uv add usdata
14
+ ```
15
+
16
+ ## Usage
17
+
18
+ ```python
19
+ from usdata import build_query, get, search
20
+ from usdata.fetch import fetch
21
+
22
+ for r in search("precipitation", location="Oklahoma"):
23
+ print(r.dataset.id, r.dataset.title)
24
+
25
+ ds = get("noaa:ghcn-daily")
26
+ query = build_query(
27
+ lat=35.39,
28
+ lon=-97.60,
29
+ radius_km=15,
30
+ start="2024-05-06",
31
+ end="2024-05-07",
32
+ variables=["PRCP", "TMAX"],
33
+ )
34
+ for item in fetch(ds, query):
35
+ print(item.path, item.provenance.checksum)
36
+ ```
37
+
38
+ ```sh
39
+ usdata search "tornado radar" --state OK
40
+ usdata info noaa:ghcn-daily
41
+ usdata fetch noaa:ghcn-daily --lat 35.39 --lon -97.60 --radius-km 15 \
42
+ --start 2024-05-06 --end 2024-05-07 --vars PRCP,TMAX
43
+ usdata fetch noaa:ghcn-daily -p stations=USW00013967 --start 2024-01-01 --end 2024-12-31
44
+ usdata fetch noaa:nexrad-level2 --lat 35.47 --lon -97.52 \
45
+ --start 2024-05-06T20:00 --end 2024-05-06T23:00 # nearest radar (KTLX)
46
+ usdata fetch noaa:nexrad-level2 -p site=KTLX --start 2024-05-06T20:00 --end 2024-05-06T20:30 --dry-run
47
+ usdata pull dataset.yaml
48
+ ```
49
+
50
+ Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
51
+ `USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
52
+ recording source URL, retrieval time, checksum, size, and license.
53
+
54
+ A manifest declares every input a project needs; `pull` fetches them and writes
55
+ a lockfile with checksums and provenance so the inputs can be reproduced:
56
+
57
+ ```yaml
58
+ name: tornado-environment
59
+ sources:
60
+ - dataset: noaa:nexrad-level2
61
+ location: oklahoma
62
+ start: 2024-05-06
63
+ end: 2024-05-07
64
+ - dataset: noaa:ghcn-daily
65
+ location: oklahoma
66
+ start: 2024-05-01
67
+ end: 2024-05-31
68
+ ```
69
+
70
+ ## Development
71
+
72
+ Requires [uv](https://docs.astral.sh/uv/) and [just](https://just.systems/).
73
+
74
+ ```sh
75
+ git clone https://github.com/jakeryderv/usdata && cd usdata
76
+ just setup # install toolchain and dependencies
77
+ just test # unit tests
78
+ just check # format, lint, typecheck, tests (what CI runs)
79
+ just run search radar
80
+ ```
81
+
82
+ Integration tests that hit live services run with `just test-integration`.
83
+
84
+ Releases: `just release minor` opens a version-bump PR; merging it publishes
85
+ to PyPI and creates the tag and GitHub release. See
86
+ [docs/versioning.md](docs/versioning.md).
87
+
88
+ See [docs/architecture.md](docs/architecture.md) for how the pieces fit,
89
+ [docs/adr/](docs/adr/) for why, and [CONTRIBUTING.md](CONTRIBUTING.md) to add
90
+ a dataset.
91
+
92
+ ## License
93
+
94
+ Apache-2.0
@@ -0,0 +1,97 @@
1
+ [project]
2
+ name = "usdata"
3
+ version = "0.2.1"
4
+ description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
+ readme = "README.md"
6
+ license = "Apache-2.0"
7
+ requires-python = ">=3.11"
8
+ keywords = [
9
+ "noaa",
10
+ "usgs",
11
+ "nasa",
12
+ "open-data",
13
+ "scientific-data",
14
+ "provenance",
15
+ ]
16
+ classifiers = [
17
+ "Development Status :: 2 - Pre-Alpha",
18
+ "Intended Audience :: Science/Research",
19
+ "Programming Language :: Python :: 3",
20
+ "Topic :: Scientific/Engineering",
21
+ ]
22
+ dependencies = [
23
+ "httpx>=0.28.1",
24
+ "pydantic>=2.7",
25
+ "pyyaml>=6.0",
26
+ "typer>=0.12",
27
+ ]
28
+
29
+ [[project.authors]]
30
+ name = "Jake Van Slyke"
31
+ email = "jakervanslyke@gmail.com"
32
+
33
+ [project.urls]
34
+ Homepage = "https://github.com/jakeryderv/usdata"
35
+ Repository = "https://github.com/jakeryderv/usdata"
36
+
37
+ [project.scripts]
38
+ usdata = "usdata.cli:app"
39
+
40
+ [dependency-groups]
41
+ dev = [
42
+ "pyright>=1.1.380",
43
+ "pytest>=8.0",
44
+ "pytest-cov>=7.1.0",
45
+ "respx>=0.23.1",
46
+ "ruff>=0.6",
47
+ ]
48
+
49
+ [build-system]
50
+ requires = ["uv_build>=0.12.5,<0.13.0"]
51
+ build-backend = "uv_build"
52
+
53
+ [tool.ruff]
54
+ line-length = 100
55
+ target-version = "py311"
56
+ src = [
57
+ "src",
58
+ "tests",
59
+ ]
60
+
61
+ [tool.ruff.lint]
62
+ select = [
63
+ "E",
64
+ "F",
65
+ "I",
66
+ "UP",
67
+ "B",
68
+ "SIM",
69
+ "RUF",
70
+ "D",
71
+ ]
72
+ ignore = [
73
+ "D105",
74
+ "D107",
75
+ ]
76
+
77
+ [tool.ruff.lint.pydocstyle]
78
+ convention = "google"
79
+
80
+ [tool.ruff.lint.per-file-ignores]
81
+ "tests/**" = ["D"]
82
+ "scripts/**" = ["D"]
83
+
84
+ [tool.pyright]
85
+ include = [
86
+ "src",
87
+ "tests",
88
+ ]
89
+ pythonVersion = "3.11"
90
+ typeCheckingMode = "standard"
91
+ venvPath = "."
92
+ venv = ".venv"
93
+
94
+ [tool.pytest.ini_options]
95
+ testpaths = ["tests"]
96
+ markers = ["integration: hits live services; skipped unless --run-integration is passed"]
97
+ addopts = "-ra"
@@ -0,0 +1,74 @@
1
+ [project]
2
+ name = "usdata"
3
+ version = "0.2.1"
4
+ description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
+ readme = "README.md"
6
+ license = "Apache-2.0"
7
+ authors = [
8
+ { name = "Jake Van Slyke", email = "jakervanslyke@gmail.com" }
9
+ ]
10
+ requires-python = ">=3.11"
11
+ keywords = ["noaa", "usgs", "nasa", "open-data", "scientific-data", "provenance"]
12
+ classifiers = [
13
+ "Development Status :: 2 - Pre-Alpha",
14
+ "Intended Audience :: Science/Research",
15
+ "Programming Language :: Python :: 3",
16
+ "Topic :: Scientific/Engineering",
17
+ ]
18
+ dependencies = [
19
+ "httpx>=0.28.1",
20
+ "pydantic>=2.7",
21
+ "pyyaml>=6.0",
22
+ "typer>=0.12",
23
+ ]
24
+
25
+ [project.urls]
26
+ Homepage = "https://github.com/jakeryderv/usdata"
27
+ Repository = "https://github.com/jakeryderv/usdata"
28
+
29
+ [project.scripts]
30
+ usdata = "usdata.cli:app"
31
+
32
+ [dependency-groups]
33
+ dev = [
34
+ "pyright>=1.1.380",
35
+ "pytest>=8.0",
36
+ "pytest-cov>=7.1.0",
37
+ "respx>=0.23.1",
38
+ "ruff>=0.6",
39
+ ]
40
+
41
+ [build-system]
42
+ requires = ["uv_build>=0.12.5,<0.13.0"]
43
+ build-backend = "uv_build"
44
+
45
+ [tool.ruff]
46
+ line-length = 100
47
+ target-version = "py311"
48
+ src = ["src", "tests"]
49
+
50
+ [tool.ruff.lint]
51
+ select = ["E", "F", "I", "UP", "B", "SIM", "RUF", "D"]
52
+ # Docstrings are required on the public API only; magic methods and __init__ are exempt.
53
+ ignore = ["D105", "D107"]
54
+
55
+ [tool.ruff.lint.pydocstyle]
56
+ convention = "google"
57
+
58
+ [tool.ruff.lint.per-file-ignores]
59
+ "tests/**" = ["D"]
60
+ "scripts/**" = ["D"]
61
+
62
+ [tool.pyright]
63
+ include = ["src", "tests"]
64
+ pythonVersion = "3.11"
65
+ typeCheckingMode = "standard"
66
+ venvPath = "."
67
+ venv = ".venv"
68
+
69
+ [tool.pytest.ini_options]
70
+ testpaths = ["tests"]
71
+ markers = [
72
+ "integration: hits live services; skipped unless --run-integration is passed",
73
+ ]
74
+ addopts = "-ra"
@@ -0,0 +1,42 @@
1
+ """usdata: unified access and provenance for U.S. public scientific data."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from importlib.metadata import PackageNotFoundError, version
6
+ from typing import Any
7
+
8
+ try:
9
+ __version__ = version("usdata")
10
+ except PackageNotFoundError: # running from a source tree without an install
11
+ __version__ = "0.0.0"
12
+
13
+ from usdata.models import Asset, BBox, Dataset, Provenance, Query, TimeRange
14
+ from usdata.query import build_query
15
+ from usdata.registry import DatasetNotFound, Registry, SearchResult, default_registry
16
+
17
+ __all__ = [
18
+ "Asset",
19
+ "BBox",
20
+ "Dataset",
21
+ "DatasetNotFound",
22
+ "Provenance",
23
+ "Query",
24
+ "Registry",
25
+ "SearchResult",
26
+ "TimeRange",
27
+ "__version__",
28
+ "build_query",
29
+ "default_registry",
30
+ "get",
31
+ "search",
32
+ ]
33
+
34
+
35
+ def search(text: str | None = None, **kwargs: Any) -> list[SearchResult]:
36
+ """Search the curated registry. Keyword arguments match ``build_query``."""
37
+ return default_registry().search(build_query(text, **kwargs))
38
+
39
+
40
+ def get(dataset_id: str) -> Dataset:
41
+ """Look up a dataset by ``provider:name`` id."""
42
+ return default_registry().get(dataset_id)
@@ -0,0 +1,36 @@
1
+ """Local file cache. Layout: ``<cache_dir>/<provider>/<dataset name>/<asset id>``."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import os
7
+ from pathlib import Path
8
+
9
+ from usdata.models import Asset
10
+
11
+ ENV_VAR = "USDATA_CACHE_DIR"
12
+
13
+
14
+ def cache_dir() -> Path:
15
+ """Cache root: $USDATA_CACHE_DIR, else $XDG_CACHE_HOME/usdata, else ~/.cache/usdata."""
16
+ if override := os.environ.get(ENV_VAR):
17
+ return Path(override).expanduser()
18
+ xdg = os.environ.get("XDG_CACHE_HOME")
19
+ base = Path(xdg).expanduser() if xdg else Path.home() / ".cache"
20
+ return base / "usdata"
21
+
22
+
23
+ def asset_path(asset: Asset, root: Path | None = None) -> Path:
24
+ """Where an asset lives in the cache: <root>/<provider>/<dataset>/<asset id>."""
25
+ provider, name = asset.dataset_id.split(":", 1)
26
+ safe_id = asset.id.replace("/", "_")
27
+ return (root or cache_dir()) / provider / name / safe_id
28
+
29
+
30
+ def sha256_file(path: Path, chunk_size: int = 1 << 20) -> str:
31
+ """Hex sha256 of a file, prefixed 'sha256:' to match Asset.checksum."""
32
+ h = hashlib.sha256()
33
+ with path.open("rb") as f:
34
+ while chunk := f.read(chunk_size):
35
+ h.update(chunk)
36
+ return f"sha256:{h.hexdigest()}"
@@ -0,0 +1,5 @@
1
+ """Command-line interface. The Typer app lives in ``usdata.cli.app``."""
2
+
3
+ from usdata.cli.app import app
4
+
5
+ __all__ = ["app"]
@@ -0,0 +1,181 @@
1
+ """Typer application: argument parsing and exit codes only; logic lives in the library."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Annotated
7
+
8
+ import httpx
9
+ import typer
10
+
11
+ from usdata import __version__, build_query, default_registry
12
+ from usdata.fetch import fetch_asset
13
+ from usdata.manifest import Manifest
14
+ from usdata.providers import load_adapter
15
+ from usdata.providers.base import NotImplementedProvider
16
+ from usdata.query import UnknownPlace
17
+ from usdata.registry import DatasetNotFound
18
+
19
+ app = typer.Typer(
20
+ name="usdata",
21
+ help="Discover, fetch, and track provenance of U.S. public scientific data.",
22
+ no_args_is_help=True,
23
+ )
24
+
25
+
26
+ def _version_callback(value: bool) -> None:
27
+ if value:
28
+ typer.echo(f"usdata {__version__}")
29
+ raise typer.Exit()
30
+
31
+
32
+ @app.callback()
33
+ def main(
34
+ version: Annotated[
35
+ bool | None,
36
+ typer.Option("--version", callback=_version_callback, is_eager=True, help="Show version."),
37
+ ] = None,
38
+ ) -> None:
39
+ """usdata: discover, fetch, and track provenance of U.S. public scientific data."""
40
+ pass
41
+
42
+
43
+ @app.command()
44
+ def search(
45
+ text: Annotated[str | None, typer.Argument(help="Free-text keywords.")] = None,
46
+ provider: Annotated[
47
+ str | None, typer.Option(help="Restrict to one provider, e.g. noaa.")
48
+ ] = None,
49
+ state: Annotated[str | None, typer.Option(help="State name or postal code.")] = None,
50
+ start: Annotated[str | None, typer.Option(help="ISO date or datetime.")] = None,
51
+ end: Annotated[str | None, typer.Option(help="ISO date or datetime.")] = None,
52
+ ) -> None:
53
+ """Search the curated dataset registry."""
54
+ try:
55
+ query = build_query(text, provider=provider, location=state, start=start, end=end)
56
+ except UnknownPlace as e:
57
+ typer.secho(f"Unknown place: {e}", err=True, fg="red")
58
+ raise typer.Exit(code=2) from None
59
+ results = default_registry().search(query)
60
+ if not results:
61
+ typer.echo("No datasets matched.")
62
+ raise typer.Exit(code=1)
63
+ width = max(len(r.dataset.id) for r in results)
64
+ for r in results:
65
+ typer.echo(f"{r.dataset.id:<{width}} {r.dataset.title}")
66
+
67
+
68
+ @app.command()
69
+ def info(
70
+ dataset_id: Annotated[str, typer.Argument(help="Dataset id, e.g. noaa:nexrad-level2")],
71
+ ) -> None:
72
+ """Show details for one dataset."""
73
+ try:
74
+ ds = default_registry().get(dataset_id)
75
+ except DatasetNotFound:
76
+ typer.secho(f"Unknown dataset: {dataset_id}", err=True, fg="red")
77
+ raise typer.Exit(code=2) from None
78
+ typer.echo(f"{ds.id}\n {ds.title}\n")
79
+ typer.echo(f" {ds.description.strip()}\n")
80
+ typer.echo(f" provider: {ds.provider}")
81
+ typer.echo(f" protocol: {ds.protocol.value}")
82
+ typer.echo(f" license: {ds.license or 'unknown'}")
83
+ typer.echo(f" homepage: {ds.homepage or '-'}")
84
+ caps = ", ".join(k for k, v in ds.capabilities.model_dump().items() if v) or "none"
85
+ typer.echo(f" subsetting: {caps}")
86
+ if ds.spatial_extent:
87
+ typer.echo(f" extent: {ds.spatial_extent.as_tuple()}")
88
+ if ds.temporal_extent:
89
+ typer.echo(
90
+ f" time: {ds.temporal_extent.start} .. {ds.temporal_extent.end or 'present'}"
91
+ )
92
+
93
+
94
+ @app.command()
95
+ def fetch(
96
+ dataset_id: Annotated[str, typer.Argument(help="Dataset id, e.g. noaa:ghcn-daily")],
97
+ state: Annotated[str | None, typer.Option(help="State name or postal code.")] = None,
98
+ bbox: Annotated[str | None, typer.Option(help="west,south,east,north in degrees.")] = None,
99
+ lat: Annotated[float | None, typer.Option()] = None,
100
+ lon: Annotated[float | None, typer.Option()] = None,
101
+ radius_km: Annotated[float, typer.Option(help="Radius around --lat/--lon.")] = 50.0,
102
+ start: Annotated[str | None, typer.Option(help="ISO date or datetime.")] = None,
103
+ end: Annotated[str | None, typer.Option(help="ISO date or datetime.")] = None,
104
+ variables: Annotated[
105
+ str | None, typer.Option("--vars", help="Comma-separated variable names.")
106
+ ] = None,
107
+ param: Annotated[
108
+ list[str] | None,
109
+ typer.Option("--param", "-p", help="Provider-specific key=value, repeatable."),
110
+ ] = None,
111
+ cache_dir: Annotated[Path | None, typer.Option(help="Override the cache directory.")] = None,
112
+ force: Annotated[bool, typer.Option(help="Re-download even if cached.")] = False,
113
+ dry_run: Annotated[
114
+ bool, typer.Option(help="List matching assets without downloading.")
115
+ ] = False,
116
+ ) -> None:
117
+ """Resolve a query against one dataset and download the matching assets."""
118
+ params: dict[str, str] = {}
119
+ for item in param or []:
120
+ key, sep, value = item.partition("=")
121
+ if not sep:
122
+ typer.secho(f"--param expects key=value, got {item!r}", err=True, fg="red")
123
+ raise typer.Exit(code=2)
124
+ params[key] = value
125
+ box = None
126
+ if bbox:
127
+ try:
128
+ w, s, e, n = (float(x) for x in bbox.split(","))
129
+ except ValueError:
130
+ typer.secho("--bbox expects west,south,east,north", err=True, fg="red")
131
+ raise typer.Exit(code=2) from None
132
+ box = (w, s, e, n)
133
+ try:
134
+ ds = default_registry().get(dataset_id)
135
+ query = build_query(
136
+ location=state,
137
+ bbox=box,
138
+ lat=lat,
139
+ lon=lon,
140
+ radius_km=radius_km,
141
+ start=start,
142
+ end=end,
143
+ variables=[v.strip() for v in variables.split(",")] if variables else None,
144
+ **params,
145
+ )
146
+ adapter = load_adapter(ds)
147
+ assets = adapter.list_assets(query)
148
+ if dry_run:
149
+ for a in assets:
150
+ typer.echo(f"{a.id}\t{a.href}")
151
+ typer.echo(f"{len(assets)} asset(s) matched", err=True)
152
+ return
153
+ fetched = [fetch_asset(ds, a, root=cache_dir, force=force) for a in assets]
154
+ except (DatasetNotFound, UnknownPlace, ValueError) as e:
155
+ typer.secho(str(e), err=True, fg="red")
156
+ raise typer.Exit(code=2) from None
157
+ except NotImplementedProvider as e:
158
+ typer.secho(str(e), err=True, fg="yellow")
159
+ raise typer.Exit(code=3) from None
160
+ except httpx.HTTPError as e:
161
+ typer.secho(f"request failed: {e}", err=True, fg="red")
162
+ raise typer.Exit(code=4) from None
163
+ if not fetched:
164
+ typer.echo("No assets matched.", err=True)
165
+ raise typer.Exit(code=1)
166
+ for f in fetched:
167
+ tag = "cached" if f.from_cache else "fetched"
168
+ typer.echo(f"{f.path}\t{tag}\t{f.provenance.size} bytes")
169
+
170
+
171
+ @app.command()
172
+ def pull(manifest: Annotated[Path, typer.Argument(exists=True, dir_okay=False)]) -> None:
173
+ """Fetch every source in a manifest and write a lockfile."""
174
+ m = Manifest.load(manifest)
175
+ missing = m.validate_against()
176
+ if missing:
177
+ typer.secho(f"Unknown datasets in manifest: {', '.join(missing)}", err=True, fg="red")
178
+ raise typer.Exit(code=2)
179
+ typer.echo(f"{m.name} v{m.version}: {len(m.sources)} source(s) validated")
180
+ typer.secho("pull is not implemented yet; no data was fetched.", err=True, fg="yellow")
181
+ raise typer.Exit(code=3)