usdata 0.2.1__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {usdata-0.2.1 → usdata-0.4.0}/PKG-INFO +30 -5
- {usdata-0.2.1 → usdata-0.4.0}/README.md +29 -4
- {usdata-0.2.1 → usdata-0.4.0}/pyproject.toml +1 -1
- {usdata-0.2.1 → usdata-0.4.0}/pyproject.toml.orig +1 -1
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/__init__.py +7 -2
- usdata-0.4.0/src/usdata/_files.py +29 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/cache.py +14 -2
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/cli/app.py +67 -19
- usdata-0.4.0/src/usdata/data/registry.yaml +666 -0
- usdata-0.4.0/src/usdata/fetch.py +79 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/manifest.py +32 -3
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/models.py +67 -4
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/protocols/http.py +2 -7
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/provenance.py +2 -1
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/base.py +21 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/noaa/coastwatch.py +1 -1
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/noaa/ghcnd.py +7 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/noaa/nexrad.py +11 -1
- usdata-0.4.0/src/usdata/pull.py +169 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/registry.py +59 -7
- usdata-0.2.1/src/usdata/data/registry.yaml +0 -50
- usdata-0.2.1/src/usdata/fetch.py +0 -54
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/cli/__init__.py +0 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/data/nexrad_sites.csv +0 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/data/places.yaml +0 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/protocols/__init__.py +0 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/protocols/s3.py +0 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/__init__.py +0 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/noaa/__init__.py +0 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/noaa/sites.py +0 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/py.typed +0 -0
- {usdata-0.2.1 → usdata-0.4.0}/src/usdata/query.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: usdata
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
|
|
5
5
|
Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
|
|
6
6
|
Author: Jake Van Slyke
|
|
@@ -28,6 +28,22 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
|
28
28
|
> data with provenance; other registry entries are stubs.
|
|
29
29
|
> See [docs/roadmap.md](docs/roadmap.md).
|
|
30
30
|
|
|
31
|
+
## Providers
|
|
32
|
+
|
|
33
|
+
<!-- registry:start -->
|
|
34
|
+
| Provider | Available | Stub | Planned | Next up (0.5) | Datasets |
|
|
35
|
+
|---|---:|---:|---:|---|---|
|
|
36
|
+
| [NOAA](docs/providers/noaa.md) | 2 | 1 | 26 | `gsom`, `gsoy`, `storm-events`, `mrms`, `goes-abi`, `hurdat2`, `ibtracs`, `climate-normals`, `coops-water-levels`, `coastwatch-sst` | `ghcn-daily`, `nexrad-level2`, _coastwatch-sst_, +26 planned |
|
|
37
|
+
| [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
|
|
38
|
+
| [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
39
|
+
| [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
|
|
40
|
+
| [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
41
|
+
| [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
|
|
42
|
+
| [USGS](docs/providers/usgs.md) | 0 | 0 | 3 | `water-daily` | +3 planned |
|
|
43
|
+
|
|
44
|
+
Available datasets are in `code`, stubs in _italics_; planned ones are counted. Each provider page has access notes and full dataset details; [docs/roadmap.md](docs/roadmap.md) lists datasets by target version.
|
|
45
|
+
<!-- registry:end -->
|
|
46
|
+
|
|
31
47
|
## Install
|
|
32
48
|
|
|
33
49
|
```sh
|
|
@@ -65,15 +81,23 @@ usdata fetch noaa:ghcn-daily -p stations=USW00013967 --start 2024-01-01 --end 20
|
|
|
65
81
|
usdata fetch noaa:nexrad-level2 --lat 35.47 --lon -97.52 \
|
|
66
82
|
--start 2024-05-06T20:00 --end 2024-05-06T23:00 # nearest radar (KTLX)
|
|
67
83
|
usdata fetch noaa:nexrad-level2 -p site=KTLX --start 2024-05-06T20:00 --end 2024-05-06T20:30 --dry-run
|
|
68
|
-
usdata pull dataset.yaml
|
|
84
|
+
usdata pull dataset.yaml # resolve, fetch, write dataset.lock.json
|
|
85
|
+
usdata verify dataset.yaml # exit 1 if any cached input drifted
|
|
69
86
|
```
|
|
70
87
|
|
|
71
88
|
Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
|
|
72
89
|
`USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
|
|
73
90
|
recording source URL, retrieval time, checksum, size, and license.
|
|
74
91
|
|
|
75
|
-
|
|
76
|
-
|
|
92
|
+
Manifest and source fields are validated strictly; unknown fields are errors.
|
|
93
|
+
Provider-specific options belong under `params`.
|
|
94
|
+
|
|
95
|
+
A manifest declares every input a project needs. `pull` resolves each source,
|
|
96
|
+
fetches it, and writes `dataset.lock.json` pinning every asset with its checksum
|
|
97
|
+
and provenance. A second `pull` restores exactly what the lockfile pins without
|
|
98
|
+
re-querying upstream, so the inputs stay reproducible even if the source
|
|
99
|
+
changes. `verify` re-hashes the cached files against the lockfile. Editing the
|
|
100
|
+
manifest after locking requires `pull --force` to re-resolve.
|
|
77
101
|
|
|
78
102
|
```yaml
|
|
79
103
|
name: tornado-environment
|
|
@@ -106,7 +130,8 @@ Releases: `just release minor` opens a version-bump PR; merging it publishes
|
|
|
106
130
|
to PyPI and creates the tag and GitHub release. See
|
|
107
131
|
[docs/versioning.md](docs/versioning.md).
|
|
108
132
|
|
|
109
|
-
See [docs/
|
|
133
|
+
See [docs/providers/](docs/providers/) for per-provider access notes,
|
|
134
|
+
[docs/architecture.md](docs/architecture.md) for how the pieces fit,
|
|
110
135
|
[docs/adr/](docs/adr/) for why, and [CONTRIBUTING.md](CONTRIBUTING.md) to add
|
|
111
136
|
a dataset.
|
|
112
137
|
|
|
@@ -7,6 +7,22 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
|
|
|
7
7
|
> data with provenance; other registry entries are stubs.
|
|
8
8
|
> See [docs/roadmap.md](docs/roadmap.md).
|
|
9
9
|
|
|
10
|
+
## Providers
|
|
11
|
+
|
|
12
|
+
<!-- registry:start -->
|
|
13
|
+
| Provider | Available | Stub | Planned | Next up (0.5) | Datasets |
|
|
14
|
+
|---|---:|---:|---:|---|---|
|
|
15
|
+
| [NOAA](docs/providers/noaa.md) | 2 | 1 | 26 | `gsom`, `gsoy`, `storm-events`, `mrms`, `goes-abi`, `hurdat2`, `ibtracs`, `climate-normals`, `coops-water-levels`, `coastwatch-sst` | `ghcn-daily`, `nexrad-level2`, _coastwatch-sst_, +26 planned |
|
|
16
|
+
| [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
|
|
17
|
+
| [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
18
|
+
| [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
|
|
19
|
+
| [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
|
|
20
|
+
| [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
|
|
21
|
+
| [USGS](docs/providers/usgs.md) | 0 | 0 | 3 | `water-daily` | +3 planned |
|
|
22
|
+
|
|
23
|
+
Available datasets are in `code`, stubs in _italics_; planned ones are counted. Each provider page has access notes and full dataset details; [docs/roadmap.md](docs/roadmap.md) lists datasets by target version.
|
|
24
|
+
<!-- registry:end -->
|
|
25
|
+
|
|
10
26
|
## Install
|
|
11
27
|
|
|
12
28
|
```sh
|
|
@@ -44,15 +60,23 @@ usdata fetch noaa:ghcn-daily -p stations=USW00013967 --start 2024-01-01 --end 20
|
|
|
44
60
|
usdata fetch noaa:nexrad-level2 --lat 35.47 --lon -97.52 \
|
|
45
61
|
--start 2024-05-06T20:00 --end 2024-05-06T23:00 # nearest radar (KTLX)
|
|
46
62
|
usdata fetch noaa:nexrad-level2 -p site=KTLX --start 2024-05-06T20:00 --end 2024-05-06T20:30 --dry-run
|
|
47
|
-
usdata pull dataset.yaml
|
|
63
|
+
usdata pull dataset.yaml # resolve, fetch, write dataset.lock.json
|
|
64
|
+
usdata verify dataset.yaml # exit 1 if any cached input drifted
|
|
48
65
|
```
|
|
49
66
|
|
|
50
67
|
Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
|
|
51
68
|
`USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
|
|
52
69
|
recording source URL, retrieval time, checksum, size, and license.
|
|
53
70
|
|
|
54
|
-
|
|
55
|
-
|
|
71
|
+
Manifest and source fields are validated strictly; unknown fields are errors.
|
|
72
|
+
Provider-specific options belong under `params`.
|
|
73
|
+
|
|
74
|
+
A manifest declares every input a project needs. `pull` resolves each source,
|
|
75
|
+
fetches it, and writes `dataset.lock.json` pinning every asset with its checksum
|
|
76
|
+
and provenance. A second `pull` restores exactly what the lockfile pins without
|
|
77
|
+
re-querying upstream, so the inputs stay reproducible even if the source
|
|
78
|
+
changes. `verify` re-hashes the cached files against the lockfile. Editing the
|
|
79
|
+
manifest after locking requires `pull --force` to re-resolve.
|
|
56
80
|
|
|
57
81
|
```yaml
|
|
58
82
|
name: tornado-environment
|
|
@@ -85,7 +109,8 @@ Releases: `just release minor` opens a version-bump PR; merging it publishes
|
|
|
85
109
|
to PyPI and creates the tag and GitHub release. See
|
|
86
110
|
[docs/versioning.md](docs/versioning.md).
|
|
87
111
|
|
|
88
|
-
See [docs/
|
|
112
|
+
See [docs/providers/](docs/providers/) for per-provider access notes,
|
|
113
|
+
[docs/architecture.md](docs/architecture.md) for how the pieces fit,
|
|
89
114
|
[docs/adr/](docs/adr/) for why, and [CONTRIBUTING.md](CONTRIBUTING.md) to add
|
|
90
115
|
a dataset.
|
|
91
116
|
|
|
@@ -11,6 +11,7 @@ except PackageNotFoundError: # running from a source tree without an install
|
|
|
11
11
|
__version__ = "0.0.0"
|
|
12
12
|
|
|
13
13
|
from usdata.models import Asset, BBox, Dataset, Provenance, Query, TimeRange
|
|
14
|
+
from usdata.pull import pull, verify
|
|
14
15
|
from usdata.query import build_query
|
|
15
16
|
from usdata.registry import DatasetNotFound, Registry, SearchResult, default_registry
|
|
16
17
|
|
|
@@ -28,13 +29,17 @@ __all__ = [
|
|
|
28
29
|
"build_query",
|
|
29
30
|
"default_registry",
|
|
30
31
|
"get",
|
|
32
|
+
"pull",
|
|
31
33
|
"search",
|
|
34
|
+
"verify",
|
|
32
35
|
]
|
|
33
36
|
|
|
34
37
|
|
|
35
|
-
def search(
|
|
38
|
+
def search(
|
|
39
|
+
text: str | None = None, *, include_planned: bool = False, **kwargs: Any
|
|
40
|
+
) -> list[SearchResult]:
|
|
36
41
|
"""Search the curated registry. Keyword arguments match ``build_query``."""
|
|
37
|
-
return default_registry().search(build_query(text, **kwargs))
|
|
42
|
+
return default_registry().search(build_query(text, **kwargs), include_planned=include_planned)
|
|
38
43
|
|
|
39
44
|
|
|
40
45
|
def get(dataset_id: str) -> Dataset:
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Atomic replacement helpers for downloads and JSON records."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterator
|
|
6
|
+
from contextlib import contextmanager
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from tempfile import NamedTemporaryFile
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@contextmanager
|
|
12
|
+
def staged_path(dest: Path) -> Iterator[Path]:
|
|
13
|
+
"""Replace dest only after successful work in a unique sibling temporary file."""
|
|
14
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
15
|
+
with NamedTemporaryFile(
|
|
16
|
+
prefix=f".{dest.name}.", suffix=".part", dir=dest.parent, delete=False
|
|
17
|
+
) as f:
|
|
18
|
+
tmp = Path(f.name)
|
|
19
|
+
try:
|
|
20
|
+
yield tmp
|
|
21
|
+
tmp.replace(dest)
|
|
22
|
+
finally:
|
|
23
|
+
tmp.unlink(missing_ok=True)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def atomic_write_text(dest: Path, text: str) -> None:
|
|
27
|
+
"""Write UTF-8 text without exposing an incomplete record to readers."""
|
|
28
|
+
with staged_path(dest) as tmp:
|
|
29
|
+
tmp.write_text(text, encoding="utf-8")
|
|
@@ -4,6 +4,7 @@ from __future__ import annotations
|
|
|
4
4
|
|
|
5
5
|
import hashlib
|
|
6
6
|
import os
|
|
7
|
+
import re
|
|
7
8
|
from pathlib import Path
|
|
8
9
|
|
|
9
10
|
from usdata.models import Asset
|
|
@@ -22,9 +23,20 @@ def cache_dir() -> Path:
|
|
|
22
23
|
|
|
23
24
|
def asset_path(asset: Asset, root: Path | None = None) -> Path:
|
|
24
25
|
"""Where an asset lives in the cache: <root>/<provider>/<dataset>/<asset id>."""
|
|
25
|
-
provider, name = asset.dataset_id.
|
|
26
|
+
provider, sep, name = asset.dataset_id.partition(":")
|
|
27
|
+
if not sep or any(
|
|
28
|
+
not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_.-]*", part) for part in (provider, name)
|
|
29
|
+
):
|
|
30
|
+
raise ValueError(f"unsafe dataset id: {asset.dataset_id!r}")
|
|
26
31
|
safe_id = asset.id.replace("/", "_")
|
|
27
|
-
|
|
32
|
+
if not safe_id or safe_id in {".", ".."} or "\\" in safe_id or "\x00" in safe_id:
|
|
33
|
+
raise ValueError(f"unsafe asset id: {asset.id!r}")
|
|
34
|
+
base = (root or cache_dir()).expanduser().resolve()
|
|
35
|
+
path = base / provider / name / safe_id
|
|
36
|
+
for candidate in (path, path.with_name(path.name + ".provenance.json")):
|
|
37
|
+
if not candidate.resolve().is_relative_to(base):
|
|
38
|
+
raise ValueError(f"cache path escapes root: {candidate}")
|
|
39
|
+
return path
|
|
28
40
|
|
|
29
41
|
|
|
30
42
|
def sha256_file(path: Path, chunk_size: int = 1 << 20) -> str:
|
|
@@ -9,10 +9,14 @@ import httpx
|
|
|
9
9
|
import typer
|
|
10
10
|
|
|
11
11
|
from usdata import __version__, build_query, default_registry
|
|
12
|
-
from usdata.fetch import
|
|
13
|
-
from usdata.
|
|
12
|
+
from usdata.fetch import ChecksumMismatch
|
|
13
|
+
from usdata.fetch import fetch as fetch_query
|
|
14
|
+
from usdata.manifest import lockfile_path
|
|
14
15
|
from usdata.providers import load_adapter
|
|
15
16
|
from usdata.providers.base import NotImplementedProvider
|
|
17
|
+
from usdata.pull import ManifestChanged, UnknownDatasets
|
|
18
|
+
from usdata.pull import pull as pull_manifest
|
|
19
|
+
from usdata.pull import verify as verify_manifest
|
|
16
20
|
from usdata.query import UnknownPlace
|
|
17
21
|
from usdata.registry import DatasetNotFound
|
|
18
22
|
|
|
@@ -49,20 +53,24 @@ def search(
|
|
|
49
53
|
state: Annotated[str | None, typer.Option(help="State name or postal code.")] = None,
|
|
50
54
|
start: Annotated[str | None, typer.Option(help="ISO date or datetime.")] = None,
|
|
51
55
|
end: Annotated[str | None, typer.Option(help="ISO date or datetime.")] = None,
|
|
56
|
+
planned: Annotated[
|
|
57
|
+
bool, typer.Option("--planned", help="Include planned datasets that have no adapter yet.")
|
|
58
|
+
] = False,
|
|
52
59
|
) -> None:
|
|
53
60
|
"""Search the curated dataset registry."""
|
|
54
61
|
try:
|
|
55
62
|
query = build_query(text, provider=provider, location=state, start=start, end=end)
|
|
56
|
-
except
|
|
57
|
-
typer.secho(
|
|
63
|
+
except ValueError as e:
|
|
64
|
+
typer.secho(str(e), err=True, fg="red")
|
|
58
65
|
raise typer.Exit(code=2) from None
|
|
59
|
-
results = default_registry().search(query)
|
|
66
|
+
results = default_registry().search(query, include_planned=planned)
|
|
60
67
|
if not results:
|
|
61
68
|
typer.echo("No datasets matched.")
|
|
62
69
|
raise typer.Exit(code=1)
|
|
63
70
|
width = max(len(r.dataset.id) for r in results)
|
|
64
71
|
for r in results:
|
|
65
|
-
|
|
72
|
+
ds = r.dataset
|
|
73
|
+
typer.echo(f"{ds.id:<{width}} {ds.status.value:<9} {ds.version_label:<12} {ds.title}")
|
|
66
74
|
|
|
67
75
|
|
|
68
76
|
@app.command()
|
|
@@ -77,6 +85,8 @@ def info(
|
|
|
77
85
|
raise typer.Exit(code=2) from None
|
|
78
86
|
typer.echo(f"{ds.id}\n {ds.title}\n")
|
|
79
87
|
typer.echo(f" {ds.description.strip()}\n")
|
|
88
|
+
typer.echo(f" status: {ds.status.value} ({ds.version_label})")
|
|
89
|
+
typer.echo(f" domain: {ds.domain}")
|
|
80
90
|
typer.echo(f" provider: {ds.provider}")
|
|
81
91
|
typer.echo(f" protocol: {ds.protocol.value}")
|
|
82
92
|
typer.echo(f" license: {ds.license or 'unknown'}")
|
|
@@ -143,21 +153,21 @@ def fetch(
|
|
|
143
153
|
variables=[v.strip() for v in variables.split(",")] if variables else None,
|
|
144
154
|
**params,
|
|
145
155
|
)
|
|
146
|
-
adapter = load_adapter(ds)
|
|
147
|
-
assets = adapter.list_assets(query)
|
|
148
156
|
if dry_run:
|
|
157
|
+
with load_adapter(ds) as adapter:
|
|
158
|
+
assets = adapter.list_assets(query)
|
|
149
159
|
for a in assets:
|
|
150
160
|
typer.echo(f"{a.id}\t{a.href}")
|
|
151
161
|
typer.echo(f"{len(assets)} asset(s) matched", err=True)
|
|
152
162
|
return
|
|
153
|
-
fetched =
|
|
163
|
+
fetched = fetch_query(ds, query, root=cache_dir, force=force)
|
|
154
164
|
except (DatasetNotFound, UnknownPlace, ValueError) as e:
|
|
155
165
|
typer.secho(str(e), err=True, fg="red")
|
|
156
166
|
raise typer.Exit(code=2) from None
|
|
157
167
|
except NotImplementedProvider as e:
|
|
158
168
|
typer.secho(str(e), err=True, fg="yellow")
|
|
159
169
|
raise typer.Exit(code=3) from None
|
|
160
|
-
except httpx.HTTPError as e:
|
|
170
|
+
except (httpx.HTTPError, ChecksumMismatch) as e:
|
|
161
171
|
typer.secho(f"request failed: {e}", err=True, fg="red")
|
|
162
172
|
raise typer.Exit(code=4) from None
|
|
163
173
|
if not fetched:
|
|
@@ -169,13 +179,51 @@ def fetch(
|
|
|
169
179
|
|
|
170
180
|
|
|
171
181
|
@app.command()
|
|
172
|
-
def pull(
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
typer.
|
|
182
|
+
def pull(
|
|
183
|
+
manifest: Annotated[Path, typer.Argument(exists=True, dir_okay=False)],
|
|
184
|
+
cache_dir: Annotated[Path | None, typer.Option(help="Override the cache directory.")] = None,
|
|
185
|
+
force: Annotated[
|
|
186
|
+
bool,
|
|
187
|
+
typer.Option(help="Ignore an existing lockfile: re-resolve every source and rewrite it."),
|
|
188
|
+
] = False,
|
|
189
|
+
) -> None:
|
|
190
|
+
"""Fetch every source in a manifest and write (or restore from) its lockfile."""
|
|
191
|
+
try:
|
|
192
|
+
result = pull_manifest(manifest, root=cache_dir, force=force)
|
|
193
|
+
except (DatasetNotFound, UnknownDatasets, ManifestChanged, UnknownPlace, ValueError) as e:
|
|
194
|
+
typer.secho(str(e), err=True, fg="red")
|
|
195
|
+
raise typer.Exit(code=2) from None
|
|
196
|
+
except NotImplementedProvider as e:
|
|
197
|
+
typer.secho(str(e), err=True, fg="yellow")
|
|
198
|
+
raise typer.Exit(code=3) from None
|
|
199
|
+
except (httpx.HTTPError, ChecksumMismatch) as e:
|
|
200
|
+
typer.secho(f"fetch failed: {e}", err=True, fg="red")
|
|
201
|
+
raise typer.Exit(code=4) from None
|
|
202
|
+
for f in result.fetched:
|
|
203
|
+
tag = "cached" if f.from_cache else "fetched"
|
|
204
|
+
typer.echo(f"{f.path}\t{tag}\t{f.provenance.size} bytes")
|
|
205
|
+
mode = "restored from" if result.from_lockfile else "wrote"
|
|
206
|
+
typer.echo(f"{len(result.fetched)} asset(s); {mode} {result.lockfile_path}", err=True)
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
@app.command()
|
|
210
|
+
def verify(
|
|
211
|
+
manifest: Annotated[Path, typer.Argument(exists=True, dir_okay=False)],
|
|
212
|
+
cache_dir: Annotated[Path | None, typer.Option(help="Override the cache directory.")] = None,
|
|
213
|
+
) -> None:
|
|
214
|
+
"""Check cached files against a manifest's lockfile. Exit 1 on any drift."""
|
|
215
|
+
lock = lockfile_path(manifest)
|
|
216
|
+
if not lock.exists():
|
|
217
|
+
typer.secho(f"no lockfile at {lock}; run pull first", err=True, fg="red")
|
|
178
218
|
raise typer.Exit(code=2)
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
219
|
+
try:
|
|
220
|
+
drift = verify_manifest(manifest, root=cache_dir)
|
|
221
|
+
except (ValueError, OSError) as e:
|
|
222
|
+
typer.secho(str(e), err=True, fg="red")
|
|
223
|
+
raise typer.Exit(code=2) from None
|
|
224
|
+
for d in drift:
|
|
225
|
+
typer.echo(f"{d.asset_id}\t{d.problem}\t{d.path}")
|
|
226
|
+
if drift:
|
|
227
|
+
typer.secho(f"{len(drift)} asset(s) drifted from {lock.name}", err=True, fg="red")
|
|
228
|
+
raise typer.Exit(code=1)
|
|
229
|
+
typer.echo(f"all assets match {lock.name}", err=True)
|