usdata 0.2.1__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {usdata-0.2.1 → usdata-0.4.0}/PKG-INFO +30 -5
  2. {usdata-0.2.1 → usdata-0.4.0}/README.md +29 -4
  3. {usdata-0.2.1 → usdata-0.4.0}/pyproject.toml +1 -1
  4. {usdata-0.2.1 → usdata-0.4.0}/pyproject.toml.orig +1 -1
  5. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/__init__.py +7 -2
  6. usdata-0.4.0/src/usdata/_files.py +29 -0
  7. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/cache.py +14 -2
  8. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/cli/app.py +67 -19
  9. usdata-0.4.0/src/usdata/data/registry.yaml +666 -0
  10. usdata-0.4.0/src/usdata/fetch.py +79 -0
  11. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/manifest.py +32 -3
  12. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/models.py +67 -4
  13. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/protocols/http.py +2 -7
  14. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/provenance.py +2 -1
  15. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/base.py +21 -0
  16. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/noaa/coastwatch.py +1 -1
  17. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/noaa/ghcnd.py +7 -0
  18. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/noaa/nexrad.py +11 -1
  19. usdata-0.4.0/src/usdata/pull.py +169 -0
  20. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/registry.py +59 -7
  21. usdata-0.2.1/src/usdata/data/registry.yaml +0 -50
  22. usdata-0.2.1/src/usdata/fetch.py +0 -54
  23. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/cli/__init__.py +0 -0
  24. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/data/nexrad_sites.csv +0 -0
  25. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/data/places.yaml +0 -0
  26. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/protocols/__init__.py +0 -0
  27. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/protocols/s3.py +0 -0
  28. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/__init__.py +0 -0
  29. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/noaa/__init__.py +0 -0
  30. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/providers/noaa/sites.py +0 -0
  31. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/py.typed +0 -0
  32. {usdata-0.2.1 → usdata-0.4.0}/src/usdata/query.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: usdata
3
- Version: 0.2.1
3
+ Version: 0.4.0
4
4
  Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
5
5
  Keywords: noaa,usgs,nasa,open-data,scientific-data,provenance
6
6
  Author: Jake Van Slyke
@@ -28,6 +28,22 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
28
28
  > data with provenance; other registry entries are stubs.
29
29
  > See [docs/roadmap.md](docs/roadmap.md).
30
30
 
31
+ ## Providers
32
+
33
+ <!-- registry:start -->
34
+ | Provider | Available | Stub | Planned | Next up (0.5) | Datasets |
35
+ |---|---:|---:|---:|---|---|
36
+ | [NOAA](docs/providers/noaa.md) | 2 | 1 | 26 | `gsom`, `gsoy`, `storm-events`, `mrms`, `goes-abi`, `hurdat2`, `ibtracs`, `climate-normals`, `coops-water-levels`, `coastwatch-sst` | `ghcn-daily`, `nexrad-level2`, _coastwatch-sst_, +26 planned |
37
+ | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
38
+ | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
39
+ | [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
40
+ | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
41
+ | [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
42
+ | [USGS](docs/providers/usgs.md) | 0 | 0 | 3 | `water-daily` | +3 planned |
43
+
44
+ Available datasets are in `code`, stubs in _italics_; planned ones are counted. Each provider page has access notes and full dataset details; [docs/roadmap.md](docs/roadmap.md) lists datasets by target version.
45
+ <!-- registry:end -->
46
+
31
47
  ## Install
32
48
 
33
49
  ```sh
@@ -65,15 +81,23 @@ usdata fetch noaa:ghcn-daily -p stations=USW00013967 --start 2024-01-01 --end 20
65
81
  usdata fetch noaa:nexrad-level2 --lat 35.47 --lon -97.52 \
66
82
  --start 2024-05-06T20:00 --end 2024-05-06T23:00 # nearest radar (KTLX)
67
83
  usdata fetch noaa:nexrad-level2 -p site=KTLX --start 2024-05-06T20:00 --end 2024-05-06T20:30 --dry-run
68
- usdata pull dataset.yaml
84
+ usdata pull dataset.yaml # resolve, fetch, write dataset.lock.json
85
+ usdata verify dataset.yaml # exit 1 if any cached input drifted
69
86
  ```
70
87
 
71
88
  Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
72
89
  `USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
73
90
  recording source URL, retrieval time, checksum, size, and license.
74
91
 
75
- A manifest declares every input a project needs; `pull` fetches them and writes
76
- a lockfile with checksums and provenance so the inputs can be reproduced:
92
+ Manifest and source fields are validated strictly; unknown fields are errors.
93
+ Provider-specific options belong under `params`.
94
+
95
+ A manifest declares every input a project needs. `pull` resolves each source,
96
+ fetches it, and writes `dataset.lock.json` pinning every asset with its checksum
97
+ and provenance. A second `pull` restores exactly what the lockfile pins without
98
+ re-querying upstream, so the inputs stay reproducible even if the source
99
+ changes. `verify` re-hashes the cached files against the lockfile. Editing the
100
+ manifest after locking requires `pull --force` to re-resolve.
77
101
 
78
102
  ```yaml
79
103
  name: tornado-environment
@@ -106,7 +130,8 @@ Releases: `just release minor` opens a version-bump PR; merging it publishes
106
130
  to PyPI and creates the tag and GitHub release. See
107
131
  [docs/versioning.md](docs/versioning.md).
108
132
 
109
- See [docs/architecture.md](docs/architecture.md) for how the pieces fit,
133
+ See [docs/providers/](docs/providers/) for per-provider access notes,
134
+ [docs/architecture.md](docs/architecture.md) for how the pieces fit,
110
135
  [docs/adr/](docs/adr/) for why, and [CONTRIBUTING.md](CONTRIBUTING.md) to add
111
136
  a dataset.
112
137
 
@@ -7,6 +7,22 @@ provenance of U.S. public scientific data (NOAA, USGS, NASA, and more).
7
7
  > data with provenance; other registry entries are stubs.
8
8
  > See [docs/roadmap.md](docs/roadmap.md).
9
9
 
10
+ ## Providers
11
+
12
+ <!-- registry:start -->
13
+ | Provider | Available | Stub | Planned | Next up (0.5) | Datasets |
14
+ |---|---:|---:|---:|---|---|
15
+ | [NOAA](docs/providers/noaa.md) | 2 | 1 | 26 | `gsom`, `gsoy`, `storm-events`, `mrms`, `goes-abi`, `hurdat2`, `ibtracs`, `climate-normals`, `coops-water-levels`, `coastwatch-sst` | `ghcn-daily`, `nexrad-level2`, _coastwatch-sst_, +26 planned |
16
+ | [Census Bureau](docs/providers/census.md) | 0 | 0 | 1 | — | +1 planned |
17
+ | [EPA](docs/providers/epa.md) | 0 | 0 | 1 | — | +1 planned |
18
+ | [FEMA](docs/providers/fema.md) | 0 | 0 | 1 | — | +1 planned |
19
+ | [NASA](docs/providers/nasa.md) | 0 | 0 | 1 | — | +1 planned |
20
+ | [USDA](docs/providers/usda.md) | 0 | 0 | 1 | — | +1 planned |
21
+ | [USGS](docs/providers/usgs.md) | 0 | 0 | 3 | `water-daily` | +3 planned |
22
+
23
+ Available datasets are in `code`, stubs in _italics_; planned ones are counted. Each provider page has access notes and full dataset details; [docs/roadmap.md](docs/roadmap.md) lists datasets by target version.
24
+ <!-- registry:end -->
25
+
10
26
  ## Install
11
27
 
12
28
  ```sh
@@ -44,15 +60,23 @@ usdata fetch noaa:ghcn-daily -p stations=USW00013967 --start 2024-01-01 --end 20
44
60
  usdata fetch noaa:nexrad-level2 --lat 35.47 --lon -97.52 \
45
61
  --start 2024-05-06T20:00 --end 2024-05-06T23:00 # nearest radar (KTLX)
46
62
  usdata fetch noaa:nexrad-level2 -p site=KTLX --start 2024-05-06T20:00 --end 2024-05-06T20:30 --dry-run
47
- usdata pull dataset.yaml
63
+ usdata pull dataset.yaml # resolve, fetch, write dataset.lock.json
64
+ usdata verify dataset.yaml # exit 1 if any cached input drifted
48
65
  ```
49
66
 
50
67
  Fetched files land in `~/.cache/usdata/<provider>/<dataset>/` (override with
51
68
  `USDATA_CACHE_DIR` or `--cache-dir`), each with a `.provenance.json` sidecar
52
69
  recording source URL, retrieval time, checksum, size, and license.
53
70
 
54
- A manifest declares every input a project needs; `pull` fetches them and writes
55
- a lockfile with checksums and provenance so the inputs can be reproduced:
71
+ Manifest and source fields are validated strictly; unknown fields are errors.
72
+ Provider-specific options belong under `params`.
73
+
74
+ A manifest declares every input a project needs. `pull` resolves each source,
75
+ fetches it, and writes `dataset.lock.json` pinning every asset with its checksum
76
+ and provenance. A second `pull` restores exactly what the lockfile pins without
77
+ re-querying upstream, so the inputs stay reproducible even if the source
78
+ changes. `verify` re-hashes the cached files against the lockfile. Editing the
79
+ manifest after locking requires `pull --force` to re-resolve.
56
80
 
57
81
  ```yaml
58
82
  name: tornado-environment
@@ -85,7 +109,8 @@ Releases: `just release minor` opens a version-bump PR; merging it publishes
85
109
  to PyPI and creates the tag and GitHub release. See
86
110
  [docs/versioning.md](docs/versioning.md).
87
111
 
88
- See [docs/architecture.md](docs/architecture.md) for how the pieces fit,
112
+ See [docs/providers/](docs/providers/) for per-provider access notes,
113
+ [docs/architecture.md](docs/architecture.md) for how the pieces fit,
89
114
  [docs/adr/](docs/adr/) for why, and [CONTRIBUTING.md](CONTRIBUTING.md) to add
90
115
  a dataset.
91
116
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.2.1"
3
+ version = "0.4.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.2.1"
3
+ version = "0.4.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -11,6 +11,7 @@ except PackageNotFoundError: # running from a source tree without an install
11
11
  __version__ = "0.0.0"
12
12
 
13
13
  from usdata.models import Asset, BBox, Dataset, Provenance, Query, TimeRange
14
+ from usdata.pull import pull, verify
14
15
  from usdata.query import build_query
15
16
  from usdata.registry import DatasetNotFound, Registry, SearchResult, default_registry
16
17
 
@@ -28,13 +29,17 @@ __all__ = [
28
29
  "build_query",
29
30
  "default_registry",
30
31
  "get",
32
+ "pull",
31
33
  "search",
34
+ "verify",
32
35
  ]
33
36
 
34
37
 
35
- def search(text: str | None = None, **kwargs: Any) -> list[SearchResult]:
38
+ def search(
39
+ text: str | None = None, *, include_planned: bool = False, **kwargs: Any
40
+ ) -> list[SearchResult]:
36
41
  """Search the curated registry. Keyword arguments match ``build_query``."""
37
- return default_registry().search(build_query(text, **kwargs))
42
+ return default_registry().search(build_query(text, **kwargs), include_planned=include_planned)
38
43
 
39
44
 
40
45
  def get(dataset_id: str) -> Dataset:
@@ -0,0 +1,29 @@
1
+ """Atomic replacement helpers for downloads and JSON records."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Iterator
6
+ from contextlib import contextmanager
7
+ from pathlib import Path
8
+ from tempfile import NamedTemporaryFile
9
+
10
+
11
+ @contextmanager
12
+ def staged_path(dest: Path) -> Iterator[Path]:
13
+ """Replace dest only after successful work in a unique sibling temporary file."""
14
+ dest.parent.mkdir(parents=True, exist_ok=True)
15
+ with NamedTemporaryFile(
16
+ prefix=f".{dest.name}.", suffix=".part", dir=dest.parent, delete=False
17
+ ) as f:
18
+ tmp = Path(f.name)
19
+ try:
20
+ yield tmp
21
+ tmp.replace(dest)
22
+ finally:
23
+ tmp.unlink(missing_ok=True)
24
+
25
+
26
+ def atomic_write_text(dest: Path, text: str) -> None:
27
+ """Write UTF-8 text without exposing an incomplete record to readers."""
28
+ with staged_path(dest) as tmp:
29
+ tmp.write_text(text, encoding="utf-8")
@@ -4,6 +4,7 @@ from __future__ import annotations
4
4
 
5
5
  import hashlib
6
6
  import os
7
+ import re
7
8
  from pathlib import Path
8
9
 
9
10
  from usdata.models import Asset
@@ -22,9 +23,20 @@ def cache_dir() -> Path:
22
23
 
23
24
  def asset_path(asset: Asset, root: Path | None = None) -> Path:
24
25
  """Where an asset lives in the cache: <root>/<provider>/<dataset>/<asset id>."""
25
- provider, name = asset.dataset_id.split(":", 1)
26
+ provider, sep, name = asset.dataset_id.partition(":")
27
+ if not sep or any(
28
+ not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_.-]*", part) for part in (provider, name)
29
+ ):
30
+ raise ValueError(f"unsafe dataset id: {asset.dataset_id!r}")
26
31
  safe_id = asset.id.replace("/", "_")
27
- return (root or cache_dir()) / provider / name / safe_id
32
+ if not safe_id or safe_id in {".", ".."} or "\\" in safe_id or "\x00" in safe_id:
33
+ raise ValueError(f"unsafe asset id: {asset.id!r}")
34
+ base = (root or cache_dir()).expanduser().resolve()
35
+ path = base / provider / name / safe_id
36
+ for candidate in (path, path.with_name(path.name + ".provenance.json")):
37
+ if not candidate.resolve().is_relative_to(base):
38
+ raise ValueError(f"cache path escapes root: {candidate}")
39
+ return path
28
40
 
29
41
 
30
42
  def sha256_file(path: Path, chunk_size: int = 1 << 20) -> str:
@@ -9,10 +9,14 @@ import httpx
9
9
  import typer
10
10
 
11
11
  from usdata import __version__, build_query, default_registry
12
- from usdata.fetch import fetch_asset
13
- from usdata.manifest import Manifest
12
+ from usdata.fetch import ChecksumMismatch
13
+ from usdata.fetch import fetch as fetch_query
14
+ from usdata.manifest import lockfile_path
14
15
  from usdata.providers import load_adapter
15
16
  from usdata.providers.base import NotImplementedProvider
17
+ from usdata.pull import ManifestChanged, UnknownDatasets
18
+ from usdata.pull import pull as pull_manifest
19
+ from usdata.pull import verify as verify_manifest
16
20
  from usdata.query import UnknownPlace
17
21
  from usdata.registry import DatasetNotFound
18
22
 
@@ -49,20 +53,24 @@ def search(
49
53
  state: Annotated[str | None, typer.Option(help="State name or postal code.")] = None,
50
54
  start: Annotated[str | None, typer.Option(help="ISO date or datetime.")] = None,
51
55
  end: Annotated[str | None, typer.Option(help="ISO date or datetime.")] = None,
56
+ planned: Annotated[
57
+ bool, typer.Option("--planned", help="Include planned datasets that have no adapter yet.")
58
+ ] = False,
52
59
  ) -> None:
53
60
  """Search the curated dataset registry."""
54
61
  try:
55
62
  query = build_query(text, provider=provider, location=state, start=start, end=end)
56
- except UnknownPlace as e:
57
- typer.secho(f"Unknown place: {e}", err=True, fg="red")
63
+ except ValueError as e:
64
+ typer.secho(str(e), err=True, fg="red")
58
65
  raise typer.Exit(code=2) from None
59
- results = default_registry().search(query)
66
+ results = default_registry().search(query, include_planned=planned)
60
67
  if not results:
61
68
  typer.echo("No datasets matched.")
62
69
  raise typer.Exit(code=1)
63
70
  width = max(len(r.dataset.id) for r in results)
64
71
  for r in results:
65
- typer.echo(f"{r.dataset.id:<{width}} {r.dataset.title}")
72
+ ds = r.dataset
73
+ typer.echo(f"{ds.id:<{width}} {ds.status.value:<9} {ds.version_label:<12} {ds.title}")
66
74
 
67
75
 
68
76
  @app.command()
@@ -77,6 +85,8 @@ def info(
77
85
  raise typer.Exit(code=2) from None
78
86
  typer.echo(f"{ds.id}\n {ds.title}\n")
79
87
  typer.echo(f" {ds.description.strip()}\n")
88
+ typer.echo(f" status: {ds.status.value} ({ds.version_label})")
89
+ typer.echo(f" domain: {ds.domain}")
80
90
  typer.echo(f" provider: {ds.provider}")
81
91
  typer.echo(f" protocol: {ds.protocol.value}")
82
92
  typer.echo(f" license: {ds.license or 'unknown'}")
@@ -143,21 +153,21 @@ def fetch(
143
153
  variables=[v.strip() for v in variables.split(",")] if variables else None,
144
154
  **params,
145
155
  )
146
- adapter = load_adapter(ds)
147
- assets = adapter.list_assets(query)
148
156
  if dry_run:
157
+ with load_adapter(ds) as adapter:
158
+ assets = adapter.list_assets(query)
149
159
  for a in assets:
150
160
  typer.echo(f"{a.id}\t{a.href}")
151
161
  typer.echo(f"{len(assets)} asset(s) matched", err=True)
152
162
  return
153
- fetched = [fetch_asset(ds, a, root=cache_dir, force=force) for a in assets]
163
+ fetched = fetch_query(ds, query, root=cache_dir, force=force)
154
164
  except (DatasetNotFound, UnknownPlace, ValueError) as e:
155
165
  typer.secho(str(e), err=True, fg="red")
156
166
  raise typer.Exit(code=2) from None
157
167
  except NotImplementedProvider as e:
158
168
  typer.secho(str(e), err=True, fg="yellow")
159
169
  raise typer.Exit(code=3) from None
160
- except httpx.HTTPError as e:
170
+ except (httpx.HTTPError, ChecksumMismatch) as e:
161
171
  typer.secho(f"request failed: {e}", err=True, fg="red")
162
172
  raise typer.Exit(code=4) from None
163
173
  if not fetched:
@@ -169,13 +179,51 @@ def fetch(
169
179
 
170
180
 
171
181
  @app.command()
172
- def pull(manifest: Annotated[Path, typer.Argument(exists=True, dir_okay=False)]) -> None:
173
- """Fetch every source in a manifest and write a lockfile."""
174
- m = Manifest.load(manifest)
175
- missing = m.validate_against()
176
- if missing:
177
- typer.secho(f"Unknown datasets in manifest: {', '.join(missing)}", err=True, fg="red")
182
+ def pull(
183
+ manifest: Annotated[Path, typer.Argument(exists=True, dir_okay=False)],
184
+ cache_dir: Annotated[Path | None, typer.Option(help="Override the cache directory.")] = None,
185
+ force: Annotated[
186
+ bool,
187
+ typer.Option(help="Ignore an existing lockfile: re-resolve every source and rewrite it."),
188
+ ] = False,
189
+ ) -> None:
190
+ """Fetch every source in a manifest and write (or restore from) its lockfile."""
191
+ try:
192
+ result = pull_manifest(manifest, root=cache_dir, force=force)
193
+ except (DatasetNotFound, UnknownDatasets, ManifestChanged, UnknownPlace, ValueError) as e:
194
+ typer.secho(str(e), err=True, fg="red")
195
+ raise typer.Exit(code=2) from None
196
+ except NotImplementedProvider as e:
197
+ typer.secho(str(e), err=True, fg="yellow")
198
+ raise typer.Exit(code=3) from None
199
+ except (httpx.HTTPError, ChecksumMismatch) as e:
200
+ typer.secho(f"fetch failed: {e}", err=True, fg="red")
201
+ raise typer.Exit(code=4) from None
202
+ for f in result.fetched:
203
+ tag = "cached" if f.from_cache else "fetched"
204
+ typer.echo(f"{f.path}\t{tag}\t{f.provenance.size} bytes")
205
+ mode = "restored from" if result.from_lockfile else "wrote"
206
+ typer.echo(f"{len(result.fetched)} asset(s); {mode} {result.lockfile_path}", err=True)
207
+
208
+
209
+ @app.command()
210
+ def verify(
211
+ manifest: Annotated[Path, typer.Argument(exists=True, dir_okay=False)],
212
+ cache_dir: Annotated[Path | None, typer.Option(help="Override the cache directory.")] = None,
213
+ ) -> None:
214
+ """Check cached files against a manifest's lockfile. Exit 1 on any drift."""
215
+ lock = lockfile_path(manifest)
216
+ if not lock.exists():
217
+ typer.secho(f"no lockfile at {lock}; run pull first", err=True, fg="red")
178
218
  raise typer.Exit(code=2)
179
- typer.echo(f"{m.name} v{m.version}: {len(m.sources)} source(s) validated")
180
- typer.secho("pull is not implemented yet; no data was fetched.", err=True, fg="yellow")
181
- raise typer.Exit(code=3)
219
+ try:
220
+ drift = verify_manifest(manifest, root=cache_dir)
221
+ except (ValueError, OSError) as e:
222
+ typer.secho(str(e), err=True, fg="red")
223
+ raise typer.Exit(code=2) from None
224
+ for d in drift:
225
+ typer.echo(f"{d.asset_id}\t{d.problem}\t{d.path}")
226
+ if drift:
227
+ typer.secho(f"{len(drift)} asset(s) drifted from {lock.name}", err=True, fg="red")
228
+ raise typer.Exit(code=1)
229
+ typer.echo(f"all assets match {lock.name}", err=True)