usdata 0.24.0__tar.gz → 0.26.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {usdata-0.24.0 → usdata-0.26.0}/PKG-INFO +2 -2
- {usdata-0.24.0 → usdata-0.26.0}/README.md +1 -1
- {usdata-0.24.0 → usdata-0.26.0}/pyproject.toml +1 -1
- {usdata-0.24.0 → usdata-0.26.0}/pyproject.toml.orig +1 -1
- usdata-0.26.0/src/usdata/_aqs.py +51 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_fetch.py +1 -1
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_hurdat2.py +8 -1
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/app.py +15 -5
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/data/registry.yaml +82 -18
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/doctor.py +38 -5
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/models.py +45 -1
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/provenance.py +11 -2
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/__init__.py +14 -1
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/base.py +80 -4
- usdata-0.26.0/src/usdata/providers/credentials.py +84 -0
- usdata-0.26.0/src/usdata/providers/epa/__init__.py +1 -0
- usdata-0.26.0/src/usdata/providers/epa/aqs.py +308 -0
- usdata-0.26.0/src/usdata/providers/http.py +139 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/goes.py +33 -19
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/hurdat2.py +32 -2
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/pull.py +43 -9
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/readers.py +22 -3
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/testing.py +199 -3
- usdata-0.24.0/src/usdata/providers/http.py +0 -44
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/__init__.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/__main__.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_files.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_grib.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_netcdf.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_progress.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_radar.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cache.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cache_ops.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cite.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/__init__.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/cache.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/cite.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/doctor.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/inspect.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/progress.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/data/nexrad_sites.csv +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/data/places.csv +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/data/places.sources.json +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/inspect.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/manifest.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/mirror.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/protocols/__init__.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/protocols/erddap.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/protocols/http.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/protocols/listing.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/protocols/s3.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/fema/__init__.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/fema/declarations.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/__init__.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/coops.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/gfs.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/ghcnd.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/glm.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/grib_index.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/gsom.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/gsoy.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/hrrr.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/ibtracs.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/lcd.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/mrms.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/nbm.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/nexrad.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/nexrad_level3.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/normals.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/nws_vtec.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/rap.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/sites.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/spc.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/storm_events.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/params.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/usgs/__init__.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/usgs/daily.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/usgs/earthquakes.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/py.typed +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/query.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/registry.py +0 -0
- {usdata-0.24.0 → usdata-0.26.0}/src/usdata/selection.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: usdata
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.26.0
|
|
4
4
|
Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
|
|
5
5
|
Keywords: noaa,usgs,open-data,scientific-data,provenance,reproducible-research,weather,climate,meteorology
|
|
6
6
|
Author: Jake Van Slyke
|
|
@@ -112,7 +112,7 @@ breaking changes.
|
|
|
112
112
|
| [docs.usdata.dev](https://docs.usdata.dev/) | How to use it: [install](https://docs.usdata.dev/install/), [getting started](https://docs.usdata.dev/getting-started/), guides, dataset notes, and reference |
|
|
113
113
|
| [Severe-weather case study](https://usdata.dev/examples/severe-weather-case-study/) | One tornado, six sources, one manifest and lockfile, ending in a citation |
|
|
114
114
|
|
|
115
|
-
Twenty-
|
|
115
|
+
Twenty-seven datasets are available today and nineteen more are planned, grouped
|
|
116
116
|
by agency and product family in the [catalog](docs/providers/README.md).
|
|
117
117
|
|
|
118
118
|
## How this compares
|
|
@@ -67,7 +67,7 @@ breaking changes.
|
|
|
67
67
|
| [docs.usdata.dev](https://docs.usdata.dev/) | How to use it: [install](https://docs.usdata.dev/install/), [getting started](https://docs.usdata.dev/getting-started/), guides, dataset notes, and reference |
|
|
68
68
|
| [Severe-weather case study](https://usdata.dev/examples/severe-weather-case-study/) | One tornado, six sources, one manifest and lockfile, ending in a citation |
|
|
69
69
|
|
|
70
|
-
Twenty-
|
|
70
|
+
Twenty-seven datasets are available today and nineteen more are planned, grouped
|
|
71
71
|
by agency and product family in the [catalog](docs/providers/README.md).
|
|
72
72
|
|
|
73
73
|
## How this compares
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Local reading of AQS daily-summary JSON into a tidy table, behind pandas.
|
|
2
|
+
|
|
3
|
+
A fetched ``epa:aqs-daily`` file is the canonical form the adapter writes: the
|
|
4
|
+
service's JSON with the echoed request left out of its header and the rows in a
|
|
5
|
+
fixed order (ADR 0039). Each element of ``Data`` is one monitor's summary for one
|
|
6
|
+
local calendar day under one pollutant standard, so the table is one row per
|
|
7
|
+
element, with the columns as the service names them.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
from importlib import import_module
|
|
14
|
+
from typing import TYPE_CHECKING, Any
|
|
15
|
+
|
|
16
|
+
from usdata.readers import MissingReaderDependency
|
|
17
|
+
|
|
18
|
+
if TYPE_CHECKING:
|
|
19
|
+
from usdata._fetch import FetchedAsset
|
|
20
|
+
|
|
21
|
+
DATE_COLUMNS = ("date_local", "date_of_last_change")
|
|
22
|
+
"""Calendar dates as the service writes them, ``YYYY-MM-DD``, parsed without a timezone.
|
|
23
|
+
|
|
24
|
+
``date_local`` is a day in the monitor's local standard time, not a UTC instant,
|
|
25
|
+
so it is left naive rather than given a zone it does not have.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def open_aqs(fetched: FetchedAsset) -> Any:
|
|
30
|
+
"""Parse a fetched AQS daily-summary file into a pandas DataFrame, one row per summary."""
|
|
31
|
+
try:
|
|
32
|
+
pandas = import_module("pandas")
|
|
33
|
+
except ModuleNotFoundError as error:
|
|
34
|
+
if error.name != "pandas":
|
|
35
|
+
raise
|
|
36
|
+
raise MissingReaderDependency(
|
|
37
|
+
'AQS reading requires pandas; install it with: pip install "usdata[pandas]" '
|
|
38
|
+
'(or uv add "usdata[pandas]")'
|
|
39
|
+
) from error
|
|
40
|
+
# Reading a fetched asset is strictly local; the cached file is never rewritten.
|
|
41
|
+
body = json.loads(fetched.path.read_text(encoding="utf-8"))
|
|
42
|
+
frame = pandas.DataFrame(body.get("Data") or [])
|
|
43
|
+
for column in DATE_COLUMNS:
|
|
44
|
+
if column in frame:
|
|
45
|
+
frame[column] = pandas.to_datetime(frame[column], format="%Y-%m-%d")
|
|
46
|
+
frame.attrs["usdata"] = {
|
|
47
|
+
"asset_id": fetched.asset.id,
|
|
48
|
+
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
49
|
+
"header": body.get("Header", []),
|
|
50
|
+
}
|
|
51
|
+
return frame
|
|
@@ -144,7 +144,7 @@ def _fetch_asset(
|
|
|
144
144
|
adapter.fetch(asset, tmp)
|
|
145
145
|
else:
|
|
146
146
|
adapter.fetch_partial(asset, tmp, partial)
|
|
147
|
-
prov = provenance.record(dataset, asset, tmp, partial)
|
|
147
|
+
prov = provenance.record(dataset, asset, tmp, partial, adapter.transformations)
|
|
148
148
|
if asset.checksum and prov.checksum != asset.checksum:
|
|
149
149
|
raise ChecksumMismatch(f"{asset.id}: expected {asset.checksum}, got {prov.checksum}")
|
|
150
150
|
# A crash between replacements leaves a detectable mismatch, never a trusted partial file.
|
|
@@ -160,7 +160,14 @@ def open_hurdat2(fetched: FetchedAsset) -> Any:
|
|
|
160
160
|
'(or uv add "usdata[pandas]")'
|
|
161
161
|
) from error
|
|
162
162
|
# Reading a fetched asset is strictly local; the source file is never rewritten.
|
|
163
|
-
|
|
163
|
+
try:
|
|
164
|
+
columns = parse(fetched.path.read_text(encoding="utf-8"))
|
|
165
|
+
except Hurdat2FormatError as error:
|
|
166
|
+
# NHC typos come and go between revisions, so name the way to pick another.
|
|
167
|
+
raise Hurdat2FormatError(
|
|
168
|
+
f"{fetched.asset.id}: {error}; an upstream typo is usually absent from the "
|
|
169
|
+
"neighbouring revisions, so select one with the 'revision' parameter"
|
|
170
|
+
) from error
|
|
164
171
|
data: dict[str, Any] = {
|
|
165
172
|
name: pandas.array(columns[name], dtype="string") for name in TEXT_COLUMNS
|
|
166
173
|
}
|
|
@@ -21,7 +21,7 @@ from usdata.cli.inspect import inspect
|
|
|
21
21
|
from usdata.cli.progress import progress
|
|
22
22
|
from usdata.manifest import lockfile_path
|
|
23
23
|
from usdata.models import READER_EXTRAS_TEXT, Dataset, Status, describe_duration
|
|
24
|
-
from usdata.providers import load_adapter
|
|
24
|
+
from usdata.providers import adapter_class, load_adapter
|
|
25
25
|
from usdata.providers.base import NotImplementedProvider
|
|
26
26
|
from usdata.pull import EmptySource, ManifestChanged, Plan, UnknownDatasets, UpstreamChanged
|
|
27
27
|
from usdata.pull import plan as plan_manifest
|
|
@@ -260,8 +260,7 @@ def info(
|
|
|
260
260
|
)
|
|
261
261
|
if ds.status is not Status.AVAILABLE:
|
|
262
262
|
return # Planned entries have no adapter or usage metadata to show.
|
|
263
|
-
|
|
264
|
-
declared = dict(adapter.accepted_params)
|
|
263
|
+
declared = dict(adapter_class(ds).accepted_params)
|
|
265
264
|
if declared:
|
|
266
265
|
typer.echo(" params:")
|
|
267
266
|
width = max(len(name) for name in declared)
|
|
@@ -285,6 +284,9 @@ def _echo_usage(ds: Dataset) -> None:
|
|
|
285
284
|
typer.echo(f" selection: {ds.selection}")
|
|
286
285
|
if ds.inputs:
|
|
287
286
|
typer.echo(f" inputs: {ds.inputs}")
|
|
287
|
+
if ds.credentials:
|
|
288
|
+
typer.echo(f" credentials: {', '.join(ds.credentials.variables)} (environment)")
|
|
289
|
+
typer.echo(f" key: {ds.credentials.signup}")
|
|
288
290
|
if ds.examples:
|
|
289
291
|
typer.echo(f" examples: {', '.join(ds.examples)}")
|
|
290
292
|
|
|
@@ -528,13 +530,21 @@ def pull(
|
|
|
528
530
|
else:
|
|
529
531
|
mode = "restored from" if result.from_lockfile else "wrote"
|
|
530
532
|
typer.echo(f"{len(result.fetched)} asset(s); {mode} {result.lockfile_path}", err=True)
|
|
531
|
-
|
|
533
|
+
changed = len(result.mirrored) - len(result.unchecked)
|
|
534
|
+
if changed:
|
|
532
535
|
typer.secho(
|
|
533
|
-
f"{
|
|
536
|
+
f"{changed} asset(s) changed upstream and were restored from the "
|
|
534
537
|
"mirror; the lockfile still pins the source. Pass --update to accept new bytes.",
|
|
535
538
|
err=True,
|
|
536
539
|
fg="yellow",
|
|
537
540
|
)
|
|
541
|
+
if result.unchecked:
|
|
542
|
+
typer.secho(
|
|
543
|
+
f"{len(result.unchecked)} asset(s) were restored from the mirror without asking "
|
|
544
|
+
"their source, whose credentials are not set; upstream was not checked for changes.",
|
|
545
|
+
err=True,
|
|
546
|
+
fg="yellow",
|
|
547
|
+
)
|
|
538
548
|
|
|
539
549
|
|
|
540
550
|
@app.command()
|
|
@@ -209,30 +209,31 @@ datasets:
|
|
|
209
209
|
system: noaa:goes-r
|
|
210
210
|
domain: weather-satellites
|
|
211
211
|
since: "0.8"
|
|
212
|
-
title: GOES-R ABI
|
|
212
|
+
title: GOES-R ABI Cloud and Moisture Imagery
|
|
213
213
|
description: >-
|
|
214
|
-
Single-channel CONUS
|
|
214
|
+
Single-channel CONUS (ABI-L2-CMIPC) and mesoscale (ABI-L2-CMIPM) imagery from
|
|
215
215
|
GOES-16, 17, 18, and 19 in anonymous NOAA S3 buckets. Select an explicit
|
|
216
|
-
satellite, channel,
|
|
216
|
+
satellite, channel, scan-start interval, and M1 or M2 for mesoscale; each asset is a complete
|
|
217
217
|
NetCDF scene with no geographic or variable subsetting.
|
|
218
|
-
keywords: [satellite, imagery, goes, abi, clouds, infrared, reflectance, netcdf, conus]
|
|
218
|
+
keywords: [satellite, imagery, goes, abi, clouds, infrared, reflectance, netcdf, conus, mesoscale]
|
|
219
219
|
protocol: s3
|
|
220
220
|
homepage: https://registry.opendata.aws/noaa-goes/
|
|
221
221
|
license: US Government Work (public domain)
|
|
222
222
|
temporal_extent: { start: "2017-02-28T00:00:00Z" }
|
|
223
223
|
capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: false }
|
|
224
|
-
summary: GOES CONUS imagery
|
|
224
|
+
summary: GOES CONUS and mesoscale imagery
|
|
225
225
|
formats: [NetCDF4]
|
|
226
|
-
selection: Whole single-channel
|
|
227
|
-
inputs: Satellite, channel, and
|
|
226
|
+
selection: Whole single-channel scenes by inclusive UTC scan-start time and explicit mesoscale sector
|
|
227
|
+
inputs: Satellite, channel, both timestamps; product and sector for mesoscale
|
|
228
228
|
reader: netcdf
|
|
229
229
|
guide: docs/providers/noaa-goes.md
|
|
230
230
|
examples:
|
|
231
231
|
- examples/goes-imagery/example.ipynb
|
|
232
|
+
- examples/goes-mesoscale/example.ipynb
|
|
232
233
|
- examples/event-context/example.ipynb
|
|
233
234
|
resolution:
|
|
234
235
|
spatial: "0.5 km to 2 km at nadir, by ABI band"
|
|
235
|
-
temporal: "
|
|
236
|
+
temporal: "CONUS every 5 minutes; two mesoscale sectors every 60 seconds or one every 30 seconds"
|
|
236
237
|
update_frequency: "New data is added as soon as it's available"
|
|
237
238
|
citation: >-
|
|
238
239
|
NOAA Geostationary Operational Environmental Satellites (GOES) 16, 17, 18 & 19 was
|
|
@@ -423,7 +424,8 @@ datasets:
|
|
|
423
424
|
intensity, pressure, and wind radii for Atlantic (since 1851) and
|
|
424
425
|
northeast/north-central Pacific (since 1949) tropical cyclones. One
|
|
425
426
|
fixed-format text file per basin, revised after each season; a basin
|
|
426
|
-
parameter picks the file and the newest revision wins
|
|
427
|
+
parameter picks the file and the newest revision wins unless a revision
|
|
428
|
+
date names another. There is no
|
|
427
429
|
query interface, so dates and geographic filters are rejected and the
|
|
428
430
|
local reader turns the whole file into one row per track point.
|
|
429
431
|
keywords: [hurricane, tropical cyclone, best track, nhc, atlantic, pacific]
|
|
@@ -434,8 +436,8 @@ datasets:
|
|
|
434
436
|
capabilities: { spatial_subset: false, temporal_subset: false, variable_subset: false }
|
|
435
437
|
summary: Tropical cyclone best tracks
|
|
436
438
|
formats: [HURDAT2 fixed-format text]
|
|
437
|
-
selection: The newest revision of one whole basin file; filter track points locally
|
|
438
|
-
inputs: Optional basin (atlantic or pacific); no dates or geographic filters
|
|
439
|
+
selection: The newest or a named revision of one whole basin file; filter track points locally
|
|
440
|
+
inputs: Optional basin (atlantic or pacific) and revision date; no dates or geographic filters
|
|
439
441
|
reader: pandas
|
|
440
442
|
guide: docs/providers/noaa-hurdat2.md
|
|
441
443
|
examples:
|
|
@@ -1449,18 +1451,80 @@ datasets:
|
|
|
1449
1451
|
# ---------------------------------------------------------------- EPA
|
|
1450
1452
|
- id: epa:aqs-daily
|
|
1451
1453
|
provider: epa
|
|
1452
|
-
status:
|
|
1454
|
+
status: available
|
|
1453
1455
|
domain: air-quality
|
|
1454
|
-
|
|
1456
|
+
since: "0.26"
|
|
1455
1457
|
title: Air Quality System Daily Summaries
|
|
1456
1458
|
description: >-
|
|
1457
|
-
Daily
|
|
1458
|
-
|
|
1459
|
-
|
|
1459
|
+
Daily summaries of regulatory air monitoring data from EPA's Air Quality
|
|
1460
|
+
System through the AQS Data API: one row per monitor, local day, and
|
|
1461
|
+
pollutant standard, for one to five AQS parameter codes such as PM2.5
|
|
1462
|
+
and ozone, over named sites, a state or county, or a box. Each request
|
|
1463
|
+
covers at most one calendar year, so a window becomes one JSON file per
|
|
1464
|
+
year. Requires a free key issued by email; the response is stored in a
|
|
1465
|
+
canonical form without the echoed request.
|
|
1466
|
+
keywords: [air quality, pollution, ozone, pm2.5, monitors, aqs, smoke, epa]
|
|
1460
1467
|
protocol: http
|
|
1461
|
-
homepage: https://aqs.epa.gov/aqsweb/documents/data_api.html
|
|
1468
|
+
homepage: https://aqs.epa.gov/aqsweb/documents/data_api.html#daily
|
|
1462
1469
|
license: US Government Work (public domain)
|
|
1463
|
-
capabilities: { spatial_subset: true, temporal_subset: true, variable_subset:
|
|
1470
|
+
capabilities: { spatial_subset: true, temporal_subset: true, variable_subset: false }
|
|
1471
|
+
credentials:
|
|
1472
|
+
variables: [USDATA_AQS_EMAIL, USDATA_AQS_KEY]
|
|
1473
|
+
signup: https://aqs.epa.gov/aqsweb/documents/data_api.html#signup
|
|
1474
|
+
summary: Daily air pollutant summaries from regulatory monitors
|
|
1475
|
+
formats: [JSON]
|
|
1476
|
+
selection: Local days within inclusive UTC calendar dates for one to five pollutants; one file per year
|
|
1477
|
+
inputs: Both dates; one to five parameter codes; site ids, a state or county, or a box
|
|
1478
|
+
reader: pandas
|
|
1479
|
+
guide: docs/providers/epa-aqs-daily.md
|
|
1480
|
+
examples:
|
|
1481
|
+
- examples/wildfire-smoke/README.md
|
|
1482
|
+
resolution:
|
|
1483
|
+
spatial: Regulatory monitoring sites operated by state, local, and tribal agencies
|
|
1484
|
+
temporal: Daily summaries of each monitor's samples, one row per pollutant standard
|
|
1485
|
+
update_frequency: >-
|
|
1486
|
+
As monitoring agencies submit each quarter (40 CFR 58.16) and certify the previous year by
|
|
1487
|
+
May 1 (40 CFR 58.15); submitted values can be revised later
|
|
1488
|
+
latency: >-
|
|
1489
|
+
Agencies must submit each calendar quarter's data within 90 days after it ends (40 CFR 58.16)
|
|
1490
|
+
citation: >-
|
|
1491
|
+
U.S. Environmental Protection Agency, Air Quality System (AQS) daily summary data, AQS Data
|
|
1492
|
+
API, accessed via usdata
|
|
1493
|
+
terms: https://aqs.epa.gov/aqsweb/documents/data_api.html#terms
|
|
1494
|
+
variables:
|
|
1495
|
+
- { name: "state_code", description: "FIPS code of the state the monitor is in; 80 for Mexico, CC for Canada at border sites" }
|
|
1496
|
+
- { name: "county_code", description: "FIPS code of the county, parish, or independent city within the state" }
|
|
1497
|
+
- { name: "site_number", description: "Four-digit site number, unique within the county" }
|
|
1498
|
+
- { name: "parameter_code", description: "AQS code of the parameter measured" }
|
|
1499
|
+
- { name: "poc", description: "Parameter occurrence code distinguishing instruments measuring the same parameter at one site" }
|
|
1500
|
+
- { name: "latitude", units: "degrees_north", description: "Site latitude, WGS84" }
|
|
1501
|
+
- { name: "longitude", units: "degrees_east", description: "Site longitude, WGS84" }
|
|
1502
|
+
- { name: "datum", description: "Datum of the coordinates, always WGS84" }
|
|
1503
|
+
- { name: "parameter", description: "Name of the parameter measured" }
|
|
1504
|
+
- { name: "sample_duration_code", description: "Code of the sample duration" }
|
|
1505
|
+
- { name: "sample_duration", description: "Averaging period: observed, such as 1 HOUR or 24 HOUR, or calculated, such as 24-HR BLK AVG" }
|
|
1506
|
+
- { name: "pollutant_standard", description: "National ambient air quality standard the row's statistics are calculated for; empty for none" }
|
|
1507
|
+
- { name: "date_local", description: "Day the sample was taken, in local standard time" }
|
|
1508
|
+
- { name: "units_of_measure", description: "Units of every statistic on the row" }
|
|
1509
|
+
- { name: "event_type", description: "Whether exceptional-event data are included: No Events, Events Included, Events Excluded, or Concurred Events Excluded" }
|
|
1510
|
+
- { name: "observation_count", description: "Number of observations in the averaging period" }
|
|
1511
|
+
- { name: "observation_percent", units: "percent", description: "Share of scheduled values for the day that were reported" }
|
|
1512
|
+
- { name: "validity_indicator", description: "Y where the value meets all completeness criteria" }
|
|
1513
|
+
- { name: "arithmetic_mean", description: "Mean of the day's values, in units_of_measure" }
|
|
1514
|
+
- { name: "first_max_value", description: "Highest value at the row's duration or standard, in units_of_measure" }
|
|
1515
|
+
- { name: "first_max_hour", description: "Hour of the day, 24-hour local standard time, of the highest value" }
|
|
1516
|
+
- { name: "aqi", description: "Air Quality Index for the day, where the pollutant has one" }
|
|
1517
|
+
- { name: "method_code", description: "Three-digit measurement method code, unique within a parameter" }
|
|
1518
|
+
- { name: "method", description: "Collection and analysis method" }
|
|
1519
|
+
- { name: "local_site_name", description: "Site name in the operating agency's own nomenclature" }
|
|
1520
|
+
- { name: "site_address", description: "Approximate street address of the site" }
|
|
1521
|
+
- { name: "county", description: "Name of the county the site is in" }
|
|
1522
|
+
- { name: "state", description: "Name of the state the site is in" }
|
|
1523
|
+
- { name: "city", description: "Incorporated city the site is in, if any" }
|
|
1524
|
+
- { name: "cbsa_code", description: "Code of the core-based statistical (metropolitan) area" }
|
|
1525
|
+
- { name: "cbsa", description: "Name of the core-based statistical (metropolitan) area" }
|
|
1526
|
+
- { name: "date_of_last_change", description: "Date the underlying data were last changed in AQS" }
|
|
1527
|
+
adapter: usdata.providers.epa.aqs:AqsDaily
|
|
1464
1528
|
|
|
1465
1529
|
# ---------------------------------------------------------------- FEMA
|
|
1466
1530
|
- id: fema:nfhl
|
|
@@ -1,7 +1,8 @@
|
|
|
1
|
-
"""Read-only environment report: interpreter, reader extras, cache, and endpoints.
|
|
1
|
+
"""Read-only environment report: interpreter, reader extras, cache, credentials, and endpoints.
|
|
2
2
|
|
|
3
3
|
``diagnose`` only observes. It imports optional modules, reads environment
|
|
4
|
-
variables, stats the cache directory,
|
|
4
|
+
variables, stats the cache directory, reports whether each dataset that needs
|
|
5
|
+
credentials has them without ever printing a value, and - when asked - makes one bounded
|
|
5
6
|
request per upstream host family. It never creates, moves, or repairs anything;
|
|
6
7
|
the CLI prints what it finds and leaves the fixing to the reader.
|
|
7
8
|
"""
|
|
@@ -30,9 +31,10 @@ from usdata import __version__
|
|
|
30
31
|
from usdata._grib import LIBRARY_HINT
|
|
31
32
|
from usdata.cache import ENV_VAR, cache_dir
|
|
32
33
|
from usdata.mirror import ENV_VAR as MIRROR_ENV_VAR
|
|
33
|
-
from usdata.models import READER_EXTRAS
|
|
34
|
+
from usdata.models import READER_EXTRAS, Status
|
|
34
35
|
from usdata.protocols import http
|
|
35
|
-
from usdata.
|
|
36
|
+
from usdata.providers.credentials import Credentials
|
|
37
|
+
from usdata.registry import Registry, default_registry
|
|
36
38
|
|
|
37
39
|
READER_MODULES: dict[str, tuple[str, ...]] = {
|
|
38
40
|
"pandas": ("pandas",),
|
|
@@ -86,7 +88,13 @@ def diagnose(*, network: bool = False) -> Report:
|
|
|
86
88
|
cache, environment, endpoints. A missing optional extra is ``warn``;
|
|
87
89
|
a cache directory that cannot be written is ``fail``.
|
|
88
90
|
"""
|
|
89
|
-
checks = [
|
|
91
|
+
checks = [
|
|
92
|
+
*_runtime_checks(),
|
|
93
|
+
*_reader_checks(),
|
|
94
|
+
*_cache_checks(),
|
|
95
|
+
*_environment_checks(),
|
|
96
|
+
*_credential_checks(),
|
|
97
|
+
]
|
|
90
98
|
if network:
|
|
91
99
|
checks.extend(_endpoint_checks())
|
|
92
100
|
return Report(checks=checks)
|
|
@@ -198,6 +206,31 @@ def _environment_checks() -> Iterator[Check]:
|
|
|
198
206
|
yield Check(name=f"env:{name}", status=CheckStatus.OK, detail=value)
|
|
199
207
|
|
|
200
208
|
|
|
209
|
+
def _credential_checks(registry: Registry | None = None) -> Iterator[Check]:
|
|
210
|
+
"""Whether each fetchable dataset that needs credentials has them; values are never read out.
|
|
211
|
+
|
|
212
|
+
An unset variable is a warning, not a failure: every other dataset still
|
|
213
|
+
works, and a locked restore can still come from the cache or the mirror.
|
|
214
|
+
"""
|
|
215
|
+
for dataset in (registry or default_registry()).list():
|
|
216
|
+
if dataset.credentials is None or dataset.status is Status.PLANNED:
|
|
217
|
+
continue
|
|
218
|
+
names = dataset.credentials.variables
|
|
219
|
+
present = Credentials.from_environment(names)
|
|
220
|
+
if missing := [name for name in names if name not in present]:
|
|
221
|
+
yield Check(
|
|
222
|
+
name=f"credentials:{dataset.id}",
|
|
223
|
+
status=CheckStatus.WARN,
|
|
224
|
+
detail=f"{', '.join(missing)} unset; request a key at {dataset.credentials.signup}",
|
|
225
|
+
)
|
|
226
|
+
else:
|
|
227
|
+
yield Check(
|
|
228
|
+
name=f"credentials:{dataset.id}",
|
|
229
|
+
status=CheckStatus.OK,
|
|
230
|
+
detail=f"{', '.join(names)} set",
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
|
|
201
234
|
def _endpoint_checks() -> Iterator[Check]:
|
|
202
235
|
"""One bounded request per upstream host; an unreachable host is a failure."""
|
|
203
236
|
hosts = probe_hosts()
|
|
@@ -230,6 +230,37 @@ class Variable(BaseModel):
|
|
|
230
230
|
return f"{self.name} ({self.units})" if self.units else self.name
|
|
231
231
|
|
|
232
232
|
|
|
233
|
+
CREDENTIAL_VARIABLE = re.compile(r"USDATA_[A-Z0-9]+(?:_[A-Z0-9]+)+")
|
|
234
|
+
"""How a credential variable is named: ``USDATA_<SYSTEM>_<FIELD>``, such as ``USDATA_AQS_KEY``."""
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
class CredentialSpec(BaseModel):
|
|
238
|
+
"""The environment variables a source needs before it can be contacted, and where to get a key.
|
|
239
|
+
|
|
240
|
+
Declared in the registry entry so the catalog, ``info``, and ``doctor`` can
|
|
241
|
+
say what a dataset needs, and so the core can check it before any request
|
|
242
|
+
(ADR 0039). Values never appear here or anywhere else the core writes.
|
|
243
|
+
"""
|
|
244
|
+
|
|
245
|
+
variables: list[str] = Field(
|
|
246
|
+
min_length=1, description="Environment variables that must be set, USDATA_<SYSTEM>_<FIELD>"
|
|
247
|
+
)
|
|
248
|
+
signup: str = Field(description="https URL where the agency issues a key")
|
|
249
|
+
|
|
250
|
+
@model_validator(mode="after")
|
|
251
|
+
def _named(self) -> CredentialSpec:
|
|
252
|
+
for name in self.variables:
|
|
253
|
+
if not CREDENTIAL_VARIABLE.fullmatch(name):
|
|
254
|
+
raise ValueError(
|
|
255
|
+
f"credential variable {name!r} must be named USDATA_<SYSTEM>_<FIELD>"
|
|
256
|
+
)
|
|
257
|
+
if len(set(self.variables)) != len(self.variables):
|
|
258
|
+
raise ValueError("credentials.variables entries must be distinct")
|
|
259
|
+
if not self.signup.startswith("https://"):
|
|
260
|
+
raise ValueError("credentials.signup must be an https URL")
|
|
261
|
+
return self
|
|
262
|
+
|
|
263
|
+
|
|
233
264
|
class Limits(BaseModel):
|
|
234
265
|
"""Request limits the adapter enforces, declared here and verified by the adapter tests."""
|
|
235
266
|
|
|
@@ -304,6 +335,10 @@ class Dataset(BaseModel):
|
|
|
304
335
|
limits: Limits | None = Field(
|
|
305
336
|
default=None, description="Request limits the adapter enforces, such as the longest window"
|
|
306
337
|
)
|
|
338
|
+
credentials: CredentialSpec | None = Field(
|
|
339
|
+
default=None,
|
|
340
|
+
description="Environment variables the source needs before it can be contacted (ADR 0039)",
|
|
341
|
+
)
|
|
307
342
|
system: str | None = Field(
|
|
308
343
|
default=None,
|
|
309
344
|
description="Id of a system declared in the registry, for datasets that belong to one",
|
|
@@ -595,7 +630,16 @@ class Provenance(BaseModel):
|
|
|
595
630
|
)
|
|
596
631
|
mirror: str | None = Field(
|
|
597
632
|
default=None,
|
|
598
|
-
description=
|
|
633
|
+
description=(
|
|
634
|
+
"Mirror object that served these bytes in place of the source (ADR 0030, ADR 0039)"
|
|
635
|
+
),
|
|
636
|
+
)
|
|
637
|
+
credentials: list[str] = Field(
|
|
638
|
+
default_factory=list,
|
|
639
|
+
description=(
|
|
640
|
+
"Environment variables the source requires to fetch this file, never their values "
|
|
641
|
+
"(ADR 0039); empty for an anonymous source"
|
|
642
|
+
),
|
|
599
643
|
)
|
|
600
644
|
|
|
601
645
|
@property
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
from collections.abc import Sequence
|
|
5
6
|
from datetime import UTC, datetime
|
|
6
7
|
from pathlib import Path
|
|
7
8
|
|
|
@@ -14,7 +15,11 @@ SIDECAR_SUFFIX = ".provenance.json"
|
|
|
14
15
|
|
|
15
16
|
|
|
16
17
|
def record(
|
|
17
|
-
dataset: Dataset,
|
|
18
|
+
dataset: Dataset,
|
|
19
|
+
asset: Asset,
|
|
20
|
+
path: Path,
|
|
21
|
+
partial: PartialFetch | None = None,
|
|
22
|
+
transformations: Sequence[str] = (),
|
|
18
23
|
) -> Provenance:
|
|
19
24
|
"""Build a provenance record for a file that was just fetched to ``path``.
|
|
20
25
|
|
|
@@ -26,6 +31,9 @@ def record(
|
|
|
26
31
|
Its index, ranges, selectors, and object identity are recorded
|
|
27
32
|
alongside the checksum of the local file, which still covers exactly
|
|
28
33
|
these bytes.
|
|
34
|
+
transformations: How the adapter's ``fetch`` changed the bytes the
|
|
35
|
+
source sent (``Provider.transformations``), recorded after any
|
|
36
|
+
partial-fetch entry.
|
|
29
37
|
|
|
30
38
|
Returns:
|
|
31
39
|
The record to write beside ``path``.
|
|
@@ -39,13 +47,14 @@ def record(
|
|
|
39
47
|
size=path.stat().st_size,
|
|
40
48
|
license=dataset.license,
|
|
41
49
|
usdata_version=__version__,
|
|
42
|
-
transformations=[] if partial is None else [partial.describe()],
|
|
50
|
+
transformations=[*([] if partial is None else [partial.describe()]), *transformations],
|
|
43
51
|
index_url=None if partial is None else partial.index_url,
|
|
44
52
|
index_checksum=None if partial is None else partial.index_checksum,
|
|
45
53
|
ranges=[] if partial is None else list(partial.ranges),
|
|
46
54
|
selectors=[] if partial is None else list(partial.selectors),
|
|
47
55
|
object_size=None if partial is None else partial.object_size,
|
|
48
56
|
object_etag=None if partial is None else partial.object_etag,
|
|
57
|
+
credentials=[] if dataset.credentials is None else list(dataset.credentials.variables),
|
|
49
58
|
)
|
|
50
59
|
|
|
51
60
|
|
|
@@ -5,11 +5,21 @@ inside this repository or outside it, is written against. ``Provider`` is the
|
|
|
5
5
|
interface, ``HttpProvider`` the client lifecycle HTTP-backed adapters inherit,
|
|
6
6
|
``QueryError`` the refusal every adapter raises, and the coercions come from
|
|
7
7
|
``usdata.providers.params`` so parameter models read the same everywhere.
|
|
8
|
+
``Credentials`` carries the values a source that needs a key receives, and
|
|
9
|
+
``MissingCredentials`` is the refusal when one is unset (ADR 0039).
|
|
8
10
|
``usdata.testing`` checks an adapter against this contract. See
|
|
9
11
|
[ADR 0027](https://github.com/jakeryderv/usdata/blob/main/docs/adr/0027-provider-contract.md).
|
|
10
12
|
"""
|
|
11
13
|
|
|
12
|
-
from usdata.providers.base import
|
|
14
|
+
from usdata.providers.base import (
|
|
15
|
+
MissingCredentials,
|
|
16
|
+
NotImplementedProvider,
|
|
17
|
+
Provider,
|
|
18
|
+
QueryError,
|
|
19
|
+
adapter_class,
|
|
20
|
+
load_adapter,
|
|
21
|
+
)
|
|
22
|
+
from usdata.providers.credentials import Credentials
|
|
13
23
|
from usdata.providers.http import HttpProvider
|
|
14
24
|
from usdata.providers.params import (
|
|
15
25
|
OptionalUpperStrList,
|
|
@@ -23,13 +33,16 @@ from usdata.providers.params import (
|
|
|
23
33
|
)
|
|
24
34
|
|
|
25
35
|
__all__ = [
|
|
36
|
+
"Credentials",
|
|
26
37
|
"HttpProvider",
|
|
38
|
+
"MissingCredentials",
|
|
27
39
|
"NotImplementedProvider",
|
|
28
40
|
"OptionalUpperStrList",
|
|
29
41
|
"Provider",
|
|
30
42
|
"QueryError",
|
|
31
43
|
"StrList",
|
|
32
44
|
"UpperStrList",
|
|
45
|
+
"adapter_class",
|
|
33
46
|
"choice",
|
|
34
47
|
"flag",
|
|
35
48
|
"int_list",
|