usdata 0.25.0__tar.gz → 0.26.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {usdata-0.25.0 → usdata-0.26.0}/PKG-INFO +2 -2
- {usdata-0.25.0 → usdata-0.26.0}/README.md +1 -1
- {usdata-0.25.0 → usdata-0.26.0}/pyproject.toml +1 -1
- {usdata-0.25.0 → usdata-0.26.0}/pyproject.toml.orig +1 -1
- usdata-0.26.0/src/usdata/_aqs.py +51 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/_fetch.py +1 -1
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/_hurdat2.py +8 -1
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/cli/app.py +15 -5
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/data/registry.yaml +73 -10
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/doctor.py +38 -5
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/models.py +45 -1
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/provenance.py +11 -2
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/__init__.py +14 -1
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/base.py +80 -4
- usdata-0.26.0/src/usdata/providers/credentials.py +84 -0
- usdata-0.26.0/src/usdata/providers/epa/__init__.py +1 -0
- usdata-0.26.0/src/usdata/providers/epa/aqs.py +308 -0
- usdata-0.26.0/src/usdata/providers/http.py +139 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/hurdat2.py +32 -2
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/pull.py +43 -9
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/readers.py +22 -3
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/testing.py +199 -3
- usdata-0.25.0/src/usdata/providers/http.py +0 -44
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/__init__.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/__main__.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/_files.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/_grib.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/_netcdf.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/_progress.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/_radar.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/cache.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/cache_ops.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/cite.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/cli/__init__.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/cli/cache.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/cli/cite.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/cli/doctor.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/cli/inspect.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/cli/progress.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/data/nexrad_sites.csv +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/data/places.csv +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/data/places.sources.json +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/inspect.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/manifest.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/mirror.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/protocols/__init__.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/protocols/erddap.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/protocols/http.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/protocols/listing.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/protocols/s3.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/fema/__init__.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/fema/declarations.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/__init__.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/coops.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/gfs.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/ghcnd.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/glm.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/goes.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/grib_index.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/gsom.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/gsoy.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/hrrr.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/ibtracs.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/lcd.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/mrms.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/nbm.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/nexrad.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/nexrad_level3.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/normals.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/nws_vtec.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/rap.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/sites.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/spc.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/noaa/storm_events.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/params.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/usgs/__init__.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/usgs/daily.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/providers/usgs/earthquakes.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/py.typed +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/query.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/registry.py +0 -0
- {usdata-0.25.0 → usdata-0.26.0}/src/usdata/selection.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: usdata
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.26.0
|
|
4
4
|
Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
|
|
5
5
|
Keywords: noaa,usgs,open-data,scientific-data,provenance,reproducible-research,weather,climate,meteorology
|
|
6
6
|
Author: Jake Van Slyke
|
|
@@ -112,7 +112,7 @@ breaking changes.
|
|
|
112
112
|
| [docs.usdata.dev](https://docs.usdata.dev/) | How to use it: [install](https://docs.usdata.dev/install/), [getting started](https://docs.usdata.dev/getting-started/), guides, dataset notes, and reference |
|
|
113
113
|
| [Severe-weather case study](https://usdata.dev/examples/severe-weather-case-study/) | One tornado, six sources, one manifest and lockfile, ending in a citation |
|
|
114
114
|
|
|
115
|
-
Twenty-
|
|
115
|
+
Twenty-seven datasets are available today and nineteen more are planned, grouped
|
|
116
116
|
by agency and product family in the [catalog](docs/providers/README.md).
|
|
117
117
|
|
|
118
118
|
## How this compares
|
|
@@ -67,7 +67,7 @@ breaking changes.
|
|
|
67
67
|
| [docs.usdata.dev](https://docs.usdata.dev/) | How to use it: [install](https://docs.usdata.dev/install/), [getting started](https://docs.usdata.dev/getting-started/), guides, dataset notes, and reference |
|
|
68
68
|
| [Severe-weather case study](https://usdata.dev/examples/severe-weather-case-study/) | One tornado, six sources, one manifest and lockfile, ending in a citation |
|
|
69
69
|
|
|
70
|
-
Twenty-
|
|
70
|
+
Twenty-seven datasets are available today and nineteen more are planned, grouped
|
|
71
71
|
by agency and product family in the [catalog](docs/providers/README.md).
|
|
72
72
|
|
|
73
73
|
## How this compares
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Local reading of AQS daily-summary JSON into a tidy table, behind pandas.
|
|
2
|
+
|
|
3
|
+
A fetched ``epa:aqs-daily`` file is the canonical form the adapter writes: the
|
|
4
|
+
service's JSON with the echoed request left out of its header and the rows in a
|
|
5
|
+
fixed order (ADR 0039). Each element of ``Data`` is one monitor's summary for one
|
|
6
|
+
local calendar day under one pollutant standard, so the table is one row per
|
|
7
|
+
element, with the columns as the service names them.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
from importlib import import_module
|
|
14
|
+
from typing import TYPE_CHECKING, Any
|
|
15
|
+
|
|
16
|
+
from usdata.readers import MissingReaderDependency
|
|
17
|
+
|
|
18
|
+
if TYPE_CHECKING:
|
|
19
|
+
from usdata._fetch import FetchedAsset
|
|
20
|
+
|
|
21
|
+
DATE_COLUMNS = ("date_local", "date_of_last_change")
|
|
22
|
+
"""Calendar dates as the service writes them, ``YYYY-MM-DD``, parsed without a timezone.
|
|
23
|
+
|
|
24
|
+
``date_local`` is a day in the monitor's local standard time, not a UTC instant,
|
|
25
|
+
so it is left naive rather than given a zone it does not have.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def open_aqs(fetched: FetchedAsset) -> Any:
|
|
30
|
+
"""Parse a fetched AQS daily-summary file into a pandas DataFrame, one row per summary."""
|
|
31
|
+
try:
|
|
32
|
+
pandas = import_module("pandas")
|
|
33
|
+
except ModuleNotFoundError as error:
|
|
34
|
+
if error.name != "pandas":
|
|
35
|
+
raise
|
|
36
|
+
raise MissingReaderDependency(
|
|
37
|
+
'AQS reading requires pandas; install it with: pip install "usdata[pandas]" '
|
|
38
|
+
'(or uv add "usdata[pandas]")'
|
|
39
|
+
) from error
|
|
40
|
+
# Reading a fetched asset is strictly local; the cached file is never rewritten.
|
|
41
|
+
body = json.loads(fetched.path.read_text(encoding="utf-8"))
|
|
42
|
+
frame = pandas.DataFrame(body.get("Data") or [])
|
|
43
|
+
for column in DATE_COLUMNS:
|
|
44
|
+
if column in frame:
|
|
45
|
+
frame[column] = pandas.to_datetime(frame[column], format="%Y-%m-%d")
|
|
46
|
+
frame.attrs["usdata"] = {
|
|
47
|
+
"asset_id": fetched.asset.id,
|
|
48
|
+
"provenance": fetched.provenance.model_dump(mode="json"),
|
|
49
|
+
"header": body.get("Header", []),
|
|
50
|
+
}
|
|
51
|
+
return frame
|
|
@@ -144,7 +144,7 @@ def _fetch_asset(
|
|
|
144
144
|
adapter.fetch(asset, tmp)
|
|
145
145
|
else:
|
|
146
146
|
adapter.fetch_partial(asset, tmp, partial)
|
|
147
|
-
prov = provenance.record(dataset, asset, tmp, partial)
|
|
147
|
+
prov = provenance.record(dataset, asset, tmp, partial, adapter.transformations)
|
|
148
148
|
if asset.checksum and prov.checksum != asset.checksum:
|
|
149
149
|
raise ChecksumMismatch(f"{asset.id}: expected {asset.checksum}, got {prov.checksum}")
|
|
150
150
|
# A crash between replacements leaves a detectable mismatch, never a trusted partial file.
|
|
@@ -160,7 +160,14 @@ def open_hurdat2(fetched: FetchedAsset) -> Any:
|
|
|
160
160
|
'(or uv add "usdata[pandas]")'
|
|
161
161
|
) from error
|
|
162
162
|
# Reading a fetched asset is strictly local; the source file is never rewritten.
|
|
163
|
-
|
|
163
|
+
try:
|
|
164
|
+
columns = parse(fetched.path.read_text(encoding="utf-8"))
|
|
165
|
+
except Hurdat2FormatError as error:
|
|
166
|
+
# NHC typos come and go between revisions, so name the way to pick another.
|
|
167
|
+
raise Hurdat2FormatError(
|
|
168
|
+
f"{fetched.asset.id}: {error}; an upstream typo is usually absent from the "
|
|
169
|
+
"neighbouring revisions, so select one with the 'revision' parameter"
|
|
170
|
+
) from error
|
|
164
171
|
data: dict[str, Any] = {
|
|
165
172
|
name: pandas.array(columns[name], dtype="string") for name in TEXT_COLUMNS
|
|
166
173
|
}
|
|
@@ -21,7 +21,7 @@ from usdata.cli.inspect import inspect
|
|
|
21
21
|
from usdata.cli.progress import progress
|
|
22
22
|
from usdata.manifest import lockfile_path
|
|
23
23
|
from usdata.models import READER_EXTRAS_TEXT, Dataset, Status, describe_duration
|
|
24
|
-
from usdata.providers import load_adapter
|
|
24
|
+
from usdata.providers import adapter_class, load_adapter
|
|
25
25
|
from usdata.providers.base import NotImplementedProvider
|
|
26
26
|
from usdata.pull import EmptySource, ManifestChanged, Plan, UnknownDatasets, UpstreamChanged
|
|
27
27
|
from usdata.pull import plan as plan_manifest
|
|
@@ -260,8 +260,7 @@ def info(
|
|
|
260
260
|
)
|
|
261
261
|
if ds.status is not Status.AVAILABLE:
|
|
262
262
|
return # Planned entries have no adapter or usage metadata to show.
|
|
263
|
-
|
|
264
|
-
declared = dict(adapter.accepted_params)
|
|
263
|
+
declared = dict(adapter_class(ds).accepted_params)
|
|
265
264
|
if declared:
|
|
266
265
|
typer.echo(" params:")
|
|
267
266
|
width = max(len(name) for name in declared)
|
|
@@ -285,6 +284,9 @@ def _echo_usage(ds: Dataset) -> None:
|
|
|
285
284
|
typer.echo(f" selection: {ds.selection}")
|
|
286
285
|
if ds.inputs:
|
|
287
286
|
typer.echo(f" inputs: {ds.inputs}")
|
|
287
|
+
if ds.credentials:
|
|
288
|
+
typer.echo(f" credentials: {', '.join(ds.credentials.variables)} (environment)")
|
|
289
|
+
typer.echo(f" key: {ds.credentials.signup}")
|
|
288
290
|
if ds.examples:
|
|
289
291
|
typer.echo(f" examples: {', '.join(ds.examples)}")
|
|
290
292
|
|
|
@@ -528,13 +530,21 @@ def pull(
|
|
|
528
530
|
else:
|
|
529
531
|
mode = "restored from" if result.from_lockfile else "wrote"
|
|
530
532
|
typer.echo(f"{len(result.fetched)} asset(s); {mode} {result.lockfile_path}", err=True)
|
|
531
|
-
|
|
533
|
+
changed = len(result.mirrored) - len(result.unchecked)
|
|
534
|
+
if changed:
|
|
532
535
|
typer.secho(
|
|
533
|
-
f"{
|
|
536
|
+
f"{changed} asset(s) changed upstream and were restored from the "
|
|
534
537
|
"mirror; the lockfile still pins the source. Pass --update to accept new bytes.",
|
|
535
538
|
err=True,
|
|
536
539
|
fg="yellow",
|
|
537
540
|
)
|
|
541
|
+
if result.unchecked:
|
|
542
|
+
typer.secho(
|
|
543
|
+
f"{len(result.unchecked)} asset(s) were restored from the mirror without asking "
|
|
544
|
+
"their source, whose credentials are not set; upstream was not checked for changes.",
|
|
545
|
+
err=True,
|
|
546
|
+
fg="yellow",
|
|
547
|
+
)
|
|
538
548
|
|
|
539
549
|
|
|
540
550
|
@app.command()
|
|
@@ -424,7 +424,8 @@ datasets:
|
|
|
424
424
|
intensity, pressure, and wind radii for Atlantic (since 1851) and
|
|
425
425
|
northeast/north-central Pacific (since 1949) tropical cyclones. One
|
|
426
426
|
fixed-format text file per basin, revised after each season; a basin
|
|
427
|
-
parameter picks the file and the newest revision wins
|
|
427
|
+
parameter picks the file and the newest revision wins unless a revision
|
|
428
|
+
date names another. There is no
|
|
428
429
|
query interface, so dates and geographic filters are rejected and the
|
|
429
430
|
local reader turns the whole file into one row per track point.
|
|
430
431
|
keywords: [hurricane, tropical cyclone, best track, nhc, atlantic, pacific]
|
|
@@ -435,8 +436,8 @@ datasets:
|
|
|
435
436
|
capabilities: { spatial_subset: false, temporal_subset: false, variable_subset: false }
|
|
436
437
|
summary: Tropical cyclone best tracks
|
|
437
438
|
formats: [HURDAT2 fixed-format text]
|
|
438
|
-
selection: The newest revision of one whole basin file; filter track points locally
|
|
439
|
-
inputs: Optional basin (atlantic or pacific); no dates or geographic filters
|
|
439
|
+
selection: The newest or a named revision of one whole basin file; filter track points locally
|
|
440
|
+
inputs: Optional basin (atlantic or pacific) and revision date; no dates or geographic filters
|
|
440
441
|
reader: pandas
|
|
441
442
|
guide: docs/providers/noaa-hurdat2.md
|
|
442
443
|
examples:
|
|
@@ -1450,18 +1451,80 @@ datasets:
|
|
|
1450
1451
|
# ---------------------------------------------------------------- EPA
|
|
1451
1452
|
- id: epa:aqs-daily
|
|
1452
1453
|
provider: epa
|
|
1453
|
-
status:
|
|
1454
|
+
status: available
|
|
1454
1455
|
domain: air-quality
|
|
1455
|
-
|
|
1456
|
+
since: "0.26"
|
|
1456
1457
|
title: Air Quality System Daily Summaries
|
|
1457
1458
|
description: >-
|
|
1458
|
-
Daily
|
|
1459
|
-
|
|
1460
|
-
|
|
1459
|
+
Daily summaries of regulatory air monitoring data from EPA's Air Quality
|
|
1460
|
+
System through the AQS Data API: one row per monitor, local day, and
|
|
1461
|
+
pollutant standard, for one to five AQS parameter codes such as PM2.5
|
|
1462
|
+
and ozone, over named sites, a state or county, or a box. Each request
|
|
1463
|
+
covers at most one calendar year, so a window becomes one JSON file per
|
|
1464
|
+
year. Requires a free key issued by email; the response is stored in a
|
|
1465
|
+
canonical form without the echoed request.
|
|
1466
|
+
keywords: [air quality, pollution, ozone, pm2.5, monitors, aqs, smoke, epa]
|
|
1461
1467
|
protocol: http
|
|
1462
|
-
homepage: https://aqs.epa.gov/aqsweb/documents/data_api.html
|
|
1468
|
+
homepage: https://aqs.epa.gov/aqsweb/documents/data_api.html#daily
|
|
1463
1469
|
license: US Government Work (public domain)
|
|
1464
|
-
capabilities: { spatial_subset: true, temporal_subset: true, variable_subset:
|
|
1470
|
+
capabilities: { spatial_subset: true, temporal_subset: true, variable_subset: false }
|
|
1471
|
+
credentials:
|
|
1472
|
+
variables: [USDATA_AQS_EMAIL, USDATA_AQS_KEY]
|
|
1473
|
+
signup: https://aqs.epa.gov/aqsweb/documents/data_api.html#signup
|
|
1474
|
+
summary: Daily air pollutant summaries from regulatory monitors
|
|
1475
|
+
formats: [JSON]
|
|
1476
|
+
selection: Local days within inclusive UTC calendar dates for one to five pollutants; one file per year
|
|
1477
|
+
inputs: Both dates; one to five parameter codes; site ids, a state or county, or a box
|
|
1478
|
+
reader: pandas
|
|
1479
|
+
guide: docs/providers/epa-aqs-daily.md
|
|
1480
|
+
examples:
|
|
1481
|
+
- examples/wildfire-smoke/README.md
|
|
1482
|
+
resolution:
|
|
1483
|
+
spatial: Regulatory monitoring sites operated by state, local, and tribal agencies
|
|
1484
|
+
temporal: Daily summaries of each monitor's samples, one row per pollutant standard
|
|
1485
|
+
update_frequency: >-
|
|
1486
|
+
As monitoring agencies submit each quarter (40 CFR 58.16) and certify the previous year by
|
|
1487
|
+
May 1 (40 CFR 58.15); submitted values can be revised later
|
|
1488
|
+
latency: >-
|
|
1489
|
+
Agencies must submit each calendar quarter's data within 90 days after it ends (40 CFR 58.16)
|
|
1490
|
+
citation: >-
|
|
1491
|
+
U.S. Environmental Protection Agency, Air Quality System (AQS) daily summary data, AQS Data
|
|
1492
|
+
API, accessed via usdata
|
|
1493
|
+
terms: https://aqs.epa.gov/aqsweb/documents/data_api.html#terms
|
|
1494
|
+
variables:
|
|
1495
|
+
- { name: "state_code", description: "FIPS code of the state the monitor is in; 80 for Mexico, CC for Canada at border sites" }
|
|
1496
|
+
- { name: "county_code", description: "FIPS code of the county, parish, or independent city within the state" }
|
|
1497
|
+
- { name: "site_number", description: "Four-digit site number, unique within the county" }
|
|
1498
|
+
- { name: "parameter_code", description: "AQS code of the parameter measured" }
|
|
1499
|
+
- { name: "poc", description: "Parameter occurrence code distinguishing instruments measuring the same parameter at one site" }
|
|
1500
|
+
- { name: "latitude", units: "degrees_north", description: "Site latitude, WGS84" }
|
|
1501
|
+
- { name: "longitude", units: "degrees_east", description: "Site longitude, WGS84" }
|
|
1502
|
+
- { name: "datum", description: "Datum of the coordinates, always WGS84" }
|
|
1503
|
+
- { name: "parameter", description: "Name of the parameter measured" }
|
|
1504
|
+
- { name: "sample_duration_code", description: "Code of the sample duration" }
|
|
1505
|
+
- { name: "sample_duration", description: "Averaging period: observed, such as 1 HOUR or 24 HOUR, or calculated, such as 24-HR BLK AVG" }
|
|
1506
|
+
- { name: "pollutant_standard", description: "National ambient air quality standard the row's statistics are calculated for; empty for none" }
|
|
1507
|
+
- { name: "date_local", description: "Day the sample was taken, in local standard time" }
|
|
1508
|
+
- { name: "units_of_measure", description: "Units of every statistic on the row" }
|
|
1509
|
+
- { name: "event_type", description: "Whether exceptional-event data are included: No Events, Events Included, Events Excluded, or Concurred Events Excluded" }
|
|
1510
|
+
- { name: "observation_count", description: "Number of observations in the averaging period" }
|
|
1511
|
+
- { name: "observation_percent", units: "percent", description: "Share of scheduled values for the day that were reported" }
|
|
1512
|
+
- { name: "validity_indicator", description: "Y where the value meets all completeness criteria" }
|
|
1513
|
+
- { name: "arithmetic_mean", description: "Mean of the day's values, in units_of_measure" }
|
|
1514
|
+
- { name: "first_max_value", description: "Highest value at the row's duration or standard, in units_of_measure" }
|
|
1515
|
+
- { name: "first_max_hour", description: "Hour of the day, 24-hour local standard time, of the highest value" }
|
|
1516
|
+
- { name: "aqi", description: "Air Quality Index for the day, where the pollutant has one" }
|
|
1517
|
+
- { name: "method_code", description: "Three-digit measurement method code, unique within a parameter" }
|
|
1518
|
+
- { name: "method", description: "Collection and analysis method" }
|
|
1519
|
+
- { name: "local_site_name", description: "Site name in the operating agency's own nomenclature" }
|
|
1520
|
+
- { name: "site_address", description: "Approximate street address of the site" }
|
|
1521
|
+
- { name: "county", description: "Name of the county the site is in" }
|
|
1522
|
+
- { name: "state", description: "Name of the state the site is in" }
|
|
1523
|
+
- { name: "city", description: "Incorporated city the site is in, if any" }
|
|
1524
|
+
- { name: "cbsa_code", description: "Code of the core-based statistical (metropolitan) area" }
|
|
1525
|
+
- { name: "cbsa", description: "Name of the core-based statistical (metropolitan) area" }
|
|
1526
|
+
- { name: "date_of_last_change", description: "Date the underlying data were last changed in AQS" }
|
|
1527
|
+
adapter: usdata.providers.epa.aqs:AqsDaily
|
|
1465
1528
|
|
|
1466
1529
|
# ---------------------------------------------------------------- FEMA
|
|
1467
1530
|
- id: fema:nfhl
|
|
@@ -1,7 +1,8 @@
|
|
|
1
|
-
"""Read-only environment report: interpreter, reader extras, cache, and endpoints.
|
|
1
|
+
"""Read-only environment report: interpreter, reader extras, cache, credentials, and endpoints.
|
|
2
2
|
|
|
3
3
|
``diagnose`` only observes. It imports optional modules, reads environment
|
|
4
|
-
variables, stats the cache directory,
|
|
4
|
+
variables, stats the cache directory, reports whether each dataset that needs
|
|
5
|
+
credentials has them without ever printing a value, and - when asked - makes one bounded
|
|
5
6
|
request per upstream host family. It never creates, moves, or repairs anything;
|
|
6
7
|
the CLI prints what it finds and leaves the fixing to the reader.
|
|
7
8
|
"""
|
|
@@ -30,9 +31,10 @@ from usdata import __version__
|
|
|
30
31
|
from usdata._grib import LIBRARY_HINT
|
|
31
32
|
from usdata.cache import ENV_VAR, cache_dir
|
|
32
33
|
from usdata.mirror import ENV_VAR as MIRROR_ENV_VAR
|
|
33
|
-
from usdata.models import READER_EXTRAS
|
|
34
|
+
from usdata.models import READER_EXTRAS, Status
|
|
34
35
|
from usdata.protocols import http
|
|
35
|
-
from usdata.
|
|
36
|
+
from usdata.providers.credentials import Credentials
|
|
37
|
+
from usdata.registry import Registry, default_registry
|
|
36
38
|
|
|
37
39
|
READER_MODULES: dict[str, tuple[str, ...]] = {
|
|
38
40
|
"pandas": ("pandas",),
|
|
@@ -86,7 +88,13 @@ def diagnose(*, network: bool = False) -> Report:
|
|
|
86
88
|
cache, environment, endpoints. A missing optional extra is ``warn``;
|
|
87
89
|
a cache directory that cannot be written is ``fail``.
|
|
88
90
|
"""
|
|
89
|
-
checks = [
|
|
91
|
+
checks = [
|
|
92
|
+
*_runtime_checks(),
|
|
93
|
+
*_reader_checks(),
|
|
94
|
+
*_cache_checks(),
|
|
95
|
+
*_environment_checks(),
|
|
96
|
+
*_credential_checks(),
|
|
97
|
+
]
|
|
90
98
|
if network:
|
|
91
99
|
checks.extend(_endpoint_checks())
|
|
92
100
|
return Report(checks=checks)
|
|
@@ -198,6 +206,31 @@ def _environment_checks() -> Iterator[Check]:
|
|
|
198
206
|
yield Check(name=f"env:{name}", status=CheckStatus.OK, detail=value)
|
|
199
207
|
|
|
200
208
|
|
|
209
|
+
def _credential_checks(registry: Registry | None = None) -> Iterator[Check]:
|
|
210
|
+
"""Whether each fetchable dataset that needs credentials has them; values are never read out.
|
|
211
|
+
|
|
212
|
+
An unset variable is a warning, not a failure: every other dataset still
|
|
213
|
+
works, and a locked restore can still come from the cache or the mirror.
|
|
214
|
+
"""
|
|
215
|
+
for dataset in (registry or default_registry()).list():
|
|
216
|
+
if dataset.credentials is None or dataset.status is Status.PLANNED:
|
|
217
|
+
continue
|
|
218
|
+
names = dataset.credentials.variables
|
|
219
|
+
present = Credentials.from_environment(names)
|
|
220
|
+
if missing := [name for name in names if name not in present]:
|
|
221
|
+
yield Check(
|
|
222
|
+
name=f"credentials:{dataset.id}",
|
|
223
|
+
status=CheckStatus.WARN,
|
|
224
|
+
detail=f"{', '.join(missing)} unset; request a key at {dataset.credentials.signup}",
|
|
225
|
+
)
|
|
226
|
+
else:
|
|
227
|
+
yield Check(
|
|
228
|
+
name=f"credentials:{dataset.id}",
|
|
229
|
+
status=CheckStatus.OK,
|
|
230
|
+
detail=f"{', '.join(names)} set",
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
|
|
201
234
|
def _endpoint_checks() -> Iterator[Check]:
|
|
202
235
|
"""One bounded request per upstream host; an unreachable host is a failure."""
|
|
203
236
|
hosts = probe_hosts()
|
|
@@ -230,6 +230,37 @@ class Variable(BaseModel):
|
|
|
230
230
|
return f"{self.name} ({self.units})" if self.units else self.name
|
|
231
231
|
|
|
232
232
|
|
|
233
|
+
CREDENTIAL_VARIABLE = re.compile(r"USDATA_[A-Z0-9]+(?:_[A-Z0-9]+)+")
|
|
234
|
+
"""How a credential variable is named: ``USDATA_<SYSTEM>_<FIELD>``, such as ``USDATA_AQS_KEY``."""
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
class CredentialSpec(BaseModel):
|
|
238
|
+
"""The environment variables a source needs before it can be contacted, and where to get a key.
|
|
239
|
+
|
|
240
|
+
Declared in the registry entry so the catalog, ``info``, and ``doctor`` can
|
|
241
|
+
say what a dataset needs, and so the core can check it before any request
|
|
242
|
+
(ADR 0039). Values never appear here or anywhere else the core writes.
|
|
243
|
+
"""
|
|
244
|
+
|
|
245
|
+
variables: list[str] = Field(
|
|
246
|
+
min_length=1, description="Environment variables that must be set, USDATA_<SYSTEM>_<FIELD>"
|
|
247
|
+
)
|
|
248
|
+
signup: str = Field(description="https URL where the agency issues a key")
|
|
249
|
+
|
|
250
|
+
@model_validator(mode="after")
|
|
251
|
+
def _named(self) -> CredentialSpec:
|
|
252
|
+
for name in self.variables:
|
|
253
|
+
if not CREDENTIAL_VARIABLE.fullmatch(name):
|
|
254
|
+
raise ValueError(
|
|
255
|
+
f"credential variable {name!r} must be named USDATA_<SYSTEM>_<FIELD>"
|
|
256
|
+
)
|
|
257
|
+
if len(set(self.variables)) != len(self.variables):
|
|
258
|
+
raise ValueError("credentials.variables entries must be distinct")
|
|
259
|
+
if not self.signup.startswith("https://"):
|
|
260
|
+
raise ValueError("credentials.signup must be an https URL")
|
|
261
|
+
return self
|
|
262
|
+
|
|
263
|
+
|
|
233
264
|
class Limits(BaseModel):
|
|
234
265
|
"""Request limits the adapter enforces, declared here and verified by the adapter tests."""
|
|
235
266
|
|
|
@@ -304,6 +335,10 @@ class Dataset(BaseModel):
|
|
|
304
335
|
limits: Limits | None = Field(
|
|
305
336
|
default=None, description="Request limits the adapter enforces, such as the longest window"
|
|
306
337
|
)
|
|
338
|
+
credentials: CredentialSpec | None = Field(
|
|
339
|
+
default=None,
|
|
340
|
+
description="Environment variables the source needs before it can be contacted (ADR 0039)",
|
|
341
|
+
)
|
|
307
342
|
system: str | None = Field(
|
|
308
343
|
default=None,
|
|
309
344
|
description="Id of a system declared in the registry, for datasets that belong to one",
|
|
@@ -595,7 +630,16 @@ class Provenance(BaseModel):
|
|
|
595
630
|
)
|
|
596
631
|
mirror: str | None = Field(
|
|
597
632
|
default=None,
|
|
598
|
-
description=
|
|
633
|
+
description=(
|
|
634
|
+
"Mirror object that served these bytes in place of the source (ADR 0030, ADR 0039)"
|
|
635
|
+
),
|
|
636
|
+
)
|
|
637
|
+
credentials: list[str] = Field(
|
|
638
|
+
default_factory=list,
|
|
639
|
+
description=(
|
|
640
|
+
"Environment variables the source requires to fetch this file, never their values "
|
|
641
|
+
"(ADR 0039); empty for an anonymous source"
|
|
642
|
+
),
|
|
599
643
|
)
|
|
600
644
|
|
|
601
645
|
@property
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
from collections.abc import Sequence
|
|
5
6
|
from datetime import UTC, datetime
|
|
6
7
|
from pathlib import Path
|
|
7
8
|
|
|
@@ -14,7 +15,11 @@ SIDECAR_SUFFIX = ".provenance.json"
|
|
|
14
15
|
|
|
15
16
|
|
|
16
17
|
def record(
|
|
17
|
-
dataset: Dataset,
|
|
18
|
+
dataset: Dataset,
|
|
19
|
+
asset: Asset,
|
|
20
|
+
path: Path,
|
|
21
|
+
partial: PartialFetch | None = None,
|
|
22
|
+
transformations: Sequence[str] = (),
|
|
18
23
|
) -> Provenance:
|
|
19
24
|
"""Build a provenance record for a file that was just fetched to ``path``.
|
|
20
25
|
|
|
@@ -26,6 +31,9 @@ def record(
|
|
|
26
31
|
Its index, ranges, selectors, and object identity are recorded
|
|
27
32
|
alongside the checksum of the local file, which still covers exactly
|
|
28
33
|
these bytes.
|
|
34
|
+
transformations: How the adapter's ``fetch`` changed the bytes the
|
|
35
|
+
source sent (``Provider.transformations``), recorded after any
|
|
36
|
+
partial-fetch entry.
|
|
29
37
|
|
|
30
38
|
Returns:
|
|
31
39
|
The record to write beside ``path``.
|
|
@@ -39,13 +47,14 @@ def record(
|
|
|
39
47
|
size=path.stat().st_size,
|
|
40
48
|
license=dataset.license,
|
|
41
49
|
usdata_version=__version__,
|
|
42
|
-
transformations=[] if partial is None else [partial.describe()],
|
|
50
|
+
transformations=[*([] if partial is None else [partial.describe()]), *transformations],
|
|
43
51
|
index_url=None if partial is None else partial.index_url,
|
|
44
52
|
index_checksum=None if partial is None else partial.index_checksum,
|
|
45
53
|
ranges=[] if partial is None else list(partial.ranges),
|
|
46
54
|
selectors=[] if partial is None else list(partial.selectors),
|
|
47
55
|
object_size=None if partial is None else partial.object_size,
|
|
48
56
|
object_etag=None if partial is None else partial.object_etag,
|
|
57
|
+
credentials=[] if dataset.credentials is None else list(dataset.credentials.variables),
|
|
49
58
|
)
|
|
50
59
|
|
|
51
60
|
|
|
@@ -5,11 +5,21 @@ inside this repository or outside it, is written against. ``Provider`` is the
|
|
|
5
5
|
interface, ``HttpProvider`` the client lifecycle HTTP-backed adapters inherit,
|
|
6
6
|
``QueryError`` the refusal every adapter raises, and the coercions come from
|
|
7
7
|
``usdata.providers.params`` so parameter models read the same everywhere.
|
|
8
|
+
``Credentials`` carries the values a source that needs a key receives, and
|
|
9
|
+
``MissingCredentials`` is the refusal when one is unset (ADR 0039).
|
|
8
10
|
``usdata.testing`` checks an adapter against this contract. See
|
|
9
11
|
[ADR 0027](https://github.com/jakeryderv/usdata/blob/main/docs/adr/0027-provider-contract.md).
|
|
10
12
|
"""
|
|
11
13
|
|
|
12
|
-
from usdata.providers.base import
|
|
14
|
+
from usdata.providers.base import (
|
|
15
|
+
MissingCredentials,
|
|
16
|
+
NotImplementedProvider,
|
|
17
|
+
Provider,
|
|
18
|
+
QueryError,
|
|
19
|
+
adapter_class,
|
|
20
|
+
load_adapter,
|
|
21
|
+
)
|
|
22
|
+
from usdata.providers.credentials import Credentials
|
|
13
23
|
from usdata.providers.http import HttpProvider
|
|
14
24
|
from usdata.providers.params import (
|
|
15
25
|
OptionalUpperStrList,
|
|
@@ -23,13 +33,16 @@ from usdata.providers.params import (
|
|
|
23
33
|
)
|
|
24
34
|
|
|
25
35
|
__all__ = [
|
|
36
|
+
"Credentials",
|
|
26
37
|
"HttpProvider",
|
|
38
|
+
"MissingCredentials",
|
|
27
39
|
"NotImplementedProvider",
|
|
28
40
|
"OptionalUpperStrList",
|
|
29
41
|
"Provider",
|
|
30
42
|
"QueryError",
|
|
31
43
|
"StrList",
|
|
32
44
|
"UpperStrList",
|
|
45
|
+
"adapter_class",
|
|
33
46
|
"choice",
|
|
34
47
|
"flag",
|
|
35
48
|
"int_list",
|
|
@@ -14,6 +14,7 @@ from pydantic import BaseModel, ValidationError
|
|
|
14
14
|
from pydantic_core import ErrorDetails
|
|
15
15
|
|
|
16
16
|
from usdata.models import Asset, Dataset, PartialFetch, Place, Provenance, Query
|
|
17
|
+
from usdata.providers.credentials import Credentials
|
|
17
18
|
|
|
18
19
|
QueryField = Literal["text", "bbox", "variables", "time"]
|
|
19
20
|
Params = TypeVar("Params", bound=BaseModel)
|
|
@@ -37,6 +38,36 @@ class QueryError(ValueError):
|
|
|
37
38
|
"""The query cannot be satisfied by this dataset (missing or unsupported constraints)."""
|
|
38
39
|
|
|
39
40
|
|
|
41
|
+
class MissingCredentials(QueryError):
|
|
42
|
+
"""A source that needs credentials was about to be contacted without them.
|
|
43
|
+
|
|
44
|
+
Raised when the adapter is built, so no request is ever sent without them.
|
|
45
|
+
``missing`` names the unset variables, and the message says where to get a key.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
def __init__(self, dataset: Dataset, missing: list[str], note: str = "") -> None:
|
|
49
|
+
self.dataset_id = dataset.id
|
|
50
|
+
self.missing = missing
|
|
51
|
+
signup = dataset.credentials.signup if dataset.credentials else None
|
|
52
|
+
message = f"{dataset.id} needs {', '.join(missing)} set in the environment"
|
|
53
|
+
if signup:
|
|
54
|
+
message = f"{message}; request a key at {signup}"
|
|
55
|
+
super().__init__(f"{message}; {note}" if note else message)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def required_credentials(dataset: Dataset, credentials: Credentials | None) -> Credentials:
|
|
59
|
+
"""``credentials``, checked against what ``dataset`` declares; empty for an anonymous source.
|
|
60
|
+
|
|
61
|
+
Raises:
|
|
62
|
+
MissingCredentials: A declared variable has no value in ``credentials``.
|
|
63
|
+
"""
|
|
64
|
+
given = credentials if credentials is not None else Credentials()
|
|
65
|
+
declared = [] if dataset.credentials is None else dataset.credentials.variables
|
|
66
|
+
if missing := [name for name in declared if not given.get(name)]:
|
|
67
|
+
raise MissingCredentials(dataset, missing)
|
|
68
|
+
return given
|
|
69
|
+
|
|
70
|
+
|
|
40
71
|
def to_utc(value: datetime) -> datetime:
|
|
41
72
|
"""Apply the shared time policy: naive datetimes mean UTC, aware ones convert to it."""
|
|
42
73
|
return value.replace(tzinfo=value.tzinfo or UTC).astimezone(UTC)
|
|
@@ -115,14 +146,34 @@ class Provider(ABC):
|
|
|
115
146
|
subclassing the parent's model.
|
|
116
147
|
"""
|
|
117
148
|
|
|
149
|
+
transformations: ClassVar[tuple[str, ...]] = ()
|
|
150
|
+
"""How ``fetch`` changes the bytes the source sent, one line each, or nothing for exact bytes.
|
|
151
|
+
|
|
152
|
+
The core records these in every provenance sidecar the adapter's fetches
|
|
153
|
+
write, beside any partial-fetch entry. Only an adapter whose source cannot be
|
|
154
|
+
pinned as received declares one: a response that echoes the request's
|
|
155
|
+
credentials, or differs between identical requests (ADR 0039).
|
|
156
|
+
"""
|
|
157
|
+
|
|
118
158
|
def __init_subclass__(cls, **kwargs: Any) -> None:
|
|
119
159
|
"""Derive ``accepted_params`` from a declared model, so one declaration feeds both."""
|
|
120
160
|
super().__init_subclass__(**kwargs)
|
|
121
161
|
if cls.params_model is not None:
|
|
122
162
|
cls.accepted_params = described_params(cls.params_model)
|
|
123
163
|
|
|
124
|
-
def __init__(self, dataset: Dataset) -> None:
|
|
164
|
+
def __init__(self, dataset: Dataset, *, credentials: Credentials | None = None) -> None:
|
|
165
|
+
"""Bind the adapter to its registry entry, refusing to exist without declared credentials.
|
|
166
|
+
|
|
167
|
+
Args:
|
|
168
|
+
dataset: The registry entry this adapter serves.
|
|
169
|
+
credentials: Values for the variables ``dataset.credentials`` declares;
|
|
170
|
+
``load_adapter`` reads them from the environment.
|
|
171
|
+
|
|
172
|
+
Raises:
|
|
173
|
+
MissingCredentials: The dataset declares a variable ``credentials`` lacks.
|
|
174
|
+
"""
|
|
125
175
|
self.dataset = dataset
|
|
176
|
+
self.credentials = required_credentials(dataset, credentials)
|
|
126
177
|
|
|
127
178
|
def __enter__(self) -> Self:
|
|
128
179
|
return self
|
|
@@ -265,8 +316,15 @@ class Provider(ABC):
|
|
|
265
316
|
)
|
|
266
317
|
|
|
267
318
|
|
|
268
|
-
def
|
|
269
|
-
"""
|
|
319
|
+
def adapter_class(dataset: Dataset) -> type[Provider]:
|
|
320
|
+
"""The Provider class a dataset's ``adapter`` dotted path names, without instantiating it.
|
|
321
|
+
|
|
322
|
+
For what a class declares, such as ``accepted_params``, which needs no
|
|
323
|
+
credentials and no transport.
|
|
324
|
+
|
|
325
|
+
Raises:
|
|
326
|
+
NotImplementedProvider: The dataset is planned and names no adapter.
|
|
327
|
+
"""
|
|
270
328
|
if dataset.adapter is None:
|
|
271
329
|
raise NotImplementedProvider(
|
|
272
330
|
f"{dataset.id} is {dataset.status.value}; no adapter exists yet"
|
|
@@ -276,4 +334,22 @@ def load_adapter(dataset: Dataset) -> Provider:
|
|
|
276
334
|
cls = getattr(module, class_name)
|
|
277
335
|
if not (isinstance(cls, type) and issubclass(cls, Provider)):
|
|
278
336
|
raise TypeError(f"{dataset.adapter} is not a Provider subclass")
|
|
279
|
-
return cls
|
|
337
|
+
return cls
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def load_adapter(dataset: Dataset) -> Provider:
|
|
341
|
+
"""Instantiate the Provider named by a dataset's ``adapter`` dotted path.
|
|
342
|
+
|
|
343
|
+
A dataset that declares credentials gets their values from the environment,
|
|
344
|
+
and one that is missing any is refused here, before a request can be made.
|
|
345
|
+
An anonymous dataset's adapter is built from the dataset alone, so an
|
|
346
|
+
adapter that predates credentials needs no change.
|
|
347
|
+
|
|
348
|
+
Raises:
|
|
349
|
+
NotImplementedProvider: The dataset is planned and names no adapter.
|
|
350
|
+
MissingCredentials: A variable the dataset declares is unset or blank.
|
|
351
|
+
"""
|
|
352
|
+
cls = adapter_class(dataset)
|
|
353
|
+
if dataset.credentials is None:
|
|
354
|
+
return cls(dataset)
|
|
355
|
+
return cls(dataset, credentials=Credentials.from_environment(dataset.credentials.variables))
|