usdata 0.24.0__tar.gz → 0.26.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. {usdata-0.24.0 → usdata-0.26.0}/PKG-INFO +2 -2
  2. {usdata-0.24.0 → usdata-0.26.0}/README.md +1 -1
  3. {usdata-0.24.0 → usdata-0.26.0}/pyproject.toml +1 -1
  4. {usdata-0.24.0 → usdata-0.26.0}/pyproject.toml.orig +1 -1
  5. usdata-0.26.0/src/usdata/_aqs.py +51 -0
  6. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_fetch.py +1 -1
  7. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_hurdat2.py +8 -1
  8. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/app.py +15 -5
  9. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/data/registry.yaml +82 -18
  10. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/doctor.py +38 -5
  11. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/models.py +45 -1
  12. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/provenance.py +11 -2
  13. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/__init__.py +14 -1
  14. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/base.py +80 -4
  15. usdata-0.26.0/src/usdata/providers/credentials.py +84 -0
  16. usdata-0.26.0/src/usdata/providers/epa/__init__.py +1 -0
  17. usdata-0.26.0/src/usdata/providers/epa/aqs.py +308 -0
  18. usdata-0.26.0/src/usdata/providers/http.py +139 -0
  19. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/goes.py +33 -19
  20. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/hurdat2.py +32 -2
  21. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/pull.py +43 -9
  22. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/readers.py +22 -3
  23. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/testing.py +199 -3
  24. usdata-0.24.0/src/usdata/providers/http.py +0 -44
  25. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/__init__.py +0 -0
  26. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/__main__.py +0 -0
  27. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_files.py +0 -0
  28. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_grib.py +0 -0
  29. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_netcdf.py +0 -0
  30. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_progress.py +0 -0
  31. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/_radar.py +0 -0
  32. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cache.py +0 -0
  33. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cache_ops.py +0 -0
  34. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cite.py +0 -0
  35. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/__init__.py +0 -0
  36. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/cache.py +0 -0
  37. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/cite.py +0 -0
  38. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/doctor.py +0 -0
  39. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/inspect.py +0 -0
  40. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/cli/progress.py +0 -0
  41. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/data/nexrad_sites.csv +0 -0
  42. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/data/places.csv +0 -0
  43. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/data/places.sources.json +0 -0
  44. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/inspect.py +0 -0
  45. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/manifest.py +0 -0
  46. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/mirror.py +0 -0
  47. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/protocols/__init__.py +0 -0
  48. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/protocols/erddap.py +0 -0
  49. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/protocols/http.py +0 -0
  50. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/protocols/listing.py +0 -0
  51. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/protocols/s3.py +0 -0
  52. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/fema/__init__.py +0 -0
  53. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/fema/declarations.py +0 -0
  54. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/__init__.py +0 -0
  55. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/coastwatch.py +0 -0
  56. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/coops.py +0 -0
  57. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/gfs.py +0 -0
  58. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/ghcnd.py +0 -0
  59. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/glm.py +0 -0
  60. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/grib_index.py +0 -0
  61. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/gsom.py +0 -0
  62. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/gsoy.py +0 -0
  63. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/hrrr.py +0 -0
  64. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/ibtracs.py +0 -0
  65. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/lcd.py +0 -0
  66. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/mrms.py +0 -0
  67. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/nbm.py +0 -0
  68. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/nexrad.py +0 -0
  69. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/nexrad_level3.py +0 -0
  70. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/normals.py +0 -0
  71. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/nws_vtec.py +0 -0
  72. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/rap.py +0 -0
  73. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/sites.py +0 -0
  74. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/spc.py +0 -0
  75. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/noaa/storm_events.py +0 -0
  76. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/params.py +0 -0
  77. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/usgs/__init__.py +0 -0
  78. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/usgs/daily.py +0 -0
  79. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/providers/usgs/earthquakes.py +0 -0
  80. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/py.typed +0 -0
  81. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/query.py +0 -0
  82. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/registry.py +0 -0
  83. {usdata-0.24.0 → usdata-0.26.0}/src/usdata/selection.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: usdata
3
- Version: 0.24.0
3
+ Version: 0.26.0
4
4
  Summary: Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data
5
5
  Keywords: noaa,usgs,open-data,scientific-data,provenance,reproducible-research,weather,climate,meteorology
6
6
  Author: Jake Van Slyke
@@ -112,7 +112,7 @@ breaking changes.
112
112
  | [docs.usdata.dev](https://docs.usdata.dev/) | How to use it: [install](https://docs.usdata.dev/install/), [getting started](https://docs.usdata.dev/getting-started/), guides, dataset notes, and reference |
113
113
  | [Severe-weather case study](https://usdata.dev/examples/severe-weather-case-study/) | One tornado, six sources, one manifest and lockfile, ending in a citation |
114
114
 
115
- Twenty-six datasets are available today and twenty more are planned, grouped
115
+ Twenty-seven datasets are available today and nineteen more are planned, grouped
116
116
  by agency and product family in the [catalog](docs/providers/README.md).
117
117
 
118
118
  ## How this compares
@@ -67,7 +67,7 @@ breaking changes.
67
67
  | [docs.usdata.dev](https://docs.usdata.dev/) | How to use it: [install](https://docs.usdata.dev/install/), [getting started](https://docs.usdata.dev/getting-started/), guides, dataset notes, and reference |
68
68
  | [Severe-weather case study](https://usdata.dev/examples/severe-weather-case-study/) | One tornado, six sources, one manifest and lockfile, ending in a citation |
69
69
 
70
- Twenty-six datasets are available today and twenty more are planned, grouped
70
+ Twenty-seven datasets are available today and nineteen more are planned, grouped
71
71
  by agency and product family in the [catalog](docs/providers/README.md).
72
72
 
73
73
  ## How this compares
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.24.0"
3
+ version = "0.26.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "usdata"
3
- version = "0.24.0"
3
+ version = "0.26.0"
4
4
  description = "Unified Python SDK and CLI for discovering, fetching, and tracking provenance of U.S. public scientific data"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -0,0 +1,51 @@
1
+ """Local reading of AQS daily-summary JSON into a tidy table, behind pandas.
2
+
3
+ A fetched ``epa:aqs-daily`` file is the canonical form the adapter writes: the
4
+ service's JSON with the echoed request left out of its header and the rows in a
5
+ fixed order (ADR 0039). Each element of ``Data`` is one monitor's summary for one
6
+ local calendar day under one pollutant standard, so the table is one row per
7
+ element, with the columns as the service names them.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ from importlib import import_module
14
+ from typing import TYPE_CHECKING, Any
15
+
16
+ from usdata.readers import MissingReaderDependency
17
+
18
+ if TYPE_CHECKING:
19
+ from usdata._fetch import FetchedAsset
20
+
21
+ DATE_COLUMNS = ("date_local", "date_of_last_change")
22
+ """Calendar dates as the service writes them, ``YYYY-MM-DD``, parsed without a timezone.
23
+
24
+ ``date_local`` is a day in the monitor's local standard time, not a UTC instant,
25
+ so it is left naive rather than given a zone it does not have.
26
+ """
27
+
28
+
29
+ def open_aqs(fetched: FetchedAsset) -> Any:
30
+ """Parse a fetched AQS daily-summary file into a pandas DataFrame, one row per summary."""
31
+ try:
32
+ pandas = import_module("pandas")
33
+ except ModuleNotFoundError as error:
34
+ if error.name != "pandas":
35
+ raise
36
+ raise MissingReaderDependency(
37
+ 'AQS reading requires pandas; install it with: pip install "usdata[pandas]" '
38
+ '(or uv add "usdata[pandas]")'
39
+ ) from error
40
+ # Reading a fetched asset is strictly local; the cached file is never rewritten.
41
+ body = json.loads(fetched.path.read_text(encoding="utf-8"))
42
+ frame = pandas.DataFrame(body.get("Data") or [])
43
+ for column in DATE_COLUMNS:
44
+ if column in frame:
45
+ frame[column] = pandas.to_datetime(frame[column], format="%Y-%m-%d")
46
+ frame.attrs["usdata"] = {
47
+ "asset_id": fetched.asset.id,
48
+ "provenance": fetched.provenance.model_dump(mode="json"),
49
+ "header": body.get("Header", []),
50
+ }
51
+ return frame
@@ -144,7 +144,7 @@ def _fetch_asset(
144
144
  adapter.fetch(asset, tmp)
145
145
  else:
146
146
  adapter.fetch_partial(asset, tmp, partial)
147
- prov = provenance.record(dataset, asset, tmp, partial)
147
+ prov = provenance.record(dataset, asset, tmp, partial, adapter.transformations)
148
148
  if asset.checksum and prov.checksum != asset.checksum:
149
149
  raise ChecksumMismatch(f"{asset.id}: expected {asset.checksum}, got {prov.checksum}")
150
150
  # A crash between replacements leaves a detectable mismatch, never a trusted partial file.
@@ -160,7 +160,14 @@ def open_hurdat2(fetched: FetchedAsset) -> Any:
160
160
  '(or uv add "usdata[pandas]")'
161
161
  ) from error
162
162
  # Reading a fetched asset is strictly local; the source file is never rewritten.
163
- columns = parse(fetched.path.read_text(encoding="utf-8"))
163
+ try:
164
+ columns = parse(fetched.path.read_text(encoding="utf-8"))
165
+ except Hurdat2FormatError as error:
166
+ # NHC typos come and go between revisions, so name the way to pick another.
167
+ raise Hurdat2FormatError(
168
+ f"{fetched.asset.id}: {error}; an upstream typo is usually absent from the "
169
+ "neighbouring revisions, so select one with the 'revision' parameter"
170
+ ) from error
164
171
  data: dict[str, Any] = {
165
172
  name: pandas.array(columns[name], dtype="string") for name in TEXT_COLUMNS
166
173
  }
@@ -21,7 +21,7 @@ from usdata.cli.inspect import inspect
21
21
  from usdata.cli.progress import progress
22
22
  from usdata.manifest import lockfile_path
23
23
  from usdata.models import READER_EXTRAS_TEXT, Dataset, Status, describe_duration
24
- from usdata.providers import load_adapter
24
+ from usdata.providers import adapter_class, load_adapter
25
25
  from usdata.providers.base import NotImplementedProvider
26
26
  from usdata.pull import EmptySource, ManifestChanged, Plan, UnknownDatasets, UpstreamChanged
27
27
  from usdata.pull import plan as plan_manifest
@@ -260,8 +260,7 @@ def info(
260
260
  )
261
261
  if ds.status is not Status.AVAILABLE:
262
262
  return # Planned entries have no adapter or usage metadata to show.
263
- with load_adapter(ds) as adapter:
264
- declared = dict(adapter.accepted_params)
263
+ declared = dict(adapter_class(ds).accepted_params)
265
264
  if declared:
266
265
  typer.echo(" params:")
267
266
  width = max(len(name) for name in declared)
@@ -285,6 +284,9 @@ def _echo_usage(ds: Dataset) -> None:
285
284
  typer.echo(f" selection: {ds.selection}")
286
285
  if ds.inputs:
287
286
  typer.echo(f" inputs: {ds.inputs}")
287
+ if ds.credentials:
288
+ typer.echo(f" credentials: {', '.join(ds.credentials.variables)} (environment)")
289
+ typer.echo(f" key: {ds.credentials.signup}")
288
290
  if ds.examples:
289
291
  typer.echo(f" examples: {', '.join(ds.examples)}")
290
292
 
@@ -528,13 +530,21 @@ def pull(
528
530
  else:
529
531
  mode = "restored from" if result.from_lockfile else "wrote"
530
532
  typer.echo(f"{len(result.fetched)} asset(s); {mode} {result.lockfile_path}", err=True)
531
- if result.mirrored:
533
+ changed = len(result.mirrored) - len(result.unchecked)
534
+ if changed:
532
535
  typer.secho(
533
- f"{len(result.mirrored)} asset(s) changed upstream and were restored from the "
536
+ f"{changed} asset(s) changed upstream and were restored from the "
534
537
  "mirror; the lockfile still pins the source. Pass --update to accept new bytes.",
535
538
  err=True,
536
539
  fg="yellow",
537
540
  )
541
+ if result.unchecked:
542
+ typer.secho(
543
+ f"{len(result.unchecked)} asset(s) were restored from the mirror without asking "
544
+ "their source, whose credentials are not set; upstream was not checked for changes.",
545
+ err=True,
546
+ fg="yellow",
547
+ )
538
548
 
539
549
 
540
550
  @app.command()
@@ -209,30 +209,31 @@ datasets:
209
209
  system: noaa:goes-r
210
210
  domain: weather-satellites
211
211
  since: "0.8"
212
- title: GOES-R ABI CONUS Cloud and Moisture Imagery
212
+ title: GOES-R ABI Cloud and Moisture Imagery
213
213
  description: >-
214
- Single-channel CONUS Cloud and Moisture Imagery (ABI-L2-CMIPC) from
214
+ Single-channel CONUS (ABI-L2-CMIPC) and mesoscale (ABI-L2-CMIPM) imagery from
215
215
  GOES-16, 17, 18, and 19 in anonymous NOAA S3 buckets. Select an explicit
216
- satellite, channel, and scan-start interval; each asset is a complete
216
+ satellite, channel, scan-start interval, and M1 or M2 for mesoscale; each asset is a complete
217
217
  NetCDF scene with no geographic or variable subsetting.
218
- keywords: [satellite, imagery, goes, abi, clouds, infrared, reflectance, netcdf, conus]
218
+ keywords: [satellite, imagery, goes, abi, clouds, infrared, reflectance, netcdf, conus, mesoscale]
219
219
  protocol: s3
220
220
  homepage: https://registry.opendata.aws/noaa-goes/
221
221
  license: US Government Work (public domain)
222
222
  temporal_extent: { start: "2017-02-28T00:00:00Z" }
223
223
  capabilities: { spatial_subset: false, temporal_subset: true, variable_subset: false }
224
- summary: GOES CONUS imagery
224
+ summary: GOES CONUS and mesoscale imagery
225
225
  formats: [NetCDF4]
226
- selection: Whole single-channel CONUS scenes by inclusive UTC scan-start time
227
- inputs: Satellite, channel, and both timestamps
226
+ selection: Whole single-channel scenes by inclusive UTC scan-start time and explicit mesoscale sector
227
+ inputs: Satellite, channel, both timestamps; product and sector for mesoscale
228
228
  reader: netcdf
229
229
  guide: docs/providers/noaa-goes.md
230
230
  examples:
231
231
  - examples/goes-imagery/example.ipynb
232
+ - examples/goes-mesoscale/example.ipynb
232
233
  - examples/event-context/example.ipynb
233
234
  resolution:
234
235
  spatial: "0.5 km to 2 km at nadir, by ABI band"
235
- temporal: "One CONUS scan every 5 minutes on average"
236
+ temporal: "CONUS every 5 minutes; two mesoscale sectors every 60 seconds or one every 30 seconds"
236
237
  update_frequency: "New data is added as soon as it's available"
237
238
  citation: >-
238
239
  NOAA Geostationary Operational Environmental Satellites (GOES) 16, 17, 18 & 19 was
@@ -423,7 +424,8 @@ datasets:
423
424
  intensity, pressure, and wind radii for Atlantic (since 1851) and
424
425
  northeast/north-central Pacific (since 1949) tropical cyclones. One
425
426
  fixed-format text file per basin, revised after each season; a basin
426
- parameter picks the file and the newest revision wins. There is no
427
+ parameter picks the file and the newest revision wins unless a revision
428
+ date names another. There is no
427
429
  query interface, so dates and geographic filters are rejected and the
428
430
  local reader turns the whole file into one row per track point.
429
431
  keywords: [hurricane, tropical cyclone, best track, nhc, atlantic, pacific]
@@ -434,8 +436,8 @@ datasets:
434
436
  capabilities: { spatial_subset: false, temporal_subset: false, variable_subset: false }
435
437
  summary: Tropical cyclone best tracks
436
438
  formats: [HURDAT2 fixed-format text]
437
- selection: The newest revision of one whole basin file; filter track points locally
438
- inputs: Optional basin (atlantic or pacific); no dates or geographic filters
439
+ selection: The newest or a named revision of one whole basin file; filter track points locally
440
+ inputs: Optional basin (atlantic or pacific) and revision date; no dates or geographic filters
439
441
  reader: pandas
440
442
  guide: docs/providers/noaa-hurdat2.md
441
443
  examples:
@@ -1449,18 +1451,80 @@ datasets:
1449
1451
  # ---------------------------------------------------------------- EPA
1450
1452
  - id: epa:aqs-daily
1451
1453
  provider: epa
1452
- status: planned
1454
+ status: available
1453
1455
  domain: air-quality
1454
- target: later
1456
+ since: "0.26"
1455
1457
  title: Air Quality System Daily Summaries
1456
1458
  description: >-
1457
- Daily pollutant summaries (ozone, PM2.5, NO2, and others) from regulatory
1458
- monitors via the AQS Data API. Requires a free API key issued by email.
1459
- keywords: [air quality, pollution, ozone, pm2.5, monitors, aqs]
1459
+ Daily summaries of regulatory air monitoring data from EPA's Air Quality
1460
+ System through the AQS Data API: one row per monitor, local day, and
1461
+ pollutant standard, for one to five AQS parameter codes such as PM2.5
1462
+ and ozone, over named sites, a state or county, or a box. Each request
1463
+ covers at most one calendar year, so a window becomes one JSON file per
1464
+ year. Requires a free key issued by email; the response is stored in a
1465
+ canonical form without the echoed request.
1466
+ keywords: [air quality, pollution, ozone, pm2.5, monitors, aqs, smoke, epa]
1460
1467
  protocol: http
1461
- homepage: https://aqs.epa.gov/aqsweb/documents/data_api.html
1468
+ homepage: https://aqs.epa.gov/aqsweb/documents/data_api.html#daily
1462
1469
  license: US Government Work (public domain)
1463
- capabilities: { spatial_subset: true, temporal_subset: true, variable_subset: true }
1470
+ capabilities: { spatial_subset: true, temporal_subset: true, variable_subset: false }
1471
+ credentials:
1472
+ variables: [USDATA_AQS_EMAIL, USDATA_AQS_KEY]
1473
+ signup: https://aqs.epa.gov/aqsweb/documents/data_api.html#signup
1474
+ summary: Daily air pollutant summaries from regulatory monitors
1475
+ formats: [JSON]
1476
+ selection: Local days within inclusive UTC calendar dates for one to five pollutants; one file per year
1477
+ inputs: Both dates; one to five parameter codes; site ids, a state or county, or a box
1478
+ reader: pandas
1479
+ guide: docs/providers/epa-aqs-daily.md
1480
+ examples:
1481
+ - examples/wildfire-smoke/README.md
1482
+ resolution:
1483
+ spatial: Regulatory monitoring sites operated by state, local, and tribal agencies
1484
+ temporal: Daily summaries of each monitor's samples, one row per pollutant standard
1485
+ update_frequency: >-
1486
+ As monitoring agencies submit each quarter (40 CFR 58.16) and certify the previous year by
1487
+ May 1 (40 CFR 58.15); submitted values can be revised later
1488
+ latency: >-
1489
+ Agencies must submit each calendar quarter's data within 90 days after it ends (40 CFR 58.16)
1490
+ citation: >-
1491
+ U.S. Environmental Protection Agency, Air Quality System (AQS) daily summary data, AQS Data
1492
+ API, accessed via usdata
1493
+ terms: https://aqs.epa.gov/aqsweb/documents/data_api.html#terms
1494
+ variables:
1495
+ - { name: "state_code", description: "FIPS code of the state the monitor is in; 80 for Mexico, CC for Canada at border sites" }
1496
+ - { name: "county_code", description: "FIPS code of the county, parish, or independent city within the state" }
1497
+ - { name: "site_number", description: "Four-digit site number, unique within the county" }
1498
+ - { name: "parameter_code", description: "AQS code of the parameter measured" }
1499
+ - { name: "poc", description: "Parameter occurrence code distinguishing instruments measuring the same parameter at one site" }
1500
+ - { name: "latitude", units: "degrees_north", description: "Site latitude, WGS84" }
1501
+ - { name: "longitude", units: "degrees_east", description: "Site longitude, WGS84" }
1502
+ - { name: "datum", description: "Datum of the coordinates, always WGS84" }
1503
+ - { name: "parameter", description: "Name of the parameter measured" }
1504
+ - { name: "sample_duration_code", description: "Code of the sample duration" }
1505
+ - { name: "sample_duration", description: "Averaging period: observed, such as 1 HOUR or 24 HOUR, or calculated, such as 24-HR BLK AVG" }
1506
+ - { name: "pollutant_standard", description: "National ambient air quality standard the row's statistics are calculated for; empty for none" }
1507
+ - { name: "date_local", description: "Day the sample was taken, in local standard time" }
1508
+ - { name: "units_of_measure", description: "Units of every statistic on the row" }
1509
+ - { name: "event_type", description: "Whether exceptional-event data are included: No Events, Events Included, Events Excluded, or Concurred Events Excluded" }
1510
+ - { name: "observation_count", description: "Number of observations in the averaging period" }
1511
+ - { name: "observation_percent", units: "percent", description: "Share of scheduled values for the day that were reported" }
1512
+ - { name: "validity_indicator", description: "Y where the value meets all completeness criteria" }
1513
+ - { name: "arithmetic_mean", description: "Mean of the day's values, in units_of_measure" }
1514
+ - { name: "first_max_value", description: "Highest value at the row's duration or standard, in units_of_measure" }
1515
+ - { name: "first_max_hour", description: "Hour of the day, 24-hour local standard time, of the highest value" }
1516
+ - { name: "aqi", description: "Air Quality Index for the day, where the pollutant has one" }
1517
+ - { name: "method_code", description: "Three-digit measurement method code, unique within a parameter" }
1518
+ - { name: "method", description: "Collection and analysis method" }
1519
+ - { name: "local_site_name", description: "Site name in the operating agency's own nomenclature" }
1520
+ - { name: "site_address", description: "Approximate street address of the site" }
1521
+ - { name: "county", description: "Name of the county the site is in" }
1522
+ - { name: "state", description: "Name of the state the site is in" }
1523
+ - { name: "city", description: "Incorporated city the site is in, if any" }
1524
+ - { name: "cbsa_code", description: "Code of the core-based statistical (metropolitan) area" }
1525
+ - { name: "cbsa", description: "Name of the core-based statistical (metropolitan) area" }
1526
+ - { name: "date_of_last_change", description: "Date the underlying data were last changed in AQS" }
1527
+ adapter: usdata.providers.epa.aqs:AqsDaily
1464
1528
 
1465
1529
  # ---------------------------------------------------------------- FEMA
1466
1530
  - id: fema:nfhl
@@ -1,7 +1,8 @@
1
- """Read-only environment report: interpreter, reader extras, cache, and endpoints.
1
+ """Read-only environment report: interpreter, reader extras, cache, credentials, and endpoints.
2
2
 
3
3
  ``diagnose`` only observes. It imports optional modules, reads environment
4
- variables, stats the cache directory, and - when asked - makes one bounded
4
+ variables, stats the cache directory, reports whether each dataset that needs
5
+ credentials has them without ever printing a value, and - when asked - makes one bounded
5
6
  request per upstream host family. It never creates, moves, or repairs anything;
6
7
  the CLI prints what it finds and leaves the fixing to the reader.
7
8
  """
@@ -30,9 +31,10 @@ from usdata import __version__
30
31
  from usdata._grib import LIBRARY_HINT
31
32
  from usdata.cache import ENV_VAR, cache_dir
32
33
  from usdata.mirror import ENV_VAR as MIRROR_ENV_VAR
33
- from usdata.models import READER_EXTRAS
34
+ from usdata.models import READER_EXTRAS, Status
34
35
  from usdata.protocols import http
35
- from usdata.registry import default_registry
36
+ from usdata.providers.credentials import Credentials
37
+ from usdata.registry import Registry, default_registry
36
38
 
37
39
  READER_MODULES: dict[str, tuple[str, ...]] = {
38
40
  "pandas": ("pandas",),
@@ -86,7 +88,13 @@ def diagnose(*, network: bool = False) -> Report:
86
88
  cache, environment, endpoints. A missing optional extra is ``warn``;
87
89
  a cache directory that cannot be written is ``fail``.
88
90
  """
89
- checks = [*_runtime_checks(), *_reader_checks(), *_cache_checks(), *_environment_checks()]
91
+ checks = [
92
+ *_runtime_checks(),
93
+ *_reader_checks(),
94
+ *_cache_checks(),
95
+ *_environment_checks(),
96
+ *_credential_checks(),
97
+ ]
90
98
  if network:
91
99
  checks.extend(_endpoint_checks())
92
100
  return Report(checks=checks)
@@ -198,6 +206,31 @@ def _environment_checks() -> Iterator[Check]:
198
206
  yield Check(name=f"env:{name}", status=CheckStatus.OK, detail=value)
199
207
 
200
208
 
209
+ def _credential_checks(registry: Registry | None = None) -> Iterator[Check]:
210
+ """Whether each fetchable dataset that needs credentials has them; values are never read out.
211
+
212
+ An unset variable is a warning, not a failure: every other dataset still
213
+ works, and a locked restore can still come from the cache or the mirror.
214
+ """
215
+ for dataset in (registry or default_registry()).list():
216
+ if dataset.credentials is None or dataset.status is Status.PLANNED:
217
+ continue
218
+ names = dataset.credentials.variables
219
+ present = Credentials.from_environment(names)
220
+ if missing := [name for name in names if name not in present]:
221
+ yield Check(
222
+ name=f"credentials:{dataset.id}",
223
+ status=CheckStatus.WARN,
224
+ detail=f"{', '.join(missing)} unset; request a key at {dataset.credentials.signup}",
225
+ )
226
+ else:
227
+ yield Check(
228
+ name=f"credentials:{dataset.id}",
229
+ status=CheckStatus.OK,
230
+ detail=f"{', '.join(names)} set",
231
+ )
232
+
233
+
201
234
  def _endpoint_checks() -> Iterator[Check]:
202
235
  """One bounded request per upstream host; an unreachable host is a failure."""
203
236
  hosts = probe_hosts()
@@ -230,6 +230,37 @@ class Variable(BaseModel):
230
230
  return f"{self.name} ({self.units})" if self.units else self.name
231
231
 
232
232
 
233
+ CREDENTIAL_VARIABLE = re.compile(r"USDATA_[A-Z0-9]+(?:_[A-Z0-9]+)+")
234
+ """How a credential variable is named: ``USDATA_<SYSTEM>_<FIELD>``, such as ``USDATA_AQS_KEY``."""
235
+
236
+
237
+ class CredentialSpec(BaseModel):
238
+ """The environment variables a source needs before it can be contacted, and where to get a key.
239
+
240
+ Declared in the registry entry so the catalog, ``info``, and ``doctor`` can
241
+ say what a dataset needs, and so the core can check it before any request
242
+ (ADR 0039). Values never appear here or anywhere else the core writes.
243
+ """
244
+
245
+ variables: list[str] = Field(
246
+ min_length=1, description="Environment variables that must be set, USDATA_<SYSTEM>_<FIELD>"
247
+ )
248
+ signup: str = Field(description="https URL where the agency issues a key")
249
+
250
+ @model_validator(mode="after")
251
+ def _named(self) -> CredentialSpec:
252
+ for name in self.variables:
253
+ if not CREDENTIAL_VARIABLE.fullmatch(name):
254
+ raise ValueError(
255
+ f"credential variable {name!r} must be named USDATA_<SYSTEM>_<FIELD>"
256
+ )
257
+ if len(set(self.variables)) != len(self.variables):
258
+ raise ValueError("credentials.variables entries must be distinct")
259
+ if not self.signup.startswith("https://"):
260
+ raise ValueError("credentials.signup must be an https URL")
261
+ return self
262
+
263
+
233
264
  class Limits(BaseModel):
234
265
  """Request limits the adapter enforces, declared here and verified by the adapter tests."""
235
266
 
@@ -304,6 +335,10 @@ class Dataset(BaseModel):
304
335
  limits: Limits | None = Field(
305
336
  default=None, description="Request limits the adapter enforces, such as the longest window"
306
337
  )
338
+ credentials: CredentialSpec | None = Field(
339
+ default=None,
340
+ description="Environment variables the source needs before it can be contacted (ADR 0039)",
341
+ )
307
342
  system: str | None = Field(
308
343
  default=None,
309
344
  description="Id of a system declared in the registry, for datasets that belong to one",
@@ -595,7 +630,16 @@ class Provenance(BaseModel):
595
630
  )
596
631
  mirror: str | None = Field(
597
632
  default=None,
598
- description="Mirror object that served these bytes after the source stopped (ADR 0030)",
633
+ description=(
634
+ "Mirror object that served these bytes in place of the source (ADR 0030, ADR 0039)"
635
+ ),
636
+ )
637
+ credentials: list[str] = Field(
638
+ default_factory=list,
639
+ description=(
640
+ "Environment variables the source requires to fetch this file, never their values "
641
+ "(ADR 0039); empty for an anonymous source"
642
+ ),
599
643
  )
600
644
 
601
645
  @property
@@ -2,6 +2,7 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ from collections.abc import Sequence
5
6
  from datetime import UTC, datetime
6
7
  from pathlib import Path
7
8
 
@@ -14,7 +15,11 @@ SIDECAR_SUFFIX = ".provenance.json"
14
15
 
15
16
 
16
17
  def record(
17
- dataset: Dataset, asset: Asset, path: Path, partial: PartialFetch | None = None
18
+ dataset: Dataset,
19
+ asset: Asset,
20
+ path: Path,
21
+ partial: PartialFetch | None = None,
22
+ transformations: Sequence[str] = (),
18
23
  ) -> Provenance:
19
24
  """Build a provenance record for a file that was just fetched to ``path``.
20
25
 
@@ -26,6 +31,9 @@ def record(
26
31
  Its index, ranges, selectors, and object identity are recorded
27
32
  alongside the checksum of the local file, which still covers exactly
28
33
  these bytes.
34
+ transformations: How the adapter's ``fetch`` changed the bytes the
35
+ source sent (``Provider.transformations``), recorded after any
36
+ partial-fetch entry.
29
37
 
30
38
  Returns:
31
39
  The record to write beside ``path``.
@@ -39,13 +47,14 @@ def record(
39
47
  size=path.stat().st_size,
40
48
  license=dataset.license,
41
49
  usdata_version=__version__,
42
- transformations=[] if partial is None else [partial.describe()],
50
+ transformations=[*([] if partial is None else [partial.describe()]), *transformations],
43
51
  index_url=None if partial is None else partial.index_url,
44
52
  index_checksum=None if partial is None else partial.index_checksum,
45
53
  ranges=[] if partial is None else list(partial.ranges),
46
54
  selectors=[] if partial is None else list(partial.selectors),
47
55
  object_size=None if partial is None else partial.object_size,
48
56
  object_etag=None if partial is None else partial.object_etag,
57
+ credentials=[] if dataset.credentials is None else list(dataset.credentials.variables),
49
58
  )
50
59
 
51
60
 
@@ -5,11 +5,21 @@ inside this repository or outside it, is written against. ``Provider`` is the
5
5
  interface, ``HttpProvider`` the client lifecycle HTTP-backed adapters inherit,
6
6
  ``QueryError`` the refusal every adapter raises, and the coercions come from
7
7
  ``usdata.providers.params`` so parameter models read the same everywhere.
8
+ ``Credentials`` carries the values a source that needs a key receives, and
9
+ ``MissingCredentials`` is the refusal when one is unset (ADR 0039).
8
10
  ``usdata.testing`` checks an adapter against this contract. See
9
11
  [ADR 0027](https://github.com/jakeryderv/usdata/blob/main/docs/adr/0027-provider-contract.md).
10
12
  """
11
13
 
12
- from usdata.providers.base import NotImplementedProvider, Provider, QueryError, load_adapter
14
+ from usdata.providers.base import (
15
+ MissingCredentials,
16
+ NotImplementedProvider,
17
+ Provider,
18
+ QueryError,
19
+ adapter_class,
20
+ load_adapter,
21
+ )
22
+ from usdata.providers.credentials import Credentials
13
23
  from usdata.providers.http import HttpProvider
14
24
  from usdata.providers.params import (
15
25
  OptionalUpperStrList,
@@ -23,13 +33,16 @@ from usdata.providers.params import (
23
33
  )
24
34
 
25
35
  __all__ = [
36
+ "Credentials",
26
37
  "HttpProvider",
38
+ "MissingCredentials",
27
39
  "NotImplementedProvider",
28
40
  "OptionalUpperStrList",
29
41
  "Provider",
30
42
  "QueryError",
31
43
  "StrList",
32
44
  "UpperStrList",
45
+ "adapter_class",
33
46
  "choice",
34
47
  "flag",
35
48
  "int_list",