forcingkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- forcingkit/__init__.py +0 -0
- forcingkit/dispatcher.py +448 -0
- forcingkit/fetchers/dbofs.py +442 -0
- forcingkit/fetchers/erddap.py +142 -0
- forcingkit/fetchers/hrrr.py +72 -0
- forcingkit/fetchers/hrrr_atmosphere.py +289 -0
- forcingkit/fetchers/hycom.py +159 -0
- forcingkit/fetchers/hydrography.py +117 -0
- forcingkit/fetchers/ndbc.py +231 -0
- forcingkit/fetchers/necofs.py +369 -0
- forcingkit/fetchers/noaa.py +87 -0
- forcingkit/fetchers/nyofs.py +458 -0
- forcingkit/settings.py +72 -0
- forcingkit/zarr_stream.py +146 -0
- forcingkit-0.1.0.dist-info/METADATA +329 -0
- forcingkit-0.1.0.dist-info/RECORD +24 -0
- forcingkit-0.1.0.dist-info/WHEEL +4 -0
- forcingkit-0.1.0.dist-info/licenses/LICENSE +201 -0
- forcingkit_serve/__init__.py +0 -0
- forcingkit_serve/main.py +544 -0
- forcingkit_serve/routers/bathymetry.py +241 -0
- forcingkit_serve/routers/plotly_api.py +150 -0
- forcingkit_serve/routers/removed.py +86 -0
- forcingkit_serve/routers/viewer.py +1199 -0
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
"""HRRR as a prescribed atmosphere for an ocean model (schema `hrrr-atm-v1`).
|
|
2
|
+
|
|
3
|
+
Each hour t is taken from the one-hour forecast (f01) of the cycle initialised at t - 1 h, so every
|
|
4
|
+
record is a short forecast from the latest analysis, and the precipitation accumulation it carries
|
|
5
|
+
covers exactly the hour ending at t. Hours are chained across cycles, so any run length works.
|
|
6
|
+
|
|
7
|
+
Eight surface fields are read from `wrfsfcf01` by byte range: 10 m wind (rotated from HRRR's
|
|
8
|
+
Lambert-conformal grid axes to east and north), 2 m air temperature and specific humidity, surface
|
|
9
|
+
pressure, downwelling shortwave and longwave radiation (instantaneous at t), and the 1 h
|
|
10
|
+
accumulated precipitation, delivered as a rate centred on t. They are regridded to a regular
|
|
11
|
+
longitude-latitude grid, which is what NumericalEarth's atmosphere regridder accepts.
|
|
12
|
+
|
|
13
|
+
A missing message or hour is an error, never a shorter record.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import concurrent.futures
|
|
17
|
+
import logging
|
|
18
|
+
import os
|
|
19
|
+
import tempfile
|
|
20
|
+
import warnings
|
|
21
|
+
from typing import Iterator
|
|
22
|
+
|
|
23
|
+
import numpy as np
|
|
24
|
+
import pandas as pd
|
|
25
|
+
import xarray as xr
|
|
26
|
+
from scipy.spatial import Delaunay
|
|
27
|
+
|
|
28
|
+
from .hrrr import _fetch_s3_byte_ranges, _parse_idx, fs
|
|
29
|
+
from .necofs import Barycentric
|
|
30
|
+
|
|
31
|
+
logger = logging.getLogger(__name__)
|
|
32
|
+
|
|
33
|
+
HRRR_ATM_SCHEMA = "hrrr-atm-v1"
|
|
34
|
+
|
|
35
|
+
# One message per field in wrfsfcf01; the leading and trailing colons anchor the whole field.
|
|
36
|
+
ATM_IDX_PATTERNS = {
|
|
37
|
+
"u10": ":UGRD:10 m above ground:1 hour fcst:",
|
|
38
|
+
"v10": ":VGRD:10 m above ground:1 hour fcst:",
|
|
39
|
+
"t2m": ":TMP:2 m above ground:1 hour fcst:",
|
|
40
|
+
"q2m": ":SPFH:2 m above ground:1 hour fcst:",
|
|
41
|
+
"sp": ":PRES:surface:1 hour fcst:",
|
|
42
|
+
"apcp": ":APCP:surface:0-1 hour acc fcst:",
|
|
43
|
+
"dswrf": ":DSWRF:surface:1 hour fcst:",
|
|
44
|
+
"dlwrf": ":DLWRF:surface:1 hour fcst:",
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
RECORD_DIMS = {
|
|
48
|
+
name: ("lat", "lon")
|
|
49
|
+
for name in ("u10", "v10", "t2m", "q2m", "sp", "dswrf", "dlwrf", "prate")
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
UNITS = {
|
|
53
|
+
"u10": ("m s-1", "eastward_wind"),
|
|
54
|
+
"v10": ("m s-1", "northward_wind"),
|
|
55
|
+
"t2m": ("K", "air_temperature"),
|
|
56
|
+
"q2m": ("kg kg-1", "specific_humidity"),
|
|
57
|
+
"sp": ("Pa", "surface_air_pressure"),
|
|
58
|
+
"dswrf": ("W m-2", "surface_downwelling_shortwave_flux_in_air"),
|
|
59
|
+
"dlwrf": ("W m-2", "surface_downwelling_longwave_flux_in_air"),
|
|
60
|
+
"prate": ("kg m-2 s-1", "precipitation_flux"),
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def cycle_for_valid_time(t: pd.Timestamp) -> pd.Timestamp:
|
|
65
|
+
"""The cycle whose one-hour forecast is valid at `t`."""
|
|
66
|
+
return pd.Timestamp(t) - pd.Timedelta(hours=1)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def s3_key(cycle: pd.Timestamp) -> str:
|
|
70
|
+
return (
|
|
71
|
+
f"noaa-hrrr-bdp-pds/hrrr.{cycle.strftime('%Y%m%d')}/conus/"
|
|
72
|
+
f"hrrr.t{cycle.strftime('%H')}z.wrfsfcf01.grib2"
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def earth_relative_winds(u, v, lon, cone, lov):
|
|
77
|
+
"""Rotate grid-relative winds on a Lambert conformal grid to east and north.
|
|
78
|
+
|
|
79
|
+
alpha = cone * (lon - lov), with cone = sin(standard parallel) for a tangent cone and both
|
|
80
|
+
longitudes in degrees east in -180..180 (NCEP's convention, as in wgrib2).
|
|
81
|
+
"""
|
|
82
|
+
lon = ((np.asarray(lon) + 180.0) % 360.0) - 180.0
|
|
83
|
+
lov = ((lov + 180.0) % 360.0) - 180.0
|
|
84
|
+
alpha = np.deg2rad(cone * (lon - lov))
|
|
85
|
+
c, s = np.cos(alpha), np.sin(alpha)
|
|
86
|
+
return c * u + s * v, -s * u + c * v
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _read_message(path: str) -> xr.DataArray:
|
|
90
|
+
with warnings.catch_warnings():
|
|
91
|
+
warnings.simplefilter("ignore")
|
|
92
|
+
ds = xr.open_dataset(path, engine="cfgrib", backend_kwargs={"indexpath": ""})
|
|
93
|
+
(name,) = list(ds.data_vars)
|
|
94
|
+
return ds[name].load()
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def fetch_hour(valid_time: pd.Timestamp, workdir: str) -> dict[str, xr.DataArray]:
|
|
98
|
+
"""The eight fields valid at `valid_time`, on HRRR's native grid. Raises if any is missing."""
|
|
99
|
+
key = s3_key(cycle_for_valid_time(valid_time))
|
|
100
|
+
try:
|
|
101
|
+
with fs.open(key + ".idx", "r") as f:
|
|
102
|
+
idx = f.read()
|
|
103
|
+
except FileNotFoundError as e:
|
|
104
|
+
raise FileNotFoundError(
|
|
105
|
+
f"HRRR index missing for {valid_time}: s3://{key}.idx"
|
|
106
|
+
) from e
|
|
107
|
+
fields = {}
|
|
108
|
+
for name, pattern in ATM_IDX_PATTERNS.items():
|
|
109
|
+
ranges = _parse_idx(idx, (pattern,))
|
|
110
|
+
if len(ranges) != 1:
|
|
111
|
+
raise FileNotFoundError(
|
|
112
|
+
f"HRRR s3://{key}: expected one message for {pattern!r}, found {len(ranges)}"
|
|
113
|
+
)
|
|
114
|
+
path = os.path.join(workdir, f"{name}_{valid_time.strftime('%Y%m%d%H')}.grib2")
|
|
115
|
+
_fetch_s3_byte_ranges(key, ranges, path)
|
|
116
|
+
try:
|
|
117
|
+
fields[name] = _read_message(path)
|
|
118
|
+
finally:
|
|
119
|
+
os.remove(path)
|
|
120
|
+
vt = pd.Timestamp(fields[name]["valid_time"].values)
|
|
121
|
+
if vt != valid_time:
|
|
122
|
+
raise RuntimeError(
|
|
123
|
+
f"HRRR {name} from s3://{key} is valid at {vt}, not {valid_time}"
|
|
124
|
+
)
|
|
125
|
+
return fields
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
class HRRRRegridder:
|
|
129
|
+
"""Native HRRR grid to a regular longitude-latitude grid, with weights built once."""
|
|
130
|
+
|
|
131
|
+
def __init__(self, native_lon, native_lat, lon_axis, lat_axis):
|
|
132
|
+
lon = ((np.asarray(native_lon) + 180.0) % 360.0) - 180.0
|
|
133
|
+
lat = np.asarray(native_lat)
|
|
134
|
+
# Native cells near the target, plus a margin of a few HRRR cells (3 km, about 0.03 deg).
|
|
135
|
+
near = (
|
|
136
|
+
(lon >= lon_axis[0] - 0.1)
|
|
137
|
+
& (lon <= lon_axis[-1] + 0.1)
|
|
138
|
+
& (lat >= lat_axis[0] - 0.1)
|
|
139
|
+
& (lat <= lat_axis[-1] + 0.1)
|
|
140
|
+
)
|
|
141
|
+
if not near.any():
|
|
142
|
+
raise ValueError("target grid lies outside the HRRR domain")
|
|
143
|
+
self.near = near
|
|
144
|
+
gx, gy = np.meshgrid(lon_axis, lat_axis)
|
|
145
|
+
targets = np.column_stack((gx.ravel(), gy.ravel()))
|
|
146
|
+
self.interp = Barycentric(
|
|
147
|
+
Delaunay(np.column_stack((lon[near], lat[near]))), targets, gx.shape
|
|
148
|
+
)
|
|
149
|
+
self.native_lon = lon
|
|
150
|
+
|
|
151
|
+
def __call__(self, field) -> np.ndarray:
|
|
152
|
+
out = self.interp(np.asarray(field)[self.near])
|
|
153
|
+
if np.isnan(out).any():
|
|
154
|
+
raise RuntimeError(
|
|
155
|
+
"regridded HRRR field has points outside the native subset"
|
|
156
|
+
)
|
|
157
|
+
return out
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def target_axes(bbox, resolution_deg: float, margin_deg: float):
|
|
161
|
+
lon_axis = np.arange(
|
|
162
|
+
bbox[0] - margin_deg, bbox[2] + margin_deg + resolution_deg / 2, resolution_deg
|
|
163
|
+
)
|
|
164
|
+
lat_axis = np.arange(
|
|
165
|
+
bbox[1] - margin_deg, bbox[3] + margin_deg + resolution_deg / 2, resolution_deg
|
|
166
|
+
)
|
|
167
|
+
return lon_axis, lat_axis
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def iter_atmosphere(
|
|
171
|
+
start_time: str,
|
|
172
|
+
hours: int,
|
|
173
|
+
bbox: list[float],
|
|
174
|
+
resolution_deg: float = 0.03,
|
|
175
|
+
margin_deg: float = 0.25,
|
|
176
|
+
max_workers: int = 6,
|
|
177
|
+
) -> Iterator[tuple]:
|
|
178
|
+
"""Yield `("static", Dataset)`, then `("record", time, fields)` for every hour from
|
|
179
|
+
start - 1 h to start + hours + 1 h, so the series brackets the run with a record to spare at
|
|
180
|
+
each end. `prate` at t is the mean of the hours ending at t and at t + 1 h (a rate centred on
|
|
181
|
+
t), so one more hour is fetched past the last record.
|
|
182
|
+
"""
|
|
183
|
+
t0 = pd.Timestamp(start_time)
|
|
184
|
+
if t0.tzinfo is not None:
|
|
185
|
+
t0 = t0.tz_convert("UTC").tz_localize(None)
|
|
186
|
+
record_times = pd.date_range(
|
|
187
|
+
t0 - pd.Timedelta(hours=1), t0 + pd.Timedelta(hours=hours + 1), freq="1h"
|
|
188
|
+
)
|
|
189
|
+
fetch_times = record_times.append(
|
|
190
|
+
pd.DatetimeIndex([record_times[-1] + pd.Timedelta(hours=1)])
|
|
191
|
+
)
|
|
192
|
+
lon_axis, lat_axis = target_axes(bbox, resolution_deg, margin_deg)
|
|
193
|
+
|
|
194
|
+
regrid = None
|
|
195
|
+
previous = (
|
|
196
|
+
None # (time, regridded fields) waiting for the next hour's precipitation
|
|
197
|
+
)
|
|
198
|
+
with (
|
|
199
|
+
tempfile.TemporaryDirectory() as workdir,
|
|
200
|
+
concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor,
|
|
201
|
+
):
|
|
202
|
+
for b in range(0, len(fetch_times), max_workers):
|
|
203
|
+
batch = list(fetch_times[b : b + max_workers])
|
|
204
|
+
for t, native in zip(
|
|
205
|
+
batch, executor.map(lambda t: fetch_hour(t, workdir), batch)
|
|
206
|
+
):
|
|
207
|
+
u = native["u10"]
|
|
208
|
+
if regrid is None:
|
|
209
|
+
a = u.attrs
|
|
210
|
+
if not (
|
|
211
|
+
a.get("GRIB_gridType") == "lambert"
|
|
212
|
+
and a.get("GRIB_Latin1InDegrees")
|
|
213
|
+
== a.get("GRIB_Latin2InDegrees")
|
|
214
|
+
):
|
|
215
|
+
raise ValueError(
|
|
216
|
+
"HRRR grid is not a tangent Lambert conformal projection: "
|
|
217
|
+
f"{a.get('GRIB_gridType')}, {a.get('GRIB_Latin1InDegrees')}, "
|
|
218
|
+
f"{a.get('GRIB_Latin2InDegrees')}"
|
|
219
|
+
)
|
|
220
|
+
cone = float(np.sin(np.deg2rad(a["GRIB_Latin1InDegrees"])))
|
|
221
|
+
lov = float(a["GRIB_LoVInDegrees"])
|
|
222
|
+
grid_relative = int(a.get("GRIB_uvRelativeToGrid", 1)) == 1
|
|
223
|
+
regrid = HRRRRegridder(
|
|
224
|
+
u["longitude"].values, u["latitude"].values, lon_axis, lat_axis
|
|
225
|
+
)
|
|
226
|
+
yield (
|
|
227
|
+
"static",
|
|
228
|
+
xr.Dataset(
|
|
229
|
+
coords={"lat": lat_axis, "lon": lon_axis},
|
|
230
|
+
attrs={
|
|
231
|
+
"type": "HRRR prescribed atmosphere",
|
|
232
|
+
"source": "NOAA HRRR via noaa-hrrr-bdp-pds (wrfsfcf01)",
|
|
233
|
+
"schema": HRRR_ATM_SCHEMA,
|
|
234
|
+
"requested_bbox": list(bbox),
|
|
235
|
+
"resolution_deg": resolution_deg,
|
|
236
|
+
"margin_deg": margin_deg,
|
|
237
|
+
"start_time": t0.strftime("%Y-%m-%dT%H:%M:%S"),
|
|
238
|
+
"first_cycle": cycle_for_valid_time(
|
|
239
|
+
fetch_times[0]
|
|
240
|
+
).strftime("%Y-%m-%dT%HZ"),
|
|
241
|
+
"last_cycle": cycle_for_valid_time(
|
|
242
|
+
fetch_times[-1]
|
|
243
|
+
).strftime("%Y-%m-%dT%HZ"),
|
|
244
|
+
"wind_frame": "earth-relative (rotated from Lambert grid axes)",
|
|
245
|
+
"radiation": "instantaneous at the record time",
|
|
246
|
+
"precipitation": "rate centred on the record time: mean of the 1 h accumulations ending at t and t + 1 h",
|
|
247
|
+
},
|
|
248
|
+
),
|
|
249
|
+
)
|
|
250
|
+
ue, ve = u.values, native["v10"].values
|
|
251
|
+
if grid_relative:
|
|
252
|
+
ue, ve = earth_relative_winds(ue, ve, regrid.native_lon, cone, lov)
|
|
253
|
+
fields = {
|
|
254
|
+
"u10": regrid(ue),
|
|
255
|
+
"v10": regrid(ve),
|
|
256
|
+
"t2m": regrid(native["t2m"].values),
|
|
257
|
+
"q2m": regrid(native["q2m"].values),
|
|
258
|
+
"sp": regrid(native["sp"].values),
|
|
259
|
+
"dswrf": regrid(native["dswrf"].values),
|
|
260
|
+
"dlwrf": regrid(native["dlwrf"].values),
|
|
261
|
+
"apcp": regrid(native["apcp"].values),
|
|
262
|
+
}
|
|
263
|
+
if previous is not None:
|
|
264
|
+
pt, pf = previous
|
|
265
|
+
pf["prate"] = (pf.pop("apcp") + fields["apcp"]) / 7200.0
|
|
266
|
+
yield ("record", pt, pf)
|
|
267
|
+
previous = (t, fields)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def fetch_hrrr_atmosphere(start_time: str, hours: int, bbox: list[float], **kwargs):
|
|
271
|
+
"""`iter_atmosphere` assembled into one in-memory Dataset."""
|
|
272
|
+
static = None
|
|
273
|
+
times, records = [], []
|
|
274
|
+
for item in iter_atmosphere(start_time, hours, bbox, **kwargs):
|
|
275
|
+
if item[0] == "static":
|
|
276
|
+
static = item[1]
|
|
277
|
+
else:
|
|
278
|
+
times.append(item[1])
|
|
279
|
+
records.append(item[2])
|
|
280
|
+
assert static is not None
|
|
281
|
+
return static.assign(
|
|
282
|
+
{
|
|
283
|
+
k: (
|
|
284
|
+
("time", "lat", "lon"),
|
|
285
|
+
np.stack([r[k] for r in records]).astype(np.float32),
|
|
286
|
+
)
|
|
287
|
+
for k in RECORD_DIMS
|
|
288
|
+
}
|
|
289
|
+
).assign_coords(time=times)
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
import xarray as xr
|
|
2
|
+
import pandas as pd
|
|
3
|
+
import logging
|
|
4
|
+
from typing import Optional
|
|
5
|
+
import numpy as np
|
|
6
|
+
|
|
7
|
+
logger = logging.getLogger(__name__)
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def get_metadata() -> dict:
|
|
11
|
+
return {
|
|
12
|
+
"id": "hycom",
|
|
13
|
+
"name": "HYCOM Global",
|
|
14
|
+
"resolution_approx_m": 9000.0,
|
|
15
|
+
"type_desc": "Global regular grid",
|
|
16
|
+
"domain_bbox": [-180.0, -90.0, 180.0, 90.0],
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def supports_bbox(bbox: list[float]) -> bool:
|
|
21
|
+
return True
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _get_hycom_url(target_dt: pd.Timestamp) -> str:
|
|
25
|
+
"""
|
|
26
|
+
Returns the appropriate HYCOM OPeNDAP URL based on the target date.
|
|
27
|
+
"""
|
|
28
|
+
# expt_93.0 covers 2018-12-04 to present
|
|
29
|
+
switch_date = pd.Timestamp("2018-12-04").tz_localize(None)
|
|
30
|
+
|
|
31
|
+
if target_dt >= switch_date:
|
|
32
|
+
return "https://tds.hycom.org/thredds/dodsC/GLBy0.08/expt_93.0"
|
|
33
|
+
else:
|
|
34
|
+
# Fallback to reanalysis expt_53.X series
|
|
35
|
+
# Note: In production, this might need further refinement based on specific 53.X sub-experiments
|
|
36
|
+
return "https://tds.hycom.org/thredds/dodsC/GLBy0.08/expt_53.X"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _normalize_lons(lons: np.ndarray) -> np.ndarray:
|
|
40
|
+
"""Convert -180/180 to 0/360."""
|
|
41
|
+
return np.where(lons < 0, lons + 360, lons)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def fetch_hycom_boundary_conditions(
|
|
45
|
+
start_date: str,
|
|
46
|
+
duration_hours: int,
|
|
47
|
+
bbox: list[float],
|
|
48
|
+
) -> Optional[xr.Dataset]:
|
|
49
|
+
"""
|
|
50
|
+
Fetches continuous historical 3D ocean boundary conditions from HYCOM.
|
|
51
|
+
"""
|
|
52
|
+
start_dt = pd.to_datetime(start_date).tz_localize(None)
|
|
53
|
+
end_dt = start_dt + pd.Timedelta(hours=duration_hours)
|
|
54
|
+
|
|
55
|
+
# Check if we need to stitch multiple experiments
|
|
56
|
+
switch_date = pd.Timestamp("2018-12-04").tz_localize(None)
|
|
57
|
+
|
|
58
|
+
if start_dt < switch_date and end_dt > switch_date:
|
|
59
|
+
logger.info(
|
|
60
|
+
"Hindcast spans HYCOM experiment boundary. Stitching expt_53.X and expt_93.0..."
|
|
61
|
+
)
|
|
62
|
+
ds_old = _fetch_hycom_data(
|
|
63
|
+
start_dt, switch_date, bbox, _get_hycom_url(start_dt)
|
|
64
|
+
)
|
|
65
|
+
ds_new = _fetch_hycom_data(switch_date, end_dt, bbox, _get_hycom_url(end_dt))
|
|
66
|
+
|
|
67
|
+
if ds_old is None or ds_new is None:
|
|
68
|
+
return ds_old or ds_new
|
|
69
|
+
|
|
70
|
+
return xr.concat([ds_old, ds_new], dim="time")
|
|
71
|
+
|
|
72
|
+
dataset_url = _get_hycom_url(start_dt)
|
|
73
|
+
return _fetch_hycom_data(start_dt, end_dt, bbox, dataset_url)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _fetch_hycom_data(
|
|
77
|
+
start_dt: pd.Timestamp,
|
|
78
|
+
end_dt: pd.Timestamp,
|
|
79
|
+
bbox: list[float],
|
|
80
|
+
dataset_url: str,
|
|
81
|
+
is_ic: bool = False,
|
|
82
|
+
) -> Optional[xr.Dataset]:
|
|
83
|
+
min_lon, min_lat, max_lon, max_lat = bbox
|
|
84
|
+
|
|
85
|
+
# Normalize longitudes for HYCOM (0 to 360)
|
|
86
|
+
hycom_min_lon = min_lon if min_lon >= 0 else 360 + min_lon
|
|
87
|
+
hycom_max_lon = max_lon if max_lon >= 0 else 360 + max_lon
|
|
88
|
+
|
|
89
|
+
logger.info(f"Fetching HYCOM data from {dataset_url} for {start_dt} to {end_dt}")
|
|
90
|
+
|
|
91
|
+
try:
|
|
92
|
+
dap_url = dataset_url.replace("https://", "dap2://").replace(
|
|
93
|
+
"http://", "dap2://"
|
|
94
|
+
)
|
|
95
|
+
ds = xr.open_dataset(dap_url, engine="pydap", decode_times=False)
|
|
96
|
+
|
|
97
|
+
# HYCOM time axis is "hours since 2000-01-01 00:00:00"
|
|
98
|
+
epoch = pd.Timestamp("2000-01-01 00:00:00")
|
|
99
|
+
start_hours = (start_dt - epoch).total_seconds() / 3600.0
|
|
100
|
+
end_hours = (end_dt - epoch).total_seconds() / 3600.0
|
|
101
|
+
|
|
102
|
+
time_var = "time"
|
|
103
|
+
if is_ic:
|
|
104
|
+
ds_subset = ds.sel({time_var: start_hours}, method="nearest")
|
|
105
|
+
else:
|
|
106
|
+
# Add a small buffer to ensure we capture the boundary times
|
|
107
|
+
ds_subset = ds.sel({time_var: slice(start_hours - 0.1, end_hours + 0.1)})
|
|
108
|
+
if ds_subset[time_var].size == 0:
|
|
109
|
+
logger.error(
|
|
110
|
+
f"Requested time range out of bounds for HYCOM. Required: {start_hours} to {end_hours}."
|
|
111
|
+
)
|
|
112
|
+
return None
|
|
113
|
+
|
|
114
|
+
lon_var = "lon" if "lon" in ds.coords else "longitude"
|
|
115
|
+
lat_var = "lat" if "lat" in ds.coords else "latitude"
|
|
116
|
+
|
|
117
|
+
# Handle 0-360 wrapping if bbox crosses prime meridian
|
|
118
|
+
if hycom_min_lon > hycom_max_lon:
|
|
119
|
+
logger.info(
|
|
120
|
+
"BBox crosses 0/360 boundary, performing dual-slice and concat."
|
|
121
|
+
)
|
|
122
|
+
part1 = ds_subset.sel({lon_var: slice(hycom_min_lon - 0.1, 360.0)})
|
|
123
|
+
part2 = ds_subset.sel({lon_var: slice(0.0, hycom_max_lon + 0.1)})
|
|
124
|
+
ds_subset = xr.concat([part1, part2], dim=lon_var)
|
|
125
|
+
else:
|
|
126
|
+
ds_subset = ds_subset.sel(
|
|
127
|
+
{
|
|
128
|
+
lat_var: slice(min_lat - 0.1, max_lat + 0.1),
|
|
129
|
+
lon_var: slice(hycom_min_lon - 0.1, hycom_max_lon + 0.1),
|
|
130
|
+
}
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
logger.info("Executing OPeNDAP download for HYCOM subset...")
|
|
134
|
+
ds_subset = ds_subset.compute()
|
|
135
|
+
|
|
136
|
+
# Rename variables to canonical names if necessary
|
|
137
|
+
rename_map = {
|
|
138
|
+
"water_u": "u",
|
|
139
|
+
"water_v": "v",
|
|
140
|
+
"water_temp": "temp",
|
|
141
|
+
"salinity": "salt",
|
|
142
|
+
"surf_el": "zeta",
|
|
143
|
+
}
|
|
144
|
+
actual_rename = {
|
|
145
|
+
k: v for k, v in rename_map.items() if k in ds_subset.data_vars
|
|
146
|
+
}
|
|
147
|
+
ds_subset = ds_subset.rename(actual_rename)
|
|
148
|
+
|
|
149
|
+
# Explicitly decode the raw float time coordinate to pandas DatetimeIndex lengths
|
|
150
|
+
if time_var in ds_subset.coords:
|
|
151
|
+
ds_subset[time_var] = epoch + pd.to_timedelta(
|
|
152
|
+
ds_subset[time_var].values, unit="h"
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
return ds_subset
|
|
156
|
+
|
|
157
|
+
except Exception as e:
|
|
158
|
+
logger.error(f"Failed to fetch from HYCOM ({dataset_url}): {e}")
|
|
159
|
+
return None
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from typing import TypedDict
|
|
3
|
+
import requests
|
|
4
|
+
|
|
5
|
+
logger = logging.getLogger(__name__)
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class HeadOfTideDict(TypedDict):
|
|
9
|
+
name: str
|
|
10
|
+
lat: float
|
|
11
|
+
lon: float
|
|
12
|
+
type: str
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def find_head_of_tide(
|
|
16
|
+
lat: float, lon: float, radius_km: float = 20.0
|
|
17
|
+
) -> list[HeadOfTideDict]:
|
|
18
|
+
"""
|
|
19
|
+
Experimental function to discover head-of-tide barriers (dams, waterfalls)
|
|
20
|
+
on tidal rivers near a target coordinate using OSM Overpass API.
|
|
21
|
+
"""
|
|
22
|
+
logger.info(
|
|
23
|
+
f"Discovering Head-of-Tide boundaries for ({lat}, {lon}) within {radius_km}km"
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
# Overpass API endpoint
|
|
27
|
+
overpass_url = "http://overpass-api.de/api/interpreter"
|
|
28
|
+
|
|
29
|
+
# Simple bounding box for roughly `radius_km`
|
|
30
|
+
# 1 deg lat ~ 111 km
|
|
31
|
+
import math
|
|
32
|
+
|
|
33
|
+
lat_offset = radius_km / 111.0
|
|
34
|
+
lon_offset = radius_km / (111.0 * math.cos(math.radians(lat)))
|
|
35
|
+
|
|
36
|
+
south = lat - lat_offset
|
|
37
|
+
west = lon - lon_offset
|
|
38
|
+
north = lat + lat_offset
|
|
39
|
+
east = lon + lon_offset
|
|
40
|
+
|
|
41
|
+
# Overpass Query: looks for dams or waterfalls near tidal sections
|
|
42
|
+
# As a proxy, we look for natural=waterfall, waterway=dam, or tidal limits.
|
|
43
|
+
# Note: OSM uses 'tidal=yes' and often maps dams explicitly.
|
|
44
|
+
overpass_query = f"""
|
|
45
|
+
[out:json][timeout:25];
|
|
46
|
+
(
|
|
47
|
+
node["waterway"="dam"]({south},{west},{north},{east});
|
|
48
|
+
way["waterway"="dam"]({south},{west},{north},{east});
|
|
49
|
+
node["natural"="waterfall"]({south},{west},{north},{east});
|
|
50
|
+
);
|
|
51
|
+
out center;
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
try:
|
|
55
|
+
import time
|
|
56
|
+
|
|
57
|
+
max_retries = 3
|
|
58
|
+
for attempt in range(max_retries):
|
|
59
|
+
try:
|
|
60
|
+
response = requests.post(
|
|
61
|
+
overpass_url, data={"data": overpass_query}, timeout=30
|
|
62
|
+
)
|
|
63
|
+
response.raise_for_status()
|
|
64
|
+
break
|
|
65
|
+
except requests.exceptions.RequestException as e:
|
|
66
|
+
if attempt == max_retries - 1:
|
|
67
|
+
raise
|
|
68
|
+
logger.warning(
|
|
69
|
+
f"OSM API timeout/error (attempt {attempt + 1}/{max_retries}): {e}. Retrying..."
|
|
70
|
+
)
|
|
71
|
+
time.sleep(2)
|
|
72
|
+
|
|
73
|
+
data = response.json()
|
|
74
|
+
|
|
75
|
+
results: list[HeadOfTideDict] = []
|
|
76
|
+
for element in data.get("elements", []):
|
|
77
|
+
if element["type"] == "node":
|
|
78
|
+
lat_c, lon_c = element["lat"], element["lon"]
|
|
79
|
+
else: # "way" has center
|
|
80
|
+
lat_c, lon_c = element["center"]["lat"], element["center"]["lon"]
|
|
81
|
+
|
|
82
|
+
tags = element.get("tags", {})
|
|
83
|
+
name = tags.get("name", "Unnamed Barrier")
|
|
84
|
+
feature_type = tags.get("waterway", tags.get("natural", "barrier"))
|
|
85
|
+
|
|
86
|
+
results.append(
|
|
87
|
+
{"name": name, "lat": lat_c, "lon": lon_c, "type": feature_type}
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
# Optional: Add hardcoded well-known fallback if results are empty or specific for Portsmouth
|
|
91
|
+
if not results:
|
|
92
|
+
logger.info(
|
|
93
|
+
"No OSM barriers found, falling back to known regional HoT catalog."
|
|
94
|
+
)
|
|
95
|
+
# E.g. Central Falls Dam for Piscataqua scale
|
|
96
|
+
if 42.5 < lat < 43.5 and -71.5 < lon < -70.0:
|
|
97
|
+
results.append(
|
|
98
|
+
{
|
|
99
|
+
"name": "Central Falls Dam",
|
|
100
|
+
"lat": 43.1979,
|
|
101
|
+
"lon": -70.8732,
|
|
102
|
+
"type": "dam",
|
|
103
|
+
}
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
return results
|
|
107
|
+
except Exception as e:
|
|
108
|
+
logger.error(f"Failed to query OSM for HoT: {e}")
|
|
109
|
+
# Return fallback for demonstration if API fails to avoid breaking AED pipeline
|
|
110
|
+
return [
|
|
111
|
+
{
|
|
112
|
+
"name": "Central Falls Dam",
|
|
113
|
+
"lat": 43.1979,
|
|
114
|
+
"lon": -70.8732,
|
|
115
|
+
"type": "dam",
|
|
116
|
+
}
|
|
117
|
+
]
|