forcingkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,231 @@
1
+ """NDBC buoy observations for model validation: standard meteorology (sea and air temperature,
2
+ pressure, wind, waves) and near-surface ADCP currents.
3
+
4
+ Files come from the NDBC data server in the order they are published: the per-month file for
5
+ months of the current year (`data/<product>/<Mon>/<id><m><yyyy>.txt.gz`), then the yearly
6
+ historical file (`data/historical/<product>/<id>h<yyyy>.txt.gz`), then the rolling 45-day
7
+ real-time file (`data/realtime2/<id>.txt` or `.adcp`). Missing-value sentinels (99, 999, 9999,
8
+ MM) become NaN. These observations validate a model; they never force it.
9
+ """
10
+
11
+ import gzip
12
+ import json
13
+ import logging
14
+ import math
15
+ import os
16
+ import re
17
+ from typing import Optional
18
+
19
+ import numpy as np
20
+ import pandas as pd
21
+ import requests
22
+
23
+ logger = logging.getLogger(__name__)
24
+
25
+ NDBC = "https://www.ndbc.noaa.gov"
26
+ PRODUCTS = ("stdmet", "adcp")
27
+ REALTIME_SUFFIX = {"stdmet": "txt", "adcp": "adcp"}
28
+ MONTH_ABBR = [
29
+ "Jan",
30
+ "Feb",
31
+ "Mar",
32
+ "Apr",
33
+ "May",
34
+ "Jun",
35
+ "Jul",
36
+ "Aug",
37
+ "Sep",
38
+ "Oct",
39
+ "Nov",
40
+ "Dec",
41
+ ]
42
+ # Column sentinels NDBC uses for "not measured".
43
+ SENTINELS = {99.0, 999.0, 9999.0, 99.00}
44
+ # stdmet columns carried through, with units.
45
+ STDMET_COLUMNS = {
46
+ "WTMP": "degC", # sea surface temperature
47
+ "ATMP": "degC", # air temperature
48
+ "PRES": "hPa", # sea level pressure
49
+ "WSPD": "m s-1", # wind speed
50
+ "WDIR": "degree from true north (direction the wind comes from)",
51
+ "GST": "m s-1",
52
+ "WVHT": "m",
53
+ "DPD": "s",
54
+ }
55
+
56
+
57
+ def month_code(month: int) -> str:
58
+ """NDBC's month code in per-month file names: 1 to 9, then a, b, c for October to December."""
59
+ return str(month) if month < 10 else "abc"[month - 10]
60
+
61
+
62
+ def candidate_urls(station: str, product: str, year: int, month: int) -> list[str]:
63
+ """URLs that may hold `product` for `station` in the given month, in the order to try."""
64
+ station = station.lower()
65
+ return [
66
+ f"{NDBC}/data/{product}/{MONTH_ABBR[month - 1]}/{station}{month_code(month)}{year}.txt.gz",
67
+ f"{NDBC}/data/historical/{product}/{station}h{year}.txt.gz",
68
+ f"{NDBC}/data/realtime2/{station.upper()}.{REALTIME_SUFFIX[product]}",
69
+ ]
70
+
71
+
72
+ def parse_ndbc_text(text: str) -> pd.DataFrame:
73
+ """Parse an NDBC whitespace-delimited file (two `#` header rows) into a frame indexed by UTC
74
+ time, sentinels and `MM` as NaN. Handles both four- and two-digit years, with or without a
75
+ minute column."""
76
+ lines = [ln for ln in text.splitlines() if ln.strip()]
77
+ if not lines or not lines[0].startswith("#"):
78
+ raise ValueError("not an NDBC data file: no header row")
79
+ names = lines[0].lstrip("#").split()
80
+ rows = [ln.split() for ln in lines if not ln.startswith("#")]
81
+ df = pd.DataFrame(rows, columns=names[: len(rows[0])] if rows else names)
82
+ year_col = names[0]
83
+ years = df[year_col].astype(int)
84
+ years = years.where(years >= 100, years + 1900)
85
+ minutes = df["mm"].astype(int) if "mm" in df else 0
86
+ df.index = pd.to_datetime(
87
+ dict(
88
+ year=years,
89
+ month=df["MM"].astype(int),
90
+ day=df["DD"].astype(int),
91
+ hour=df["hh"].astype(int),
92
+ minute=minutes,
93
+ )
94
+ )
95
+ df = df.drop(columns=[c for c in (year_col, "MM", "DD", "hh", "mm") if c in df])
96
+ df = df.replace("MM", np.nan).apply(pd.to_numeric, errors="coerce")
97
+ return df.mask(df.isin(SENTINELS))
98
+
99
+
100
+ def _get(url: str, timeout: float = 60.0) -> Optional[str]:
101
+ r = requests.get(url, timeout=timeout)
102
+ if r.status_code == 404:
103
+ return None
104
+ r.raise_for_status()
105
+ if url.endswith(".gz"):
106
+ return gzip.decompress(r.content).decode("utf-8", errors="replace")
107
+ return r.text
108
+
109
+
110
+ def _utc_naive(t) -> pd.Timestamp:
111
+ t = pd.Timestamp(t)
112
+ return t if t.tzinfo is None else t.tz_convert("UTC").tz_localize(None)
113
+
114
+
115
+ def station_position(station: str, get=_get) -> tuple[float, float]:
116
+ """(lat, lon) of an NDBC station from its station page."""
117
+ html = get(f"{NDBC}/station_page.php?station={station.lower()}") or ""
118
+ m = re.search(r"(\d+\.\d+)\s*([NS])\s+(\d+\.\d+)\s*([EW])", html)
119
+ if not m:
120
+ raise ValueError(f"No position found for NDBC station {station}")
121
+ lat = float(m.group(1)) * (1 if m.group(2) == "N" else -1)
122
+ lon = float(m.group(3)) * (1 if m.group(4) == "E" else -1)
123
+ return lat, lon
124
+
125
+
126
+ def fetch_product(
127
+ station: str, product: str, start: pd.Timestamp, end: pd.Timestamp, get=_get
128
+ ):
129
+ """`product` for `station` over [start, end]: (frame, source URLs). Each month is taken from
130
+ the first source that has it."""
131
+ frames, sources = [], []
132
+ for month_start in pd.date_range(start.normalize().replace(day=1), end, freq="MS"):
133
+ for url in candidate_urls(
134
+ station, product, month_start.year, month_start.month
135
+ ):
136
+ text = get(url)
137
+ if text is None:
138
+ continue
139
+ df = parse_ndbc_text(text)
140
+ df = df[(df.index >= start) & (df.index <= end)]
141
+ if len(df):
142
+ frames.append(df)
143
+ sources.append(url)
144
+ break
145
+ if not frames:
146
+ return pd.DataFrame(), sources
147
+ df = pd.concat(frames)
148
+ return df[~df.index.duplicated(keep="first")].sort_index(), sources
149
+
150
+
151
+ def _list(values) -> list:
152
+ return [
153
+ None if (v is None or (isinstance(v, float) and math.isnan(v))) else float(v)
154
+ for v in values
155
+ ]
156
+
157
+
158
+ def fetch_ndbc(
159
+ station: str,
160
+ start_time: str,
161
+ end_time: str,
162
+ products: tuple[str, ...] = PRODUCTS,
163
+ cache_dir: Optional[str] = None,
164
+ get=_get,
165
+ position: Optional[tuple[float, float]] = None,
166
+ ) -> dict:
167
+ """Observations at an NDBC station over [start_time, end_time] (UTC) as a JSON-ready dict.
168
+
169
+ stdmet: times plus the columns of `STDMET_COLUMNS`. adcp: times, the bin depth and the
170
+ current as eastward and northward components (m/s); NDBC reports speed in cm/s and the
171
+ direction the current flows toward, in degrees true. Missing values are null. A window that
172
+ ended more than two days ago is cached as JSON under `cache_dir`. `get` (a URL to text or
173
+ None) and `position` exist so tests can run without the network.
174
+ """
175
+ start, end = _utc_naive(start_time), _utc_naive(end_time)
176
+ unknown = set(products) - set(PRODUCTS)
177
+ if unknown:
178
+ raise ValueError(f"unknown NDBC products {sorted(unknown)}")
179
+
180
+ cache_path = None
181
+ if cache_dir:
182
+ os.makedirs(os.path.join(cache_dir, "ndbc"), exist_ok=True)
183
+ key = f"{station.lower()}_{start:%Y%m%dT%H%M}_{end:%Y%m%dT%H%M}_{'-'.join(sorted(products))}"
184
+ cache_path = os.path.join(cache_dir, "ndbc", f"{key}.json")
185
+ if os.path.exists(cache_path):
186
+ with open(cache_path) as f:
187
+ return json.load(f)
188
+
189
+ lat, lon = position if position is not None else station_position(station, get)
190
+ out: dict = {
191
+ "station_id": station.upper(),
192
+ "lat": lat,
193
+ "lon": lon,
194
+ "start_time": f"{start:%Y-%m-%dT%H:%M:%SZ}",
195
+ "end_time": f"{end:%Y-%m-%dT%H:%M:%SZ}",
196
+ "products": {},
197
+ "sources": [],
198
+ }
199
+ for product in products:
200
+ df, sources = fetch_product(station, product, start, end, get=get)
201
+ out["sources"] += sources
202
+ times = [f"{t:%Y-%m-%dT%H:%M:%SZ}" for t in df.index]
203
+ if product == "stdmet":
204
+ out["products"]["stdmet"] = {
205
+ "times": times,
206
+ "units": {c: u for c, u in STDMET_COLUMNS.items() if c in df},
207
+ **{c: _list(df[c]) for c in STDMET_COLUMNS if c in df},
208
+ }
209
+ else:
210
+ speed = df["SPD01"] / 100.0 if "SPD01" in df else pd.Series(dtype=float)
211
+ toward = (
212
+ np.deg2rad(df["DIR01"]) if "DIR01" in df else pd.Series(dtype=float)
213
+ )
214
+ out["products"]["adcp"] = {
215
+ "times": times,
216
+ "depth_m": _list(df["DEP01"]) if "DEP01" in df else [],
217
+ "u": _list(speed * np.sin(toward)),
218
+ "v": _list(speed * np.cos(toward)),
219
+ "units": {
220
+ "u": "m s-1 eastward",
221
+ "v": "m s-1 northward",
222
+ "depth_m": "m",
223
+ },
224
+ }
225
+
226
+ if cache_path and end < pd.Timestamp.utcnow().tz_localize(None) - pd.Timedelta(
227
+ days=2
228
+ ):
229
+ with open(cache_path, "w") as f:
230
+ json.dump(out, f)
231
+ return out
@@ -0,0 +1,369 @@
1
+ import warnings
2
+ import logging
3
+ import pandas as pd
4
+ import numpy as np
5
+ import xarray as xr
6
+ from typing import Optional
7
+ from scipy.spatial import Delaunay
8
+ from forcingkit import settings
9
+
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+
14
+ def get_metadata() -> dict:
15
+ return {
16
+ "id": "necofs",
17
+ "name": "NECOFS FVCOM GOM7",
18
+ "resolution_approx_m": 200.0,
19
+ "type_desc": "Unstructured Triangular Mesh",
20
+ "domain_bbox": [-77.0, 35.0, -65.0, 46.0],
21
+ }
22
+
23
+
24
+ def supports_bbox(bbox: list[float]) -> bool:
25
+ min_lon, min_lat, max_lon, max_lat = bbox
26
+ # Rough check for GOM3 bounds (NECOFS FVCOM)
27
+ if max_lat < 35.0 or min_lat > 46.0 or max_lon < -77.0 or min_lon > -65.0:
28
+ return False
29
+ return True
30
+
31
+
32
+ NECOFS_GOM7_URL = "http://www.smast.umassd.edu:8080/thredds/dodsC/models/fvcom/NECOFS/Forecasts/NECOFS_GOM7_FORECAST.nc"
33
+
34
+
35
+ def get_necofs_url(target_dt: pd.Timestamp) -> str:
36
+ """
37
+ Determine best NECOFS GOM7 URL by falling back to daily history archives.
38
+ SMAST daily archives (e.g. 2026_03_03.nc) contain [Mar 2 01:00 to Mar 3 00:00].
39
+ """
40
+ import requests
41
+
42
+ # NECOFS GOM7 daily history files contain data for the PREVIOUS day up to 00:00 of the CURRENT day.
43
+ # Therefore to get data forward-looking from target_dt, we must ALWAYS fetch the file for the NEXT day.
44
+ file_dt = target_dt.normalize() + pd.Timedelta(days=1)
45
+
46
+ date_str = file_dt.strftime("%Y_%m_%d")
47
+ history_url = f"http://www.smast.umassd.edu:8080/thredds/dodsC/models/fvcom/NECOFS/Archive/necofs_history/NECOFS_GOM7_{date_str}.nc"
48
+
49
+ try:
50
+ resp = requests.get(history_url + ".dds", timeout=5)
51
+ if resp.status_code == 200:
52
+ logger.info(f"Using historic NECOFS GOM7 archive: {history_url}")
53
+ return history_url
54
+ except requests.RequestException:
55
+ pass
56
+
57
+ logger.info("Falling back to NECOFS GOM7 rolling forecast.")
58
+ return NECOFS_GOM7_URL
59
+
60
+
61
+ # Parent-ocean delivery schema. Bump when the layout or meaning of the boundary store changes, so
62
+ # a cached store built under an older schema can never be served for a newer request.
63
+ OBC_SCHEMA = "z-v2"
64
+
65
+
66
+ def necofs_archive_file_date(t: pd.Timestamp) -> pd.Timestamp:
67
+ """Date stamp of the GOM7 daily history file that holds the record at time `t`.
68
+
69
+ File D holds (D-1 01:00, D 00:00], so the record at exactly midnight lives in the file stamped
70
+ with that same day, and every other hour in the file stamped the following day.
71
+ """
72
+ return (t - pd.Timedelta(hours=1)).normalize() + pd.Timedelta(days=1)
73
+
74
+
75
+ def z_levels_for_depth(
76
+ max_depth_m: float, spacing_m: float
77
+ ) -> tuple[np.ndarray, np.ndarray]:
78
+ """Fixed z levels covering 0 to -max_depth_m: (centres, faces), both ordered bottom to top.
79
+
80
+ The bottom face sits at or below -max_depth_m, so every wet donor column is spanned.
81
+ """
82
+ if spacing_m <= 0:
83
+ raise ValueError(f"vertical spacing must be positive, got {spacing_m}")
84
+ nz = max(1, int(np.ceil(max_depth_m / spacing_m)))
85
+ faces = -spacing_m * np.arange(nz, -1, -1, dtype=np.float64)
86
+ centres = 0.5 * (faces[:-1] + faces[1:])
87
+ return centres, faces
88
+
89
+
90
+ def sigma_to_z(
91
+ values: np.ndarray,
92
+ z_layers: np.ndarray,
93
+ z_bottom: np.ndarray,
94
+ z_targets: np.ndarray,
95
+ ) -> np.ndarray:
96
+ """Interpolate sigma-layer columns onto fixed z levels.
97
+
98
+ values, z_layers: (nsig, ny, nx), layer index 0 at the surface (FVCOM convention), z negative.
99
+ z_bottom: (ny, nx), the sea floor (-h); NaN marks land.
100
+ z_targets: (nz,) target level centres, any order.
101
+
102
+ Returns (nz, ny, nx). Between the surface and the shallowest layer centre the shallowest value
103
+ is used, and between the deepest layer centre and the sea floor the deepest value; levels below
104
+ the sea floor, and land columns, are NaN.
105
+ """
106
+ nsig = values.shape[0]
107
+ out = np.full((len(z_targets),) + values.shape[1:], np.nan, dtype=np.float64)
108
+ land = ~np.isfinite(z_bottom)
109
+ for n, zt in enumerate(z_targets):
110
+ # Number of layer centres at or above this level; layers are ordered surface first.
111
+ above = np.sum(z_layers >= zt, axis=0)
112
+ upper = np.clip(above - 1, 0, nsig - 1)
113
+ lower = np.clip(above, 0, nsig - 1)
114
+ zu = np.take_along_axis(z_layers, upper[None], axis=0)[0]
115
+ zl = np.take_along_axis(z_layers, lower[None], axis=0)[0]
116
+ vu = np.take_along_axis(values, upper[None], axis=0)[0]
117
+ vl = np.take_along_axis(values, lower[None], axis=0)[0]
118
+ span = zu - zl
119
+ w = np.where(span > 0, (zu - zt) / np.where(span > 0, span, 1.0), 0.0)
120
+ w = np.clip(w, 0.0, 1.0)
121
+ level = (1.0 - w) * vu + w * vl
122
+ level[zt < z_bottom] = np.nan
123
+ level[land] = np.nan
124
+ out[n] = level
125
+ return out
126
+
127
+
128
+ class Barycentric:
129
+ """Linear interpolation from scattered points onto fixed targets, with the simplex search and
130
+ barycentric weights computed once.
131
+
132
+ Equivalent to `LinearNDInterpolator(tri, values)(targets)`, which repeats the search for every
133
+ call: per parent hour that is one search per layer per variable (181 for LIS), each over every
134
+ target point. Values may carry leading dimensions: (..., npoints) -> (..., *shape). Targets
135
+ outside the triangulation are NaN.
136
+ """
137
+
138
+ def __init__(self, tri: Delaunay, targets: np.ndarray, shape: tuple[int, ...]):
139
+ simplex = tri.find_simplex(targets)
140
+ self.inside = simplex >= 0
141
+ s = np.where(self.inside, simplex, 0)
142
+ transform = tri.transform[s]
143
+ b = np.einsum("ijk,ik->ij", transform[:, :2], targets - transform[:, 2])
144
+ self.weights = np.column_stack([b, 1.0 - b.sum(axis=1)])
145
+ self.vertices = tri.simplices[s]
146
+ self.shape = shape
147
+
148
+ def __call__(self, values: np.ndarray) -> np.ndarray:
149
+ values = np.asarray(values, dtype=np.float64)
150
+ out = np.einsum("...pk,pk->...p", values[..., self.vertices], self.weights)
151
+ out[..., ~self.inside] = np.nan
152
+ return out.reshape(values.shape[:-1] + self.shape)
153
+
154
+
155
+ PARENT_RECORD_DIMS = {
156
+ "u": ("z", "lat", "lon"),
157
+ "v": ("z", "lat", "lon"),
158
+ "temp": ("z", "lat", "lon"),
159
+ "salt": ("z", "lat", "lon"),
160
+ "zeta": ("lat", "lon"),
161
+ }
162
+
163
+ PARENT_UNITS = {
164
+ "u": ("m s-1", "eastward_sea_water_velocity"),
165
+ "v": ("m s-1", "northward_sea_water_velocity"),
166
+ "temp": ("degree_Celsius", "sea_water_temperature"),
167
+ "salt": ("1e-3", "sea_water_practical_salinity"),
168
+ "zeta": ("m", "sea_surface_height_above_geoid"),
169
+ "h": ("m", "sea_floor_depth_below_geoid"),
170
+ }
171
+
172
+
173
+ def _parent_static(lon_axis, lat_axis, z_centres, z_faces, h_grid, attrs) -> xr.Dataset:
174
+ mask = np.isfinite(h_grid).astype(np.int8)
175
+ ds = xr.Dataset(
176
+ data_vars={
177
+ "h": (("lat", "lon"), h_grid.astype(np.float32)),
178
+ "mask": (("lat", "lon"), mask),
179
+ "z_face": (("z_face",), z_faces),
180
+ },
181
+ coords={"z": z_centres, "lat": lat_axis, "lon": lon_axis},
182
+ attrs=attrs,
183
+ )
184
+ ds["h"].attrs.update(units="m", standard_name="sea_floor_depth_below_geoid")
185
+ ds["mask"].attrs.update(
186
+ long_name="1 where NECOFS has ocean, 0 on land or outside the mesh"
187
+ )
188
+ ds["z"].attrs.update(
189
+ units="m", positive="up", axis="Z", long_name="level centre depth"
190
+ )
191
+ ds["z_face"].attrs.update(
192
+ units="m", positive="up", long_name="level faces, bottom to top"
193
+ )
194
+ ds["lat"].attrs.update(units="degrees_north", standard_name="latitude", axis="Y")
195
+ ds["lon"].attrs.update(units="degrees_east", standard_name="longitude", axis="X")
196
+ return ds
197
+
198
+
199
+ def iter_parent(
200
+ start_date: str,
201
+ duration_hours: int,
202
+ bbox: list[float],
203
+ pad_cells: int = 3,
204
+ vertical_spacing_m: float = 2.0,
205
+ ):
206
+ """NECOFS GOM7 as a parent ocean, one hourly record at a time.
207
+
208
+ Yields `("static", Dataset)` once (lon, lat, z, z_face, h, mask and the store attributes),
209
+ then `("record", time, {u, v, temp, salt, zeta})` for each of `duration_hours` hours starting
210
+ at `start_date`, in order. u, v, temp and salt are (z, lat, lon) and zeta (lat, lon) on a
211
+ regular 0.002 degree grid covering `bbox` plus `pad_cells` donor cells on every side, so the
212
+ parent brackets the child. The vertical axis is true depth: each FVCOM sigma layer is placed at
213
+ z = siglay * (h + zeta) + zeta for its column and interpolated onto fixed levels every
214
+ `vertical_spacing_m` metres, ordered bottom to top. Land, points outside the mesh and levels
215
+ below the sea floor are NaN.
216
+
217
+ Raises rather than returning a short record: an unreachable archive file, a missing hour or a
218
+ failed interpolation ends the delivery with an error.
219
+ """
220
+ import concurrent.futures
221
+
222
+ min_lon, min_lat, max_lon, max_lat = bbox
223
+ if max_lat < 35.0 or min_lat > 46.0 or max_lon < -77.0 or min_lon > -65.0:
224
+ raise ValueError(f"Bounding box {bbox} outside NECOFS domain.")
225
+
226
+ target_dt = pd.to_datetime(start_date)
227
+ if target_dt.tzinfo is not None:
228
+ target_dt = target_dt.tz_convert("UTC").tz_localize(None)
229
+
230
+ d_spacing = 0.002
231
+ pad = pad_cells * d_spacing
232
+ lon_axis = np.arange(min_lon - pad, max_lon + pad + d_spacing / 2, d_spacing)
233
+ lat_axis = np.arange(min_lat - pad, max_lat + pad + d_spacing / 2, d_spacing)
234
+ lon_grid, lat_grid = np.meshgrid(lon_axis, lat_axis)
235
+ target_pts = np.column_stack((lon_grid.ravel(), lat_grid.ravel()))
236
+ shape = lon_grid.shape
237
+
238
+ at_node = at_elem = None
239
+ h_grid = siglay_grid = None
240
+ z_centres = None
241
+ max_workers = settings.max_workers()
242
+
243
+ current_dt = target_dt
244
+ end_dt = target_dt + pd.Timedelta(hours=duration_hours - 1)
245
+ while current_dt <= end_dt:
246
+ # get_necofs_url(t) opens the file stamped t.normalize() + 1 day.
247
+ file_date = necofs_archive_file_date(current_dt)
248
+ dap_url = get_necofs_url(file_date - pd.Timedelta(days=1))
249
+ with warnings.catch_warnings():
250
+ warnings.simplefilter("ignore")
251
+ ds_raw = xr.open_dataset(dap_url, engine="pydap", decode_times=False)
252
+ drop_vars = [v for v in ["Itime", "Itime2"] if v in ds_raw.variables]
253
+ ds = xr.decode_cf(ds_raw.drop_vars(drop_vars))
254
+ ds_times = pd.DatetimeIndex(ds.time.values)
255
+ if ds_times.tz is not None:
256
+ ds_times = ds_times.tz_convert("UTC").tz_localize(None)
257
+
258
+ if at_node is None:
259
+ logger.info(
260
+ "Building NECOFS parent interpolation weights and vertical grid..."
261
+ )
262
+ at_node = Barycentric(
263
+ Delaunay(np.column_stack((ds["lon"].values, ds["lat"].values))),
264
+ target_pts,
265
+ shape,
266
+ )
267
+ at_elem = Barycentric(
268
+ Delaunay(np.column_stack((ds["lonc"].values, ds["latc"].values))),
269
+ target_pts,
270
+ shape,
271
+ )
272
+ h_grid = at_node(ds["h"].values)
273
+ siglay_grid = at_node(np.asarray(ds["siglay"].values)) # surface first
274
+ if not np.any(np.isfinite(h_grid)):
275
+ raise ValueError(f"No NECOFS ocean inside {bbox}")
276
+ z_centres, z_faces = z_levels_for_depth(
277
+ float(np.nanmax(h_grid)), vertical_spacing_m
278
+ )
279
+ yield (
280
+ "static",
281
+ _parent_static(
282
+ lon_axis,
283
+ lat_axis,
284
+ z_centres,
285
+ z_faces,
286
+ h_grid,
287
+ {
288
+ "type": "NECOFS/FVCOM GOM7 parent ocean",
289
+ "source": "UMass Dartmouth SMAST",
290
+ "schema": OBC_SCHEMA,
291
+ "pad_cells": pad_cells,
292
+ "vertical_spacing_m": vertical_spacing_m,
293
+ "requested_bbox": list(bbox),
294
+ },
295
+ ),
296
+ )
297
+
298
+ wanted = pd.date_range(current_dt, min(end_dt, ds_times[-1]), freq="1h")
299
+ idx = ds_times.get_indexer(wanted)
300
+ if len(wanted) == 0 or (idx < 0).any():
301
+ raise RuntimeError(
302
+ f"{dap_url} lacks hourly records from {current_dt} "
303
+ f"(holds {ds_times[0]} to {ds_times[-1]})"
304
+ )
305
+ logger.info(f"Extracting {len(wanted)} hours from {dap_url}...")
306
+
307
+ def process(t_idx):
308
+ snap = ds.isel(time=int(t_idx))
309
+ zeta_t = at_node(snap["zeta"].values)
310
+ z_bottom = np.where(np.isfinite(zeta_t), -h_grid, np.nan)
311
+ z_layers = siglay_grid * (h_grid + zeta_t)[None] + zeta_t[None]
312
+ fields = {
313
+ "u": at_elem(snap["u"].values),
314
+ "v": at_elem(snap["v"].values),
315
+ "temp": at_node(snap["temp"].values),
316
+ "salt": at_node(snap["salinity"].values),
317
+ }
318
+ out = {
319
+ k: sigma_to_z(v, z_layers, z_bottom, z_centres).astype(np.float32)
320
+ for k, v in fields.items()
321
+ }
322
+ out["zeta"] = zeta_t.astype(np.float32)
323
+ return out
324
+
325
+ # In batches of max_workers, so at most that many records are held at once.
326
+ with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
327
+ for b in range(0, len(idx), max_workers):
328
+ batch = idx[b : b + max_workers]
329
+ for t, out in zip(
330
+ wanted[b : b + len(batch)], executor.map(process, batch)
331
+ ):
332
+ yield ("record", t, out)
333
+
334
+ current_dt = wanted[-1] + pd.Timedelta(hours=1)
335
+
336
+
337
+ def fetch_necofs_boundary_conditions(
338
+ start_date: str,
339
+ duration_hours: int,
340
+ bbox: list[float],
341
+ pad_cells: int = 3,
342
+ vertical_spacing_m: float = 2.0,
343
+ ) -> Optional[xr.Dataset]:
344
+ """`iter_parent` assembled into one in-memory Dataset, for callers that want it whole. The
345
+ service streams the records to disk instead (`dispatcher.dispatch_obc_request`)."""
346
+ static = None
347
+ times: list = []
348
+ collected: dict[str, list[np.ndarray]] = {k: [] for k in PARENT_RECORD_DIMS}
349
+ for item in iter_parent(
350
+ start_date, duration_hours, bbox, pad_cells, vertical_spacing_m
351
+ ):
352
+ if item[0] == "static":
353
+ static = item[1]
354
+ continue
355
+ _, t, rec = item
356
+ times.append(t)
357
+ for k in collected:
358
+ collected[k].append(rec[k])
359
+ if static is None:
360
+ return None
361
+ ds_out = static.assign(
362
+ {
363
+ k: (("time",) + dims, np.stack(collected[k]))
364
+ for k, dims in PARENT_RECORD_DIMS.items()
365
+ }
366
+ ).assign_coords(time=times)
367
+ for name, (unit, standard_name) in PARENT_UNITS.items():
368
+ ds_out[name].attrs.update(units=unit, standard_name=standard_name)
369
+ return ds_out
@@ -0,0 +1,87 @@
1
+ import os
2
+ import logging
3
+ import requests
4
+ import json
5
+ from datetime import datetime
6
+ from forcingkit import settings
7
+
8
+ logger = logging.getLogger(__name__)
9
+
10
+
11
+ def fetch_noaa_tide_data(
12
+ station_id: str,
13
+ start_time: str,
14
+ end_time: str,
15
+ cache_dir: str = os.path.join(
16
+ settings.cache_dir(),
17
+ "noaa",
18
+ ),
19
+ cache_bust: bool = False,
20
+ ) -> dict:
21
+ """
22
+ Fetches water level data from NOAA CO-OPS API for a specific station and time window.
23
+ Supports caching to avoid redundant API calls.
24
+
25
+ Args:
26
+ station_id: NOAA station ID (e.g., '8516945' for Kings Point)
27
+ start_time: ISO8601 string (YYYY-MM-DDTHH:MM:SSZ)
28
+ end_time: ISO8601 string (YYYY-MM-DDTHH:MM:SSZ)
29
+ cache_dir: Local directory for caching JSON responses
30
+ """
31
+ os.makedirs(cache_dir, exist_ok=True)
32
+
33
+ # Generate a cache key based on station and time
34
+ cache_key = f"{station_id}_{start_time}_{end_time}".replace(":", "-")
35
+ cache_path = os.path.join(cache_dir, f"{cache_key}.json")
36
+
37
+ if not cache_bust and os.path.exists(cache_path):
38
+ logger.info(f"Cache hit for NOAA station {station_id}: {cache_path}")
39
+ with open(cache_path, "r") as f:
40
+ return json.load(f)
41
+
42
+ logger.info(
43
+ f"Fetching NOAA tide data for station {station_id} ({start_time} to {end_time})..."
44
+ )
45
+
46
+ # NOAA API expects yyyyMMdd HH:mm
47
+ # Note: We assume the input strings are ISO8601 UTC
48
+ s_dt = datetime.fromisoformat(start_time.replace("Z", ""))
49
+ e_dt = datetime.fromisoformat(end_time.replace("Z", ""))
50
+
51
+ begin_str = s_dt.strftime("%Y%m%d %H:%M")
52
+ end_str = e_dt.strftime("%Y%m%d %H:%M")
53
+
54
+ url = "https://api.tidesandcurrents.noaa.gov/api/prod/datagetter"
55
+ params = {
56
+ "begin_date": begin_str,
57
+ "end_date": end_str,
58
+ "station": station_id,
59
+ "product": "water_level",
60
+ "datum": "NAVD",
61
+ "units": "metric",
62
+ "time_zone": "gmt",
63
+ "format": "json",
64
+ "application": "lhzn_coastal_sim",
65
+ }
66
+
67
+ try:
68
+ response = requests.get(url, params=params, timeout=30)
69
+ response.raise_for_status()
70
+ data = response.json()
71
+
72
+ if "error" in data:
73
+ logger.warning(
74
+ f"NOAA API returned error: {data['error'].get('message', 'Unknown error')}"
75
+ )
76
+ # Return a dummy fallback or raise? For now, we'll raise to let dispatcher handle it
77
+ raise RuntimeError(f"NOAA API Error: {data['error'].get('message')}")
78
+
79
+ # Cache the successful response
80
+ with open(cache_path, "w") as f:
81
+ json.dump(data, f, indent=4)
82
+
83
+ return data
84
+
85
+ except Exception as e:
86
+ logger.error(f"Failed to fetch NOAA tide data: {e}")
87
+ raise e