forcingkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,442 @@
1
+ """
2
+ DBOFS (NOAA Delaware Bay Operational Forecast System) fetcher.
3
+
4
+ Provides Initial Conditions (IC) and Open Boundary Conditions (OBC) from a
5
+ ROMS structured grid via OPeNDAP. Covers Delaware Bay and the adjacent
6
+ Mid-Atlantic Bight continental shelf, including offshore NJ south of 40°N.
7
+
8
+ Domain: [-76.5, 37.5, -73.0, 40.0] - domain area 8.75°², finer than the
9
+ NECOFS/MARACOOS 132°² footprint, so DBOFS wins the dispatcher ranking for
10
+ any bbox fully contained in this region.
11
+ """
12
+
13
+ import numpy as np
14
+ import xarray as xr
15
+ import pandas as pd
16
+ import logging
17
+ from typing import Optional, Tuple
18
+ from forcingkit import settings
19
+
20
+ logger = logging.getLogger(__name__)
21
+
22
+ # Abort NCEI file enumeration after this many consecutive open failures.
23
+ _NCEI_CONSECUTIVE_FAIL_LIMIT = 3
24
+
25
+
26
+ def get_metadata() -> dict:
27
+ """Returns metadata for the DBOFS system."""
28
+ return {
29
+ "id": "dbofs",
30
+ "name": "NOAA DBOFS (Delaware Bay / Offshore NJ)",
31
+ "resolution_approx_m": 100.0,
32
+ "type_desc": "Structured curvilinear ROMS grid",
33
+ "domain_bbox": [-75.875, 37.810, -73.264, 40.206],
34
+ }
35
+
36
+
37
+ def supports_bbox(bbox: list[float]) -> bool:
38
+ """Check if the requested bbox is fully within the DBOFS active mask domain."""
39
+ min_lon, min_lat, max_lon, max_lat = bbox
40
+ domain_bbox = get_metadata()["domain_bbox"]
41
+ domain_min_lon, domain_min_lat, domain_max_lon, domain_max_lat = domain_bbox
42
+
43
+ # 1. Check strict rectangle bounds
44
+ if not (
45
+ min_lon >= domain_min_lon
46
+ and max_lon <= domain_max_lon
47
+ and min_lat >= domain_min_lat
48
+ and max_lat <= domain_max_lat
49
+ ):
50
+ return False
51
+
52
+ # 2. Check curvilinear offshore empty corner (e.g. Barnegat Light Area)
53
+ # The active mask drops out completely for regions east of -74.0 and north of 39.03
54
+ if min_lon > -74.0 and min_lat > 39.03:
55
+ return False
56
+
57
+ return True
58
+
59
+
60
+ def _to_dap_url(url: str) -> str:
61
+ """Convert http(s) URL to pydap dap2:// scheme."""
62
+ return url.replace("https://", "dap2://").replace("http://", "dap2://")
63
+
64
+
65
+ def _get_dbofs_url(target_dt: pd.Timestamp) -> Tuple[str, str]:
66
+ """
67
+ Resolve DBOFS data access URL and mode.
68
+
69
+ Returns (access_mode, url_or_pattern) where access_mode is one of:
70
+ - "fmrc": single FMRC aggregated OPeNDAP URL (recent data, < 31 days)
71
+ - "ncei": NCEI THREDDS file pattern (historical data, > 31 days)
72
+ """
73
+ now = pd.Timestamp.now(tz="UTC").tz_localize(None)
74
+ age_days = (now - target_dt).total_seconds() / 86400
75
+
76
+ if 0 <= age_days <= 6:
77
+ fmrc_url = (
78
+ "https://opendap.co-ops.nos.noaa.gov/thredds/dodsC/"
79
+ "DBOFS/fmrc/Aggregated_7_day_DBOFS_Fields_Forecast_best.ncd"
80
+ )
81
+ return ("fmrc", fmrc_url)
82
+
83
+ # Historical: AWS S3 or NCEI file-per-hour. Naming convention changed 2024-09-09
84
+ # and the archive migrated to AWS S3 starting 2024-01-01.
85
+ if target_dt >= pd.Timestamp("2024-01-01"):
86
+ pattern = (
87
+ "s3://noaa-nos-ofs-pds/dbofs/netcdf/"
88
+ "{yyyy}/{mm}/{dd}/dbofs.t{cc}z.{yyyymmdd}.fields.{type}{hhh:03d}.nc"
89
+ )
90
+ return ("aws_s3", pattern)
91
+ else:
92
+ pattern = (
93
+ "https://www.ncei.noaa.gov/thredds/dodsC/model-dbofs-files/"
94
+ "{yyyy}/{mm}/nos.dbofs.fields.{type}{hhh:03d}.{yyyymmdd}.t{cc}z.nc"
95
+ )
96
+ return ("ncei", pattern)
97
+
98
+
99
+ def _enumerate_ncei_dbofs_files(
100
+ pattern: str, start_dt: pd.Timestamp, end_dt: pd.Timestamp
101
+ ) -> list[str]:
102
+ """
103
+ Enumerate NCEI DBOFS file URLs for a time range.
104
+
105
+ Picks the best 6-hourly cycle (00/06/12/18Z) and builds per-hour file URLs.
106
+ """
107
+ cycle_hours = [0, 6, 12, 18]
108
+ cycle_before = None
109
+ for ch in reversed(cycle_hours):
110
+ test_dt = start_dt.replace(hour=ch, minute=0, second=0, microsecond=0)
111
+ if test_dt <= start_dt:
112
+ cycle_before = test_dt
113
+ break
114
+
115
+ if cycle_before is None:
116
+ cycle_before = (start_dt - pd.Timedelta(days=1)).replace(
117
+ hour=18, minute=0, second=0, microsecond=0
118
+ )
119
+
120
+ files = []
121
+ current_dt = cycle_before
122
+ current_cycle_dt = cycle_before
123
+
124
+ while current_dt <= end_dt:
125
+ if (current_dt - current_cycle_dt).total_seconds() >= 6 * 3600:
126
+ current_cycle_dt = current_dt.replace(minute=0, second=0, microsecond=0)
127
+ cycle_hour = (current_cycle_dt.hour // 6) * 6
128
+ current_cycle_dt = current_cycle_dt.replace(hour=cycle_hour)
129
+
130
+ hour_offset = int((current_dt - current_cycle_dt).total_seconds() / 3600) + 1
131
+ forecast_or_nowcast = "f" if current_dt > current_cycle_dt else "n"
132
+
133
+ fmt_vars = {
134
+ "yyyy": current_cycle_dt.strftime("%Y"),
135
+ "mm": current_cycle_dt.strftime("%m"),
136
+ "dd": current_cycle_dt.strftime("%d"),
137
+ "yyyymmdd": current_cycle_dt.strftime("%Y%m%d"),
138
+ "cc": current_cycle_dt.strftime("%H"),
139
+ "hhh": hour_offset,
140
+ "type": forecast_or_nowcast,
141
+ }
142
+
143
+ files.append(pattern.format(**fmt_vars))
144
+ current_dt += pd.Timedelta(hours=1)
145
+
146
+ return files
147
+
148
+
149
+ def _open_dbofs_dataset(
150
+ access_mode: str,
151
+ url_or_pattern: str,
152
+ target_dt: pd.Timestamp,
153
+ end_dt: Optional[pd.Timestamp] = None,
154
+ ) -> Optional[xr.Dataset]:
155
+ """Open DBOFS dataset via OPeNDAP (FMRC or NCEI)."""
156
+ try:
157
+ if access_mode == "fmrc":
158
+ logger.info(f"Opening DBOFS FMRC aggregation: {url_or_pattern}")
159
+ dap_url = _to_dap_url(url_or_pattern)
160
+ ds = xr.open_dataset(dap_url, engine="pydap")
161
+
162
+ time_var = "time" if "time" in ds.coords else "ocean_time"
163
+ ds = ds.sortby(time_var)
164
+
165
+ if end_dt is None:
166
+ return ds.sel({time_var: target_dt}, method="nearest")
167
+ else:
168
+ ds_t = ds.sel({time_var: slice(target_dt, end_dt)})
169
+ if ds_t.sizes[time_var] == 0:
170
+ logger.warning(
171
+ "DBOFS FMRC: exact time range empty, fell back to nearest."
172
+ )
173
+ return ds.sel({time_var: target_dt}, method="nearest").expand_dims(
174
+ time_var
175
+ )
176
+ return ds_t
177
+
178
+ elif access_mode in ("ncei", "aws_s3"):
179
+ mode_name = "AWS S3" if access_mode == "aws_s3" else "NCEI"
180
+ logger.info(
181
+ f"Enumerating DBOFS {mode_name} files from {target_dt} to {end_dt}"
182
+ )
183
+ if end_dt is None:
184
+ end_dt = target_dt + pd.Timedelta(hours=1)
185
+
186
+ files = _enumerate_ncei_dbofs_files(url_or_pattern, target_dt, end_dt)
187
+ logger.info(f"Opening {len(files)} DBOFS {mode_name} files...")
188
+
189
+ if access_mode == "aws_s3":
190
+ try:
191
+ logger.info("Using xarray.open_mfdataset for parallel S3 access...")
192
+ ds_t = xr.open_mfdataset(
193
+ files,
194
+ engine="h5netcdf",
195
+ parallel=True,
196
+ storage_options={"anon": True},
197
+ data_vars="minimal",
198
+ coords="minimal",
199
+ compat="override",
200
+ )
201
+ return ds_t
202
+ except Exception as e:
203
+ logger.error(
204
+ f"Failed to open/concat DBOFS S3 files via mfdataset: {e}"
205
+ )
206
+ return None
207
+ else:
208
+ import concurrent.futures
209
+
210
+ datasets: list[Optional[xr.Dataset]] = [None] * len(files)
211
+ fail_counts = [0]
212
+
213
+ def _fetch_file(args):
214
+ i, f = args
215
+ # Optional short-circuit if another thread hit the failure limit
216
+ if fail_counts[0] >= _NCEI_CONSECUTIVE_FAIL_LIMIT:
217
+ return i, None
218
+ try:
219
+ if i % 10 == 0 or i == 1 or i == len(files):
220
+ logger.info(
221
+ f"[{i}/{len(files)}] Fetching/Opening DBOFS {mode_name} file: {f.split('/')[-1] if 's3' in f else f}"
222
+ )
223
+ ds_file = xr.open_dataset(_to_dap_url(f), engine="pydap")
224
+ return i, ds_file
225
+ except Exception as e:
226
+ logger.warning(
227
+ f"Failed to open DBOFS {mode_name} file {f}: {e}"
228
+ )
229
+ fail_counts[0] += 1
230
+ return i, None
231
+
232
+ max_workers = settings.max_workers()
233
+ with concurrent.futures.ThreadPoolExecutor(
234
+ max_workers=max_workers
235
+ ) as executor:
236
+ for i, ds_file in executor.map(_fetch_file, enumerate(files, 1)):
237
+ datasets[i - 1] = ds_file
238
+
239
+ # Filter out failures
240
+ opened = [ds for ds in datasets if ds is not None]
241
+
242
+ if not opened:
243
+ logger.error(f"No DBOFS {mode_name} files could be opened.")
244
+ return None
245
+
246
+ return xr.concat(opened, dim="time", join="override")
247
+
248
+ else:
249
+ logger.error(f"Unknown access_mode: {access_mode}")
250
+ return None
251
+
252
+ except Exception as e:
253
+ logger.error(f"Failed to open DBOFS dataset ({access_mode}): {e}")
254
+ return None
255
+
256
+
257
+ _VAR_CANDIDATES: dict[str, list[str]] = {
258
+ "u": ["u", "water_u", "u_eastward"],
259
+ "v": ["v", "water_v", "v_northward"],
260
+ "temp": ["temp", "water_temp", "temperature", "sea_water_temperature"],
261
+ "salt": ["salt", "salinity", "sea_water_salinity"],
262
+ "zeta": ["zeta", "sea_surface_height", "ssh"],
263
+ }
264
+
265
+
266
+ def _resolve_var(ds: xr.Dataset, role: str) -> str:
267
+ """Return the first candidate name for *role* that exists in *ds*."""
268
+ for name in _VAR_CANDIDATES[role]:
269
+ if name in ds:
270
+ return name
271
+ raise KeyError(
272
+ f"No variable found for role '{role}'. "
273
+ f"Tried: {_VAR_CANDIDATES[role]}. Available: {list(ds.data_vars)}"
274
+ )
275
+
276
+
277
+ def _c_grid_to_rho(
278
+ u_raw: np.ndarray,
279
+ v_raw: np.ndarray,
280
+ ) -> Tuple[np.ndarray, np.ndarray]:
281
+ """
282
+ Interpolate Arakawa C-grid u,v face values to rho-points by averaging
283
+ adjacent pairs. Output shape always matches input shape (boundary filled
284
+ by copy-edge extrapolation).
285
+ """
286
+ u_rho = np.empty_like(u_raw)
287
+ u_rho[..., :-1] = 0.5 * (u_raw[..., :-1] + u_raw[..., 1:])
288
+ u_rho[..., -1] = u_raw[..., -1]
289
+
290
+ v_rho = np.empty_like(v_raw)
291
+ v_rho[..., :-1, :] = 0.5 * (v_raw[..., :-1, :] + v_raw[..., 1:, :])
292
+ v_rho[..., -1, :] = v_raw[..., -1, :]
293
+
294
+ return u_rho, v_rho
295
+
296
+
297
+ def fetch_dbofs_boundary_conditions(
298
+ start_date: str, duration_hours: int, bbox: list[float]
299
+ ) -> Optional[xr.Dataset]:
300
+ """
301
+ Fetch 4D Ocean State (u, v) from NOAA DBOFS over a time range for OBC.
302
+
303
+ Output dimensions: (time, depth, eta, xi) matching the standard OBC contract.
304
+
305
+ Args:
306
+ start_date: ISO format datetime string
307
+ duration_hours: Duration of the boundary condition period
308
+ bbox: [min_lon, min_lat, max_lon, max_lat]
309
+
310
+ Returns:
311
+ xr.Dataset with dims (time, depth, eta, xi) or None if fetch fails
312
+ """
313
+ min_lon, min_lat, max_lon, max_lat = bbox
314
+
315
+ if not supports_bbox(bbox):
316
+ logger.info(f"Bounding box {bbox} outside DBOFS domain.")
317
+ return None
318
+
319
+ target_dt = pd.to_datetime(start_date)
320
+ if target_dt.tzinfo is not None:
321
+ target_dt = target_dt.tz_convert("UTC").tz_localize(None)
322
+
323
+ end_dt = target_dt + pd.Timedelta(hours=duration_hours)
324
+
325
+ access_mode, url_or_pattern = _get_dbofs_url(target_dt)
326
+ logger.info(f"Attempting DBOFS OBC ({duration_hours}h) from {access_mode.upper()}")
327
+
328
+ ds_t = _open_dbofs_dataset(access_mode, url_or_pattern, target_dt, end_dt)
329
+ if ds_t is None:
330
+ return None
331
+
332
+ try:
333
+ lon_var = "lon_rho" if "lon_rho" in ds_t else "lon"
334
+ lat_var = "lat_rho" if "lat_rho" in ds_t else "lat"
335
+
336
+ lon_arr = ds_t[lon_var].values
337
+ lat_arr = ds_t[lat_var].values
338
+
339
+ mask_var = (
340
+ "mask_rho" if "mask_rho" in ds_t else ("mask" if "mask" in ds_t else None)
341
+ )
342
+ mask_arr = ds_t[mask_var].values if mask_var else np.ones_like(lon_arr)
343
+
344
+ valid_indices = np.where(
345
+ (lon_arr >= min_lon)
346
+ & (lon_arr <= max_lon)
347
+ & (lat_arr >= min_lat)
348
+ & (lat_arr <= max_lat)
349
+ & (mask_arr == 1)
350
+ )
351
+
352
+ if len(valid_indices[0]) == 0:
353
+ logger.warning("No valid DBOFS ocean points found in bounding box.")
354
+ return None
355
+
356
+ eta_min, eta_max = int(np.min(valid_indices[0])), int(np.max(valid_indices[0]))
357
+ xi_min, xi_max = int(np.min(valid_indices[1])), int(np.max(valid_indices[1]))
358
+
359
+ # Buffer slightly for interpolation
360
+ eta_min = max(0, eta_min - 2)
361
+ eta_max = min(lon_arr.shape[0] - 1, eta_max + 2)
362
+ xi_min = max(0, xi_min - 2)
363
+ xi_max = min(lon_arr.shape[1] - 1, xi_max + 2)
364
+
365
+ slice_dict = {}
366
+ for d in list(ds_t.dims):
367
+ if "eta" in str(d):
368
+ c_max = min(eta_max + 1, ds_t.sizes[d])
369
+ slice_dict[str(d)] = slice(eta_min, c_max)
370
+ elif "xi" in str(d):
371
+ c_max = min(xi_max + 1, ds_t.sizes[d])
372
+ slice_dict[str(d)] = slice(xi_min, c_max)
373
+
374
+ ds_sub = ds_t.isel(slice_dict) # type: ignore
375
+
376
+ # Check if the bounding box yielded zero valid water points
377
+ if any(size == 0 for size in ds_sub.sizes.values()):
378
+ logger.warning(
379
+ "No valid DBOFS ocean points found in bounding box (size is 0)."
380
+ )
381
+ return None
382
+
383
+ logger.info("Executing OPeNDAP download for DBOFS OBC subset...")
384
+ ds_sub = ds_sub.compute()
385
+
386
+ u_var = _resolve_var(ds_sub, "u")
387
+ v_var = _resolve_var(ds_sub, "v")
388
+ logger.info(f"DBOFS OBC variable mapping: u={u_var}, v={v_var}")
389
+
390
+ all_dims = list(ds_sub[u_var].dims)
391
+ time_var = "time" if "time" in ds_sub.coords else "ocean_time"
392
+ sigma_dim = None
393
+ for candidate in ["s_rho", "sigma", "depth", "siglay"]:
394
+ if candidate in all_dims:
395
+ sigma_dim = candidate
396
+ break
397
+
398
+ if sigma_dim is None:
399
+ logger.error(
400
+ f"No recognized sigma dimension in DBOFS OBC. Dims: {all_dims}"
401
+ )
402
+ return None
403
+
404
+ spatial_dims = [d for d in all_dims if d != time_var and d != sigma_dim]
405
+ if len(spatial_dims) < 2:
406
+ logger.error(f"Unexpected spatial dims after subset: {spatial_dims}")
407
+ return None
408
+
409
+ u_rho, v_rho = _c_grid_to_rho(
410
+ ds_sub[u_var].values.astype(np.float32),
411
+ ds_sub[v_var].values.astype(np.float32),
412
+ )
413
+
414
+ n_sigma = u_rho.shape[1]
415
+ depths = np.linspace(-50, 0, n_sigma).astype(np.float32)
416
+ out_times = ds_sub[time_var].values
417
+ n_eta = u_rho.shape[2]
418
+ n_xi = u_rho.shape[3]
419
+
420
+ ds_out = xr.Dataset(
421
+ data_vars={
422
+ "u": (("time", "depth", "eta", "xi"), u_rho),
423
+ "v": (("time", "depth", "eta", "xi"), v_rho),
424
+ },
425
+ coords={
426
+ "time": out_times,
427
+ "depth": depths,
428
+ "eta": np.arange(n_eta, dtype=np.float32),
429
+ "xi": np.arange(n_xi, dtype=np.float32),
430
+ },
431
+ attrs={"type": "NOAA DBOFS OBC", "source": "NOAA CO-OPS"},
432
+ )
433
+
434
+ logger.info(
435
+ f"Successfully processed DBOFS OBC. "
436
+ f"Shape: u{ds_out['u'].shape}, {len(out_times)} time steps"
437
+ )
438
+ return ds_out
439
+
440
+ except Exception as e:
441
+ logger.error(f"Failed to process DBOFS OBC data: {e}")
442
+ return None
@@ -0,0 +1,142 @@
1
+ import os
2
+ import logging
3
+ import pandas as pd
4
+ from forcingkit import settings
5
+
6
+ logger = logging.getLogger(__name__)
7
+
8
+ BASE_ERDDAP_URL = "http://merlin.dms.uconn.edu:8080/erddap"
9
+
10
+
11
+ def fetch_erddap_station_profiles(
12
+ station_id: str,
13
+ start_time: str,
14
+ end_time: str,
15
+ cache_dir: str = os.path.join(
16
+ settings.cache_dir(),
17
+ "erddap",
18
+ ),
19
+ cache_bust: bool = False,
20
+ ) -> dict:
21
+ """
22
+ Fetches 3-depth temperature profiles for a given station.
23
+ Targeting LIS stations via UConn ERDDAP.
24
+
25
+ Args:
26
+ station_id: Station ID, e.g., "WLIS", "EXRX"
27
+ start_time: ISO8601 string (e.g., "YYYY-MM-DDTHH:MM:SSZ")
28
+ end_time: ISO8601 string
29
+
30
+ Returns:
31
+ dictionary with keys: "surface", "mid", "bottom", each containing a dictionary of time (ISO) -> temperature (float)
32
+ """
33
+ os.makedirs(cache_dir, exist_ok=True)
34
+
35
+ station_upper = station_id.upper()
36
+ datasets = {
37
+ "surface": f"{station_upper}_WQ_SFC",
38
+ "mid": f"{station_upper}_WQ_MID",
39
+ "bottom": f"{station_upper}_WQ_BTM",
40
+ }
41
+
42
+ # Handle irregular dataset names in LISICOS
43
+ if station_upper == "ARTG":
44
+ datasets["bottom"] = "ARTG_WQ_BTM_1"
45
+
46
+ profiles = {}
47
+
48
+ for level, dataset_name in datasets.items():
49
+ cache_key = f"{dataset_name}_{start_time}_{end_time}".replace(":", "-").replace(
50
+ "Z", ""
51
+ )
52
+ cache_path = os.path.join(cache_dir, f"{cache_key}.csv")
53
+
54
+ if not cache_bust and os.path.exists(cache_path):
55
+ logger.info(f"Cache hit for {dataset_name}: {cache_path}")
56
+ df = pd.read_csv(cache_path, index_col=0, parse_dates=True)
57
+ else:
58
+ query_url = f"{BASE_ERDDAP_URL}/tabledap/{dataset_name}.csvp?time,sea_water_temperature&time%3E={start_time}&time%3C={end_time}"
59
+
60
+ logger.info(
61
+ f"Fetching ERDDAP data for {dataset_name} ({start_time} to {end_time})..."
62
+ )
63
+ try:
64
+ df = pd.read_csv(query_url)
65
+ if df is not None and not df.empty:
66
+ # The headers will be like "time (UTC)" and "sea_water_temperature (celsius)"
67
+ # We can normalize them
68
+ df.columns = ["time", "sea_water_temperature"]
69
+ df = df.set_index("time")
70
+ df.index = pd.to_datetime(df.index, utc=True)
71
+ df.to_csv(cache_path)
72
+ except Exception as e:
73
+ logger.error(
74
+ f"Failed fetching or parsing ERDDAP for {dataset_name}: {e}"
75
+ )
76
+ df = None
77
+
78
+ if df is not None and not df.empty:
79
+ df = df.dropna(subset=["sea_water_temperature"])
80
+ # We must convert timestamp back to isoformat strings to match json expectations if returned by API
81
+ time_dict = {
82
+ ts.isoformat(): temp for ts, temp in df["sea_water_temperature"].items()
83
+ }
84
+ profiles[level] = time_dict
85
+
86
+ return profiles
87
+
88
+
89
+ KNOWN_STATIONS = {
90
+ "WLIS": {"lat": 40.96, "lon": -73.58},
91
+ "EXRX": {"lat": 40.88, "lon": -73.73},
92
+ "CLIS": {"lat": 41.14, "lon": -72.66},
93
+ "ARTG": {"lat": 41.01, "lon": -73.29},
94
+ }
95
+
96
+
97
+ def fetch_erddap_stations_in_bbox(
98
+ bbox: list[float],
99
+ start_time: str,
100
+ end_time: str,
101
+ cache_dir: str = os.path.join(
102
+ settings.cache_dir(),
103
+ "erddap",
104
+ ),
105
+ cache_bust: bool = False,
106
+ ) -> dict:
107
+ """
108
+ Finds all known LIS stations within the bounding box and fetches their profiles.
109
+ bbox: [max_lat, min_lon, min_lat, max_lon] or [min_lon, min_lat, max_lon, max_lat]
110
+ """
111
+ # Accept standard Julia order (min_lon, min_lat, max_lon, max_lat)
112
+ if len(bbox) == 4:
113
+ min_lon, min_lat, max_lon, max_lat = bbox
114
+ else:
115
+ return {}
116
+
117
+ stations_to_fetch = []
118
+ for st_id, coords in KNOWN_STATIONS.items():
119
+ if min_lat <= coords["lat"] <= max_lat and min_lon <= coords["lon"] <= max_lon:
120
+ stations_to_fetch.append(st_id)
121
+
122
+ # fallback to EXRX if none in bounding box for testing
123
+ if not stations_to_fetch:
124
+ logger.warning(
125
+ f"No stations found in bbox {bbox}. Falling back to default proxy stations (EXRX, WLIS)"
126
+ )
127
+ stations_to_fetch = ["EXRX", "WLIS"]
128
+
129
+ results = {}
130
+ for st_id in stations_to_fetch:
131
+ logger.info(f"Fetching station {st_id} for bounding box request...")
132
+ profile = fetch_erddap_station_profiles(
133
+ st_id, start_time, end_time, cache_dir, cache_bust
134
+ )
135
+ if profile and any(v for v in profile.values()):
136
+ results[st_id] = {
137
+ "lat": KNOWN_STATIONS.get(st_id, {}).get("lat", 0.0),
138
+ "lon": KNOWN_STATIONS.get(st_id, {}).get("lon", 0.0),
139
+ "profiles": profile,
140
+ }
141
+
142
+ return results
@@ -0,0 +1,72 @@
1
+ import s3fs
2
+ import logging
3
+
4
+
5
+ logger = logging.getLogger(__name__)
6
+
7
+ # NOAA makes HRRR freely available on AWS S3 without authentication.
8
+ # The same bucket (noaa-hrrr-bdp-pds) holds both the rolling operational
9
+ # window and the full historical archive back to 2014-07-30.
10
+ fs = s3fs.S3FileSystem(anon=True)
11
+
12
+
13
+ # First date for which HRRR data exists in the noaa-hrrr-bdp-pds archive.
14
+ # GRIB2 .idx variable patterns for 10-meter wind components.
15
+ # Each HRRR .grib2 on S3 has a sidecar .idx listing byte offsets per message.
16
+ def _parse_idx(
17
+ idx_text: str, patterns: tuple[str, ...]
18
+ ) -> list[tuple[int, int | None]]:
19
+ """
20
+ Parse a GRIB2 .idx sidecar file and return byte ranges for matching messages.
21
+
22
+ Each .idx line has the format:
23
+ message_number:byte_offset:date:variable:level:forecast_type:
24
+
25
+ Returns a list of (start_byte, end_byte) tuples. end_byte is None for the
26
+ last message in the file (meaning read-to-EOF).
27
+ """
28
+ lines = [line.strip() for line in idx_text.strip().splitlines() if line.strip()]
29
+
30
+ # Build list of (message_number, byte_offset, raw_line)
31
+ entries = []
32
+ for line in lines:
33
+ parts = line.split(":")
34
+ if len(parts) < 3:
35
+ continue
36
+ try:
37
+ offset = int(parts[1])
38
+ except ValueError:
39
+ continue
40
+ entries.append((offset, line))
41
+
42
+ # Find matching messages and compute byte ranges
43
+ ranges = []
44
+ for i, (offset, line) in enumerate(entries):
45
+ if any(pattern in line for pattern in patterns):
46
+ # End byte is the start of the next message, or None for EOF
47
+ end = entries[i + 1][0] if i + 1 < len(entries) else None
48
+ ranges.append((offset, end))
49
+
50
+ return ranges
51
+
52
+
53
+ def _fetch_s3_byte_ranges(
54
+ s3_path: str, ranges: list[tuple[int, int | None]], output_path: str
55
+ ) -> None:
56
+ """
57
+ Download specific byte ranges from an S3 object and concatenate into a
58
+ single valid GRIB2 file. Each GRIB2 message is self-contained, so
59
+ concatenating selected messages produces a valid file.
60
+ """
61
+ with open(output_path, "wb") as f:
62
+ for start, end in ranges:
63
+ if end is not None:
64
+ length = end - start
65
+ with fs.open(s3_path, "rb") as s3f:
66
+ s3f.seek(start)
67
+ f.write(s3f.read(length))
68
+ else:
69
+ # Read from start to EOF
70
+ with fs.open(s3_path, "rb") as s3f:
71
+ s3f.seek(start)
72
+ f.write(s3f.read())