pyPRMS 0.9.7__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pyPRMS/Exceptions_custom.py +31 -0
- pyPRMS/__init__.py +52 -0
- pyPRMS/cbh/Cbh.py +431 -0
- pyPRMS/cbh/CbhAscii.py +458 -0
- pyPRMS/cbh/CbhNetcdf.py +199 -0
- pyPRMS/cbh/__init__.py +3 -0
- pyPRMS/constants.py +131 -0
- pyPRMS/control/Control.py +362 -0
- pyPRMS/control/ControlFile.py +161 -0
- pyPRMS/control/ControlVariable.py +208 -0
- pyPRMS/control/__init__.py +3 -0
- pyPRMS/dimensions/Dimension.py +154 -0
- pyPRMS/dimensions/Dimensions.py +256 -0
- pyPRMS/dimensions/__init__.py +2 -0
- pyPRMS/input/DataFile.py +354 -0
- pyPRMS/input/InputVariable.py +61 -0
- pyPRMS/input/__init__.py +0 -0
- pyPRMS/metadata/__init__.py +1 -0
- pyPRMS/metadata/metadata.py +430 -0
- pyPRMS/parameters/ParamDb.py +73 -0
- pyPRMS/parameters/Parameter.py +624 -0
- pyPRMS/parameters/ParameterFile.py +190 -0
- pyPRMS/parameters/ParameterNetCDF.py +74 -0
- pyPRMS/parameters/ParameterSet.py +96 -0
- pyPRMS/parameters/Parameters.py +1506 -0
- pyPRMS/parameters/__init__.py +5 -0
- pyPRMS/plot_helpers.py +305 -0
- pyPRMS/prms_helpers.py +235 -0
- pyPRMS/py.typed +0 -0
- pyPRMS/summary/OutputCSV.py +64 -0
- pyPRMS/summary/OutputVariable.py +163 -0
- pyPRMS/summary/OutputVariables.py +227 -0
- pyPRMS/summary/__init__.py +2 -0
- pyPRMS/utilities/__init__.py +0 -0
- pyPRMS/utilities/convert_cbh.py +107 -0
- pyPRMS/utilities/convert_model_output.py +91 -0
- pyPRMS/utilities/convert_params.py +60 -0
- pyPRMS/version.py +13 -0
- pyPRMS/xml/cbh.xml +163 -0
- pyPRMS/xml/control.xml +1447 -0
- pyPRMS/xml/dimensions.xml +311 -0
- pyPRMS/xml/modules.xml +251 -0
- pyPRMS/xml/parameters.xml +6932 -0
- pyPRMS/xml/time_series_input.xml +198 -0
- pyPRMS/xml/variables.xml +8173 -0
- pyprms-0.9.7.dist-info/LICENSE.md +21 -0
- pyprms-0.9.7.dist-info/METADATA +67 -0
- pyprms-0.9.7.dist-info/RECORD +51 -0
- pyprms-0.9.7.dist-info/WHEEL +5 -0
- pyprms-0.9.7.dist-info/entry_points.txt +3 -0
- pyprms-0.9.7.dist-info/top_level.txt +1 -0
pyPRMS/cbh/CbhAscii.py
ADDED
|
@@ -0,0 +1,458 @@
|
|
|
1
|
+
import datetime
|
|
2
|
+
import os
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd # type: ignore
|
|
5
|
+
import netCDF4 as nc # type: ignore
|
|
6
|
+
# from collections import OrderedDict
|
|
7
|
+
|
|
8
|
+
from typing import Dict, List, Union, Optional
|
|
9
|
+
|
|
10
|
+
from ..constants import REGIONS
|
|
11
|
+
|
|
12
|
+
CBH_VARNAMES = ['prcp', 'tmin', 'tmax']
|
|
13
|
+
CBH_INDEX_COLS = [0, 1, 2, 3, 4, 5]
|
|
14
|
+
TS_FORMAT = '%Y %m %d %H %M %S' # 1915 1 13 0 0 0
|
|
15
|
+
NA_VALS_DEFAULT = ['-99.0', '-999.0', 'NaN', 'inf']
|
|
16
|
+
|
|
17
|
+
class CbhAscii(object):
|
|
18
|
+
|
|
19
|
+
"""Class for handling classic climate-by-hru (CBH) files.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
# Author: Parker Norton (pnorton@usgs.gov)
|
|
23
|
+
# Create date: 2019-04
|
|
24
|
+
|
|
25
|
+
# This class assumes it is dealing with regional cbh files (not a CONUS-level NHM file)
|
|
26
|
+
# TODO: As currently written the type of data (e.g. tmax, tmin, prcp) is ignored.
|
|
27
|
+
# TODO: Verify that given data type size matches number of columns
|
|
28
|
+
|
|
29
|
+
# 2016-12-20 PAN:
|
|
30
|
+
# As written this works with CBH files that were created with
|
|
31
|
+
# java class gov.usgs.mows.GCMtoPRMS.GDPtoCBH
|
|
32
|
+
# This program creates the CBH files in the format needed by PRMS
|
|
33
|
+
# and also verifies the correctness of the data including:
|
|
34
|
+
# tmax is never less than tmin
|
|
35
|
+
# prcp is never negative
|
|
36
|
+
# any missing data/missing date is filled with (?? avg of bracketing dates??)
|
|
37
|
+
|
|
38
|
+
def __init__(self, src_path: Optional[str] = None,
|
|
39
|
+
st_date: Optional[datetime.datetime] = None,
|
|
40
|
+
en_date: Optional[datetime.datetime] = None,
|
|
41
|
+
indices: Optional[Dict] = None,
|
|
42
|
+
nhm_hrus: Optional[List] = None,
|
|
43
|
+
mapping: Optional[Dict] = None):
|
|
44
|
+
"""Create CbhAscii object.
|
|
45
|
+
|
|
46
|
+
:param src_path: path to by-region CBH ASCII source files
|
|
47
|
+
:param st_date: start date for extraction
|
|
48
|
+
:param en_date: end date for extraction
|
|
49
|
+
:param indices: ordered dictionary of nhm_id, local_id pairs
|
|
50
|
+
:param nhm_hrus: list NHM HRUs to extract
|
|
51
|
+
:param mapping: dictionary mapping regions to nhm_id ranges
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
self.__src_path = src_path
|
|
55
|
+
|
|
56
|
+
# self.__indices = [str(kk) for kk in indices]
|
|
57
|
+
self.__indices = indices # OrdereDict: nhm_ids -> local_ids
|
|
58
|
+
|
|
59
|
+
self.__data = None
|
|
60
|
+
self.__stdate = st_date
|
|
61
|
+
self.__endate = en_date
|
|
62
|
+
self.__nhm_hrus = nhm_hrus
|
|
63
|
+
self.__mapping = mapping
|
|
64
|
+
self.__dataframe = None
|
|
65
|
+
|
|
66
|
+
def read_cbh(self):
|
|
67
|
+
"""Reads an entire CBH file.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
# incl_cols = list(self.__indices.values())
|
|
71
|
+
# for xx in CBH_INDEX_COLS[:-1]:
|
|
72
|
+
# incl_cols.insert(0, xx)
|
|
73
|
+
|
|
74
|
+
incl_cols = list(CBH_INDEX_COLS)
|
|
75
|
+
|
|
76
|
+
for xx in self.__indices.values():
|
|
77
|
+
incl_cols.append(xx+5) # include an offset for the datetime info
|
|
78
|
+
# print(incl_cols)
|
|
79
|
+
|
|
80
|
+
# Columns 0-5 always represent date/time information
|
|
81
|
+
self.__data = pd.read_csv(self.__src_path, sep=' ', skipinitialspace=True, usecols=incl_cols,
|
|
82
|
+
skiprows=3, engine='c', memory_map=True,
|
|
83
|
+
parse_dates={'time': CBH_INDEX_COLS},
|
|
84
|
+
index_col='time', header=None, na_values=[-99.0, -999.0])
|
|
85
|
+
|
|
86
|
+
self.__data.index = pd.to_datetime(self.__data.index, exact=True, cache=True, format=TS_FORMAT)
|
|
87
|
+
|
|
88
|
+
if self.__stdate is not None and self.__endate is not None:
|
|
89
|
+
self.__data = self.__data[self.__stdate:self.__endate]
|
|
90
|
+
|
|
91
|
+
# self.__data.reset_index(drop=True, inplace=True)
|
|
92
|
+
|
|
93
|
+
# Rename columns with NHM HRU ids
|
|
94
|
+
ren_dict = {v + 5: k for k, v in self.__indices.items()}
|
|
95
|
+
|
|
96
|
+
# NOTE: The rename is an expensive operation
|
|
97
|
+
self.__data.rename(columns=ren_dict, inplace=True)
|
|
98
|
+
|
|
99
|
+
def read_cbh_full(self):
|
|
100
|
+
"""Read entire CBH file.
|
|
101
|
+
"""
|
|
102
|
+
|
|
103
|
+
# incl_cols = list(self.__indices.values())
|
|
104
|
+
# for xx in CBH_INDEX_COLS[:-1]:
|
|
105
|
+
# incl_cols.insert(0, xx)
|
|
106
|
+
|
|
107
|
+
print('READING')
|
|
108
|
+
# Columns 0-5 always represent date/time information
|
|
109
|
+
self.__data = pd.read_csv(self.__src_path, sep=' ', skipinitialspace=True,
|
|
110
|
+
skiprows=3, engine='c', memory_map=True,
|
|
111
|
+
parse_dates={'time': CBH_INDEX_COLS},
|
|
112
|
+
# date_parser=dparse, parse_dates={'time': CBH_INDEX_COLS},
|
|
113
|
+
index_col='time', header=None, na_values=[-99.0, -999.0])
|
|
114
|
+
|
|
115
|
+
self.__data.index = pd.to_datetime(self.__data.index, exact=True, cache=True, format=TS_FORMAT)
|
|
116
|
+
|
|
117
|
+
if self.__stdate is not None and self.__endate is not None:
|
|
118
|
+
self.__data = self.__data[self.__stdate:self.__endate]
|
|
119
|
+
|
|
120
|
+
# self.__data.reset_index(drop=True, inplace=True)
|
|
121
|
+
|
|
122
|
+
# Rename columns with NHM HRU ids
|
|
123
|
+
# ren_dict = {v + 5: k for k, v in self.__indices.items()}
|
|
124
|
+
|
|
125
|
+
# NOTE: The rename is an expensive operation
|
|
126
|
+
# self.__data.rename(columns=ren_dict, inplace=True)
|
|
127
|
+
self.__data['year'] = self.__data.index.year
|
|
128
|
+
self.__data['month'] = self.__data.index.month
|
|
129
|
+
self.__data['day'] = self.__data.index.day
|
|
130
|
+
self.__data['hour'] = 0
|
|
131
|
+
self.__data['minute'] = 0
|
|
132
|
+
self.__data['second'] = 0
|
|
133
|
+
|
|
134
|
+
def read_ascii_file(self, filename: str,
|
|
135
|
+
columns: Optional[List] = None) -> pd.DataFrame:
|
|
136
|
+
"""Reads a single CBH file.
|
|
137
|
+
|
|
138
|
+
:param filename: name of the CBH file
|
|
139
|
+
:param columns: columns to read
|
|
140
|
+
:returns: dataframe of CBH variable
|
|
141
|
+
"""
|
|
142
|
+
# Columns 0-5 always represent date/time information
|
|
143
|
+
time_col_names = {0: 'year', 1: 'month', 2: 'day', 3: 'hour', 4: 'minute', 5: 'second'}
|
|
144
|
+
|
|
145
|
+
if columns is not None:
|
|
146
|
+
df = pd.read_csv(filename, sep=' ', skipinitialspace=True,
|
|
147
|
+
skiprows=3, engine='c', memory_map=True,
|
|
148
|
+
header=None, na_values=NA_VALS_DEFAULT,
|
|
149
|
+
usecols=columns)
|
|
150
|
+
else:
|
|
151
|
+
df = pd.read_csv(filename, sep=' ', skipinitialspace=True,
|
|
152
|
+
skiprows=3, engine='c', memory_map=True,
|
|
153
|
+
header=None, na_values=NA_VALS_DEFAULT)
|
|
154
|
+
|
|
155
|
+
# Rename columns with time information
|
|
156
|
+
df.rename(columns=time_col_names, inplace=True)
|
|
157
|
+
|
|
158
|
+
# Rename columns with local model indices
|
|
159
|
+
ren_dict = {k + 6: k + 1 for k in range(len(df.columns))}
|
|
160
|
+
df.rename(columns=ren_dict, inplace=True)
|
|
161
|
+
|
|
162
|
+
df['time'] = pd.to_datetime(df[time_col_names.values()], yearfirst=True)
|
|
163
|
+
df.drop(columns=time_col_names.values(), inplace=True)
|
|
164
|
+
df.set_index('time', inplace=True)
|
|
165
|
+
|
|
166
|
+
return df
|
|
167
|
+
|
|
168
|
+
def check_region(self, region: str) -> Union[Dict[int, int], None]:
|
|
169
|
+
"""Get the range of nhm_id values for selected region.
|
|
170
|
+
|
|
171
|
+
:param region: HUC2 region number (1 to 18)
|
|
172
|
+
:returns: dictionary of local_id, nhm_id pairs
|
|
173
|
+
"""
|
|
174
|
+
if self.__indices is not None:
|
|
175
|
+
# Get the range of nhm_ids for the region
|
|
176
|
+
rvals = self.__mapping[region]
|
|
177
|
+
|
|
178
|
+
# print('Examining {} ({} to {})'.format(rr, rvals[0], rvals[1]))
|
|
179
|
+
if rvals[0] >= rvals[1]:
|
|
180
|
+
raise ValueError('Lower HRU bound is greater than upper HRU bound.')
|
|
181
|
+
|
|
182
|
+
idx_retrieve = OrderedDict()
|
|
183
|
+
|
|
184
|
+
for yy in self.__indices.keys():
|
|
185
|
+
if rvals[0] <= yy <= rvals[1]:
|
|
186
|
+
idx_retrieve[self.__indices[yy]] = yy # {local_ids: nhm_ids}
|
|
187
|
+
|
|
188
|
+
return idx_retrieve
|
|
189
|
+
return None
|
|
190
|
+
|
|
191
|
+
def read_cbh_multifile(self, var: Optional[str] = None) -> Union[pd.DataFrame, None]:
|
|
192
|
+
"""Read cbh data from multiple csv files.
|
|
193
|
+
|
|
194
|
+
:param var: name of variable to read
|
|
195
|
+
:returns: dataframe of extracted variable
|
|
196
|
+
"""
|
|
197
|
+
|
|
198
|
+
if var is None:
|
|
199
|
+
raise ValueError('Variable name (var) must be provided')
|
|
200
|
+
|
|
201
|
+
first = True
|
|
202
|
+
self.__dataframe = None
|
|
203
|
+
|
|
204
|
+
for rr in REGIONS:
|
|
205
|
+
idx_retrieve = self.check_region(region=rr)
|
|
206
|
+
|
|
207
|
+
if len(idx_retrieve) > 0:
|
|
208
|
+
# Build the list of columns to load
|
|
209
|
+
# The given local ids must be adjusted by 5 to reflect:
|
|
210
|
+
# 1) the presence of 6 columns of time information
|
|
211
|
+
# 2) 0-based column names
|
|
212
|
+
load_cols = list(CBH_INDEX_COLS)
|
|
213
|
+
load_cols.extend([xx+5 for xx in idx_retrieve.keys()])
|
|
214
|
+
else:
|
|
215
|
+
load_cols = None
|
|
216
|
+
|
|
217
|
+
if len(idx_retrieve) > 0:
|
|
218
|
+
# The current region contains HRUs in the model subset
|
|
219
|
+
# Read in the data for those HRUs
|
|
220
|
+
cbh_file = f'{self.__src_path}/{rr}_{var}.cbh.gz'
|
|
221
|
+
|
|
222
|
+
print(f'\tLoad {len(idx_retrieve)} HRUs from {rr}')
|
|
223
|
+
|
|
224
|
+
if not os.path.isfile(cbh_file):
|
|
225
|
+
# Missing data file for this variable and region
|
|
226
|
+
raise IOError(f'Required CBH file, {cbh_file}, is missing.')
|
|
227
|
+
|
|
228
|
+
# df = self.read_ascii_file(cbh_file, columns=load_cols)
|
|
229
|
+
|
|
230
|
+
# Small read to get number of columns
|
|
231
|
+
df = pd.read_csv(cbh_file, sep=' ', skipinitialspace=True,
|
|
232
|
+
usecols=load_cols, nrows=2,
|
|
233
|
+
skiprows=3, engine='c', memory_map=True,
|
|
234
|
+
parse_dates={'time': CBH_INDEX_COLS},
|
|
235
|
+
# date_parser=dparse, parse_dates={'time': CBH_INDEX_COLS},
|
|
236
|
+
index_col='time', header=None, na_values=[-99.0, -999.0, 'NaN', 'inf'])
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
# Override Pandas' rather stupid default of float64
|
|
240
|
+
col_dtypes = {xx: np.float32 for xx in df.columns}
|
|
241
|
+
|
|
242
|
+
# Now read the whole file using float32 instead of float64
|
|
243
|
+
df = pd.read_csv(cbh_file, sep=' ', skipinitialspace=True,
|
|
244
|
+
usecols=load_cols, dtype=col_dtypes,
|
|
245
|
+
skiprows=3, engine='c', memory_map=True,
|
|
246
|
+
parse_dates={'time': CBH_INDEX_COLS},
|
|
247
|
+
# date_parser=dparse, parse_dates={'time': CBH_INDEX_COLS},
|
|
248
|
+
index_col='time', header=None, na_values=[-99.0, -999.0, 'NaN', 'inf'])
|
|
249
|
+
|
|
250
|
+
df.index = pd.to_datetime(df.index, exact=True, cache=True, format=TS_FORMAT)
|
|
251
|
+
|
|
252
|
+
if self.__stdate is not None and self.__endate is not None:
|
|
253
|
+
# Restrict the date range
|
|
254
|
+
df = df[self.__stdate:self.__endate]
|
|
255
|
+
|
|
256
|
+
# Rename columns with NHM HRU ids
|
|
257
|
+
ren_dict = {k+5: v for k, v in idx_retrieve.items()}
|
|
258
|
+
|
|
259
|
+
# NOTE: The rename is an expensive operation
|
|
260
|
+
df.rename(columns=ren_dict, inplace=True)
|
|
261
|
+
|
|
262
|
+
if first:
|
|
263
|
+
self.__dataframe = df.copy()
|
|
264
|
+
first = False
|
|
265
|
+
else:
|
|
266
|
+
self.__dataframe = self.__dataframe.join(df, how='left')
|
|
267
|
+
return self.__dataframe
|
|
268
|
+
|
|
269
|
+
def get_var(self, var: str) -> Union[pd.DataFrame, None]:
|
|
270
|
+
"""Get CBH variable.
|
|
271
|
+
|
|
272
|
+
:param var: name of CBH variable
|
|
273
|
+
:returns: dataframe of CBH variable values
|
|
274
|
+
"""
|
|
275
|
+
|
|
276
|
+
data = self.read_cbh_multifile(var=var)
|
|
277
|
+
return data
|
|
278
|
+
|
|
279
|
+
def write_ascii(self, pathname: Optional[str] = None,
|
|
280
|
+
fileprefix: Optional[str] = None,
|
|
281
|
+
variables: Optional[List[str]] = None):
|
|
282
|
+
"""Write ASCII CBH file for selected variable.
|
|
283
|
+
|
|
284
|
+
By default CBH filenames are saved in the current working directory and
|
|
285
|
+
are named for the selected variable with an extension of .cbh
|
|
286
|
+
|
|
287
|
+
:param pathname: path to save files to
|
|
288
|
+
:param fileprefix: prefix to add to CBH output filename
|
|
289
|
+
:param variables: variables to write to CBH files
|
|
290
|
+
"""
|
|
291
|
+
|
|
292
|
+
# For out_order the first six columns contain the time information and
|
|
293
|
+
# are always output for the cbh files
|
|
294
|
+
out_order = [kk for kk in self.__nhm_hrus]
|
|
295
|
+
for cc in ['second', 'minute', 'hour', 'day', 'month', 'year']:
|
|
296
|
+
out_order.insert(0, cc)
|
|
297
|
+
|
|
298
|
+
var_list = []
|
|
299
|
+
if variables is None:
|
|
300
|
+
var_list = CBH_VARNAMES
|
|
301
|
+
elif isinstance(list, variables):
|
|
302
|
+
var_list = variables
|
|
303
|
+
|
|
304
|
+
for cvar in var_list:
|
|
305
|
+
data = self.get_var(var=cvar)
|
|
306
|
+
|
|
307
|
+
# Add time information as columns
|
|
308
|
+
data['year'] = data.index.year
|
|
309
|
+
data['month'] = data.index.month
|
|
310
|
+
data['day'] = data.index.day
|
|
311
|
+
data['hour'] = 0
|
|
312
|
+
data['minute'] = 0
|
|
313
|
+
data['second'] = 0
|
|
314
|
+
|
|
315
|
+
# Output ASCII CBH files
|
|
316
|
+
if fileprefix is None:
|
|
317
|
+
outfile = f'{cvar}.cbh'
|
|
318
|
+
else:
|
|
319
|
+
outfile = f'{fileprefix}_{cvar}.cbh'
|
|
320
|
+
|
|
321
|
+
if pathname is not None:
|
|
322
|
+
outfile = f'{pathname}/{outfile}'
|
|
323
|
+
|
|
324
|
+
out_cbh = open(outfile, 'w')
|
|
325
|
+
out_cbh.write('Written by Bandit\n')
|
|
326
|
+
out_cbh.write(f'{cvar} {len(self.__nhm_hrus)}\n')
|
|
327
|
+
out_cbh.write('########################################\n')
|
|
328
|
+
|
|
329
|
+
data.to_csv(out_cbh, columns=out_order, na_rep='-999', float_format='%0.3f',
|
|
330
|
+
sep=' ', index=False, header=False, encoding=None, chunksize=50)
|
|
331
|
+
out_cbh.close()
|
|
332
|
+
|
|
333
|
+
def write_netcdf(self, filename: str,
|
|
334
|
+
variables: Optional[List[str]] = None):
|
|
335
|
+
"""Write CBH to netcdf format file
|
|
336
|
+
|
|
337
|
+
:param filename: name of netCDF output file
|
|
338
|
+
:param variables: list of variables to write to output file
|
|
339
|
+
"""
|
|
340
|
+
|
|
341
|
+
# NetCDF-related variables
|
|
342
|
+
var_desc = {'tmax': 'Maximum Temperature', 'tmin': 'Minimum temperature', 'prcp': 'Precipitation'}
|
|
343
|
+
var_units = {'tmax': 'C', 'tmin': 'C', 'prcp': 'inches'}
|
|
344
|
+
|
|
345
|
+
# Create a netCDF file for the CBH data
|
|
346
|
+
nco = nc.Dataset(filename, 'w', clobber=True)
|
|
347
|
+
nco.createDimension('hru', len(self.__nhm_hrus))
|
|
348
|
+
nco.createDimension('time', None)
|
|
349
|
+
|
|
350
|
+
timeo = nco.createVariable('time', 'f4', ('time'))
|
|
351
|
+
timeo.calendar = 'standard'
|
|
352
|
+
# timeo.bounds = 'time_bnds'
|
|
353
|
+
|
|
354
|
+
# FIXME: Days since needs to be set to the starting date of the model pull
|
|
355
|
+
timeo.units = 'days since 1980-01-01 00:00:00'
|
|
356
|
+
|
|
357
|
+
hruo = nco.createVariable('hru', 'i4', ('hru'))
|
|
358
|
+
hruo.long_name = 'Hydrologic Response Unit ID (HRU)'
|
|
359
|
+
|
|
360
|
+
var_list = []
|
|
361
|
+
if variables is None:
|
|
362
|
+
var_list = CBH_VARNAMES
|
|
363
|
+
elif isinstance(list, variables):
|
|
364
|
+
var_list = variables
|
|
365
|
+
|
|
366
|
+
for cvar in var_list:
|
|
367
|
+
varo = nco.createVariable(cvar, 'f4', ('time', 'hru'), fill_value=nc.default_fillvals['f4'], zlib=True)
|
|
368
|
+
varo.long_name = var_desc[cvar]
|
|
369
|
+
varo.units = var_units[cvar]
|
|
370
|
+
|
|
371
|
+
nco.setncattr('Description', 'Climate by HRU')
|
|
372
|
+
# nco.setncattr('Bandit_version', __version__)
|
|
373
|
+
# nco.setncattr('NHM_version', nhmparamdb_revision)
|
|
374
|
+
|
|
375
|
+
# Write the HRU ids
|
|
376
|
+
hruo[:] = self.__nhm_hrus
|
|
377
|
+
|
|
378
|
+
first = True
|
|
379
|
+
for cvar in var_list:
|
|
380
|
+
data = self.get_var(var=cvar)
|
|
381
|
+
|
|
382
|
+
if first:
|
|
383
|
+
timeo[:] = nc.date2num(data.index.tolist(),
|
|
384
|
+
units='days since 1980-01-01 00:00:00',
|
|
385
|
+
calendar='standard')
|
|
386
|
+
first = False
|
|
387
|
+
|
|
388
|
+
# Write the CBH values
|
|
389
|
+
nco.variables[cvar][:, :] = data[self.__nhm_hrus].values
|
|
390
|
+
|
|
391
|
+
nco.close()
|
|
392
|
+
|
|
393
|
+
# def write_cbh_subset(self, outdir):
|
|
394
|
+
# outdata = None
|
|
395
|
+
# first = True
|
|
396
|
+
#
|
|
397
|
+
# for vv in CBH_VARNAMES:
|
|
398
|
+
# outorder = list(CBH_INDEX_COLS)
|
|
399
|
+
#
|
|
400
|
+
# for rr, rvals in iteritems(self.__mapping):
|
|
401
|
+
# idx_retrieve = {}
|
|
402
|
+
#
|
|
403
|
+
# for yy in self.__nhm_hrus.keys():
|
|
404
|
+
# if rvals[0] <= yy <= rvals[1]:
|
|
405
|
+
# idx_retrieve[yy] = self.__nhm_hrus[yy]
|
|
406
|
+
#
|
|
407
|
+
# if len(idx_retrieve) > 0:
|
|
408
|
+
# self.__src_path = '{}/{}_{}.cbh.gz'.format(self.__cbhdb_dir, rr, vv)
|
|
409
|
+
# self.read_cbh()
|
|
410
|
+
# if first:
|
|
411
|
+
# outdata = self.__data
|
|
412
|
+
# first = False
|
|
413
|
+
# else:
|
|
414
|
+
# outdata = pd.merge(outdata, self.__data, how='left', left_index=True, right_index=True)
|
|
415
|
+
#
|
|
416
|
+
# # Append the HRUs as ordered for the subset
|
|
417
|
+
# outorder.extend(self.__nhm_hrus)
|
|
418
|
+
#
|
|
419
|
+
# out_cbh = open('{}/{}.cbh'.format(outdir, vv), 'w')
|
|
420
|
+
# out_cbh.write('Written by pyPRMS.Cbh\n')
|
|
421
|
+
# out_cbh.write('{} {}\n'.format(vv, len()))
|
|
422
|
+
|
|
423
|
+
# def read_cbh_parq(self, src_dir):
|
|
424
|
+
# """Read CBH files stored in the parquet format"""
|
|
425
|
+
# if self.__indices:
|
|
426
|
+
# pfile = fp.ParquetFile('{}/daymet_{}.parq'.format(src_dir, self.__var))
|
|
427
|
+
# self.__data = pfile.to_pandas(self.__indices)
|
|
428
|
+
#
|
|
429
|
+
# if self.__stdate is not None and self.__endate is not None:
|
|
430
|
+
# # Given a date range to restrict the output
|
|
431
|
+
# self.__data = self.__data[self.__stdate:self.__endate]
|
|
432
|
+
#
|
|
433
|
+
# self.__data['year'] = self.__data.index.year
|
|
434
|
+
# self.__data['month'] = self.__data.index.month
|
|
435
|
+
# self.__data['day'] = self.__data.index.day
|
|
436
|
+
# self.__data['hour'] = 0
|
|
437
|
+
# self.__data['minute'] = 0
|
|
438
|
+
# self.__data['second'] = 0
|
|
439
|
+
#
|
|
440
|
+
# def read_cbh_hdf(self, src_dir):
|
|
441
|
+
# """Read CBH files stored in HDF5 format"""
|
|
442
|
+
# if self.__indices:
|
|
443
|
+
# # self.__data = pd.read_hdf('{}/daymet_{}.h5'.format(src_dir, self.__var), columns=self.__indices)
|
|
444
|
+
# self.__data = pd.read_hdf('{}/daymet_{}.h5'.format(src_dir, self.__var))
|
|
445
|
+
#
|
|
446
|
+
# if self.__stdate is not None and self.__endate is not None:
|
|
447
|
+
# # Given a date range to restrict the output
|
|
448
|
+
# self.__data = self.__data[self.__stdate:self.__endate]
|
|
449
|
+
#
|
|
450
|
+
# self.__data = self.__data[self.__indices]
|
|
451
|
+
#
|
|
452
|
+
# self.__data['year'] = self.__data.index.year
|
|
453
|
+
# self.__data['month'] = self.__data.index.month
|
|
454
|
+
# self.__data['day'] = self.__data.index.day
|
|
455
|
+
# self.__data['hour'] = 0
|
|
456
|
+
# self.__data['minute'] = 0
|
|
457
|
+
# self.__data['second'] = 0
|
|
458
|
+
#
|
pyPRMS/cbh/CbhNetcdf.py
ADDED
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
import datetime
|
|
2
|
+
import fsspec # type: ignore
|
|
3
|
+
import numpy as np
|
|
4
|
+
import os
|
|
5
|
+
import pandas as pd # type: ignore
|
|
6
|
+
import netCDF4 as nc # type: ignore
|
|
7
|
+
import xarray as xr
|
|
8
|
+
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Dict, List, Optional, Union
|
|
11
|
+
|
|
12
|
+
__author__ = 'Parker Norton (pnorton@usgs.gov)'
|
|
13
|
+
|
|
14
|
+
CBH_VARNAMES = ['prcp', 'tmin', 'tmax']
|
|
15
|
+
CBH_INDEX_COLS = [0, 1, 2, 3, 4, 5]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class CbhNetcdf(object):
|
|
19
|
+
"""Climate-By-HRU (CBH) files in netCDF format."""
|
|
20
|
+
|
|
21
|
+
def __init__(self, src_path: Union[str, Path],
|
|
22
|
+
nhm_hrus: List[int],
|
|
23
|
+
st_date: Optional[datetime.datetime] = None,
|
|
24
|
+
en_date: Optional[datetime.datetime] = None):
|
|
25
|
+
"""
|
|
26
|
+
:param src_path: Full path to netCDF file
|
|
27
|
+
:param st_date: The starting date for restricting CBH results
|
|
28
|
+
:param en_date: The ending date for restricting CBH results
|
|
29
|
+
:param nhm_hrus: List of NHM HRU IDs to extract from CBH
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
if isinstance(src_path, str):
|
|
33
|
+
src_path = Path(src_path)
|
|
34
|
+
|
|
35
|
+
self.__src_path = src_path
|
|
36
|
+
self.__stdate = st_date
|
|
37
|
+
self.__endate = en_date
|
|
38
|
+
self.__nhm_hrus = nhm_hrus
|
|
39
|
+
|
|
40
|
+
self.__dataset = self.read_netcdf()
|
|
41
|
+
|
|
42
|
+
@property
|
|
43
|
+
def data(self) -> xr.Dataset:
|
|
44
|
+
"""Returns the CBH dataset.
|
|
45
|
+
|
|
46
|
+
:returns: xarray dataset
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
return self.__dataset
|
|
50
|
+
|
|
51
|
+
def read_netcdf(self) -> xr.Dataset:
|
|
52
|
+
"""Read CBH files stored in netCDF format.
|
|
53
|
+
|
|
54
|
+
:returns: xarray dataset
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
filepath = str(self.__src_path.resolve())
|
|
58
|
+
match self.__src_path.suffix:
|
|
59
|
+
case '.nc':
|
|
60
|
+
ds = xr.open_mfdataset(filepath, chunks={}, combine='by_coords',
|
|
61
|
+
data_vars='minimal', decode_cf=True, engine='netcdf4',
|
|
62
|
+
parallel=True)
|
|
63
|
+
case '.zarr':
|
|
64
|
+
ds = xr.open_zarr(filepath, consolidated=True)
|
|
65
|
+
case '.json':
|
|
66
|
+
fs = fsspec.filesystem('reference', fo=filepath)
|
|
67
|
+
m = fs.get_mapper('')
|
|
68
|
+
|
|
69
|
+
ds = xr.open_dataset(m, engine='zarr', chunks={}, backend_kwargs={'consolidated': False})
|
|
70
|
+
|
|
71
|
+
if 'nhm_id' in ds.data_vars:
|
|
72
|
+
# dataset has nhm_id variable so use it as the nhru dimension
|
|
73
|
+
ds = ds.assign_coords(nhru=ds.nhm_id)
|
|
74
|
+
|
|
75
|
+
if self.__stdate is None:
|
|
76
|
+
# Use first date from the dataset
|
|
77
|
+
self.__stdate = pd.to_datetime(str(ds.time[0].values))
|
|
78
|
+
|
|
79
|
+
if self.__endate is None:
|
|
80
|
+
# Use last date from the dataset
|
|
81
|
+
self.__endate = pd.to_datetime(str(ds.time[-1].values))
|
|
82
|
+
|
|
83
|
+
return ds
|
|
84
|
+
|
|
85
|
+
def get_var(self, var: str) -> pd.DataFrame:
|
|
86
|
+
"""Get a variable from the netCDF file.
|
|
87
|
+
|
|
88
|
+
:param var: Name of the variable
|
|
89
|
+
:returns: dataframe of variable values
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
try:
|
|
93
|
+
data = self.__dataset[var].sel(time=slice(self.__stdate, self.__endate),
|
|
94
|
+
nhru=self.__nhm_hrus).to_pandas()
|
|
95
|
+
except IndexError:
|
|
96
|
+
print(f'ERROR: Dimensions (time, nhru) were used to subset {var} which expects' +
|
|
97
|
+
f'dimensions ({" ".join(map(str, self.__dataset[var].coords))})')
|
|
98
|
+
raise
|
|
99
|
+
except KeyError:
|
|
100
|
+
# Happens when older hruid dimension is used instead of nhru
|
|
101
|
+
data = self.__dataset[var].sel(time=slice(self.__stdate, self.__endate),
|
|
102
|
+
hruid=self.__nhm_hrus).to_pandas()
|
|
103
|
+
|
|
104
|
+
return data
|
|
105
|
+
|
|
106
|
+
def write_ascii(self, filename: Union[str, os.PathLike, Path],
|
|
107
|
+
variable: str):
|
|
108
|
+
"""Write CBH data for variable to PRMS ASCII-formatted file.
|
|
109
|
+
|
|
110
|
+
:param filename: Climate-by-HRU filename
|
|
111
|
+
:param variable: CBH variable to write
|
|
112
|
+
"""
|
|
113
|
+
|
|
114
|
+
# For out_order the first six columns contain the time information and
|
|
115
|
+
# are always output for the cbh files
|
|
116
|
+
out_order: List[Union[int, str]] = [kk for kk in self.__nhm_hrus]
|
|
117
|
+
for cc in ['second', 'minute', 'hour', 'day', 'month', 'year']:
|
|
118
|
+
out_order.insert(0, cc)
|
|
119
|
+
|
|
120
|
+
if variable in self.__dataset.data_vars:
|
|
121
|
+
data = self.get_var(var=variable)
|
|
122
|
+
|
|
123
|
+
# Add time information as columns
|
|
124
|
+
data['year'] = data.index.year
|
|
125
|
+
data['month'] = data.index.month
|
|
126
|
+
data['day'] = data.index.day
|
|
127
|
+
data['hour'] = 0
|
|
128
|
+
data['minute'] = 0
|
|
129
|
+
data['second'] = 0
|
|
130
|
+
|
|
131
|
+
out_cbh = open(filename, 'w')
|
|
132
|
+
out_cbh.write('Written by Bandit\n')
|
|
133
|
+
out_cbh.write(f'{variable} {len(self.__nhm_hrus)}\n')
|
|
134
|
+
out_cbh.write('########################################\n')
|
|
135
|
+
# data.to_csv(out_cbh, columns=out_order, na_rep='-999', float_format='%0.3f',
|
|
136
|
+
data.to_csv(out_cbh, columns=out_order, na_rep='-999', float_format='%0.2f',
|
|
137
|
+
sep=' ', index=False, header=False, lineterminator='\n', encoding=None, chunksize=50)
|
|
138
|
+
out_cbh.close()
|
|
139
|
+
else:
|
|
140
|
+
print(f'WARNING: {variable} does not exist in source CBH files..skipping')
|
|
141
|
+
|
|
142
|
+
def write_netcdf(self, filename: Union[str, os.PathLike, Path],
|
|
143
|
+
variables: Optional[List[str]] = None,
|
|
144
|
+
global_attrs: Optional[Dict] = None):
|
|
145
|
+
"""Write CBH variables to netCDF file.
|
|
146
|
+
|
|
147
|
+
:param filename: name of netCDF output file
|
|
148
|
+
:param variables: list of CBH variables to write
|
|
149
|
+
:param global_attrs: optional dictionary of attributes to include in netcdf file
|
|
150
|
+
"""
|
|
151
|
+
|
|
152
|
+
ds = self.__dataset
|
|
153
|
+
ds = ds.sel(time=slice(self.__stdate, self.__endate), nhru=self.__nhm_hrus)
|
|
154
|
+
|
|
155
|
+
if variables is None:
|
|
156
|
+
pass
|
|
157
|
+
elif isinstance(variables, list):
|
|
158
|
+
ds = ds[variables]
|
|
159
|
+
|
|
160
|
+
# Remove _FillValue from coordinate variables
|
|
161
|
+
for vv in list(ds.coords):
|
|
162
|
+
ds[vv].encoding.update({'_FillValue': None})
|
|
163
|
+
|
|
164
|
+
ds['crs'] = self.__dataset['crs']
|
|
165
|
+
ds['crs'].encoding.update({'_FillValue': None,
|
|
166
|
+
'contiguous': True})
|
|
167
|
+
|
|
168
|
+
ds['time'].attrs['standard_name'] = 'time'
|
|
169
|
+
ds['time'].attrs['long_name'] = 'time'
|
|
170
|
+
|
|
171
|
+
# Add nhm_id variable which will be the global NHM IDs
|
|
172
|
+
ds['nhm_id'] = ds['nhru']
|
|
173
|
+
ds['nhm_id'].attrs['long_name'] = 'Global model Hydrologic Response Unit ID (HRU)'
|
|
174
|
+
|
|
175
|
+
# Change the nhru coordinate variable values to reflect the local model HRU IDs
|
|
176
|
+
ds = ds.assign_coords(nhru=np.arange(1, ds.nhru.values.size+1, dtype=ds.nhru.dtype))
|
|
177
|
+
ds['nhru'].attrs['long_name'] = 'Local model Hydrologic Response Unit ID (HRU)'
|
|
178
|
+
ds['nhru'].attrs['cf_role'] = 'timeseries_id'
|
|
179
|
+
|
|
180
|
+
# Add/update global attributes
|
|
181
|
+
ds.attrs['Description'] = 'Climate-by-HRU'
|
|
182
|
+
|
|
183
|
+
if global_attrs is not None:
|
|
184
|
+
for kk, vv in global_attrs.items():
|
|
185
|
+
ds.attrs[kk] = vv
|
|
186
|
+
|
|
187
|
+
encoding = {}
|
|
188
|
+
|
|
189
|
+
for cvar in ds.variables:
|
|
190
|
+
if ds[cvar].ndim > 1:
|
|
191
|
+
encoding[cvar] = dict(_FillValue=ds[cvar].encoding['_FillValue'],
|
|
192
|
+
compression='zlib',
|
|
193
|
+
complevel=2,
|
|
194
|
+
fletcher32=True)
|
|
195
|
+
else:
|
|
196
|
+
encoding[cvar] = dict(_FillValue=None,
|
|
197
|
+
contiguous=True)
|
|
198
|
+
|
|
199
|
+
ds.load().to_netcdf(filename, engine='netcdf4', format='NETCDF4', encoding=encoding)
|
pyPRMS/cbh/__init__.py
ADDED