pyPRMS 0.9.7__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. pyPRMS/Exceptions_custom.py +31 -0
  2. pyPRMS/__init__.py +52 -0
  3. pyPRMS/cbh/Cbh.py +431 -0
  4. pyPRMS/cbh/CbhAscii.py +458 -0
  5. pyPRMS/cbh/CbhNetcdf.py +199 -0
  6. pyPRMS/cbh/__init__.py +3 -0
  7. pyPRMS/constants.py +131 -0
  8. pyPRMS/control/Control.py +362 -0
  9. pyPRMS/control/ControlFile.py +161 -0
  10. pyPRMS/control/ControlVariable.py +208 -0
  11. pyPRMS/control/__init__.py +3 -0
  12. pyPRMS/dimensions/Dimension.py +154 -0
  13. pyPRMS/dimensions/Dimensions.py +256 -0
  14. pyPRMS/dimensions/__init__.py +2 -0
  15. pyPRMS/input/DataFile.py +354 -0
  16. pyPRMS/input/InputVariable.py +61 -0
  17. pyPRMS/input/__init__.py +0 -0
  18. pyPRMS/metadata/__init__.py +1 -0
  19. pyPRMS/metadata/metadata.py +430 -0
  20. pyPRMS/parameters/ParamDb.py +73 -0
  21. pyPRMS/parameters/Parameter.py +624 -0
  22. pyPRMS/parameters/ParameterFile.py +190 -0
  23. pyPRMS/parameters/ParameterNetCDF.py +74 -0
  24. pyPRMS/parameters/ParameterSet.py +96 -0
  25. pyPRMS/parameters/Parameters.py +1506 -0
  26. pyPRMS/parameters/__init__.py +5 -0
  27. pyPRMS/plot_helpers.py +305 -0
  28. pyPRMS/prms_helpers.py +235 -0
  29. pyPRMS/py.typed +0 -0
  30. pyPRMS/summary/OutputCSV.py +64 -0
  31. pyPRMS/summary/OutputVariable.py +163 -0
  32. pyPRMS/summary/OutputVariables.py +227 -0
  33. pyPRMS/summary/__init__.py +2 -0
  34. pyPRMS/utilities/__init__.py +0 -0
  35. pyPRMS/utilities/convert_cbh.py +107 -0
  36. pyPRMS/utilities/convert_model_output.py +91 -0
  37. pyPRMS/utilities/convert_params.py +60 -0
  38. pyPRMS/version.py +13 -0
  39. pyPRMS/xml/cbh.xml +163 -0
  40. pyPRMS/xml/control.xml +1447 -0
  41. pyPRMS/xml/dimensions.xml +311 -0
  42. pyPRMS/xml/modules.xml +251 -0
  43. pyPRMS/xml/parameters.xml +6932 -0
  44. pyPRMS/xml/time_series_input.xml +198 -0
  45. pyPRMS/xml/variables.xml +8173 -0
  46. pyprms-0.9.7.dist-info/LICENSE.md +21 -0
  47. pyprms-0.9.7.dist-info/METADATA +67 -0
  48. pyprms-0.9.7.dist-info/RECORD +51 -0
  49. pyprms-0.9.7.dist-info/WHEEL +5 -0
  50. pyprms-0.9.7.dist-info/entry_points.txt +3 -0
  51. pyprms-0.9.7.dist-info/top_level.txt +1 -0
pyPRMS/cbh/CbhAscii.py ADDED
@@ -0,0 +1,458 @@
1
+ import datetime
2
+ import os
3
+ import numpy as np
4
+ import pandas as pd # type: ignore
5
+ import netCDF4 as nc # type: ignore
6
+ # from collections import OrderedDict
7
+
8
+ from typing import Dict, List, Union, Optional
9
+
10
+ from ..constants import REGIONS
11
+
12
+ CBH_VARNAMES = ['prcp', 'tmin', 'tmax']
13
+ CBH_INDEX_COLS = [0, 1, 2, 3, 4, 5]
14
+ TS_FORMAT = '%Y %m %d %H %M %S' # 1915 1 13 0 0 0
15
+ NA_VALS_DEFAULT = ['-99.0', '-999.0', 'NaN', 'inf']
16
+
17
+ class CbhAscii(object):
18
+
19
+ """Class for handling classic climate-by-hru (CBH) files.
20
+ """
21
+
22
+ # Author: Parker Norton (pnorton@usgs.gov)
23
+ # Create date: 2019-04
24
+
25
+ # This class assumes it is dealing with regional cbh files (not a CONUS-level NHM file)
26
+ # TODO: As currently written the type of data (e.g. tmax, tmin, prcp) is ignored.
27
+ # TODO: Verify that given data type size matches number of columns
28
+
29
+ # 2016-12-20 PAN:
30
+ # As written this works with CBH files that were created with
31
+ # java class gov.usgs.mows.GCMtoPRMS.GDPtoCBH
32
+ # This program creates the CBH files in the format needed by PRMS
33
+ # and also verifies the correctness of the data including:
34
+ # tmax is never less than tmin
35
+ # prcp is never negative
36
+ # any missing data/missing date is filled with (?? avg of bracketing dates??)
37
+
38
+ def __init__(self, src_path: Optional[str] = None,
39
+ st_date: Optional[datetime.datetime] = None,
40
+ en_date: Optional[datetime.datetime] = None,
41
+ indices: Optional[Dict] = None,
42
+ nhm_hrus: Optional[List] = None,
43
+ mapping: Optional[Dict] = None):
44
+ """Create CbhAscii object.
45
+
46
+ :param src_path: path to by-region CBH ASCII source files
47
+ :param st_date: start date for extraction
48
+ :param en_date: end date for extraction
49
+ :param indices: ordered dictionary of nhm_id, local_id pairs
50
+ :param nhm_hrus: list NHM HRUs to extract
51
+ :param mapping: dictionary mapping regions to nhm_id ranges
52
+ """
53
+
54
+ self.__src_path = src_path
55
+
56
+ # self.__indices = [str(kk) for kk in indices]
57
+ self.__indices = indices # OrdereDict: nhm_ids -> local_ids
58
+
59
+ self.__data = None
60
+ self.__stdate = st_date
61
+ self.__endate = en_date
62
+ self.__nhm_hrus = nhm_hrus
63
+ self.__mapping = mapping
64
+ self.__dataframe = None
65
+
66
+ def read_cbh(self):
67
+ """Reads an entire CBH file.
68
+ """
69
+
70
+ # incl_cols = list(self.__indices.values())
71
+ # for xx in CBH_INDEX_COLS[:-1]:
72
+ # incl_cols.insert(0, xx)
73
+
74
+ incl_cols = list(CBH_INDEX_COLS)
75
+
76
+ for xx in self.__indices.values():
77
+ incl_cols.append(xx+5) # include an offset for the datetime info
78
+ # print(incl_cols)
79
+
80
+ # Columns 0-5 always represent date/time information
81
+ self.__data = pd.read_csv(self.__src_path, sep=' ', skipinitialspace=True, usecols=incl_cols,
82
+ skiprows=3, engine='c', memory_map=True,
83
+ parse_dates={'time': CBH_INDEX_COLS},
84
+ index_col='time', header=None, na_values=[-99.0, -999.0])
85
+
86
+ self.__data.index = pd.to_datetime(self.__data.index, exact=True, cache=True, format=TS_FORMAT)
87
+
88
+ if self.__stdate is not None and self.__endate is not None:
89
+ self.__data = self.__data[self.__stdate:self.__endate]
90
+
91
+ # self.__data.reset_index(drop=True, inplace=True)
92
+
93
+ # Rename columns with NHM HRU ids
94
+ ren_dict = {v + 5: k for k, v in self.__indices.items()}
95
+
96
+ # NOTE: The rename is an expensive operation
97
+ self.__data.rename(columns=ren_dict, inplace=True)
98
+
99
+ def read_cbh_full(self):
100
+ """Read entire CBH file.
101
+ """
102
+
103
+ # incl_cols = list(self.__indices.values())
104
+ # for xx in CBH_INDEX_COLS[:-1]:
105
+ # incl_cols.insert(0, xx)
106
+
107
+ print('READING')
108
+ # Columns 0-5 always represent date/time information
109
+ self.__data = pd.read_csv(self.__src_path, sep=' ', skipinitialspace=True,
110
+ skiprows=3, engine='c', memory_map=True,
111
+ parse_dates={'time': CBH_INDEX_COLS},
112
+ # date_parser=dparse, parse_dates={'time': CBH_INDEX_COLS},
113
+ index_col='time', header=None, na_values=[-99.0, -999.0])
114
+
115
+ self.__data.index = pd.to_datetime(self.__data.index, exact=True, cache=True, format=TS_FORMAT)
116
+
117
+ if self.__stdate is not None and self.__endate is not None:
118
+ self.__data = self.__data[self.__stdate:self.__endate]
119
+
120
+ # self.__data.reset_index(drop=True, inplace=True)
121
+
122
+ # Rename columns with NHM HRU ids
123
+ # ren_dict = {v + 5: k for k, v in self.__indices.items()}
124
+
125
+ # NOTE: The rename is an expensive operation
126
+ # self.__data.rename(columns=ren_dict, inplace=True)
127
+ self.__data['year'] = self.__data.index.year
128
+ self.__data['month'] = self.__data.index.month
129
+ self.__data['day'] = self.__data.index.day
130
+ self.__data['hour'] = 0
131
+ self.__data['minute'] = 0
132
+ self.__data['second'] = 0
133
+
134
+ def read_ascii_file(self, filename: str,
135
+ columns: Optional[List] = None) -> pd.DataFrame:
136
+ """Reads a single CBH file.
137
+
138
+ :param filename: name of the CBH file
139
+ :param columns: columns to read
140
+ :returns: dataframe of CBH variable
141
+ """
142
+ # Columns 0-5 always represent date/time information
143
+ time_col_names = {0: 'year', 1: 'month', 2: 'day', 3: 'hour', 4: 'minute', 5: 'second'}
144
+
145
+ if columns is not None:
146
+ df = pd.read_csv(filename, sep=' ', skipinitialspace=True,
147
+ skiprows=3, engine='c', memory_map=True,
148
+ header=None, na_values=NA_VALS_DEFAULT,
149
+ usecols=columns)
150
+ else:
151
+ df = pd.read_csv(filename, sep=' ', skipinitialspace=True,
152
+ skiprows=3, engine='c', memory_map=True,
153
+ header=None, na_values=NA_VALS_DEFAULT)
154
+
155
+ # Rename columns with time information
156
+ df.rename(columns=time_col_names, inplace=True)
157
+
158
+ # Rename columns with local model indices
159
+ ren_dict = {k + 6: k + 1 for k in range(len(df.columns))}
160
+ df.rename(columns=ren_dict, inplace=True)
161
+
162
+ df['time'] = pd.to_datetime(df[time_col_names.values()], yearfirst=True)
163
+ df.drop(columns=time_col_names.values(), inplace=True)
164
+ df.set_index('time', inplace=True)
165
+
166
+ return df
167
+
168
+ def check_region(self, region: str) -> Union[Dict[int, int], None]:
169
+ """Get the range of nhm_id values for selected region.
170
+
171
+ :param region: HUC2 region number (1 to 18)
172
+ :returns: dictionary of local_id, nhm_id pairs
173
+ """
174
+ if self.__indices is not None:
175
+ # Get the range of nhm_ids for the region
176
+ rvals = self.__mapping[region]
177
+
178
+ # print('Examining {} ({} to {})'.format(rr, rvals[0], rvals[1]))
179
+ if rvals[0] >= rvals[1]:
180
+ raise ValueError('Lower HRU bound is greater than upper HRU bound.')
181
+
182
+ idx_retrieve = OrderedDict()
183
+
184
+ for yy in self.__indices.keys():
185
+ if rvals[0] <= yy <= rvals[1]:
186
+ idx_retrieve[self.__indices[yy]] = yy # {local_ids: nhm_ids}
187
+
188
+ return idx_retrieve
189
+ return None
190
+
191
+ def read_cbh_multifile(self, var: Optional[str] = None) -> Union[pd.DataFrame, None]:
192
+ """Read cbh data from multiple csv files.
193
+
194
+ :param var: name of variable to read
195
+ :returns: dataframe of extracted variable
196
+ """
197
+
198
+ if var is None:
199
+ raise ValueError('Variable name (var) must be provided')
200
+
201
+ first = True
202
+ self.__dataframe = None
203
+
204
+ for rr in REGIONS:
205
+ idx_retrieve = self.check_region(region=rr)
206
+
207
+ if len(idx_retrieve) > 0:
208
+ # Build the list of columns to load
209
+ # The given local ids must be adjusted by 5 to reflect:
210
+ # 1) the presence of 6 columns of time information
211
+ # 2) 0-based column names
212
+ load_cols = list(CBH_INDEX_COLS)
213
+ load_cols.extend([xx+5 for xx in idx_retrieve.keys()])
214
+ else:
215
+ load_cols = None
216
+
217
+ if len(idx_retrieve) > 0:
218
+ # The current region contains HRUs in the model subset
219
+ # Read in the data for those HRUs
220
+ cbh_file = f'{self.__src_path}/{rr}_{var}.cbh.gz'
221
+
222
+ print(f'\tLoad {len(idx_retrieve)} HRUs from {rr}')
223
+
224
+ if not os.path.isfile(cbh_file):
225
+ # Missing data file for this variable and region
226
+ raise IOError(f'Required CBH file, {cbh_file}, is missing.')
227
+
228
+ # df = self.read_ascii_file(cbh_file, columns=load_cols)
229
+
230
+ # Small read to get number of columns
231
+ df = pd.read_csv(cbh_file, sep=' ', skipinitialspace=True,
232
+ usecols=load_cols, nrows=2,
233
+ skiprows=3, engine='c', memory_map=True,
234
+ parse_dates={'time': CBH_INDEX_COLS},
235
+ # date_parser=dparse, parse_dates={'time': CBH_INDEX_COLS},
236
+ index_col='time', header=None, na_values=[-99.0, -999.0, 'NaN', 'inf'])
237
+
238
+
239
+ # Override Pandas' rather stupid default of float64
240
+ col_dtypes = {xx: np.float32 for xx in df.columns}
241
+
242
+ # Now read the whole file using float32 instead of float64
243
+ df = pd.read_csv(cbh_file, sep=' ', skipinitialspace=True,
244
+ usecols=load_cols, dtype=col_dtypes,
245
+ skiprows=3, engine='c', memory_map=True,
246
+ parse_dates={'time': CBH_INDEX_COLS},
247
+ # date_parser=dparse, parse_dates={'time': CBH_INDEX_COLS},
248
+ index_col='time', header=None, na_values=[-99.0, -999.0, 'NaN', 'inf'])
249
+
250
+ df.index = pd.to_datetime(df.index, exact=True, cache=True, format=TS_FORMAT)
251
+
252
+ if self.__stdate is not None and self.__endate is not None:
253
+ # Restrict the date range
254
+ df = df[self.__stdate:self.__endate]
255
+
256
+ # Rename columns with NHM HRU ids
257
+ ren_dict = {k+5: v for k, v in idx_retrieve.items()}
258
+
259
+ # NOTE: The rename is an expensive operation
260
+ df.rename(columns=ren_dict, inplace=True)
261
+
262
+ if first:
263
+ self.__dataframe = df.copy()
264
+ first = False
265
+ else:
266
+ self.__dataframe = self.__dataframe.join(df, how='left')
267
+ return self.__dataframe
268
+
269
+ def get_var(self, var: str) -> Union[pd.DataFrame, None]:
270
+ """Get CBH variable.
271
+
272
+ :param var: name of CBH variable
273
+ :returns: dataframe of CBH variable values
274
+ """
275
+
276
+ data = self.read_cbh_multifile(var=var)
277
+ return data
278
+
279
+ def write_ascii(self, pathname: Optional[str] = None,
280
+ fileprefix: Optional[str] = None,
281
+ variables: Optional[List[str]] = None):
282
+ """Write ASCII CBH file for selected variable.
283
+
284
+ By default CBH filenames are saved in the current working directory and
285
+ are named for the selected variable with an extension of .cbh
286
+
287
+ :param pathname: path to save files to
288
+ :param fileprefix: prefix to add to CBH output filename
289
+ :param variables: variables to write to CBH files
290
+ """
291
+
292
+ # For out_order the first six columns contain the time information and
293
+ # are always output for the cbh files
294
+ out_order = [kk for kk in self.__nhm_hrus]
295
+ for cc in ['second', 'minute', 'hour', 'day', 'month', 'year']:
296
+ out_order.insert(0, cc)
297
+
298
+ var_list = []
299
+ if variables is None:
300
+ var_list = CBH_VARNAMES
301
+ elif isinstance(list, variables):
302
+ var_list = variables
303
+
304
+ for cvar in var_list:
305
+ data = self.get_var(var=cvar)
306
+
307
+ # Add time information as columns
308
+ data['year'] = data.index.year
309
+ data['month'] = data.index.month
310
+ data['day'] = data.index.day
311
+ data['hour'] = 0
312
+ data['minute'] = 0
313
+ data['second'] = 0
314
+
315
+ # Output ASCII CBH files
316
+ if fileprefix is None:
317
+ outfile = f'{cvar}.cbh'
318
+ else:
319
+ outfile = f'{fileprefix}_{cvar}.cbh'
320
+
321
+ if pathname is not None:
322
+ outfile = f'{pathname}/{outfile}'
323
+
324
+ out_cbh = open(outfile, 'w')
325
+ out_cbh.write('Written by Bandit\n')
326
+ out_cbh.write(f'{cvar} {len(self.__nhm_hrus)}\n')
327
+ out_cbh.write('########################################\n')
328
+
329
+ data.to_csv(out_cbh, columns=out_order, na_rep='-999', float_format='%0.3f',
330
+ sep=' ', index=False, header=False, encoding=None, chunksize=50)
331
+ out_cbh.close()
332
+
333
+ def write_netcdf(self, filename: str,
334
+ variables: Optional[List[str]] = None):
335
+ """Write CBH to netcdf format file
336
+
337
+ :param filename: name of netCDF output file
338
+ :param variables: list of variables to write to output file
339
+ """
340
+
341
+ # NetCDF-related variables
342
+ var_desc = {'tmax': 'Maximum Temperature', 'tmin': 'Minimum temperature', 'prcp': 'Precipitation'}
343
+ var_units = {'tmax': 'C', 'tmin': 'C', 'prcp': 'inches'}
344
+
345
+ # Create a netCDF file for the CBH data
346
+ nco = nc.Dataset(filename, 'w', clobber=True)
347
+ nco.createDimension('hru', len(self.__nhm_hrus))
348
+ nco.createDimension('time', None)
349
+
350
+ timeo = nco.createVariable('time', 'f4', ('time'))
351
+ timeo.calendar = 'standard'
352
+ # timeo.bounds = 'time_bnds'
353
+
354
+ # FIXME: Days since needs to be set to the starting date of the model pull
355
+ timeo.units = 'days since 1980-01-01 00:00:00'
356
+
357
+ hruo = nco.createVariable('hru', 'i4', ('hru'))
358
+ hruo.long_name = 'Hydrologic Response Unit ID (HRU)'
359
+
360
+ var_list = []
361
+ if variables is None:
362
+ var_list = CBH_VARNAMES
363
+ elif isinstance(list, variables):
364
+ var_list = variables
365
+
366
+ for cvar in var_list:
367
+ varo = nco.createVariable(cvar, 'f4', ('time', 'hru'), fill_value=nc.default_fillvals['f4'], zlib=True)
368
+ varo.long_name = var_desc[cvar]
369
+ varo.units = var_units[cvar]
370
+
371
+ nco.setncattr('Description', 'Climate by HRU')
372
+ # nco.setncattr('Bandit_version', __version__)
373
+ # nco.setncattr('NHM_version', nhmparamdb_revision)
374
+
375
+ # Write the HRU ids
376
+ hruo[:] = self.__nhm_hrus
377
+
378
+ first = True
379
+ for cvar in var_list:
380
+ data = self.get_var(var=cvar)
381
+
382
+ if first:
383
+ timeo[:] = nc.date2num(data.index.tolist(),
384
+ units='days since 1980-01-01 00:00:00',
385
+ calendar='standard')
386
+ first = False
387
+
388
+ # Write the CBH values
389
+ nco.variables[cvar][:, :] = data[self.__nhm_hrus].values
390
+
391
+ nco.close()
392
+
393
+ # def write_cbh_subset(self, outdir):
394
+ # outdata = None
395
+ # first = True
396
+ #
397
+ # for vv in CBH_VARNAMES:
398
+ # outorder = list(CBH_INDEX_COLS)
399
+ #
400
+ # for rr, rvals in iteritems(self.__mapping):
401
+ # idx_retrieve = {}
402
+ #
403
+ # for yy in self.__nhm_hrus.keys():
404
+ # if rvals[0] <= yy <= rvals[1]:
405
+ # idx_retrieve[yy] = self.__nhm_hrus[yy]
406
+ #
407
+ # if len(idx_retrieve) > 0:
408
+ # self.__src_path = '{}/{}_{}.cbh.gz'.format(self.__cbhdb_dir, rr, vv)
409
+ # self.read_cbh()
410
+ # if first:
411
+ # outdata = self.__data
412
+ # first = False
413
+ # else:
414
+ # outdata = pd.merge(outdata, self.__data, how='left', left_index=True, right_index=True)
415
+ #
416
+ # # Append the HRUs as ordered for the subset
417
+ # outorder.extend(self.__nhm_hrus)
418
+ #
419
+ # out_cbh = open('{}/{}.cbh'.format(outdir, vv), 'w')
420
+ # out_cbh.write('Written by pyPRMS.Cbh\n')
421
+ # out_cbh.write('{} {}\n'.format(vv, len()))
422
+
423
+ # def read_cbh_parq(self, src_dir):
424
+ # """Read CBH files stored in the parquet format"""
425
+ # if self.__indices:
426
+ # pfile = fp.ParquetFile('{}/daymet_{}.parq'.format(src_dir, self.__var))
427
+ # self.__data = pfile.to_pandas(self.__indices)
428
+ #
429
+ # if self.__stdate is not None and self.__endate is not None:
430
+ # # Given a date range to restrict the output
431
+ # self.__data = self.__data[self.__stdate:self.__endate]
432
+ #
433
+ # self.__data['year'] = self.__data.index.year
434
+ # self.__data['month'] = self.__data.index.month
435
+ # self.__data['day'] = self.__data.index.day
436
+ # self.__data['hour'] = 0
437
+ # self.__data['minute'] = 0
438
+ # self.__data['second'] = 0
439
+ #
440
+ # def read_cbh_hdf(self, src_dir):
441
+ # """Read CBH files stored in HDF5 format"""
442
+ # if self.__indices:
443
+ # # self.__data = pd.read_hdf('{}/daymet_{}.h5'.format(src_dir, self.__var), columns=self.__indices)
444
+ # self.__data = pd.read_hdf('{}/daymet_{}.h5'.format(src_dir, self.__var))
445
+ #
446
+ # if self.__stdate is not None and self.__endate is not None:
447
+ # # Given a date range to restrict the output
448
+ # self.__data = self.__data[self.__stdate:self.__endate]
449
+ #
450
+ # self.__data = self.__data[self.__indices]
451
+ #
452
+ # self.__data['year'] = self.__data.index.year
453
+ # self.__data['month'] = self.__data.index.month
454
+ # self.__data['day'] = self.__data.index.day
455
+ # self.__data['hour'] = 0
456
+ # self.__data['minute'] = 0
457
+ # self.__data['second'] = 0
458
+ #
@@ -0,0 +1,199 @@
1
+ import datetime
2
+ import fsspec # type: ignore
3
+ import numpy as np
4
+ import os
5
+ import pandas as pd # type: ignore
6
+ import netCDF4 as nc # type: ignore
7
+ import xarray as xr
8
+
9
+ from pathlib import Path
10
+ from typing import Dict, List, Optional, Union
11
+
12
+ __author__ = 'Parker Norton (pnorton@usgs.gov)'
13
+
14
+ CBH_VARNAMES = ['prcp', 'tmin', 'tmax']
15
+ CBH_INDEX_COLS = [0, 1, 2, 3, 4, 5]
16
+
17
+
18
+ class CbhNetcdf(object):
19
+ """Climate-By-HRU (CBH) files in netCDF format."""
20
+
21
+ def __init__(self, src_path: Union[str, Path],
22
+ nhm_hrus: List[int],
23
+ st_date: Optional[datetime.datetime] = None,
24
+ en_date: Optional[datetime.datetime] = None):
25
+ """
26
+ :param src_path: Full path to netCDF file
27
+ :param st_date: The starting date for restricting CBH results
28
+ :param en_date: The ending date for restricting CBH results
29
+ :param nhm_hrus: List of NHM HRU IDs to extract from CBH
30
+ """
31
+
32
+ if isinstance(src_path, str):
33
+ src_path = Path(src_path)
34
+
35
+ self.__src_path = src_path
36
+ self.__stdate = st_date
37
+ self.__endate = en_date
38
+ self.__nhm_hrus = nhm_hrus
39
+
40
+ self.__dataset = self.read_netcdf()
41
+
42
+ @property
43
+ def data(self) -> xr.Dataset:
44
+ """Returns the CBH dataset.
45
+
46
+ :returns: xarray dataset
47
+ """
48
+
49
+ return self.__dataset
50
+
51
+ def read_netcdf(self) -> xr.Dataset:
52
+ """Read CBH files stored in netCDF format.
53
+
54
+ :returns: xarray dataset
55
+ """
56
+
57
+ filepath = str(self.__src_path.resolve())
58
+ match self.__src_path.suffix:
59
+ case '.nc':
60
+ ds = xr.open_mfdataset(filepath, chunks={}, combine='by_coords',
61
+ data_vars='minimal', decode_cf=True, engine='netcdf4',
62
+ parallel=True)
63
+ case '.zarr':
64
+ ds = xr.open_zarr(filepath, consolidated=True)
65
+ case '.json':
66
+ fs = fsspec.filesystem('reference', fo=filepath)
67
+ m = fs.get_mapper('')
68
+
69
+ ds = xr.open_dataset(m, engine='zarr', chunks={}, backend_kwargs={'consolidated': False})
70
+
71
+ if 'nhm_id' in ds.data_vars:
72
+ # dataset has nhm_id variable so use it as the nhru dimension
73
+ ds = ds.assign_coords(nhru=ds.nhm_id)
74
+
75
+ if self.__stdate is None:
76
+ # Use first date from the dataset
77
+ self.__stdate = pd.to_datetime(str(ds.time[0].values))
78
+
79
+ if self.__endate is None:
80
+ # Use last date from the dataset
81
+ self.__endate = pd.to_datetime(str(ds.time[-1].values))
82
+
83
+ return ds
84
+
85
+ def get_var(self, var: str) -> pd.DataFrame:
86
+ """Get a variable from the netCDF file.
87
+
88
+ :param var: Name of the variable
89
+ :returns: dataframe of variable values
90
+ """
91
+
92
+ try:
93
+ data = self.__dataset[var].sel(time=slice(self.__stdate, self.__endate),
94
+ nhru=self.__nhm_hrus).to_pandas()
95
+ except IndexError:
96
+ print(f'ERROR: Dimensions (time, nhru) were used to subset {var} which expects' +
97
+ f'dimensions ({" ".join(map(str, self.__dataset[var].coords))})')
98
+ raise
99
+ except KeyError:
100
+ # Happens when older hruid dimension is used instead of nhru
101
+ data = self.__dataset[var].sel(time=slice(self.__stdate, self.__endate),
102
+ hruid=self.__nhm_hrus).to_pandas()
103
+
104
+ return data
105
+
106
+ def write_ascii(self, filename: Union[str, os.PathLike, Path],
107
+ variable: str):
108
+ """Write CBH data for variable to PRMS ASCII-formatted file.
109
+
110
+ :param filename: Climate-by-HRU filename
111
+ :param variable: CBH variable to write
112
+ """
113
+
114
+ # For out_order the first six columns contain the time information and
115
+ # are always output for the cbh files
116
+ out_order: List[Union[int, str]] = [kk for kk in self.__nhm_hrus]
117
+ for cc in ['second', 'minute', 'hour', 'day', 'month', 'year']:
118
+ out_order.insert(0, cc)
119
+
120
+ if variable in self.__dataset.data_vars:
121
+ data = self.get_var(var=variable)
122
+
123
+ # Add time information as columns
124
+ data['year'] = data.index.year
125
+ data['month'] = data.index.month
126
+ data['day'] = data.index.day
127
+ data['hour'] = 0
128
+ data['minute'] = 0
129
+ data['second'] = 0
130
+
131
+ out_cbh = open(filename, 'w')
132
+ out_cbh.write('Written by Bandit\n')
133
+ out_cbh.write(f'{variable} {len(self.__nhm_hrus)}\n')
134
+ out_cbh.write('########################################\n')
135
+ # data.to_csv(out_cbh, columns=out_order, na_rep='-999', float_format='%0.3f',
136
+ data.to_csv(out_cbh, columns=out_order, na_rep='-999', float_format='%0.2f',
137
+ sep=' ', index=False, header=False, lineterminator='\n', encoding=None, chunksize=50)
138
+ out_cbh.close()
139
+ else:
140
+ print(f'WARNING: {variable} does not exist in source CBH files..skipping')
141
+
142
+ def write_netcdf(self, filename: Union[str, os.PathLike, Path],
143
+ variables: Optional[List[str]] = None,
144
+ global_attrs: Optional[Dict] = None):
145
+ """Write CBH variables to netCDF file.
146
+
147
+ :param filename: name of netCDF output file
148
+ :param variables: list of CBH variables to write
149
+ :param global_attrs: optional dictionary of attributes to include in netcdf file
150
+ """
151
+
152
+ ds = self.__dataset
153
+ ds = ds.sel(time=slice(self.__stdate, self.__endate), nhru=self.__nhm_hrus)
154
+
155
+ if variables is None:
156
+ pass
157
+ elif isinstance(variables, list):
158
+ ds = ds[variables]
159
+
160
+ # Remove _FillValue from coordinate variables
161
+ for vv in list(ds.coords):
162
+ ds[vv].encoding.update({'_FillValue': None})
163
+
164
+ ds['crs'] = self.__dataset['crs']
165
+ ds['crs'].encoding.update({'_FillValue': None,
166
+ 'contiguous': True})
167
+
168
+ ds['time'].attrs['standard_name'] = 'time'
169
+ ds['time'].attrs['long_name'] = 'time'
170
+
171
+ # Add nhm_id variable which will be the global NHM IDs
172
+ ds['nhm_id'] = ds['nhru']
173
+ ds['nhm_id'].attrs['long_name'] = 'Global model Hydrologic Response Unit ID (HRU)'
174
+
175
+ # Change the nhru coordinate variable values to reflect the local model HRU IDs
176
+ ds = ds.assign_coords(nhru=np.arange(1, ds.nhru.values.size+1, dtype=ds.nhru.dtype))
177
+ ds['nhru'].attrs['long_name'] = 'Local model Hydrologic Response Unit ID (HRU)'
178
+ ds['nhru'].attrs['cf_role'] = 'timeseries_id'
179
+
180
+ # Add/update global attributes
181
+ ds.attrs['Description'] = 'Climate-by-HRU'
182
+
183
+ if global_attrs is not None:
184
+ for kk, vv in global_attrs.items():
185
+ ds.attrs[kk] = vv
186
+
187
+ encoding = {}
188
+
189
+ for cvar in ds.variables:
190
+ if ds[cvar].ndim > 1:
191
+ encoding[cvar] = dict(_FillValue=ds[cvar].encoding['_FillValue'],
192
+ compression='zlib',
193
+ complevel=2,
194
+ fletcher32=True)
195
+ else:
196
+ encoding[cvar] = dict(_FillValue=None,
197
+ contiguous=True)
198
+
199
+ ds.load().to_netcdf(filename, engine='netcdf4', format='NETCDF4', encoding=encoding)
pyPRMS/cbh/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ from .Cbh import Cbh
2
+ from .CbhAscii import CbhAscii
3
+ from .CbhNetcdf import CbhNetcdf