pyVPRM 3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. pyVPRM/VPRM.py +1118 -0
  2. pyVPRM/__init__.py +1 -0
  3. pyVPRM/lib/__init__.py +0 -0
  4. pyVPRM/lib/downmodis.py +1012 -0
  5. pyVPRM/lib/fancy_plot.py +75 -0
  6. pyVPRM/lib/flux_tower_class.py +444 -0
  7. pyVPRM/lib/fluxnet_info/fluxnet_sites.pkl +0 -0
  8. pyVPRM/lib/fluxnet_info/site_infos.txt +218 -0
  9. pyVPRM/lib/functions.py +471 -0
  10. pyVPRM/meteorologies/__init__.py +0 -0
  11. pyVPRM/meteorologies/era5_class_dkrz.py +279 -0
  12. pyVPRM/meteorologies/era5_class_draft.py +60 -0
  13. pyVPRM/meteorologies/era5_monthly_xr.py +149 -0
  14. pyVPRM/meteorologies/met_base_class.py +67 -0
  15. pyVPRM/meteorologies/met_local_measurement.py +56 -0
  16. pyVPRM/sat_managers/__init__.py +0 -0
  17. pyVPRM/sat_managers/base_manager.py +430 -0
  18. pyVPRM/sat_managers/city.py +21 -0
  19. pyVPRM/sat_managers/copernicus.py +29 -0
  20. pyVPRM/sat_managers/esa_world_cover.py +21 -0
  21. pyVPRM/sat_managers/mapbiomas.py +35 -0
  22. pyVPRM/sat_managers/modis.py +266 -0
  23. pyVPRM/sat_managers/proba_v.py +86 -0
  24. pyVPRM/sat_managers/sentinel2.py +192 -0
  25. pyVPRM/sat_managers/synmap.py +21 -0
  26. pyVPRM/sat_managers/viirs.py +232 -0
  27. pyVPRM/sat_managers/viirs09ga.py +173 -0
  28. pyVPRM/vprm_configs/__init__.py +0 -0
  29. pyVPRM/vprm_configs/copernicus_land_cover.yaml +94 -0
  30. pyVPRM/vprm_configs/esa_world_cover.yaml +71 -0
  31. pyVPRM/vprm_configs/synmap.yaml +106 -0
  32. pyVPRM/vprm_models/__init__.py +1 -0
  33. pyVPRM/vprm_models/model_params/__init__.py +0 -0
  34. pyVPRM/vprm_models/vprm_base.py +691 -0
  35. pyVPRM/vprm_models/vprm_base_no_xeric.py +585 -0
  36. pyVPRM/vprm_models/vprm_modified.py +530 -0
  37. pyVPRM/vprm_models/vprm_nn.py +130 -0
  38. pyVPRM-3.0.dist-info/LICENSE +21 -0
  39. pyVPRM-3.0.dist-info/METADATA +40 -0
  40. pyVPRM-3.0.dist-info/RECORD +42 -0
  41. pyVPRM-3.0.dist-info/WHEEL +5 -0
  42. pyVPRM-3.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1012 @@
1
+ #!/usr/bin/env python
2
+ # class to download modis data
3
+ #
4
+ # (c) Copyright Luca Delucchi 2010-2016
5
+ # (c) Copyright Logan C Byers 2014
6
+ # Authors: Luca Delucchi
7
+ # Logan C Byers
8
+ # Email: luca dot delucchi at fmach dot it
9
+ # loganbyers@ku.edu
10
+ #
11
+ ##################################################################
12
+ #
13
+ # This MODIS Python class is licensed under the terms of GNU GPL 2.
14
+ # This program is free software; you can redistribute it and/or
15
+ # modify it under the terms of the GNU General Public License as
16
+ # published by the Free Software Foundation; either version 2 of
17
+ # the License, or (at your option) any later version.
18
+ # This program is distributed in the hope that it will be useful,
19
+ # but WITHOUT ANY WARRANTY; without even implied warranty of
20
+ # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.
21
+ # See the GNU General Public License for more details.
22
+ #
23
+ ##################################################################
24
+ """Module to download MODIS HDF files from NASA repository.
25
+ It supports both FTP and HTTP repositories
26
+
27
+ Classes:
28
+
29
+ * :class:`modisHtmlParser`
30
+ * :class:`downModis`
31
+
32
+ Functions:
33
+
34
+ * :func:`urljoin`
35
+ * :func:`getNewerVersion`
36
+ * :func:`str2date`
37
+
38
+ """
39
+
40
+ from builtins import dict
41
+
42
+ from datetime import date
43
+ from datetime import timedelta
44
+ import os
45
+ import sys
46
+ import glob
47
+ import logging
48
+ import socket
49
+ from ftplib import FTP
50
+ import ftplib
51
+
52
+ import requests
53
+
54
+ # urllib in python 2 and 3
55
+ from urllib.request import urlopen
56
+ import urllib.request
57
+ import urllib.error
58
+ from base64 import b64encode
59
+ from html.parser import HTMLParser
60
+ import re
61
+ import netrc
62
+ import warnings
63
+
64
+ # urlparse in python 2 and 3
65
+ try:
66
+ from urlparse import urlparse
67
+
68
+ URLPARSE = True
69
+ except ImportError:
70
+ try:
71
+ from urllib.parse import urlparse
72
+
73
+ URLPARSE = True
74
+ except ImportError:
75
+ URLPARSE = False
76
+ warnings.warn(
77
+ "urlparse not found, it is not possible to use" " netrc file", ImportError
78
+ )
79
+ # global GDAL
80
+
81
+ # try:
82
+ # import osgeo.gdal as gdal
83
+ # GDAL = True
84
+ # except ImportError:
85
+ # try:
86
+ # import gdal
87
+ # GDAL = True
88
+ # except ImportError:
89
+ # GDAL = False
90
+ # warnings.warn('Python GDAL library not found, please install'
91
+ # ' it to check data downloaded with pyModis', ImportError)
92
+ # # setup gdal
93
+ # if GDAL:
94
+ # gdal.UseExceptions()
95
+ # gdalDriver = gdal.GetDriverByName('HDF4')
96
+ # if not gdalDriver:
97
+ # GDAL = False
98
+ # warnings.warn("GDAL installation has no support for HDF4, "
99
+ # "please update GDAL", ImportError)
100
+
101
+
102
+ def urljoin(*args):
103
+ """Joins given arguments into a url. Trailing but not leading slashes are
104
+ stripped for each argument.
105
+ http://stackoverflow.com/a/11326230
106
+
107
+ :return: a string
108
+ """
109
+
110
+ return "/".join([str(x).rstrip("/") for x in args])
111
+
112
+
113
+ def getNewerVersion(oldFile, newFile):
114
+ """Check two files to determine which is newer
115
+
116
+ :param str oldFile: one of the two similar files
117
+ :param str newFile: one of the two similar files
118
+
119
+ :return: the name of newer file
120
+ """
121
+ # get the processing date (YYYYDDDHHMMSS) from the file strings
122
+ if oldFile.split(".")[4] > newFile.split(".")[4]:
123
+ return oldFile
124
+ else:
125
+ return newFile
126
+
127
+
128
+ def str2date(datestring):
129
+ """Convert to datetime.date object from a string
130
+
131
+ :param str datestring string with format (YYYY-MM-DD)
132
+ :return: a datetime.date object representing datestring
133
+ """
134
+ if "-" in datestring:
135
+ stringSplit = datestring.split("-")
136
+ elif "." in datestring:
137
+ stringSplit = datestring.split(".")
138
+ elif " " in datestring:
139
+ stringSplit = datestring.split(" ")
140
+ return date(int(stringSplit[0]), int(stringSplit[1]), int(stringSplit[2]))
141
+
142
+
143
+ class ModisHTTPRedirectHandler(urllib.request.HTTPRedirectHandler):
144
+ """Class to return 302 error"""
145
+
146
+ def http_error_302(self, req, fp, code, msg, headers):
147
+ return urllib.request.HTTPRedirectHandler.http_error_302(
148
+ self, req, fp, code, msg, headers
149
+ )
150
+
151
+
152
+ class modisHtmlParser(HTMLParser):
153
+ """A class to parse HTML
154
+
155
+ :param fh: content of http request
156
+ """
157
+
158
+ def __init__(self, fh):
159
+ """Function to initialize the object"""
160
+ HTMLParser.__init__(self)
161
+ self.fileids = []
162
+ self.feed(str(fh))
163
+
164
+ def handle_starttag(self, tag, attrs):
165
+ if tag == "a":
166
+ attrD = dict(attrs)
167
+ self.fileids.append(attrD["href"].replace("/", ""))
168
+
169
+ def get_all(self):
170
+ """Return everything"""
171
+ return self.fileids
172
+
173
+ def get_dates(self):
174
+ """Return a list of directories with date"""
175
+ regex = re.compile(r"(\d{4})[/.-](\d{2})[/.-](\d{2})$")
176
+ alldata = set([elem for elem in self.fileids if regex.match(elem)])
177
+ return sorted(list(alldata))
178
+
179
+ def get_tiles(self, prod, tiles, jpeg=False):
180
+ """Return a list of files to download
181
+
182
+ :param str prod: the code of MODIS product that we are going to
183
+ analyze
184
+ :param list tiles: the list of tiles to consider
185
+ :param bool jpeg: True to also check for jpeg data
186
+ """
187
+ finalList = []
188
+ for i in self.fileids:
189
+ # distinguish jpg from hdf by where the tileID is within the string
190
+ # jpgs have the tileID at index 3, hdf have tileID at index 2
191
+ name = i.split(".")
192
+ # if product is not in the filename, move to next filename in list
193
+ if not name.count(prod):
194
+ continue
195
+ # if tiles are not specified and the file is not a jpg, add to list
196
+ if not tiles and not (name.count("jpg") or name.count("BROWSE")):
197
+ finalList.append(i)
198
+ # if tiles are specified
199
+ if tiles:
200
+ # if a tileID is at index 3 and jpgs are to be downloaded
201
+ if tiles.count(name[3]) == 1 and jpeg:
202
+ finalList.append(i)
203
+ # if a tileID is at in index 2, it is known to be HDF
204
+ elif tiles.count(name[2]) == 1:
205
+ finalList.append(i)
206
+ return finalList
207
+
208
+
209
+ class downModis:
210
+ """A class to download MODIS data from NASA FTP or HTTP repositories
211
+
212
+ :param str destinationFolder: where the files will be stored
213
+ :param str password: the password required by NASA authentication system
214
+ :param str user: the user namerequired by NASA authentication system
215
+ :param str url: the base url from where to download the MODIS data,
216
+ it can be FTP or HTTP but it has to start with
217
+ 'ftp://' or 'http://' or 'https://'
218
+ :param str path: the directory where the data that you want to
219
+ download are stored on the FTP server. For HTTP
220
+ requests, this is the part of the url between the 'url'
221
+ parameter and the 'product' parameter.
222
+ :param str product: the code of the product to download, the code
223
+ should be idential to the one of the url
224
+ :param str tiles: a set of tiles to be downloaded, None == all tiles.
225
+ This can be passed as a string of tileIDs separated
226
+ by commas, or as a list of individual tileIDs
227
+ :param str today: the day to start downloading; in order to pass a
228
+ date different from today use the format YYYY-MM-DD
229
+ :param str enddate: the day to end downloading; in order to pass a
230
+ date use the format YYYY-MM-DD. This day must be
231
+ before the 'today' parameter. Downloading happens
232
+ in reverse order (currently)
233
+
234
+ :param int delta: timelag i.e. the number of days starting from
235
+ today backwards. Will be overwritten if
236
+ 'enddate' is specifed during instantiation
237
+ :param bool jpeg: set to True if you want to download the JPG overview
238
+ file in addition to the HDF
239
+ :param bool debug: set to True if you want to obtain debug information
240
+ :param int timeout: Timeout value for HTTP server (seconds)
241
+ :param bool checkgdal: variable to set the GDAL check
242
+ """
243
+
244
+ def __init__(
245
+ self,
246
+ destinationFolder,
247
+ password=None,
248
+ user=None,
249
+ token=None,
250
+ url="https://e4ftl01.cr.usgs.gov",
251
+ tiles=None,
252
+ path="MOLT",
253
+ product="MOD11A1.006",
254
+ today=None,
255
+ enddate=None,
256
+ delta=10,
257
+ jpg=False,
258
+ debug=False,
259
+ timeout=30,
260
+ checkgdal=True,
261
+ ):
262
+ """Function to initialize the object"""
263
+
264
+ self.token = None
265
+ self.user = None
266
+ self.password = None
267
+
268
+ # prepare the base url and set the url type (ftp/http)
269
+ if "ftp://" in url:
270
+ self.url = url.replace("ftp://", "").rstrip("/")
271
+ self.urltype = "ftp"
272
+ elif "http://" in url:
273
+ self.url = url
274
+ self.urltype = "http"
275
+ elif "https://" in url:
276
+ self.url = url
277
+ self.urltype = "http"
278
+ else:
279
+ raise IOError("The url should contain 'ftp://' or 'http://'")
280
+
281
+ # token case
282
+ if token:
283
+ # token for download
284
+ self.token = token
285
+ # user and password case
286
+ elif user and password:
287
+ # user for download
288
+ self.user = user
289
+ # password for download
290
+ self.password = password
291
+ # netrc case
292
+ else:
293
+ if not URLPARSE:
294
+ raise IOError("Please use 'user' and 'password' parameters")
295
+ self.domain = urlparse(self.url).hostname
296
+ try:
297
+ nt = netrc.netrc()
298
+ except:
299
+ raise IOError(
300
+ "Please set 'user' and 'password' parameters netrc file does not exist"
301
+ )
302
+ try:
303
+ account = nt.hosts[self.domain]
304
+ except:
305
+ try:
306
+ account = nt.hosts["urs.earthdata.nasa.gov"]
307
+ except:
308
+ raise IOError(
309
+ "Please set 'user' and 'password' parameters netrc file does not contain parameter for NASA url"
310
+ )
311
+ # user for download
312
+ self.user = account[0]
313
+ # password for download
314
+ self.password = account[2]
315
+ # token for download from password
316
+ self.token = self.password if self.user == "token" else None
317
+
318
+ if not self.user and not self.password and not self.token:
319
+ raise IOError("You must provide either a token or a user and password")
320
+
321
+ # set the http header
322
+ if self.token:
323
+ self.http_header = {"Authorization": f"Bearer {self.token}"}
324
+ else:
325
+ self.userpwd = "{us}:{pw}".format(us=self.user, pw=self.password)
326
+ userAndPass = b64encode(str.encode(self.userpwd)).decode("ascii")
327
+ self.http_header = {"Authorization": "Basic %s" % userAndPass}
328
+
329
+ cookieprocessor = urllib.request.HTTPCookieProcessor()
330
+ opener = urllib.request.build_opener(ModisHTTPRedirectHandler, cookieprocessor)
331
+ urllib.request.install_opener(opener)
332
+ # the product (product_code.004 or product_cod.005)
333
+ self.product = product
334
+ self.product_code = product.split(".")[0]
335
+ # url directory where data are located
336
+ self.path = urljoin(path, self.product)
337
+ # tiles to downloads
338
+ if isinstance(tiles, str):
339
+ self.tiles = tiles.split(",")
340
+ else: # tiles are list, tuple, or None
341
+ self.tiles = tiles
342
+ # set destination folder
343
+ if not os.path.isdir(destinationFolder):
344
+ os.makedirs(destinationFolder)
345
+ self.writeFilePath = destinationFolder
346
+ elif os.access(destinationFolder, os.W_OK):
347
+ self.writeFilePath = destinationFolder
348
+ else:
349
+ try:
350
+ os.mkdir(destinationFolder)
351
+ self.writeFilePath = destinationFolder
352
+ except:
353
+ raise Exception(
354
+ "Folder to store downloaded files does not "
355
+ "exist or is not writeable"
356
+ )
357
+ # return the name of product
358
+ if len(self.path.split("/")) == 2:
359
+ self.product = self.path.split("/")[1]
360
+ elif len(self.path.split("/")) == 3:
361
+ self.product = self.path.split("/")[2]
362
+ # write a file with the name of file to be downloaded
363
+ self.filelist = open(
364
+ os.path.join(
365
+ self.writeFilePath, "listfile{pro}.txt".format(pro=self.product)
366
+ ),
367
+ "w",
368
+ )
369
+ # set if to download jpgs
370
+ self.jpeg = jpg
371
+ # today, or the last day in the download series chronologically
372
+ self.today = today
373
+ # chronologically the first day in the download series
374
+ self.enday = enddate
375
+ # default number of days to consider if enddate not specified
376
+ self.delta = delta
377
+ # status of tile download
378
+ self.status = True
379
+ # for debug, you can download only xml files
380
+ self.debug = debug
381
+ # for logging
382
+ log_filename = os.path.join(
383
+ self.writeFilePath, "modis{pro}.log".format(pro=self.product)
384
+ )
385
+ log_format = "%(asctime)s - %(levelname)s - %(message)s"
386
+ logging.basicConfig(
387
+ filename=log_filename, level=logging.DEBUG, format=log_format
388
+ )
389
+ logging.captureWarnings(True)
390
+ # global connection attempt counter
391
+ self.nconnection = 0
392
+ # timeout for HTTP connection before failing (seconds)
393
+ self.timeout = timeout
394
+ # files within the directory where data will be saved
395
+ self.fileInPath = []
396
+ for f in os.listdir(self.writeFilePath):
397
+ if os.path.isfile(os.path.join(self.writeFilePath, f)):
398
+ self.fileInPath.append(f)
399
+ # global GDAL
400
+ # if not GDAL and checkgdal:
401
+ # logging.warning("WARNING: Python GDAL library not found")
402
+ # elif GDAL and not checkgdal:
403
+ # GDAL = False
404
+ self.dirData = []
405
+ # set today and enday dates
406
+ self._getToday()
407
+
408
+ def removeEmptyFiles(self):
409
+ """Function to remove files in the download directory that have
410
+ filesize equal to 0
411
+ """
412
+ year = str(self.today.year)
413
+ prefix = self.product.split(".")[0]
414
+ files = glob.glob1(self.writeFilePath, "%s.A%s*" % (prefix, year))
415
+ for f in files:
416
+ fil = os.path.join(self.writeFilePath, f)
417
+ if os.path.getsize(fil) == 0:
418
+ os.remove(fil)
419
+
420
+ def connect(self, ncon=20):
421
+ """Connect to the server and fill the dirData variable
422
+
423
+ :param int ncon: maximum number of attempts to connect to the HTTP
424
+ server before failing
425
+ """
426
+ if self.urltype == "ftp":
427
+ self._connectFTP(ncon)
428
+ elif self.urltype == "http":
429
+ self._connectHTTP(ncon)
430
+ if len(self.dirData) == 0:
431
+ raise Exception(
432
+ "There are some troubles with the server. "
433
+ "The directory seems to be empty"
434
+ )
435
+
436
+ def _connectHTTP(self, ncon=20):
437
+ """Connect to HTTP server, create a list of directories for all days
438
+
439
+ :param int ncon: maximum number of attempts to connect to the HTTP
440
+ server before failing. If ncon < 0, connection
441
+ attempts are unlimited in number
442
+ """
443
+ self.nconnection += 1
444
+ try:
445
+ url = urljoin(self.url, self.path)
446
+ try:
447
+ req = urllib.request.Request(url, headers=self.http_header)
448
+ http = urllib.request.urlopen(req)
449
+ self.dirData = modisHtmlParser(http.read()).get_dates()
450
+ except Exception as e:
451
+ logging.error(
452
+ "Error in connection. Code {code}, "
453
+ "reason {re}".format(code=e.code, re=e.reason)
454
+ )
455
+ http = urlopen(url, timeout=self.timeout)
456
+ self.dirData = modisHtmlParser(http.read()).get_dates()
457
+ self.dirData.reverse()
458
+ except Exception as e:
459
+ try:
460
+ logging.error(
461
+ "Error in connection. Code {code}, "
462
+ "reason {re}".format(code=e.code, re=e.reason)
463
+ )
464
+ except:
465
+ logging.error("Error {er}".format(er=e))
466
+ if self.nconnection <= ncon or ncon < 0:
467
+ self._connectHTTP()
468
+
469
+ def _connectFTP(self, ncon=20):
470
+ """Set connection to ftp server, move to path where data are stored,
471
+ and create a list of directories for all days
472
+
473
+ :param int ncon: maximum number of attempts to connect to the FTP
474
+ server before failing.
475
+
476
+ """
477
+ if not self.user and not self.password:
478
+ raise IOError("You must provide a user and password to connect.")
479
+ self.nconnection += 1
480
+ try:
481
+ # connect to ftp server
482
+ self.ftp = FTP(self.url)
483
+ self.ftp.login(self.user, self.password)
484
+ # enter in directory
485
+ self.ftp.cwd(self.path)
486
+ self.dirData = []
487
+ # return data inside directory
488
+ self.ftp.dir(self.dirData.append)
489
+ # reverse order of data for have first the nearest to today
490
+ self.dirData.reverse()
491
+ # ensure dirData contains only directories, remove all references to files
492
+ self.dirData = [
493
+ elem.split()[-1] for elem in self.dirData if elem.startswith("d")
494
+ ]
495
+ if self.debug:
496
+ logging.debug("Open connection {url}".format(url=self.url))
497
+ except (EOFError, ftplib.error_perm) as e:
498
+ logging.error("Error in connection: {err}".format(err=e))
499
+ if self.nconnection <= ncon:
500
+ self._connectFTP()
501
+
502
+ def closeFTP(self):
503
+ """Close ftp connection and close the file list document"""
504
+ self.ftp.quit()
505
+ self.closeFilelist()
506
+ if self.debug:
507
+ logging.debug("Close connection {url}".format(url=self.url))
508
+
509
+ def closeFilelist(self):
510
+ """Function to close the file list of where the files are downloaded"""
511
+ self.filelist.close()
512
+
513
+ def setDirectoryIn(self, day):
514
+ """Enter into the file directory of a specified day
515
+
516
+ :param str day: a string representing a day in format YYYY.MM.DD
517
+ """
518
+ try:
519
+ self.ftp.cwd(day)
520
+ except (ftplib.error_reply, socket.error) as e:
521
+ logging.error(
522
+ "Error {err} entering in directory " "{name}".format(err=e, name=day)
523
+ )
524
+ self.setDirectoryIn(day)
525
+
526
+ def setDirectoryOver(self):
527
+ """Move up within the file directory"""
528
+ try:
529
+ self.ftp.cwd("..")
530
+ except (ftplib.error_reply, socket.error) as e:
531
+ logging.error("Error {err} when trying to come back".format(err=e))
532
+ self.setDirectoryOver()
533
+
534
+ def _getToday(self):
535
+ """Set the dates for the start and end of downloading"""
536
+ if self.today is None:
537
+ # set today variable from datetime.date method
538
+ self.today = date.today()
539
+ elif isinstance(self.today, str):
540
+ # set today variable from string data passed by user
541
+ self.today = str2date(self.today)
542
+ # set enday variable to data passed by user
543
+ if isinstance(self.enday, str):
544
+ self.enday = str2date(self.enday)
545
+ # set delta
546
+ if self.today and self.enday:
547
+ if self.today < self.enday:
548
+ self.today, self.enday = self.enday, self.today
549
+ delta = self.today - self.enday
550
+ self.delta = abs(delta.days) + 1
551
+
552
+ def getListDays(self):
553
+ """Return a list of all selected days"""
554
+ today_s = self.today.strftime("%Y.%m.%d")
555
+ # dirData is reverse sorted
556
+ for i, d in enumerate(self.dirData):
557
+ if d <= today_s:
558
+ today_index = i
559
+ break
560
+ # else:
561
+ # logging.error("No data available for requested days")
562
+ # import sys
563
+ # sys.exit()
564
+ days = self.dirData[today_index:][: self.delta]
565
+ # this is useful for 8/16 days data, delta could download more images
566
+ # that you want
567
+ if self.enday is not None:
568
+ enday_s = self.enday.strftime("%Y.%m.%d")
569
+ delta = 0
570
+ # make a full cycle from the last index and find
571
+ # it make a for cicle from the last value and find the internal
572
+ # delta to remove file outside temporaly range
573
+ for i in range(0, len(days)):
574
+ if days[i] < enday_s:
575
+ break
576
+ else:
577
+ delta = delta + 1
578
+ # remove days outside new delta
579
+ days = days[:delta]
580
+ return days
581
+
582
+ def getAllDays(self):
583
+ """Return a list of all days"""
584
+ return self.dirData
585
+
586
+ def getFilesList(self, day=None):
587
+ """Returns a list of files to download. HDF and XML files are
588
+ downloaded by default. JPG files will be downloaded if
589
+ self.jpeg == True.
590
+
591
+ :param str day: the date of data in format YYYY.MM.DD
592
+
593
+ :return: a list of files to download for the day
594
+ """
595
+ if self.urltype == "http":
596
+ return self._getFilesListHTTP(day)
597
+ elif self.urltype == "ftp":
598
+ return self._getFilesListFTP()
599
+
600
+ def _getFilesListHTTP(self, day):
601
+ """Returns a list of files to download from http server, which will
602
+ be HDF and XML files, and optionally JPG files if specified by
603
+ self.jpeg
604
+
605
+ :param str day: the date of data in format YYYY.MM.DD
606
+ """
607
+ # return the files list inside the directory of each day
608
+ try:
609
+ url = urljoin(self.url, self.path, day)
610
+ if self.debug:
611
+ logging.debug("The url is: {url}".format(url=url))
612
+ try:
613
+ http = modisHtmlParser(requests.get(url, timeout=self.timeout).content)
614
+ except:
615
+ http = modisHtmlParser(urlopen(url, timeout=self.timeout).read())
616
+ # download JPG files also
617
+ if self.jpeg:
618
+ # if tiles not specified, download all files
619
+ if not self.tiles:
620
+ finalList = http.get_all()
621
+ # if tiles specified, download all files with jpegs
622
+ else:
623
+ finalList = http.get_tiles(self.product_code, self.tiles, jpeg=True)
624
+ # if JPG files should not be downloaded, get only HDF and XML
625
+ else:
626
+ finalList = http.get_tiles(self.product_code, self.tiles)
627
+ if self.debug:
628
+ logging.debug(
629
+ "The number of file to download is: "
630
+ "{num}".format(num=len(finalList))
631
+ )
632
+
633
+ return finalList
634
+ except socket.error as e:
635
+ logging.error(
636
+ "Error {err} when try to receive list of " "files".format(err=e)
637
+ )
638
+ self._getFilesListHTTP(day)
639
+
640
+ def _getFilesListFTP(self):
641
+ """Create a list of files to download from FTP server, it is possible
642
+ choose to download also the JPG overview files or only the HDF files
643
+ """
644
+
645
+ def cicle_file(jpeg=False):
646
+ """Check the type of file"""
647
+ finalList = []
648
+ for i in self.listfiles:
649
+ name = i.split(".")
650
+ # distinguish jpeg files from hdf files by the number of index
651
+ # where find the tile index
652
+ if not self.tiles and not (name.count("jpg") or name.count("BROWSE")):
653
+ finalList.append(i)
654
+ # is a jpeg of tiles number
655
+ if self.tiles:
656
+ if self.tiles.count(name[3]) == 1 and jpeg:
657
+ finalList.append(i)
658
+ # is a hdf of tiles number
659
+ elif self.tiles.count(name[2]) == 1:
660
+ finalList.append(i)
661
+ return finalList
662
+
663
+ # return the file's list inside the directory of each day
664
+ try:
665
+ self.listfiles = self.ftp.nlst()
666
+ # download also jpeg
667
+ if self.jpeg:
668
+ # finallist is ugual to all file with jpeg file
669
+ if not self.tiles:
670
+ finalList = self.listfiles
671
+ # finallist is ugual to tiles file with jpeg file
672
+ else:
673
+ finalList = cicle_file(jpeg=True)
674
+ # not download jpeg
675
+ else:
676
+ finalList = cicle_file()
677
+ if self.debug:
678
+ logging.debug(
679
+ "The number of file to download is: "
680
+ "{num}".format(num=len(finalList))
681
+ )
682
+ return finalList
683
+ except (ftplib.error_reply, socket.error) as e:
684
+ logging.error(
685
+ "Error {err} when trying to receive list of " "files".format(err=e)
686
+ )
687
+ self._getFilesListFTP()
688
+
689
+ def checkDataExist(self, listNewFile, move=False):
690
+ """Check if a file already exists in the local download directory
691
+
692
+ :param list listNewFile: list of all files, returned by getFilesList
693
+ function
694
+ :param bool move: it is useful to know if a function is called from
695
+ download or move function
696
+ :return: list of files to download
697
+ """
698
+ # different return if this method is used from downloadsAllDay() or
699
+ # moveFile()
700
+ if not listNewFile and not self.fileInPath:
701
+ logging.error("checkDataExist both lists are empty")
702
+ elif not listNewFile:
703
+ listNewFile = list()
704
+ elif not self.fileInPath:
705
+ self.fileInPath = list()
706
+ if not move:
707
+ listOfDifferent = list(set(listNewFile) - set(self.fileInPath))
708
+ elif move:
709
+ listOfDifferent = list(set(self.fileInPath) - set(listNewFile))
710
+ return listOfDifferent
711
+
712
+ def checkFile(self, filHdf):
713
+ """Check by using GDAL to be sure that the download went ok
714
+
715
+ :param str filHdf: name of the HDF file to check
716
+
717
+ :return: 0 if file is correct, 1 for error
718
+ """
719
+ return 0
720
+ # try:
721
+ # gdal.Open(filHdf)
722
+ # return 0
723
+ # except (RuntimeError) as e:
724
+ # logging.error(e)
725
+ # return 1
726
+
727
+ def downloadFile(self, filDown, filHdf, day):
728
+ """Download a single file
729
+
730
+ :param str filDown: name of the file to download
731
+ :param str filHdf: name of the file to write to
732
+ :param str day: the day in format YYYY.MM.DD
733
+ """
734
+ if self.urltype == "http":
735
+ self._downloadFileHTTP(filDown, filHdf, day)
736
+ elif self.urltype == "ftp":
737
+ self._downloadFileFTP(filDown, filHdf)
738
+
739
+ def _downloadFileHTTP(self, filDown, filHdf, day):
740
+ """Download a single file from the http server
741
+
742
+ :param str filDown: name of the file to download
743
+ :param str filHdf: name of the file to write to
744
+ :param str day: the day in format YYYY.MM.DD
745
+ """
746
+ filSave = open(filHdf, "wb")
747
+ url = urljoin(self.url, self.path, day, filDown)
748
+ orig_size = None
749
+ try: # download and write the file
750
+ req = urllib.request.Request(url, headers=self.http_header)
751
+ http = urllib.request.urlopen(req)
752
+ orig_size = http.headers["Content-Length"]
753
+ filSave.write(http.read())
754
+ # if local file has an error, try to download the file again
755
+ except Exception as e:
756
+ logging.warning(
757
+ "Tried to downlaod with urllib but got this "
758
+ "error {co}, reason {re}".format(co=e.code, re=e.reason)
759
+ )
760
+ try:
761
+ http = requests.get(url, timeout=self.timeout)
762
+ orig_size = http.headers["Content-Length"]
763
+ filSave.write(http.content)
764
+ except Exception as e:
765
+ logging.warning(
766
+ "Tried to downlaod with requests but got this "
767
+ "error {co}, reason {re}".format(co=e.code, re=e.reason)
768
+ )
769
+ logging.error(
770
+ "Cannot download {name}. " "Retrying...".format(name=filDown)
771
+ )
772
+ filSave.close()
773
+ os.remove(filSave.name)
774
+ import time
775
+
776
+ time.sleep(5)
777
+ self._downloadFileHTTP(filDown, filHdf, day)
778
+ filSave.close()
779
+ transf_size = os.path.getsize(filSave.name)
780
+ if not orig_size:
781
+ self.filelist.write("{name}\n".format(name=filDown))
782
+ self.filelist.flush()
783
+ if self.debug:
784
+ logging.debug(
785
+ "File {name} downloaded but not "
786
+ "check the size".format(name=filDown)
787
+ )
788
+ return 0
789
+ if int(orig_size) == int(transf_size):
790
+ # if no xml file, delete the HDF and redownload
791
+ if filHdf.find(".xml") == -1:
792
+ test = False
793
+ # if GDAL:
794
+ # test = self.checkFile(filHdf)
795
+ test = True
796
+ if test:
797
+ os.remove(filSave.name)
798
+ self._downloadFileHTTP(filDown, filHdf, day)
799
+ else:
800
+ self.filelist.write("{name}\n".format(name=filDown))
801
+ self.filelist.flush()
802
+ if self.debug:
803
+ logging.debug(
804
+ "File {name} downloaded " "correctly".format(name=filDown)
805
+ )
806
+ return 0
807
+ else: # xml exists
808
+ self.filelist.write("{name}\n".format(name=filDown))
809
+ self.filelist.flush()
810
+ if self.debug:
811
+ logging.debug(
812
+ "File {name} downloaded " "correctly".format(name=filDown)
813
+ )
814
+ return 0
815
+ # if filesizes are different, delete and try again
816
+ else:
817
+ logging.warning(
818
+ "Different size for file {name} - original data: "
819
+ "{orig}, downloaded: {down}".format(
820
+ name=filDown, orig=orig_size, down=transf_size
821
+ )
822
+ )
823
+ os.remove(filSave.name)
824
+ self._downloadFileHTTP(filDown, filHdf, day)
825
+
826
+ def _downloadFileFTP(self, filDown, filHdf):
827
+ """Download a single file from ftp server
828
+
829
+ :param str filDown: name of the file to download
830
+ :param str filHdf: name of the file to write to
831
+ """
832
+ filSave = open(filHdf, "wb")
833
+ try: # transfer file from ftp
834
+ self.ftp.retrbinary("RETR " + filDown, filSave.write)
835
+ self.filelist.write("{name}\n".format(name=filDown))
836
+ self.filelist.flush()
837
+ if self.debug:
838
+ logging.debug("File {name} downloaded".format(name=filDown))
839
+ # if error during download process, try to redownload the file
840
+ except (ftplib.error_reply, socket.error, ftplib.error_temp, EOFError) as e:
841
+ logging.error(
842
+ "Cannot download {name}, the error was '{err}'. "
843
+ "Retrying...".format(name=filDown, err=e)
844
+ )
845
+ filSave.close()
846
+ os.remove(filSave.name)
847
+ try:
848
+ self.ftp.pwd()
849
+ except (ftplib.error_temp, EOFError) as e:
850
+ self._connectFTP()
851
+ self._downloadFileFTP(filDown, filHdf)
852
+ filSave.close()
853
+ orig_size = self.ftp.size(filDown)
854
+ transf_size = os.path.getsize(filSave.name)
855
+ if orig_size == transf_size:
856
+ return 0
857
+ else:
858
+ logging.warning(
859
+ "Different size for file {name} - original data: "
860
+ "{orig}, downloaded: {down}".format(
861
+ name=filDown, orig=orig_size, down=transf_size
862
+ )
863
+ )
864
+ os.remove(filSave.name)
865
+ self._downloadFileFTP(filDown, filHdf)
866
+
867
+ def dayDownload(self, day, listFilesDown):
868
+ """Downloads tiles for the selected day
869
+
870
+ :param str day: the day in format YYYY.MM.DD
871
+ :param list listFilesDown: list of the files to download, returned
872
+ by checkDataExist function
873
+ """
874
+ # for each file in files' list
875
+ for i in listFilesDown:
876
+ fileSplit = i.split(".")
877
+ filePrefix = "{a}.{b}.{c}.{d}".format(
878
+ a=fileSplit[0], b=fileSplit[1], c=fileSplit[2], d=fileSplit[3]
879
+ )
880
+
881
+ # check if this file already exists in the save directory
882
+ oldFile = glob.glob1(self.writeFilePath, filePrefix + "*" + fileSplit[-1])
883
+ numFiles = len(oldFile)
884
+ # if it doesn't exist
885
+ if numFiles == 0:
886
+ file_hdf = os.path.join(self.writeFilePath, i)
887
+ # if one does exist
888
+ elif numFiles == 1:
889
+ # check the version of file, delete local file if it is older
890
+ fileDown = getNewerVersion(oldFile[0], i)
891
+ if fileDown != oldFile[0]:
892
+ os.remove(os.path.join(self.writeFilePath, oldFile[0]))
893
+ file_hdf = os.path.join(self.writeFilePath, fileDown)
894
+ elif numFiles > 1:
895
+ logging.error("There are to many files for " "{name}".format(name=i))
896
+ if numFiles == 0 or (numFiles == 1 and fileDown != oldFile[0]):
897
+ self.downloadFile(i, file_hdf, day)
898
+
899
+ def downloadsAllDay(self, clean=False, allDays=False):
900
+ """Download all requested days
901
+
902
+ :param bool clean: if True remove the empty files, they could have
903
+ some problems in the previous download
904
+ :param bool allDays: download all passable days
905
+ """
906
+ if clean:
907
+ self.removeEmptyFiles()
908
+ # get the days to download
909
+ if allDays:
910
+ days = self.getAllDays()
911
+ else:
912
+ days = self.getListDays()
913
+ # log the days to download
914
+ if self.debug:
915
+ logging.debug(
916
+ "The number of days to download is: " "{num}".format(num=len(days))
917
+ )
918
+ # download the data
919
+ if self.urltype == "http":
920
+ self._downloadAllDaysHTTP(days)
921
+ elif self.urltype == "ftp":
922
+ self._downloadAllDaysFTP(days)
923
+
924
+ def _downloadAllDaysHTTP(self, days):
925
+ """Downloads all the tiles considered from HTTP server
926
+
927
+ :param list days: the list of days to download
928
+ """
929
+ # for each day
930
+ for day in days:
931
+ # obtain list of all files
932
+ listAllFiles = self.getFilesList(day)
933
+ # filter files based on local files in save directory
934
+ listFilesDown = self.checkDataExist(listAllFiles)
935
+ # download files for a day
936
+ self.dayDownload(day, listFilesDown)
937
+ self.closeFilelist()
938
+ if self.debug:
939
+ logging.debug("Download terminated")
940
+ return 0
941
+
942
+ def _downloadAllDaysFTP(self, days):
943
+ """Downloads all the tiles considered from FTP server
944
+
945
+ :param list days: the list of days to download
946
+ """
947
+ # for each day
948
+ for day in days:
949
+ # enter in the directory of day
950
+ self.setDirectoryIn(day)
951
+ # obtain list of all files
952
+ listAllFiles = self.getFilesList()
953
+ # filter files based on local files in save directory
954
+ listFilesDown = self.checkDataExist(listAllFiles)
955
+ # download files for a day
956
+ self.dayDownload(day, listFilesDown)
957
+ self.setDirectoryOver()
958
+ self.closeFTP()
959
+ if self.debug:
960
+ logging.debug("Download terminated")
961
+ return 0
962
+
963
+ def debugLog(self):
964
+ """Function to create the debug file
965
+
966
+ :return: a Logger object to use to write debug info
967
+ """
968
+ # create logger
969
+ logger = logging.getLogger("PythonLibModis debug")
970
+ logger.setLevel(logging.DEBUG)
971
+ # create console handler and set level to debug
972
+ ch = logging.StreamHandler()
973
+ ch.setLevel(logging.DEBUG)
974
+ # create formatter
975
+ formatter = logging.Formatter(
976
+ "%(asctime)s - %(name)s - " "%(levelname)s - %(message)s"
977
+ )
978
+ # add formatter to console handler
979
+ ch.setFormatter(formatter)
980
+ # add console handler to logger
981
+ logger.addHandler(ch)
982
+ return logger
983
+
984
+ def debugDays(self):
985
+ """This function is useful to debug the number of days"""
986
+ logger = self.debugLog()
987
+ days = self.getListDays()
988
+ # if length of list of days and the delta of days are different
989
+ if len(days) != self.delta:
990
+ # for each day
991
+ for i in range(1, self.delta + 1):
992
+ # calculate the current day using datetime.timedelta
993
+ delta = timedelta(days=i)
994
+ day = self.today - delta
995
+ day = day.strftime("%Y.%m.%d")
996
+ # check if day is in the days list
997
+ if day not in days:
998
+ logger.critical(
999
+ "This day {day} is not present on " "list".format(day=day)
1000
+ )
1001
+ # the length of list of days and delta are equal
1002
+ else:
1003
+ logger.info("debugDays() : getListDays() and self.delta are same " "length")
1004
+
1005
+ def debugMaps(self):
1006
+ """Prints the files to download to the debug stream"""
1007
+ logger = self.debugLog()
1008
+ days = self.getListDays()
1009
+ for day in days:
1010
+ listAllFiles = self.getFilesList(day)
1011
+ string = day + ": " + str(len(listAllFiles)) + "\n"
1012
+ logger.debug(string)