scdata 1.0.3__tar.gz → 1.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. {scdata-1.0.3/scdata.egg-info → scdata-1.2.3}/PKG-INFO +1 -2
  2. {scdata-1.0.3 → scdata-1.2.3}/requirements.txt +0 -2
  3. {scdata-1.0.3 → scdata-1.2.3}/scdata/__init__.py +1 -1
  4. {scdata-1.0.3 → scdata-1.2.3}/scdata/_config/config.py +95 -12
  5. {scdata-1.0.3 → scdata-1.2.3}/scdata/device/device.py +147 -131
  6. {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/__init__.py +1 -0
  7. {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/alphasense.py +48 -55
  8. scdata-1.2.3/scdata/device/process/error_codes.py +25 -0
  9. {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/params.py +24 -0
  10. {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/regression.py +3 -3
  11. {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/timeseries.py +21 -13
  12. {scdata-1.0.3 → scdata-1.2.3}/scdata/models/models.py +4 -0
  13. scdata-1.2.3/scdata/test/checks/__init__.py +1 -0
  14. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/checks/checks.py +0 -44
  15. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/test.py +4 -95
  16. scdata-1.2.3/scdata/tools/zenodo_templates/README.md +0 -0
  17. {scdata-1.0.3 → scdata-1.2.3/scdata.egg-info}/PKG-INFO +1 -2
  18. {scdata-1.0.3 → scdata-1.2.3}/scdata.egg-info/SOURCES.txt +2 -1
  19. {scdata-1.0.3 → scdata-1.2.3}/scdata.egg-info/requires.txt +0 -1
  20. {scdata-1.0.3 → scdata-1.2.3}/setup.py +1 -1
  21. scdata-1.0.3/scdata/_config/custom_logger.py +0 -40
  22. scdata-1.0.3/scdata/test/checks/__init__.py +0 -1
  23. {scdata-1.0.3 → scdata-1.2.3}/LICENSE +0 -0
  24. {scdata-1.0.3 → scdata-1.2.3}/MANIFEST.in +0 -0
  25. {scdata-1.0.3 → scdata-1.2.3}/README.md +0 -0
  26. {scdata-1.0.3 → scdata-1.2.3}/scdata/_config/__init__.py +0 -0
  27. {scdata-1.0.3 → scdata-1.2.3}/scdata/device/__init__.py +0 -0
  28. {scdata-1.0.3/scdata/tools → scdata-1.2.3/scdata/device/plot}/__init__.py +0 -0
  29. {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/baseline.py +0 -0
  30. {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/formulae.py +0 -0
  31. {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/geoseries.py +0 -0
  32. {scdata-1.0.3 → scdata-1.2.3}/scdata/io/__init__.py +0 -0
  33. {scdata-1.0.3 → scdata-1.2.3}/scdata/io/device_api.py +0 -0
  34. {scdata-1.0.3 → scdata-1.2.3}/scdata/io/device_file.py +0 -0
  35. {scdata-1.0.3 → scdata-1.2.3}/scdata/io/model.py +0 -0
  36. {scdata-1.0.3 → scdata-1.2.3}/scdata/models/__init__.py +0 -0
  37. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/__init__.py +0 -0
  38. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/dispersion/__init__.py +0 -0
  39. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/dispersion/dispersion.py +0 -0
  40. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/export/__init__.py +0 -0
  41. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/export/templates/sc_template.html +0 -0
  42. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/export/to_file.py +0 -0
  43. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/__init__.py +0 -0
  44. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/box_plot.py +0 -0
  45. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/heatmap_iplot.py +0 -0
  46. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/heatmap_plot.py +0 -0
  47. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/maps.py +0 -0
  48. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/plot_tools.py +0 -0
  49. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/scatter_dispersion_grid.py +0 -0
  50. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/scatter_iplot.py +0 -0
  51. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/scatter_plot.py +0 -0
  52. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/target_diagram.py +0 -0
  53. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_dendrogram.py +0 -0
  54. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_dispersion_grid.py +0 -0
  55. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_dispersion_plot.py +0 -0
  56. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_dispersion_uplot.py +0 -0
  57. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_iplot.py +0 -0
  58. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_plot.py +0 -0
  59. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_scatter.py +0 -0
  60. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_uplot.py +0 -0
  61. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/tools/__init__.py +0 -0
  62. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/tools/combine.py +0 -0
  63. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/tools/history.py +0 -0
  64. {scdata-1.0.3 → scdata-1.2.3}/scdata/test/tools/prepare.py +0 -0
  65. /scdata-1.0.3/scdata/tools/zenodo_templates/README.md → /scdata-1.2.3/scdata/tools/__init__.py +0 -0
  66. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/cleaning.py +0 -0
  67. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/custom_logger.py +0 -0
  68. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/date.py +0 -0
  69. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/dictmerge.py +0 -0
  70. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/find.py +0 -0
  71. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/gets.py +0 -0
  72. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/interim/example.csv +0 -0
  73. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/interim/geodata.csv +0 -0
  74. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/lazy.py +0 -0
  75. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/location.py +0 -0
  76. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/report.py +0 -0
  77. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/stats.py +0 -0
  78. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/units.py +0 -0
  79. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/uploads/example_upload_1.json +0 -0
  80. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/uploads/example_zenodo_upload.yaml +0 -0
  81. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/uploads/report.pdf +0 -0
  82. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/url_check.py +0 -0
  83. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/zenodo.py +0 -0
  84. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/zenodo_templates/template_zenodo_dataset.json +0 -0
  85. {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/zenodo_templates/template_zenodo_publication.json +0 -0
  86. {scdata-1.0.3 → scdata-1.2.3}/scdata.egg-info/dependency_links.txt +0 -0
  87. {scdata-1.0.3 → scdata-1.2.3}/scdata.egg-info/not-zip-safe +0 -0
  88. {scdata-1.0.3 → scdata-1.2.3}/scdata.egg-info/top_level.txt +0 -0
  89. {scdata-1.0.3 → scdata-1.2.3}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: scdata
3
- Version: 1.0.3
3
+ Version: 1.2.3
4
4
  Summary: Analysis of sensors and time series data
5
5
  Home-page: https://github.com/fablabbcn/smartcitizen-data
6
6
  Author: oscgonfer
@@ -24,7 +24,6 @@ Requires-Dist: folium~=0.12.1
24
24
  Requires-Dist: geopy~=1.21.0
25
25
  Requires-Dist: Jinja2~=3.1.2
26
26
  Requires-Dist: matplotlib
27
- Requires-Dist: missingno~=0.5.2
28
27
  Requires-Dist: numpy~=1.25.2
29
28
  Requires-Dist: pandas~=2.2.2
30
29
  Requires-Dist: pydantic
@@ -9,8 +9,6 @@ geopy~=1.21.0
9
9
  # TODO To be updated?
10
10
  Jinja2~=3.1.2
11
11
  matplotlib
12
- # TODO To be updated?
13
- missingno~=0.5.2
14
12
  numpy~=1.25.2
15
13
  pandas~=2.2.2
16
14
  pydantic
@@ -3,4 +3,4 @@ from .device import Device
3
3
  from .test import Test
4
4
  from .models import Source, TestOptions, DeviceOptions, APIParams, FileParams, CSVParams
5
5
 
6
- __version__ = '1.0.3'
6
+ __version__ = '1.2.3'
@@ -83,7 +83,7 @@ class Config(object):
83
83
  ### -------------SMART CITIZEN-------------
84
84
  ### ---------------------------------------
85
85
  # # Urls
86
- _base_postprocessing_url = 'https://raw.githubusercontent.com/fablabbcn/smartcitizen-data/enhacement/flexible-handlers/'
86
+ _base_postprocessing_url = 'https://raw.githubusercontent.com/fablabbcn/smartcitizen-data/master/'
87
87
  _default_file_type = 'json'
88
88
 
89
89
  calibrations_urls = [
@@ -118,7 +118,7 @@ class Config(object):
118
118
  names_urls = [
119
119
  # Revert to base postprocessing url
120
120
  # f'{_base_postprocessing_url}names/SCDevice.json'
121
- 'https://raw.githubusercontent.com/fablabbcn/smartcitizen-data/enhacement/flexible-handlers/names/SCDevice.json'
121
+ 'https://raw.githubusercontent.com/fablabbcn/smartcitizen-data/master/names/SCDevice.json'
122
122
  ]
123
123
 
124
124
 
@@ -359,14 +359,6 @@ class Config(object):
359
359
  }
360
360
  }
361
361
 
362
- _missingno_def_fmt = {
363
- 'height': 6,
364
- 'width': 6,
365
- 'fontsize': 8.,
366
- 'title_fontsize': 14
367
- }
368
-
369
-
370
362
  ### ---------------------------------------
371
363
  ### ----------------MODELS-----------------
372
364
  ### ---------------------------------------
@@ -419,9 +411,97 @@ class Config(object):
419
411
  }
420
412
 
421
413
  ### ---------------------------------------
422
- ### ---------------------------------------
414
+ ### ------------VALUES-CHECK---------------
423
415
  ### ---------------------------------------
424
416
 
417
+ _default_sampling_rate = {
418
+ 'AMS AS7731 - UVA': 1,
419
+ 'AMS AS7731 - UVB': 1,
420
+ 'AMS AS7731 - UVC': 1,
421
+ 'LIGHT': 1,
422
+ 'BATT': 1,
423
+ 'NOISE_A': 1,
424
+ 'SCD30_CO2': 1,
425
+ 'SCD30_HUM': 1,
426
+ 'SCD30_TEMP': 1,
427
+ 'SD-card': 1,
428
+ 'ST LPS33 - Barometric Pressure': 1,
429
+ 'PRESS': 1,
430
+ 'PMS5003_PM_1': 5,
431
+ 'PMS5003_PM_25': 5,
432
+ 'PMS5003_PM_10': 5,
433
+ 'PMS5003_PN_03': 5,
434
+ 'PMS5003_PN_03': 5,
435
+ 'PMS5003_PN_05':5,
436
+ 'PMS5003_PN_1':5,
437
+ 'PMS5003_PN_10':5,
438
+ 'PMS5003_PN_25':5,
439
+ 'PMS5003_PN_5':5,
440
+ 'SEN5X_HUM': 5,
441
+ 'SEN5X_PM_1': 5,
442
+ 'SEN5X_PM_10': 5,
443
+ 'SEN5X_PM_25': 5,
444
+ 'SEN5X_PM_40': 5,
445
+ 'SEN5X_PN_05': 5,
446
+ 'SEN5X_PN_1': 5,
447
+ 'SEN5X_PN_10': 5,
448
+ 'SEN5X_PN_25': 5,
449
+ 'SEN5X_PN_40': 5,
450
+ 'SEN5X_TPS': 5,
451
+ 'SEN5X_TEMP': 5,
452
+ 'SFA30_HCHO': 1,
453
+ 'SFA30_HUM': 1,
454
+ 'SFA30_TEMP': 1,
455
+ 'ADC_48_0': 1,
456
+ 'ADC_48_1': 1,
457
+ 'ADC_48_2': 1,
458
+ 'ADC_48_3': 1,
459
+ 'ADC_49_0': 1,
460
+ 'ADC_49_1': 1,
461
+ 'ADC_49_2': 1,
462
+ 'ADC_49_3': 1,
463
+ 'CCS811_VOCS': 1,
464
+ 'CCS811_ECO2': 1,
465
+ 'HUM': 1,
466
+ 'TEMP': 1,
467
+ 'RSSI': 1,
468
+ 'NO2': 1,
469
+ 'O3': 1
470
+ }
471
+
472
+ _default_unplausible_values = {
473
+ 'NOISE_A': [20, 99],
474
+ 'SCD30_CO2': [300, 2000],
475
+ 'SCD30_HUM': [20, 99],
476
+ 'SCD30_TEMP': [-20, 50],
477
+ 'ST LPS33 - Barometric Pressure': [50, 110],
478
+ 'PRESS': [50, 110],
479
+ 'PMS5003_PM_1': [0, 500],
480
+ 'PMS5003_PM_25': [0, 500],
481
+ 'PMS5003_PM_10': [0, 500],
482
+ 'SEN5X_HUM': [20, 99],
483
+ 'SEN5X_PM_1': [0, 500],
484
+ 'SEN5X_PM_10': [0, 500],
485
+ 'SEN5X_PM_25': [0, 500],
486
+ 'SEN5X_PM_40': [0, 500],
487
+ 'SEN5X_TEMP': [-20, 50],
488
+ 'SFA30_HCHO': [00, 1000],
489
+ 'SFA30_HUM': [20, 99],
490
+ 'SFA30_TEMP': [-20, 50],
491
+ 'ADC_48_0': [0, 3],
492
+ 'ADC_48_1': [0, 3],
493
+ 'ADC_48_2': [0, 3],
494
+ 'ADC_48_3': [0, 3],
495
+ 'ADC_49_0': [0, 3],
496
+ 'ADC_49_1': [0, 3],
497
+ 'ADC_49_2': [0, 3],
498
+ 'ADC_49_3': [0, 3],
499
+ 'HUM': [20, 99],
500
+ 'TEMP': [-20, 50],
501
+ 'NO2': [0, 1000],
502
+ 'O3': [0, 1000]
503
+ }
504
+
425
505
  def __init__(self):
426
506
  self._env_file = None
427
507
  self.paths = self.get_paths()
@@ -695,7 +775,7 @@ class Config(object):
695
775
  with open(namespath, 'w') as file:
696
776
  json.dump(names_dump, file)
697
777
 
698
- # Find environment file in root or in scdata/ for clones
778
+ # Find environment file in root
699
779
  if exists(join(self.paths['data'],'.env')):
700
780
  self._env_file = join(self.paths['data'],'.env')
701
781
  print(f'Found Environment file at: {self._env_file}')
@@ -704,6 +784,9 @@ class Config(object):
704
784
  print(f'No environment file found. If you had an environment file (.env) before, make sure its now here')
705
785
  print(join(self.paths['data'],'.env'))
706
786
 
787
+ if 'SC_BEARER' not in environ:
788
+ print('SC_BEARER not in environment variables. You may get throttled when requesting to api.smartcitizen.me')
789
+
707
790
  def load(self):
708
791
  """ Override config if config file exists. """
709
792
  _sccpath = join(self.paths['config'], 'config.yaml')
@@ -12,7 +12,7 @@ from scdata._config import config
12
12
  from scdata.io.device_api import *
13
13
  from scdata.models import Blueprint, Metric, Source, APIParams, CSVParams, DeviceOptions, Sensor
14
14
 
15
- from os.path import join, basename
15
+ from os.path import join, basename, exists
16
16
  from urllib.parse import urlparse
17
17
  from pandas import DataFrame, to_timedelta, Timedelta
18
18
  from numpy import nan
@@ -256,7 +256,8 @@ class Device(BaseModel):
256
256
  return True
257
257
  return False
258
258
 
259
- async def load(self, cache=None, convert_units=True, convert_names=True, max_amount=None):
259
+ async def load(self, cache=None, convert_units=True,
260
+ convert_names=True, ignore_error = True):
260
261
  '''
261
262
  Loads the device with some options
262
263
 
@@ -271,9 +272,9 @@ class Device(BaseModel):
271
272
  convert_names: bool
272
273
  Default: True
273
274
  Convert names for channels based on ids
274
- max_amount: int
275
- Default: None
276
- Trim dataframe to this amount for processing and forwarding purposes (workaround)
275
+ ignore_error: bool
276
+ Default: True
277
+ Ignore if the cache does not exist
277
278
  Returns
278
279
  ----------
279
280
  True if loaded correctly
@@ -284,21 +285,29 @@ class Device(BaseModel):
284
285
  frequency = self.options.frequency
285
286
  clean_na = self.options.clean_na
286
287
  resample = self.options.resample
288
+ limit = self.options.limit
289
+ channels = self.options.channels
287
290
  cached_data = DataFrame()
288
291
 
289
292
  # Only case where cache makes sense
290
293
  if self.source.type == 'api':
291
294
  if cache is not None and cache:
292
- if cache.endswith('.csv'):
293
- cached_data = read_csv_file(
294
- path = cache,
295
- timezone = timezone,
296
- frequency = frequency,
297
- clean_na = clean_na,
298
- resample = resample,
299
- index_name = 'TIME')
295
+ if not exists(cache):
296
+ if not ignore_error:
297
+ raise FileExistsError(f'Cache does not exist: {cache}')
298
+ else:
299
+ logger.warning(f'Cache file does not exist: {cache}')
300
300
  else:
301
- raise NotImplementedError(f'Cache needs to be a .csv file. Got {cache}.')
301
+ if cache.endswith('.csv'):
302
+ cached_data = read_csv_file(
303
+ path = cache,
304
+ timezone = timezone,
305
+ frequency = frequency,
306
+ clean_na = clean_na,
307
+ resample = resample,
308
+ index_name = 'TIME')
309
+ else:
310
+ raise NotImplementedError(f'Cache needs to be a .csv file. Got {cache}.')
302
311
 
303
312
  # Make request with a logical min_date
304
313
  if not cached_data.empty:
@@ -316,6 +325,8 @@ class Device(BaseModel):
316
325
  max_date = max_date,
317
326
  frequency = frequency,
318
327
  clean_na = clean_na,
328
+ limit = limit,
329
+ channels = channels,
319
330
  resample = resample)
320
331
  else:
321
332
  self.handler.get_data(
@@ -327,27 +338,24 @@ class Device(BaseModel):
327
338
 
328
339
  # In principle this links both dataframes as they are unmutable
329
340
  self.data = self.handler.data
330
- # Wrap it all up
331
- # TODO Avoid doing this if not needed?
332
- self.loaded = self.__load_wrapup__(max_amount, convert_units=convert_units, convert_names=convert_names, cached_data=cached_data)
333
341
 
342
+ # Wrap it all up
343
+ self.loaded = self.__load_wrapup__(cached_data=cached_data)
334
344
  self.processed = False
345
+
335
346
  return self.loaded
336
347
 
337
- def __load_wrapup__(self, max_amount, convert_units=True, convert_names=True, cached_data=None):
348
+ def __load_wrapup__(self, cached_data=None):
349
+
338
350
  if self.data is not None:
339
351
  if not self.data.empty:
340
- if max_amount is not None:
341
- # TODO Dirty workaround
342
- logger.info(f'Trimming dataframe to {max_amount} rows')
343
- self.data=self.data.dropna(axis = 0, how='all').head(max_amount)
344
352
  # Convert names
345
- if convert_names:
346
- self.__convert_names__()
353
+ self.__convert_names__()
347
354
  # Convert units
348
- if convert_units:
349
- self.__convert_units__()
355
+ self.__convert_units__()
356
+
350
357
  self.postprocessing_updated = False
358
+
351
359
  else:
352
360
  logger.info('Empty dataframe in loaded data. Waiting for cache...')
353
361
 
@@ -358,6 +366,7 @@ class Device(BaseModel):
358
366
  return not self.data.empty
359
367
 
360
368
  def __convert_names__(self):
369
+ if not self.options.convert_names: return
361
370
  logger.info('Converting names...')
362
371
 
363
372
  self.data.rename(columns=self._rename, inplace=True)
@@ -370,6 +379,8 @@ class Device(BaseModel):
370
379
  The files are with original units, and then converted in the device only
371
380
  for the data but never chached like so.
372
381
  '''
382
+ if not self.options.convert_units: return
383
+
373
384
  logger.info('Checking if units need to be converted...')
374
385
  for sensor in self.data.columns:
375
386
  _rename_inv = {v: k for k, v in self._rename.items()}
@@ -468,20 +479,31 @@ class Device(BaseModel):
468
479
  if metric.kwargs is not None: kwargs = metric.kwargs
469
480
 
470
481
  try:
471
- result = funct(self.data, *args, **kwargs)
482
+ process_result = funct(self.data, *args, **kwargs)
472
483
  except KeyError:
473
484
  logger.error('Cannot process requested function with data provided')
474
485
  process_ok = False
475
486
  pass
476
487
  else:
477
- if result is not None:
478
- self.data[metric.name] = result
479
- process_ok &= True
480
488
  # If the metric is None, might be for many reasons and shouldn't collapse the process_ok
489
+ if process_result is not None:
490
+ if 'ERROR' in process_result.status_code.name:
491
+ # We got an error during the processing
492
+ logger.error(process_result.status_code.name)
493
+ process_ok &= False
494
+ elif 'WARNING' in process_result.status_code.name:
495
+ # In this case there is no data to put into the metric
496
+ # but there is no reason to make deny process_ok
497
+ logger.warning(process_result.status_code.name)
498
+ process_ok &= True
499
+ elif 'SUCCESS' in process_result.status_code.name:
500
+ self.data[metric.name] = process_result.data
501
+ logger.info(process_result.status_code.name)
502
+ process_ok &= True
481
503
 
482
504
  if process_ok:
483
505
  logger.info(f"Device {self.paramsParsed.id} processed")
484
- self.processed = process_ok & self.update_postprocessing_date()
506
+ self.processed = process_ok
485
507
 
486
508
  return self.processed
487
509
 
@@ -490,114 +512,108 @@ class Device(BaseModel):
490
512
  return self._sensors
491
513
 
492
514
  def update_postprocessing_date(self):
493
- latest_postprocessing = localise_date(self.data.index[-1]+\
494
- to_timedelta(self.options.frequency), 'UTC')
515
+ # This function updates the postprocessing date with the latest logical date
516
+ latest_postprocessing = None
517
+ if self.loaded:
518
+ # If device was loaded (data not empty)
519
+ if self.processed:
520
+ # If device was processed, new postprocessing is the last reading rounded up with frequency
521
+ latest_postprocessing = localise_date(self.data.index[-1] + to_timedelta(self.options.frequency), 'UTC')
522
+ logger.info(f'Updating latest_postprocessing to {latest_postprocessing}')
523
+ else:
524
+ logger.info(f'Cannot update latest_postprocessing. Device was loaded but not processed')
525
+ else:
526
+ # If device was not loaded, increase the postprocessing limited to last_reading_at
527
+ latest_postprocessing = min(self.handler.json.last_reading_at, self.options.max_date)
528
+ logger.info(f'Updating latest_postprocessing to {latest_postprocessing}')
529
+
530
+ if latest_postprocessing is None:
531
+ return False
532
+
495
533
  if self.handler.update_latest_postprocessing(latest_postprocessing):
496
534
  # Consider the case of no postprocessing, to avoid making the whole thing false
497
535
  if latest_postprocessing.to_pydatetime() == self.handler.latest_postprocessing or self.handler.json.postprocessing is None:
498
536
  self.postprocessing_updated = True
499
537
  else:
500
538
  self.postprocessing_updated = False
539
+
501
540
  return self.postprocessing_updated
502
541
 
503
- # TODO
504
- def health_check(self):
505
- return True
506
-
507
- # TODO - Decide if we keep it
508
- # def forward(self, chunk_size = 500, dry_run = False, max_retries = 2):
509
- # '''
510
- # Forwards data to another api.
511
- # Parameters
512
- # ----------
513
- # chunk_size: int
514
- # 500
515
- # Chunk size to be sent to device.post_data_to_device in question
516
- # dry_run: boolean
517
- # False
518
- # Post the payload to the API or just return it
519
- # max_retries: int
520
- # 2
521
- # Maximum number of retries per chunk
522
- # Returns
523
- # ----------
524
- # boolean
525
- # True if posted ok, False otherwise
526
- # '''
527
-
528
- # # Import requested handler
529
- # hmod = __import__('scdata.io.device_api', fromlist = ['io.device_api'])
530
- # Hclass = getattr(hmod, config.connectors[self.forwarding_request]['handler'])
531
-
532
- # # Create new device in target API if it hasn't been created yet
533
- # if self.forwarding_params is None:
534
- # std_out('Empty forwarding information, attemping creating a new device', 'WARNING')
535
- # # We assume the device has never been posted
536
- # # Construct new device kwargs dictionary
537
- # kwargs = dict()
538
- # for item in config.connectors[self.forwarding_request]['kwargs']:
539
- # val = config.connectors[self.forwarding_request]['kwargs'][item]
540
- # if val == 'options':
541
- # kitem = self.options[item]
542
- # elif val == 'config':
543
- # # Items in config should be underscored
544
- # kitem = config.__getattr__(f'_{item}')
545
- # elif isinstance(val, Iterable):
546
- # if 'same' in val:
547
- # if 'as_device' in val:
548
- # if item == 'sensors':
549
- # kitem = self.merge_sensor_metrics(ignore_empty = True)
550
- # elif item == 'description':
551
- # kitem = self.blueprint.replace('_', ' ')
552
- # elif 'as_api' in val:
553
- # if item == 'sensors':
554
- # kitem = self.api_device.get_device_sensors()
555
- # elif item == 'description':
556
- # kitem = self.api_device.get_device_description()
557
- # else:
558
- # kitem = val
559
- # kwargs[item] = kitem
560
-
561
- # response = Hclass.new_device(name = config.connectors[self.forwarding_request]['name_prepend']\
562
- # + str(self.params.id),
563
- # location = self.location,
564
- # dry_run = dry_run,
565
- # **kwargs)
566
- # if response:
567
- # if 'message' in response:
568
- # if response['message'] == 'Created':
569
- # if 'sensorid' in response:
570
- # self.forwarding_params = response['sensorid']
571
- # self.api_device.postprocessing['forwarding_params'] = self.forwarding_params
572
- # std_out(f'New sensor ID in {self.forwarding_request}\
573
- # is {self.forwarding_params}. Updating')
574
-
575
- # if self.forwarding_params is not None:
576
- # df = self.data.copy()
577
- # df = df[df.columns.intersection(list(self.merge_sensor_metrics(ignore_empty=True).keys()))]
578
- # df = clean(df, 'drop', how = 'all')
579
-
580
- # if df.empty:
581
- # std_out('Empty dataframe, ignoring', 'WARNING')
582
- # return False
583
-
584
- # # Create object
585
- # ndev = Hclass(did = self.forwarding_params)
586
- # post_ok = ndev.post_data_to_device(df, chunk_size = chunk_size,
587
- # dry_run = dry_run, max_retries = 2)
588
-
589
- # if post_ok:
590
- # # TODO Check if we like this
591
- # if self.source == 'api':
592
- # self.update_latest_postprocessing()
593
- # std_out(f'Posted data for {self.params.id}', 'SUCCESS')
594
- # else:
595
- # std_out(f'Error posting data for {self.params.id}', 'ERROR')
596
- # return post_ok
597
-
598
- # else:
599
- # std_out('Empty forwarding information', 'ERROR')
600
- # return False
542
+ # Check nans per column, return dict
543
+ # TODO-DOCUMENT
544
+ def get_nan_ratio(self, **kwargs):
545
+ if not self.loaded:
546
+ logger.error('Need to load first (device.load())')
547
+ return False
548
+
549
+ if 'sampling_rate' not in kwargs:
550
+ sampling_rate = config._default_sampling_rate
551
+ else:
552
+ sampling_rate = kwargs['sampling_rate']
553
+ result = {}
554
+
555
+ for column in self.data.columns:
556
+ if column not in sampling_rate: continue
557
+ df = self.data[column].resample(f'{sampling_rate[column]}Min').mean()
558
+ minutes = df.groupby(df.index.date).mean().index.to_series().diff()/Timedelta('60s')
559
+ result[column] = (1-(minutes-df.isna().groupby(df.index.date).sum())/minutes)*sampling_rate[column]
560
+
561
+ return result
562
+
563
+ # Check plausibility per column, return dict. Doesn't take into account nans
564
+ # TODO-DOCUMENT
565
+ def get_plausible_ratio(self, **kwargs):
566
+ if not self.loaded:
567
+ logger.error('Need to load first (device.load())')
568
+ return False
569
+
570
+ if 'unplausible_values' not in kwargs:
571
+ unplausible_values = config._default_unplausible_values
572
+ else:
573
+ unplausible_values = kwargs['unplausible_values']
574
+
575
+ if 'sampling_rate' not in kwargs:
576
+ sampling_rate = config._default_sampling_rate
577
+ else:
578
+ sampling_rate = kwargs['sampling_rate']
579
+
580
+ return {column: self.data[column].between(left=unplausible_values[column][0], right=unplausible_values[column][1]).groupby(self.data[column].index.date).sum()/self.data.groupby(self.data.index.date).count()[column] for column in self.data.columns if column in unplausible_values}
581
+
582
+ # Check plausibility per column, return dict. Doesn't take into account nans
583
+ def get_outlier_ratio(self, **kwargs):
584
+ if not self.loaded:
585
+ logger.error('Need to load first (device.load())')
586
+ return False
587
+ result = {}
588
+ resample = '360h'
589
+
590
+ for column in self.data.columns:
591
+ Q1 = self.data[column].resample(resample).mean().quantile(0.25)
592
+ Q3 = self.data[column].resample(resample).mean().quantile(0.75)
593
+ IQR = Q3 - Q1
594
+
595
+ mask = (self.data[column] < (Q1 - 1.5 * IQR)) | (self.data[column] > (Q3 + 1.5 * IQR))
596
+ result[column] = mask.groupby(mask.index.date).mean()
597
+
598
+ return result
599
+
600
+ # Check plausibility per column, return dict. Doesn't take into account nans
601
+ def get_outliers(self, **kwargs):
602
+ if not self.loaded:
603
+ logger.error('Need to load first (device.load())')
604
+ return False
605
+ result = {}
606
+ resample = '360h'
607
+
608
+ for column in self.data.columns:
609
+ Q1 = self.data[column].resample(resample).mean().quantile(0.25)
610
+ Q3 = self.data[column].resample(resample).mean().quantile(0.75)
611
+ IQR = Q3 - Q1
612
+
613
+ mask = (self.data[column] < (Q1 - 1.5 * IQR)) | (self.data[column] > (Q3 + 1.5 * IQR))
614
+ result[column] = mask
615
+
616
+ return result
601
617
 
602
618
  def export(self, path, forced_overwrite = False, file_format = 'csv'):
603
619
  '''
@@ -1,5 +1,6 @@
1
1
  ''' Implementation of different processes to be done in each device '''
2
2
 
3
+ # TODO UPDATE all these functions to make them comply with StatusCode types
3
4
  from scdata.tools.lazy import LazyCallable
4
5
  from .formulae import absolute_humidity, exp_f, fit_exp_f
5
6
  from .geoseries import is_within_circle