scdata 1.0.3__tar.gz → 1.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scdata-1.0.3/scdata.egg-info → scdata-1.2.3}/PKG-INFO +1 -2
- {scdata-1.0.3 → scdata-1.2.3}/requirements.txt +0 -2
- {scdata-1.0.3 → scdata-1.2.3}/scdata/__init__.py +1 -1
- {scdata-1.0.3 → scdata-1.2.3}/scdata/_config/config.py +95 -12
- {scdata-1.0.3 → scdata-1.2.3}/scdata/device/device.py +147 -131
- {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/__init__.py +1 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/alphasense.py +48 -55
- scdata-1.2.3/scdata/device/process/error_codes.py +25 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/params.py +24 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/regression.py +3 -3
- {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/timeseries.py +21 -13
- {scdata-1.0.3 → scdata-1.2.3}/scdata/models/models.py +4 -0
- scdata-1.2.3/scdata/test/checks/__init__.py +1 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/checks/checks.py +0 -44
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/test.py +4 -95
- scdata-1.2.3/scdata/tools/zenodo_templates/README.md +0 -0
- {scdata-1.0.3 → scdata-1.2.3/scdata.egg-info}/PKG-INFO +1 -2
- {scdata-1.0.3 → scdata-1.2.3}/scdata.egg-info/SOURCES.txt +2 -1
- {scdata-1.0.3 → scdata-1.2.3}/scdata.egg-info/requires.txt +0 -1
- {scdata-1.0.3 → scdata-1.2.3}/setup.py +1 -1
- scdata-1.0.3/scdata/_config/custom_logger.py +0 -40
- scdata-1.0.3/scdata/test/checks/__init__.py +0 -1
- {scdata-1.0.3 → scdata-1.2.3}/LICENSE +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/MANIFEST.in +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/README.md +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/_config/__init__.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/device/__init__.py +0 -0
- {scdata-1.0.3/scdata/tools → scdata-1.2.3/scdata/device/plot}/__init__.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/baseline.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/formulae.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/device/process/geoseries.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/io/__init__.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/io/device_api.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/io/device_file.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/io/model.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/models/__init__.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/__init__.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/dispersion/__init__.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/dispersion/dispersion.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/export/__init__.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/export/templates/sc_template.html +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/export/to_file.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/__init__.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/box_plot.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/heatmap_iplot.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/heatmap_plot.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/maps.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/plot_tools.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/scatter_dispersion_grid.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/scatter_iplot.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/scatter_plot.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/target_diagram.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_dendrogram.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_dispersion_grid.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_dispersion_plot.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_dispersion_uplot.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_iplot.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_plot.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_scatter.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/plot/ts_uplot.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/tools/__init__.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/tools/combine.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/tools/history.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/test/tools/prepare.py +0 -0
- /scdata-1.0.3/scdata/tools/zenodo_templates/README.md → /scdata-1.2.3/scdata/tools/__init__.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/cleaning.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/custom_logger.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/date.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/dictmerge.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/find.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/gets.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/interim/example.csv +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/interim/geodata.csv +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/lazy.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/location.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/report.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/stats.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/units.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/uploads/example_upload_1.json +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/uploads/example_zenodo_upload.yaml +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/uploads/report.pdf +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/url_check.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/zenodo.py +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/zenodo_templates/template_zenodo_dataset.json +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata/tools/zenodo_templates/template_zenodo_publication.json +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata.egg-info/dependency_links.txt +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata.egg-info/not-zip-safe +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/scdata.egg-info/top_level.txt +0 -0
- {scdata-1.0.3 → scdata-1.2.3}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: scdata
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.3
|
|
4
4
|
Summary: Analysis of sensors and time series data
|
|
5
5
|
Home-page: https://github.com/fablabbcn/smartcitizen-data
|
|
6
6
|
Author: oscgonfer
|
|
@@ -24,7 +24,6 @@ Requires-Dist: folium~=0.12.1
|
|
|
24
24
|
Requires-Dist: geopy~=1.21.0
|
|
25
25
|
Requires-Dist: Jinja2~=3.1.2
|
|
26
26
|
Requires-Dist: matplotlib
|
|
27
|
-
Requires-Dist: missingno~=0.5.2
|
|
28
27
|
Requires-Dist: numpy~=1.25.2
|
|
29
28
|
Requires-Dist: pandas~=2.2.2
|
|
30
29
|
Requires-Dist: pydantic
|
|
@@ -83,7 +83,7 @@ class Config(object):
|
|
|
83
83
|
### -------------SMART CITIZEN-------------
|
|
84
84
|
### ---------------------------------------
|
|
85
85
|
# # Urls
|
|
86
|
-
_base_postprocessing_url = 'https://raw.githubusercontent.com/fablabbcn/smartcitizen-data/
|
|
86
|
+
_base_postprocessing_url = 'https://raw.githubusercontent.com/fablabbcn/smartcitizen-data/master/'
|
|
87
87
|
_default_file_type = 'json'
|
|
88
88
|
|
|
89
89
|
calibrations_urls = [
|
|
@@ -118,7 +118,7 @@ class Config(object):
|
|
|
118
118
|
names_urls = [
|
|
119
119
|
# Revert to base postprocessing url
|
|
120
120
|
# f'{_base_postprocessing_url}names/SCDevice.json'
|
|
121
|
-
'https://raw.githubusercontent.com/fablabbcn/smartcitizen-data/
|
|
121
|
+
'https://raw.githubusercontent.com/fablabbcn/smartcitizen-data/master/names/SCDevice.json'
|
|
122
122
|
]
|
|
123
123
|
|
|
124
124
|
|
|
@@ -359,14 +359,6 @@ class Config(object):
|
|
|
359
359
|
}
|
|
360
360
|
}
|
|
361
361
|
|
|
362
|
-
_missingno_def_fmt = {
|
|
363
|
-
'height': 6,
|
|
364
|
-
'width': 6,
|
|
365
|
-
'fontsize': 8.,
|
|
366
|
-
'title_fontsize': 14
|
|
367
|
-
}
|
|
368
|
-
|
|
369
|
-
|
|
370
362
|
### ---------------------------------------
|
|
371
363
|
### ----------------MODELS-----------------
|
|
372
364
|
### ---------------------------------------
|
|
@@ -419,9 +411,97 @@ class Config(object):
|
|
|
419
411
|
}
|
|
420
412
|
|
|
421
413
|
### ---------------------------------------
|
|
422
|
-
###
|
|
414
|
+
### ------------VALUES-CHECK---------------
|
|
423
415
|
### ---------------------------------------
|
|
424
416
|
|
|
417
|
+
_default_sampling_rate = {
|
|
418
|
+
'AMS AS7731 - UVA': 1,
|
|
419
|
+
'AMS AS7731 - UVB': 1,
|
|
420
|
+
'AMS AS7731 - UVC': 1,
|
|
421
|
+
'LIGHT': 1,
|
|
422
|
+
'BATT': 1,
|
|
423
|
+
'NOISE_A': 1,
|
|
424
|
+
'SCD30_CO2': 1,
|
|
425
|
+
'SCD30_HUM': 1,
|
|
426
|
+
'SCD30_TEMP': 1,
|
|
427
|
+
'SD-card': 1,
|
|
428
|
+
'ST LPS33 - Barometric Pressure': 1,
|
|
429
|
+
'PRESS': 1,
|
|
430
|
+
'PMS5003_PM_1': 5,
|
|
431
|
+
'PMS5003_PM_25': 5,
|
|
432
|
+
'PMS5003_PM_10': 5,
|
|
433
|
+
'PMS5003_PN_03': 5,
|
|
434
|
+
'PMS5003_PN_03': 5,
|
|
435
|
+
'PMS5003_PN_05':5,
|
|
436
|
+
'PMS5003_PN_1':5,
|
|
437
|
+
'PMS5003_PN_10':5,
|
|
438
|
+
'PMS5003_PN_25':5,
|
|
439
|
+
'PMS5003_PN_5':5,
|
|
440
|
+
'SEN5X_HUM': 5,
|
|
441
|
+
'SEN5X_PM_1': 5,
|
|
442
|
+
'SEN5X_PM_10': 5,
|
|
443
|
+
'SEN5X_PM_25': 5,
|
|
444
|
+
'SEN5X_PM_40': 5,
|
|
445
|
+
'SEN5X_PN_05': 5,
|
|
446
|
+
'SEN5X_PN_1': 5,
|
|
447
|
+
'SEN5X_PN_10': 5,
|
|
448
|
+
'SEN5X_PN_25': 5,
|
|
449
|
+
'SEN5X_PN_40': 5,
|
|
450
|
+
'SEN5X_TPS': 5,
|
|
451
|
+
'SEN5X_TEMP': 5,
|
|
452
|
+
'SFA30_HCHO': 1,
|
|
453
|
+
'SFA30_HUM': 1,
|
|
454
|
+
'SFA30_TEMP': 1,
|
|
455
|
+
'ADC_48_0': 1,
|
|
456
|
+
'ADC_48_1': 1,
|
|
457
|
+
'ADC_48_2': 1,
|
|
458
|
+
'ADC_48_3': 1,
|
|
459
|
+
'ADC_49_0': 1,
|
|
460
|
+
'ADC_49_1': 1,
|
|
461
|
+
'ADC_49_2': 1,
|
|
462
|
+
'ADC_49_3': 1,
|
|
463
|
+
'CCS811_VOCS': 1,
|
|
464
|
+
'CCS811_ECO2': 1,
|
|
465
|
+
'HUM': 1,
|
|
466
|
+
'TEMP': 1,
|
|
467
|
+
'RSSI': 1,
|
|
468
|
+
'NO2': 1,
|
|
469
|
+
'O3': 1
|
|
470
|
+
}
|
|
471
|
+
|
|
472
|
+
_default_unplausible_values = {
|
|
473
|
+
'NOISE_A': [20, 99],
|
|
474
|
+
'SCD30_CO2': [300, 2000],
|
|
475
|
+
'SCD30_HUM': [20, 99],
|
|
476
|
+
'SCD30_TEMP': [-20, 50],
|
|
477
|
+
'ST LPS33 - Barometric Pressure': [50, 110],
|
|
478
|
+
'PRESS': [50, 110],
|
|
479
|
+
'PMS5003_PM_1': [0, 500],
|
|
480
|
+
'PMS5003_PM_25': [0, 500],
|
|
481
|
+
'PMS5003_PM_10': [0, 500],
|
|
482
|
+
'SEN5X_HUM': [20, 99],
|
|
483
|
+
'SEN5X_PM_1': [0, 500],
|
|
484
|
+
'SEN5X_PM_10': [0, 500],
|
|
485
|
+
'SEN5X_PM_25': [0, 500],
|
|
486
|
+
'SEN5X_PM_40': [0, 500],
|
|
487
|
+
'SEN5X_TEMP': [-20, 50],
|
|
488
|
+
'SFA30_HCHO': [00, 1000],
|
|
489
|
+
'SFA30_HUM': [20, 99],
|
|
490
|
+
'SFA30_TEMP': [-20, 50],
|
|
491
|
+
'ADC_48_0': [0, 3],
|
|
492
|
+
'ADC_48_1': [0, 3],
|
|
493
|
+
'ADC_48_2': [0, 3],
|
|
494
|
+
'ADC_48_3': [0, 3],
|
|
495
|
+
'ADC_49_0': [0, 3],
|
|
496
|
+
'ADC_49_1': [0, 3],
|
|
497
|
+
'ADC_49_2': [0, 3],
|
|
498
|
+
'ADC_49_3': [0, 3],
|
|
499
|
+
'HUM': [20, 99],
|
|
500
|
+
'TEMP': [-20, 50],
|
|
501
|
+
'NO2': [0, 1000],
|
|
502
|
+
'O3': [0, 1000]
|
|
503
|
+
}
|
|
504
|
+
|
|
425
505
|
def __init__(self):
|
|
426
506
|
self._env_file = None
|
|
427
507
|
self.paths = self.get_paths()
|
|
@@ -695,7 +775,7 @@ class Config(object):
|
|
|
695
775
|
with open(namespath, 'w') as file:
|
|
696
776
|
json.dump(names_dump, file)
|
|
697
777
|
|
|
698
|
-
# Find environment file in root
|
|
778
|
+
# Find environment file in root
|
|
699
779
|
if exists(join(self.paths['data'],'.env')):
|
|
700
780
|
self._env_file = join(self.paths['data'],'.env')
|
|
701
781
|
print(f'Found Environment file at: {self._env_file}')
|
|
@@ -704,6 +784,9 @@ class Config(object):
|
|
|
704
784
|
print(f'No environment file found. If you had an environment file (.env) before, make sure its now here')
|
|
705
785
|
print(join(self.paths['data'],'.env'))
|
|
706
786
|
|
|
787
|
+
if 'SC_BEARER' not in environ:
|
|
788
|
+
print('SC_BEARER not in environment variables. You may get throttled when requesting to api.smartcitizen.me')
|
|
789
|
+
|
|
707
790
|
def load(self):
|
|
708
791
|
""" Override config if config file exists. """
|
|
709
792
|
_sccpath = join(self.paths['config'], 'config.yaml')
|
|
@@ -12,7 +12,7 @@ from scdata._config import config
|
|
|
12
12
|
from scdata.io.device_api import *
|
|
13
13
|
from scdata.models import Blueprint, Metric, Source, APIParams, CSVParams, DeviceOptions, Sensor
|
|
14
14
|
|
|
15
|
-
from os.path import join, basename
|
|
15
|
+
from os.path import join, basename, exists
|
|
16
16
|
from urllib.parse import urlparse
|
|
17
17
|
from pandas import DataFrame, to_timedelta, Timedelta
|
|
18
18
|
from numpy import nan
|
|
@@ -256,7 +256,8 @@ class Device(BaseModel):
|
|
|
256
256
|
return True
|
|
257
257
|
return False
|
|
258
258
|
|
|
259
|
-
async def load(self, cache=None, convert_units=True,
|
|
259
|
+
async def load(self, cache=None, convert_units=True,
|
|
260
|
+
convert_names=True, ignore_error = True):
|
|
260
261
|
'''
|
|
261
262
|
Loads the device with some options
|
|
262
263
|
|
|
@@ -271,9 +272,9 @@ class Device(BaseModel):
|
|
|
271
272
|
convert_names: bool
|
|
272
273
|
Default: True
|
|
273
274
|
Convert names for channels based on ids
|
|
274
|
-
|
|
275
|
-
Default:
|
|
276
|
-
|
|
275
|
+
ignore_error: bool
|
|
276
|
+
Default: True
|
|
277
|
+
Ignore if the cache does not exist
|
|
277
278
|
Returns
|
|
278
279
|
----------
|
|
279
280
|
True if loaded correctly
|
|
@@ -284,21 +285,29 @@ class Device(BaseModel):
|
|
|
284
285
|
frequency = self.options.frequency
|
|
285
286
|
clean_na = self.options.clean_na
|
|
286
287
|
resample = self.options.resample
|
|
288
|
+
limit = self.options.limit
|
|
289
|
+
channels = self.options.channels
|
|
287
290
|
cached_data = DataFrame()
|
|
288
291
|
|
|
289
292
|
# Only case where cache makes sense
|
|
290
293
|
if self.source.type == 'api':
|
|
291
294
|
if cache is not None and cache:
|
|
292
|
-
if cache
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
clean_na = clean_na,
|
|
298
|
-
resample = resample,
|
|
299
|
-
index_name = 'TIME')
|
|
295
|
+
if not exists(cache):
|
|
296
|
+
if not ignore_error:
|
|
297
|
+
raise FileExistsError(f'Cache does not exist: {cache}')
|
|
298
|
+
else:
|
|
299
|
+
logger.warning(f'Cache file does not exist: {cache}')
|
|
300
300
|
else:
|
|
301
|
-
|
|
301
|
+
if cache.endswith('.csv'):
|
|
302
|
+
cached_data = read_csv_file(
|
|
303
|
+
path = cache,
|
|
304
|
+
timezone = timezone,
|
|
305
|
+
frequency = frequency,
|
|
306
|
+
clean_na = clean_na,
|
|
307
|
+
resample = resample,
|
|
308
|
+
index_name = 'TIME')
|
|
309
|
+
else:
|
|
310
|
+
raise NotImplementedError(f'Cache needs to be a .csv file. Got {cache}.')
|
|
302
311
|
|
|
303
312
|
# Make request with a logical min_date
|
|
304
313
|
if not cached_data.empty:
|
|
@@ -316,6 +325,8 @@ class Device(BaseModel):
|
|
|
316
325
|
max_date = max_date,
|
|
317
326
|
frequency = frequency,
|
|
318
327
|
clean_na = clean_na,
|
|
328
|
+
limit = limit,
|
|
329
|
+
channels = channels,
|
|
319
330
|
resample = resample)
|
|
320
331
|
else:
|
|
321
332
|
self.handler.get_data(
|
|
@@ -327,27 +338,24 @@ class Device(BaseModel):
|
|
|
327
338
|
|
|
328
339
|
# In principle this links both dataframes as they are unmutable
|
|
329
340
|
self.data = self.handler.data
|
|
330
|
-
# Wrap it all up
|
|
331
|
-
# TODO Avoid doing this if not needed?
|
|
332
|
-
self.loaded = self.__load_wrapup__(max_amount, convert_units=convert_units, convert_names=convert_names, cached_data=cached_data)
|
|
333
341
|
|
|
342
|
+
# Wrap it all up
|
|
343
|
+
self.loaded = self.__load_wrapup__(cached_data=cached_data)
|
|
334
344
|
self.processed = False
|
|
345
|
+
|
|
335
346
|
return self.loaded
|
|
336
347
|
|
|
337
|
-
def __load_wrapup__(self,
|
|
348
|
+
def __load_wrapup__(self, cached_data=None):
|
|
349
|
+
|
|
338
350
|
if self.data is not None:
|
|
339
351
|
if not self.data.empty:
|
|
340
|
-
if max_amount is not None:
|
|
341
|
-
# TODO Dirty workaround
|
|
342
|
-
logger.info(f'Trimming dataframe to {max_amount} rows')
|
|
343
|
-
self.data=self.data.dropna(axis = 0, how='all').head(max_amount)
|
|
344
352
|
# Convert names
|
|
345
|
-
|
|
346
|
-
self.__convert_names__()
|
|
353
|
+
self.__convert_names__()
|
|
347
354
|
# Convert units
|
|
348
|
-
|
|
349
|
-
|
|
355
|
+
self.__convert_units__()
|
|
356
|
+
|
|
350
357
|
self.postprocessing_updated = False
|
|
358
|
+
|
|
351
359
|
else:
|
|
352
360
|
logger.info('Empty dataframe in loaded data. Waiting for cache...')
|
|
353
361
|
|
|
@@ -358,6 +366,7 @@ class Device(BaseModel):
|
|
|
358
366
|
return not self.data.empty
|
|
359
367
|
|
|
360
368
|
def __convert_names__(self):
|
|
369
|
+
if not self.options.convert_names: return
|
|
361
370
|
logger.info('Converting names...')
|
|
362
371
|
|
|
363
372
|
self.data.rename(columns=self._rename, inplace=True)
|
|
@@ -370,6 +379,8 @@ class Device(BaseModel):
|
|
|
370
379
|
The files are with original units, and then converted in the device only
|
|
371
380
|
for the data but never chached like so.
|
|
372
381
|
'''
|
|
382
|
+
if not self.options.convert_units: return
|
|
383
|
+
|
|
373
384
|
logger.info('Checking if units need to be converted...')
|
|
374
385
|
for sensor in self.data.columns:
|
|
375
386
|
_rename_inv = {v: k for k, v in self._rename.items()}
|
|
@@ -468,20 +479,31 @@ class Device(BaseModel):
|
|
|
468
479
|
if metric.kwargs is not None: kwargs = metric.kwargs
|
|
469
480
|
|
|
470
481
|
try:
|
|
471
|
-
|
|
482
|
+
process_result = funct(self.data, *args, **kwargs)
|
|
472
483
|
except KeyError:
|
|
473
484
|
logger.error('Cannot process requested function with data provided')
|
|
474
485
|
process_ok = False
|
|
475
486
|
pass
|
|
476
487
|
else:
|
|
477
|
-
if result is not None:
|
|
478
|
-
self.data[metric.name] = result
|
|
479
|
-
process_ok &= True
|
|
480
488
|
# If the metric is None, might be for many reasons and shouldn't collapse the process_ok
|
|
489
|
+
if process_result is not None:
|
|
490
|
+
if 'ERROR' in process_result.status_code.name:
|
|
491
|
+
# We got an error during the processing
|
|
492
|
+
logger.error(process_result.status_code.name)
|
|
493
|
+
process_ok &= False
|
|
494
|
+
elif 'WARNING' in process_result.status_code.name:
|
|
495
|
+
# In this case there is no data to put into the metric
|
|
496
|
+
# but there is no reason to make deny process_ok
|
|
497
|
+
logger.warning(process_result.status_code.name)
|
|
498
|
+
process_ok &= True
|
|
499
|
+
elif 'SUCCESS' in process_result.status_code.name:
|
|
500
|
+
self.data[metric.name] = process_result.data
|
|
501
|
+
logger.info(process_result.status_code.name)
|
|
502
|
+
process_ok &= True
|
|
481
503
|
|
|
482
504
|
if process_ok:
|
|
483
505
|
logger.info(f"Device {self.paramsParsed.id} processed")
|
|
484
|
-
self.processed = process_ok
|
|
506
|
+
self.processed = process_ok
|
|
485
507
|
|
|
486
508
|
return self.processed
|
|
487
509
|
|
|
@@ -490,114 +512,108 @@ class Device(BaseModel):
|
|
|
490
512
|
return self._sensors
|
|
491
513
|
|
|
492
514
|
def update_postprocessing_date(self):
|
|
493
|
-
|
|
494
|
-
|
|
515
|
+
# This function updates the postprocessing date with the latest logical date
|
|
516
|
+
latest_postprocessing = None
|
|
517
|
+
if self.loaded:
|
|
518
|
+
# If device was loaded (data not empty)
|
|
519
|
+
if self.processed:
|
|
520
|
+
# If device was processed, new postprocessing is the last reading rounded up with frequency
|
|
521
|
+
latest_postprocessing = localise_date(self.data.index[-1] + to_timedelta(self.options.frequency), 'UTC')
|
|
522
|
+
logger.info(f'Updating latest_postprocessing to {latest_postprocessing}')
|
|
523
|
+
else:
|
|
524
|
+
logger.info(f'Cannot update latest_postprocessing. Device was loaded but not processed')
|
|
525
|
+
else:
|
|
526
|
+
# If device was not loaded, increase the postprocessing limited to last_reading_at
|
|
527
|
+
latest_postprocessing = min(self.handler.json.last_reading_at, self.options.max_date)
|
|
528
|
+
logger.info(f'Updating latest_postprocessing to {latest_postprocessing}')
|
|
529
|
+
|
|
530
|
+
if latest_postprocessing is None:
|
|
531
|
+
return False
|
|
532
|
+
|
|
495
533
|
if self.handler.update_latest_postprocessing(latest_postprocessing):
|
|
496
534
|
# Consider the case of no postprocessing, to avoid making the whole thing false
|
|
497
535
|
if latest_postprocessing.to_pydatetime() == self.handler.latest_postprocessing or self.handler.json.postprocessing is None:
|
|
498
536
|
self.postprocessing_updated = True
|
|
499
537
|
else:
|
|
500
538
|
self.postprocessing_updated = False
|
|
539
|
+
|
|
501
540
|
return self.postprocessing_updated
|
|
502
541
|
|
|
503
|
-
#
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
#
|
|
525
|
-
#
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
#
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
#
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
# df = clean(df, 'drop', how = 'all')
|
|
579
|
-
|
|
580
|
-
# if df.empty:
|
|
581
|
-
# std_out('Empty dataframe, ignoring', 'WARNING')
|
|
582
|
-
# return False
|
|
583
|
-
|
|
584
|
-
# # Create object
|
|
585
|
-
# ndev = Hclass(did = self.forwarding_params)
|
|
586
|
-
# post_ok = ndev.post_data_to_device(df, chunk_size = chunk_size,
|
|
587
|
-
# dry_run = dry_run, max_retries = 2)
|
|
588
|
-
|
|
589
|
-
# if post_ok:
|
|
590
|
-
# # TODO Check if we like this
|
|
591
|
-
# if self.source == 'api':
|
|
592
|
-
# self.update_latest_postprocessing()
|
|
593
|
-
# std_out(f'Posted data for {self.params.id}', 'SUCCESS')
|
|
594
|
-
# else:
|
|
595
|
-
# std_out(f'Error posting data for {self.params.id}', 'ERROR')
|
|
596
|
-
# return post_ok
|
|
597
|
-
|
|
598
|
-
# else:
|
|
599
|
-
# std_out('Empty forwarding information', 'ERROR')
|
|
600
|
-
# return False
|
|
542
|
+
# Check nans per column, return dict
|
|
543
|
+
# TODO-DOCUMENT
|
|
544
|
+
def get_nan_ratio(self, **kwargs):
|
|
545
|
+
if not self.loaded:
|
|
546
|
+
logger.error('Need to load first (device.load())')
|
|
547
|
+
return False
|
|
548
|
+
|
|
549
|
+
if 'sampling_rate' not in kwargs:
|
|
550
|
+
sampling_rate = config._default_sampling_rate
|
|
551
|
+
else:
|
|
552
|
+
sampling_rate = kwargs['sampling_rate']
|
|
553
|
+
result = {}
|
|
554
|
+
|
|
555
|
+
for column in self.data.columns:
|
|
556
|
+
if column not in sampling_rate: continue
|
|
557
|
+
df = self.data[column].resample(f'{sampling_rate[column]}Min').mean()
|
|
558
|
+
minutes = df.groupby(df.index.date).mean().index.to_series().diff()/Timedelta('60s')
|
|
559
|
+
result[column] = (1-(minutes-df.isna().groupby(df.index.date).sum())/minutes)*sampling_rate[column]
|
|
560
|
+
|
|
561
|
+
return result
|
|
562
|
+
|
|
563
|
+
# Check plausibility per column, return dict. Doesn't take into account nans
|
|
564
|
+
# TODO-DOCUMENT
|
|
565
|
+
def get_plausible_ratio(self, **kwargs):
|
|
566
|
+
if not self.loaded:
|
|
567
|
+
logger.error('Need to load first (device.load())')
|
|
568
|
+
return False
|
|
569
|
+
|
|
570
|
+
if 'unplausible_values' not in kwargs:
|
|
571
|
+
unplausible_values = config._default_unplausible_values
|
|
572
|
+
else:
|
|
573
|
+
unplausible_values = kwargs['unplausible_values']
|
|
574
|
+
|
|
575
|
+
if 'sampling_rate' not in kwargs:
|
|
576
|
+
sampling_rate = config._default_sampling_rate
|
|
577
|
+
else:
|
|
578
|
+
sampling_rate = kwargs['sampling_rate']
|
|
579
|
+
|
|
580
|
+
return {column: self.data[column].between(left=unplausible_values[column][0], right=unplausible_values[column][1]).groupby(self.data[column].index.date).sum()/self.data.groupby(self.data.index.date).count()[column] for column in self.data.columns if column in unplausible_values}
|
|
581
|
+
|
|
582
|
+
# Check plausibility per column, return dict. Doesn't take into account nans
|
|
583
|
+
def get_outlier_ratio(self, **kwargs):
|
|
584
|
+
if not self.loaded:
|
|
585
|
+
logger.error('Need to load first (device.load())')
|
|
586
|
+
return False
|
|
587
|
+
result = {}
|
|
588
|
+
resample = '360h'
|
|
589
|
+
|
|
590
|
+
for column in self.data.columns:
|
|
591
|
+
Q1 = self.data[column].resample(resample).mean().quantile(0.25)
|
|
592
|
+
Q3 = self.data[column].resample(resample).mean().quantile(0.75)
|
|
593
|
+
IQR = Q3 - Q1
|
|
594
|
+
|
|
595
|
+
mask = (self.data[column] < (Q1 - 1.5 * IQR)) | (self.data[column] > (Q3 + 1.5 * IQR))
|
|
596
|
+
result[column] = mask.groupby(mask.index.date).mean()
|
|
597
|
+
|
|
598
|
+
return result
|
|
599
|
+
|
|
600
|
+
# Check plausibility per column, return dict. Doesn't take into account nans
|
|
601
|
+
def get_outliers(self, **kwargs):
|
|
602
|
+
if not self.loaded:
|
|
603
|
+
logger.error('Need to load first (device.load())')
|
|
604
|
+
return False
|
|
605
|
+
result = {}
|
|
606
|
+
resample = '360h'
|
|
607
|
+
|
|
608
|
+
for column in self.data.columns:
|
|
609
|
+
Q1 = self.data[column].resample(resample).mean().quantile(0.25)
|
|
610
|
+
Q3 = self.data[column].resample(resample).mean().quantile(0.75)
|
|
611
|
+
IQR = Q3 - Q1
|
|
612
|
+
|
|
613
|
+
mask = (self.data[column] < (Q1 - 1.5 * IQR)) | (self.data[column] > (Q3 + 1.5 * IQR))
|
|
614
|
+
result[column] = mask
|
|
615
|
+
|
|
616
|
+
return result
|
|
601
617
|
|
|
602
618
|
def export(self, path, forced_overwrite = False, file_format = 'csv'):
|
|
603
619
|
'''
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
''' Implementation of different processes to be done in each device '''
|
|
2
2
|
|
|
3
|
+
# TODO UPDATE all these functions to make them comply with StatusCode types
|
|
3
4
|
from scdata.tools.lazy import LazyCallable
|
|
4
5
|
from .formulae import absolute_humidity, exp_f, fit_exp_f
|
|
5
6
|
from .geoseries import is_within_circle
|