scdata 1.3.0__tar.gz → 1.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scdata-1.3.0/scdata.egg-info → scdata-1.3.2}/PKG-INFO +1 -1
- {scdata-1.3.0 → scdata-1.3.2}/scdata/__init__.py +1 -1
- {scdata-1.3.0 → scdata-1.3.2}/scdata/_config/config.py +13 -7
- {scdata-1.3.0 → scdata-1.3.2}/scdata/device/device.py +232 -48
- {scdata-1.3.0 → scdata-1.3.2}/scdata/io/device_file.py +7 -4
- scdata-1.3.2/scdata/tools/series.py +61 -0
- {scdata-1.3.0 → scdata-1.3.2/scdata.egg-info}/PKG-INFO +1 -1
- {scdata-1.3.0 → scdata-1.3.2}/scdata.egg-info/SOURCES.txt +1 -0
- {scdata-1.3.0 → scdata-1.3.2}/setup.py +1 -1
- {scdata-1.3.0 → scdata-1.3.2}/LICENSE +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/MANIFEST.in +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/README.md +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/requirements.txt +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/_config/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/device/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/device/plot/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/alphasense.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/baseline.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/error_codes.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/formulae.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/geoseries.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/params.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/regression.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/timeseries.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/io/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/io/device_api.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/io/model.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/models/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/models/models.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/checks/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/checks/checks.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/dispersion/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/dispersion/dispersion.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/export/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/export/templates/sc_template.html +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/export/to_file.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/box_plot.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/heatmap_iplot.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/heatmap_plot.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/maps.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/plot_tools.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/scatter_dispersion_grid.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/scatter_iplot.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/scatter_plot.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/target_diagram.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_dendrogram.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_dispersion_grid.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_dispersion_plot.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_dispersion_uplot.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_iplot.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_plot.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_scatter.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_uplot.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/test.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/tools/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/tools/combine.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/tools/history.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/test/tools/prepare.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/__init__.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/cleaning.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/custom_logger.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/date.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/dictmerge.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/find.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/gets.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/interim/example.csv +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/interim/geodata.csv +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/lazy.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/location.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/report.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/stats.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/units.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/uploads/example_upload_1.json +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/uploads/example_zenodo_upload.yaml +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/uploads/report.pdf +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/url_check.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/zenodo.py +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/zenodo_templates/README.md +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/zenodo_templates/template_zenodo_dataset.json +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/zenodo_templates/template_zenodo_publication.json +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata.egg-info/dependency_links.txt +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata.egg-info/not-zip-safe +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata.egg-info/requires.txt +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/scdata.egg-info/top_level.txt +0 -0
- {scdata-1.3.0 → scdata-1.3.2}/setup.cfg +0 -0
|
@@ -42,9 +42,6 @@ class Config(object):
|
|
|
42
42
|
_timeout = 3
|
|
43
43
|
_max_http_retries = 2
|
|
44
44
|
|
|
45
|
-
# Max concurrent requests
|
|
46
|
-
_max_concurrent_requests = 30
|
|
47
|
-
|
|
48
45
|
### ---------------------------------------
|
|
49
46
|
### -----------------DATA------------------
|
|
50
47
|
### ---------------------------------------
|
|
@@ -121,7 +118,6 @@ class Config(object):
|
|
|
121
118
|
'https://raw.githubusercontent.com/fablabbcn/smartcitizen-data/master/names/SCDevice.json'
|
|
122
119
|
]
|
|
123
120
|
|
|
124
|
-
|
|
125
121
|
### ---------------------------------------
|
|
126
122
|
### -------------METRICS DATA--------------
|
|
127
123
|
### ---------------------------------------
|
|
@@ -466,7 +462,12 @@ class Config(object):
|
|
|
466
462
|
'TEMP': 1,
|
|
467
463
|
'RSSI': 1,
|
|
468
464
|
'NO2': 1,
|
|
469
|
-
'O3': 1
|
|
465
|
+
'O3': 1,
|
|
466
|
+
'AS_TEMP': 1,
|
|
467
|
+
'AS_PH': 1,
|
|
468
|
+
'AS_COND': 1,
|
|
469
|
+
'AS_DO_SAT': 1,
|
|
470
|
+
'AS_DO': 1
|
|
470
471
|
}
|
|
471
472
|
|
|
472
473
|
_default_unplausible_values = {
|
|
@@ -474,6 +475,7 @@ class Config(object):
|
|
|
474
475
|
'SCD30_CO2': [300, 2000],
|
|
475
476
|
'SCD30_HUM': [20, 99],
|
|
476
477
|
'SCD30_TEMP': [-20, 50],
|
|
478
|
+
'BATT': [0, 100],
|
|
477
479
|
'ST LPS33 - Barometric Pressure': [50, 110],
|
|
478
480
|
'PRESS': [50, 110],
|
|
479
481
|
'PMS5003_PM_1': [0, 500],
|
|
@@ -499,7 +501,12 @@ class Config(object):
|
|
|
499
501
|
'HUM': [20, 99],
|
|
500
502
|
'TEMP': [-20, 50],
|
|
501
503
|
'NO2': [0, 1000],
|
|
502
|
-
'O3': [0, 1000]
|
|
504
|
+
'O3': [0, 1000],
|
|
505
|
+
'AS_TEMP': [-20, 50],
|
|
506
|
+
'AS_PH': [0, 14],
|
|
507
|
+
'AS_COND': [0, 100000],
|
|
508
|
+
'AS_DO_SAT': [0, 100],
|
|
509
|
+
'AS_DO': [0, 15]
|
|
503
510
|
}
|
|
504
511
|
|
|
505
512
|
def __init__(self):
|
|
@@ -508,7 +515,6 @@ class Config(object):
|
|
|
508
515
|
self.load()
|
|
509
516
|
self.get_meta_data()
|
|
510
517
|
|
|
511
|
-
|
|
512
518
|
def __getattr__(self, name):
|
|
513
519
|
try:
|
|
514
520
|
return self[name]
|
|
@@ -8,19 +8,20 @@ from scdata.tools.date import localise_date
|
|
|
8
8
|
from scdata.tools.dictmerge import dict_fmerge
|
|
9
9
|
from scdata.tools.units import get_units_convf
|
|
10
10
|
from scdata.tools.find import find_by_field
|
|
11
|
+
from scdata.tools.series import count_nas, infer_sampling_rate, mode_ratio, normalize_central, rolling_deltas
|
|
11
12
|
from scdata._config import config
|
|
12
13
|
from scdata.io.device_api import *
|
|
13
14
|
from scdata.models import Blueprint, Metric, Source, APIParams, CSVParams, DeviceOptions, Sensor
|
|
14
15
|
|
|
15
16
|
from os.path import join, basename, exists
|
|
16
17
|
from urllib.parse import urlparse
|
|
17
|
-
from pandas import DataFrame, to_timedelta, Timedelta
|
|
18
|
+
from pandas import DataFrame, Series, to_timedelta, Timedelta
|
|
18
19
|
from numpy import nan
|
|
19
20
|
from collections.abc import Iterable
|
|
20
21
|
from importlib import import_module
|
|
21
22
|
from pydantic import TypeAdapter, BaseModel, ConfigDict
|
|
22
23
|
from pydantic_core import ValidationError
|
|
23
|
-
from typing import Optional, List
|
|
24
|
+
from typing import Optional, List, Dict
|
|
24
25
|
from json import dumps
|
|
25
26
|
|
|
26
27
|
import os
|
|
@@ -312,7 +313,7 @@ class Device(BaseModel):
|
|
|
312
313
|
else:
|
|
313
314
|
logger.warning(f'Cache file does not exist: {cache}')
|
|
314
315
|
else:
|
|
315
|
-
if cache.endswith('.csv'):
|
|
316
|
+
if cache.endswith('.csv') or cache.endswith('.csv.gz'):
|
|
316
317
|
cached_data = read_csv_file(
|
|
317
318
|
path = cache,
|
|
318
319
|
timezone = timezone,
|
|
@@ -554,83 +555,266 @@ class Device(BaseModel):
|
|
|
554
555
|
|
|
555
556
|
return self.postprocessing_updated
|
|
556
557
|
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
558
|
+
def get_nan_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_nan_ratio", sampling_rates:Dict[str, int]=None) -> DataFrame:
|
|
559
|
+
'''
|
|
560
|
+
Check NaN ratio per column, return pd.DataFrame with the same index as
|
|
561
|
+
self.data with the NaN ratio per rolling window.
|
|
562
|
+
Parameters
|
|
563
|
+
----------
|
|
564
|
+
period: str
|
|
565
|
+
"1h"
|
|
566
|
+
Rolling window width.
|
|
567
|
+
subset: List[str]
|
|
568
|
+
Columns to apply the stuck ratio calculation to. If None, all columns
|
|
569
|
+
are used.
|
|
570
|
+
sampling_rates: Dict[str, int]
|
|
571
|
+
Dictionary with sampling rates per column in minutes. If None, will get
|
|
572
|
+
sampling rates from config._default_sampling_rate or try to infer them.
|
|
573
|
+
|
|
574
|
+
Returns
|
|
575
|
+
----------
|
|
576
|
+
result: DataFrame
|
|
577
|
+
DataFrame with rolling nan ratio.
|
|
578
|
+
'''
|
|
560
579
|
if not self.loaded:
|
|
561
580
|
logger.error('Need to load first (device.load())')
|
|
562
581
|
return False
|
|
563
582
|
|
|
564
|
-
if
|
|
565
|
-
|
|
583
|
+
if subset is not None:
|
|
584
|
+
data = self.data[subset]
|
|
566
585
|
else:
|
|
567
|
-
|
|
586
|
+
data = self.data
|
|
587
|
+
|
|
568
588
|
result = {}
|
|
569
589
|
|
|
570
|
-
for column in
|
|
571
|
-
if
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
590
|
+
for column in data.columns:
|
|
591
|
+
if sampling_rates is not None and column in sampling_rates:
|
|
592
|
+
sampling_rate = sampling_rates[column]
|
|
593
|
+
elif column in config._default_sampling_rate:
|
|
594
|
+
sampling_rate = config._default_sampling_rate[column]
|
|
595
|
+
else:
|
|
596
|
+
sampling_rate = infer_sampling_rate(data[column])
|
|
575
597
|
|
|
576
|
-
|
|
598
|
+
if sampling_rate is None:
|
|
599
|
+
logger.warning(f'Cannot infer sampling rate for column {column}. Skipping NaN ratio calculation')
|
|
600
|
+
continue
|
|
601
|
+
|
|
602
|
+
sampling_rate_timedelta = Timedelta(f'{sampling_rate}min')
|
|
603
|
+
max_possible_count = Timedelta(period) / sampling_rate_timedelta
|
|
604
|
+
datapoints = data[column].resample(sampling_rate_timedelta).mean() # Need to aggregate somehow
|
|
605
|
+
|
|
606
|
+
nan_count = datapoints.rolling(period).apply(count_nas).fillna(max_possible_count) # If we didn't fillna, we would get nas as count when the whole period is missing.
|
|
607
|
+
|
|
608
|
+
nan_ratio = nan_count / max_possible_count
|
|
609
|
+
|
|
610
|
+
result[column + suffix] = nan_ratio
|
|
611
|
+
|
|
612
|
+
return DataFrame(result)
|
|
613
|
+
|
|
614
|
+
def get_implausible_values(self, column: str, plausible_interval:List[int]=None) -> Series:
|
|
615
|
+
'''Get a boolean series indicating which values are outside the plausible
|
|
616
|
+
interval for the physical magnitude, as defined by plausible_interval
|
|
617
|
+
or config._default_unplausible_values.
|
|
618
|
+
|
|
619
|
+
For example, we consider values of NOISE_A below 20dB or above 99dB to
|
|
620
|
+
be implausible.
|
|
621
|
+
|
|
622
|
+
Parameters
|
|
623
|
+
----------
|
|
624
|
+
column: str
|
|
625
|
+
Column to apply the plausible values check to.
|
|
626
|
+
plausible_interval: List[int]
|
|
627
|
+
List with two elements defining the plausible interval [min, max].
|
|
628
|
+
If None, will try to get the plausible interval from
|
|
629
|
+
config._default_unplausible_values.'''
|
|
630
|
+
|
|
631
|
+
series = self.data[column]
|
|
577
632
|
|
|
578
|
-
# Check plausibility per column, return dict. Doesn't take into account nans
|
|
579
|
-
# TODO-DOCUMENT
|
|
580
|
-
def get_plausible_ratio(self, **kwargs):
|
|
581
633
|
if not self.loaded:
|
|
582
634
|
logger.error('Need to load first (device.load())')
|
|
583
635
|
return False
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
636
|
+
if plausible_interval is not None:
|
|
637
|
+
left, right = plausible_interval
|
|
638
|
+
elif column in config._default_unplausible_values:
|
|
639
|
+
left, right = config._default_unplausible_values[series.name] # I don't like this name
|
|
587
640
|
else:
|
|
588
|
-
|
|
641
|
+
copy = series.copy()
|
|
642
|
+
copy[:] = nan
|
|
589
643
|
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
644
|
+
return copy
|
|
645
|
+
|
|
646
|
+
implausible = (series < left) | (series > right)
|
|
647
|
+
nullable = implausible.convert_dtypes()
|
|
648
|
+
nullable[series.isna()] = None # Propagate NaNs
|
|
649
|
+
|
|
650
|
+
return nullable
|
|
594
651
|
|
|
595
|
-
return {column: self.data[column].between(left=unplausible_values[column][0], right=unplausible_values[column][1]).groupby(self.data[column].index.date).sum()/self.data.groupby(self.data.index.date).count()[column] for column in self.data.columns if column in unplausible_values}
|
|
596
652
|
|
|
597
|
-
|
|
598
|
-
|
|
653
|
+
def get_implausible_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_implausible_ratio", plausible_intervals:Dict[str, List[int]]=None) -> DataFrame:
|
|
654
|
+
'''Scan the series for values outside the plausible interval for the
|
|
655
|
+
physical magnitude, as defined by plausible_interval or
|
|
656
|
+
config._default_unplausible_values. For example, we find values of
|
|
657
|
+
NOISE_A below 20dB or above 99dB to be implausible.
|
|
658
|
+
|
|
659
|
+
Parameters
|
|
660
|
+
----------
|
|
661
|
+
period: str
|
|
662
|
+
Rolling window width.
|
|
663
|
+
subset: List[str]
|
|
664
|
+
Columns to apply the plausible ratio calculation to. If None, all columns
|
|
665
|
+
are used.
|
|
666
|
+
plausible_intervals: Dict[str, List[int]]
|
|
667
|
+
Dictionary with plausible intervals per column. Optional.
|
|
668
|
+
|
|
669
|
+
Returns
|
|
670
|
+
----------
|
|
671
|
+
result: DataFrame
|
|
672
|
+
DataFrame with rolling calculation of plausible ratio.
|
|
673
|
+
'''
|
|
674
|
+
|
|
599
675
|
if not self.loaded:
|
|
600
676
|
logger.error('Need to load first (device.load())')
|
|
601
677
|
return False
|
|
678
|
+
|
|
679
|
+
if subset is not None:
|
|
680
|
+
data = self.data[subset]
|
|
681
|
+
else:
|
|
682
|
+
data = self.data
|
|
602
683
|
result = {}
|
|
603
|
-
resample = '360h'
|
|
604
684
|
|
|
605
|
-
for column in
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
685
|
+
for column in data.columns:
|
|
686
|
+
if plausible_intervals is not None and column in plausible_intervals:
|
|
687
|
+
interval = plausible_intervals[column]
|
|
688
|
+
else:
|
|
689
|
+
interval = None
|
|
609
690
|
|
|
610
|
-
|
|
611
|
-
result[column] = mask.groupby(mask.index.date).mean()
|
|
691
|
+
implausible_values = self.get_implausible_values(column, plausible_interval=interval)
|
|
612
692
|
|
|
613
|
-
|
|
693
|
+
result[column + suffix] = implausible_values.rolling(period).mean() # Only consider actual observed data
|
|
694
|
+
|
|
695
|
+
return DataFrame(result)
|
|
696
|
+
|
|
697
|
+
def get_outlier_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_outlier_ratio", sigma=5, pct=0.05) -> DataFrame:
|
|
698
|
+
'''Get the percentage of outlier values based on the rate of increase. When sensors
|
|
699
|
+
report sudden jumps, these are likely to be erroneous values.
|
|
700
|
+
|
|
701
|
+
Parameters
|
|
702
|
+
----------
|
|
703
|
+
period: str
|
|
704
|
+
"1h"
|
|
705
|
+
Rolling window width.
|
|
706
|
+
subset: List[str]
|
|
707
|
+
Columns to apply the outlier ratio calculation to. If None, all columns
|
|
708
|
+
are used.
|
|
709
|
+
sigma: int
|
|
710
|
+
5
|
|
711
|
+
Number of standard deviations to consider a point an outlier. A higher
|
|
712
|
+
value would be more restrictive, detecting only worse malfunctions.
|
|
713
|
+
pct: float
|
|
714
|
+
0.05
|
|
715
|
+
Percentage of top and bottom values to ignore when normalizing. We
|
|
716
|
+
assume the outliers will always be a minority of the data, so ignoring
|
|
717
|
+
a small percentage of extreme values should help get a better estimate.
|
|
718
|
+
|
|
719
|
+
Returns
|
|
720
|
+
----------
|
|
721
|
+
result: DataFrame
|
|
722
|
+
DataFrame with rolling outlier ratio.
|
|
723
|
+
'''
|
|
614
724
|
|
|
615
|
-
# Check plausibility per column, return dict. Doesn't take into account nans
|
|
616
|
-
def get_outliers(self, **kwargs):
|
|
617
725
|
if not self.loaded:
|
|
618
726
|
logger.error('Need to load first (device.load())')
|
|
619
727
|
return False
|
|
728
|
+
|
|
729
|
+
if subset is not None:
|
|
730
|
+
data = self.data[subset]
|
|
731
|
+
else:
|
|
732
|
+
data = self.data
|
|
733
|
+
|
|
620
734
|
result = {}
|
|
621
|
-
resample = '360h'
|
|
622
735
|
|
|
623
|
-
for column in
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
736
|
+
for column in data.columns:
|
|
737
|
+
outlier_values = self.get_outlier_values(column, sigma=sigma, pct=pct)
|
|
738
|
+
|
|
739
|
+
result[column + suffix] = outlier_values.rolling(period).mean()
|
|
740
|
+
|
|
741
|
+
return DataFrame(result)
|
|
742
|
+
|
|
743
|
+
def get_outlier_values(self, column: str, sigma=5, pct=0.05) -> Series:
|
|
744
|
+
'''
|
|
745
|
+
Get a boolean series indicating which values are outliers based on
|
|
746
|
+
the rate of increase. We calculate the first derivative implied by
|
|
747
|
+
each datapoint. Then we normalize the derivative series after removing
|
|
748
|
+
the top and bottom pct% of values to avoid outliers impacting the mean
|
|
749
|
+
too much. Finally, we consider outliers those points where the
|
|
750
|
+
normalized derivative is above `sigma` standard deviations.
|
|
751
|
+
|
|
752
|
+
Parameters
|
|
753
|
+
----------
|
|
754
|
+
column: str
|
|
755
|
+
Column to apply the outlier values check to.
|
|
756
|
+
sigma: int
|
|
757
|
+
Number of standard deviations to consider a point an outlier.
|
|
758
|
+
pct: float
|
|
759
|
+
Percentage of top and bottom values to ignore when normalizing.
|
|
760
|
+
|
|
761
|
+
Returns
|
|
762
|
+
----------
|
|
763
|
+
Series
|
|
764
|
+
Boolean series indicating outlier values.
|
|
765
|
+
'''
|
|
766
|
+
|
|
767
|
+
if not self.loaded:
|
|
768
|
+
logger.error('Need to load first (device.load())')
|
|
769
|
+
return False
|
|
770
|
+
|
|
771
|
+
series = self.data[column]
|
|
772
|
+
|
|
773
|
+
deltas = rolling_deltas(series)
|
|
774
|
+
normalized_deltas = normalize_central(deltas, pct=pct)
|
|
775
|
+
outliers = normalized_deltas.abs() > sigma
|
|
627
776
|
|
|
628
|
-
|
|
629
|
-
|
|
777
|
+
return outliers
|
|
778
|
+
|
|
779
|
+
def get_top_value_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_top_value_ratio", ignore_zeroes=True) -> DataFrame:
|
|
780
|
+
'''
|
|
781
|
+
Check the frequency of the mode (top value) per column, return
|
|
782
|
+
pd.DataFrame with the same index as self.data with the mode ratio
|
|
783
|
+
per rolling window.
|
|
784
|
+
|
|
785
|
+
Parameters
|
|
786
|
+
----------
|
|
787
|
+
period: str
|
|
788
|
+
"1h"
|
|
789
|
+
Rolling window width.
|
|
790
|
+
subset: List[str]
|
|
791
|
+
Columns to apply the stuck ratio calculation to. If None, all columns
|
|
792
|
+
are used.
|
|
793
|
+
ignore_zeroes: boolean
|
|
794
|
+
True
|
|
795
|
+
Ignore zeroes when checking for stuck values. Passed through to mode_ratio()
|
|
796
|
+
Returns
|
|
797
|
+
----------
|
|
798
|
+
result: DataFrame
|
|
799
|
+
DataFrame with rolling mode ratio.
|
|
800
|
+
'''
|
|
801
|
+
if not self.loaded:
|
|
802
|
+
logger.error('Need to load first (device.load())')
|
|
803
|
+
return False
|
|
804
|
+
|
|
805
|
+
if subset is not None:
|
|
806
|
+
data = self.data[subset]
|
|
807
|
+
else:
|
|
808
|
+
data = self.data
|
|
809
|
+
|
|
810
|
+
rolling = data.rolling(period)
|
|
811
|
+
result = rolling.apply(lambda w: mode_ratio(w, ignore_zeroes), raw=False)
|
|
812
|
+
result.columns = [col + suffix for col in result.columns]
|
|
630
813
|
|
|
631
814
|
return result
|
|
632
815
|
|
|
633
|
-
|
|
816
|
+
|
|
817
|
+
def export(self, path, forced_overwrite = False, file_format = 'csv', gzip=False):
|
|
634
818
|
'''
|
|
635
819
|
Exports Device.data to file
|
|
636
820
|
Parameters
|
|
@@ -653,7 +837,7 @@ class Device(BaseModel):
|
|
|
653
837
|
logger.error('Cannot export null data')
|
|
654
838
|
return False
|
|
655
839
|
if file_format == 'csv':
|
|
656
|
-
return export_csv_file(path, str(self.paramsParsed.id), self.data, forced_overwrite = forced_overwrite)
|
|
840
|
+
return export_csv_file(path, str(self.paramsParsed.id), self.data, forced_overwrite = forced_overwrite, gzip=gzip)
|
|
657
841
|
else:
|
|
658
842
|
# TODO Make a list of supported formats
|
|
659
843
|
return NotImplementedError (f'Not supported format. Formats: [csv]')
|
|
@@ -64,7 +64,7 @@ class CSVHandler:
|
|
|
64
64
|
|
|
65
65
|
return self.data
|
|
66
66
|
|
|
67
|
-
def export_csv_file(path, file_name, df, forced_overwrite=False):
|
|
67
|
+
def export_csv_file(path, file_name, df, forced_overwrite=False, gzip=False):
|
|
68
68
|
'''
|
|
69
69
|
Exports pandas dataframe to a csv file
|
|
70
70
|
Parameters
|
|
@@ -87,10 +87,13 @@ def export_csv_file(path, file_name, df, forced_overwrite=False):
|
|
|
87
87
|
if not exists(path):
|
|
88
88
|
makedirs(path)
|
|
89
89
|
|
|
90
|
+
full_path = path + '/' + str(file_name) + '.csv'
|
|
91
|
+
full_path += '.gz' if gzip else ''
|
|
92
|
+
|
|
90
93
|
# If file does not exist
|
|
91
|
-
if not exists(
|
|
92
|
-
df.to_csv(
|
|
93
|
-
logger.info('File saved to: \n' +
|
|
94
|
+
if not exists(full_path) or forced_overwrite:
|
|
95
|
+
df.to_csv(full_path, sep=",")
|
|
96
|
+
logger.info('File saved to: \n' + full_path)
|
|
94
97
|
else:
|
|
95
98
|
logger.error("File Already exists - delete it first, I was not asked to overwrite anything!")
|
|
96
99
|
return False
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from scdata.tools.custom_logger import logger
|
|
4
|
+
|
|
5
|
+
from pandas import Series, Timedelta
|
|
6
|
+
import numpy as np
|
|
7
|
+
|
|
8
|
+
def infer_sampling_rate(series: Series) -> int | None:
|
|
9
|
+
'''Infer the sampling rate of the given timeseries, rounded to the
|
|
10
|
+
closest minute.
|
|
11
|
+
'''
|
|
12
|
+
|
|
13
|
+
time_differences = series.index.diff().value_counts()
|
|
14
|
+
most_common = time_differences.index[0]
|
|
15
|
+
|
|
16
|
+
minutes = most_common / Timedelta("1min")
|
|
17
|
+
integer_minutes = round(minutes)
|
|
18
|
+
|
|
19
|
+
if abs(integer_minutes - minutes) > 0.05:
|
|
20
|
+
logger.warning('Rounded a time difference with more than 5% error')
|
|
21
|
+
return None
|
|
22
|
+
|
|
23
|
+
return integer_minutes
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def mode_ratio(series: Series, ignore_zeroes=True) -> int:
|
|
27
|
+
'''Count the percentage of times the most common value appears in the series,
|
|
28
|
+
ignoring zeroes and NaNs.'''
|
|
29
|
+
|
|
30
|
+
if ignore_zeroes:
|
|
31
|
+
# Replace zeroes with random so that they don't impact value count
|
|
32
|
+
series = series.where(series!=0.0, np.random.random(size=series.size))
|
|
33
|
+
|
|
34
|
+
mode_count = series.value_counts().iloc[0]
|
|
35
|
+
|
|
36
|
+
return mode_count / series.count()
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def count_nas(series: Series) -> int:
|
|
40
|
+
'''Count the number of NaN values in the series.'''
|
|
41
|
+
return series.isna().sum()
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def rolling_deltas(series: Series) -> Series:
|
|
45
|
+
'''Compute the first derivative of the series at each datapoint.'''
|
|
46
|
+
|
|
47
|
+
dys = series.rolling(window=2).apply(lambda ys: ys.iloc[1] - ys.iloc[0])
|
|
48
|
+
dxs = series.index.diff().total_seconds()
|
|
49
|
+
|
|
50
|
+
return dys / dxs
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def normalize_central(series: Series, pct=0.05) -> Series:
|
|
54
|
+
'''Normalize the series by removing the mean and scaling to unit variance,
|
|
55
|
+
ignroring the top and bottom `pct` percent of values. This should be more
|
|
56
|
+
robust to outliers than standard normalization.'''
|
|
57
|
+
|
|
58
|
+
central = series[((series > series.quantile(pct)) | (series > series.quantile(1 - pct)))]
|
|
59
|
+
normalized = (series - central.mean()) / central.std()
|
|
60
|
+
|
|
61
|
+
return normalized
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scdata-1.3.0 → scdata-1.3.2}/scdata/tools/zenodo_templates/template_zenodo_publication.json
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|