scdata 1.3.0__tar.gz → 1.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. {scdata-1.3.0/scdata.egg-info → scdata-1.3.2}/PKG-INFO +1 -1
  2. {scdata-1.3.0 → scdata-1.3.2}/scdata/__init__.py +1 -1
  3. {scdata-1.3.0 → scdata-1.3.2}/scdata/_config/config.py +13 -7
  4. {scdata-1.3.0 → scdata-1.3.2}/scdata/device/device.py +232 -48
  5. {scdata-1.3.0 → scdata-1.3.2}/scdata/io/device_file.py +7 -4
  6. scdata-1.3.2/scdata/tools/series.py +61 -0
  7. {scdata-1.3.0 → scdata-1.3.2/scdata.egg-info}/PKG-INFO +1 -1
  8. {scdata-1.3.0 → scdata-1.3.2}/scdata.egg-info/SOURCES.txt +1 -0
  9. {scdata-1.3.0 → scdata-1.3.2}/setup.py +1 -1
  10. {scdata-1.3.0 → scdata-1.3.2}/LICENSE +0 -0
  11. {scdata-1.3.0 → scdata-1.3.2}/MANIFEST.in +0 -0
  12. {scdata-1.3.0 → scdata-1.3.2}/README.md +0 -0
  13. {scdata-1.3.0 → scdata-1.3.2}/requirements.txt +0 -0
  14. {scdata-1.3.0 → scdata-1.3.2}/scdata/_config/__init__.py +0 -0
  15. {scdata-1.3.0 → scdata-1.3.2}/scdata/device/__init__.py +0 -0
  16. {scdata-1.3.0 → scdata-1.3.2}/scdata/device/plot/__init__.py +0 -0
  17. {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/__init__.py +0 -0
  18. {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/alphasense.py +0 -0
  19. {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/baseline.py +0 -0
  20. {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/error_codes.py +0 -0
  21. {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/formulae.py +0 -0
  22. {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/geoseries.py +0 -0
  23. {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/params.py +0 -0
  24. {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/regression.py +0 -0
  25. {scdata-1.3.0 → scdata-1.3.2}/scdata/device/process/timeseries.py +0 -0
  26. {scdata-1.3.0 → scdata-1.3.2}/scdata/io/__init__.py +0 -0
  27. {scdata-1.3.0 → scdata-1.3.2}/scdata/io/device_api.py +0 -0
  28. {scdata-1.3.0 → scdata-1.3.2}/scdata/io/model.py +0 -0
  29. {scdata-1.3.0 → scdata-1.3.2}/scdata/models/__init__.py +0 -0
  30. {scdata-1.3.0 → scdata-1.3.2}/scdata/models/models.py +0 -0
  31. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/__init__.py +0 -0
  32. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/checks/__init__.py +0 -0
  33. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/checks/checks.py +0 -0
  34. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/dispersion/__init__.py +0 -0
  35. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/dispersion/dispersion.py +0 -0
  36. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/export/__init__.py +0 -0
  37. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/export/templates/sc_template.html +0 -0
  38. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/export/to_file.py +0 -0
  39. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/__init__.py +0 -0
  40. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/box_plot.py +0 -0
  41. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/heatmap_iplot.py +0 -0
  42. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/heatmap_plot.py +0 -0
  43. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/maps.py +0 -0
  44. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/plot_tools.py +0 -0
  45. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/scatter_dispersion_grid.py +0 -0
  46. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/scatter_iplot.py +0 -0
  47. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/scatter_plot.py +0 -0
  48. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/target_diagram.py +0 -0
  49. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_dendrogram.py +0 -0
  50. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_dispersion_grid.py +0 -0
  51. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_dispersion_plot.py +0 -0
  52. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_dispersion_uplot.py +0 -0
  53. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_iplot.py +0 -0
  54. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_plot.py +0 -0
  55. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_scatter.py +0 -0
  56. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/plot/ts_uplot.py +0 -0
  57. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/test.py +0 -0
  58. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/tools/__init__.py +0 -0
  59. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/tools/combine.py +0 -0
  60. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/tools/history.py +0 -0
  61. {scdata-1.3.0 → scdata-1.3.2}/scdata/test/tools/prepare.py +0 -0
  62. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/__init__.py +0 -0
  63. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/cleaning.py +0 -0
  64. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/custom_logger.py +0 -0
  65. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/date.py +0 -0
  66. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/dictmerge.py +0 -0
  67. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/find.py +0 -0
  68. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/gets.py +0 -0
  69. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/interim/example.csv +0 -0
  70. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/interim/geodata.csv +0 -0
  71. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/lazy.py +0 -0
  72. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/location.py +0 -0
  73. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/report.py +0 -0
  74. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/stats.py +0 -0
  75. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/units.py +0 -0
  76. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/uploads/example_upload_1.json +0 -0
  77. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/uploads/example_zenodo_upload.yaml +0 -0
  78. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/uploads/report.pdf +0 -0
  79. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/url_check.py +0 -0
  80. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/zenodo.py +0 -0
  81. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/zenodo_templates/README.md +0 -0
  82. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/zenodo_templates/template_zenodo_dataset.json +0 -0
  83. {scdata-1.3.0 → scdata-1.3.2}/scdata/tools/zenodo_templates/template_zenodo_publication.json +0 -0
  84. {scdata-1.3.0 → scdata-1.3.2}/scdata.egg-info/dependency_links.txt +0 -0
  85. {scdata-1.3.0 → scdata-1.3.2}/scdata.egg-info/not-zip-safe +0 -0
  86. {scdata-1.3.0 → scdata-1.3.2}/scdata.egg-info/requires.txt +0 -0
  87. {scdata-1.3.0 → scdata-1.3.2}/scdata.egg-info/top_level.txt +0 -0
  88. {scdata-1.3.0 → scdata-1.3.2}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scdata
3
- Version: 1.3.0
3
+ Version: 1.3.2
4
4
  Summary: Analysis of sensors and time series data
5
5
  Home-page: https://github.com/fablabbcn/smartcitizen-data
6
6
  Author: oscgonfer
@@ -3,4 +3,4 @@ from .device import Device
3
3
  from .test import Test
4
4
  from .models import Source, TestOptions, DeviceOptions, APIParams, FileParams, CSVParams
5
5
 
6
- __version__ = '1.3.0'
6
+ __version__ = '1.3.2'
@@ -42,9 +42,6 @@ class Config(object):
42
42
  _timeout = 3
43
43
  _max_http_retries = 2
44
44
 
45
- # Max concurrent requests
46
- _max_concurrent_requests = 30
47
-
48
45
  ### ---------------------------------------
49
46
  ### -----------------DATA------------------
50
47
  ### ---------------------------------------
@@ -121,7 +118,6 @@ class Config(object):
121
118
  'https://raw.githubusercontent.com/fablabbcn/smartcitizen-data/master/names/SCDevice.json'
122
119
  ]
123
120
 
124
-
125
121
  ### ---------------------------------------
126
122
  ### -------------METRICS DATA--------------
127
123
  ### ---------------------------------------
@@ -466,7 +462,12 @@ class Config(object):
466
462
  'TEMP': 1,
467
463
  'RSSI': 1,
468
464
  'NO2': 1,
469
- 'O3': 1
465
+ 'O3': 1,
466
+ 'AS_TEMP': 1,
467
+ 'AS_PH': 1,
468
+ 'AS_COND': 1,
469
+ 'AS_DO_SAT': 1,
470
+ 'AS_DO': 1
470
471
  }
471
472
 
472
473
  _default_unplausible_values = {
@@ -474,6 +475,7 @@ class Config(object):
474
475
  'SCD30_CO2': [300, 2000],
475
476
  'SCD30_HUM': [20, 99],
476
477
  'SCD30_TEMP': [-20, 50],
478
+ 'BATT': [0, 100],
477
479
  'ST LPS33 - Barometric Pressure': [50, 110],
478
480
  'PRESS': [50, 110],
479
481
  'PMS5003_PM_1': [0, 500],
@@ -499,7 +501,12 @@ class Config(object):
499
501
  'HUM': [20, 99],
500
502
  'TEMP': [-20, 50],
501
503
  'NO2': [0, 1000],
502
- 'O3': [0, 1000]
504
+ 'O3': [0, 1000],
505
+ 'AS_TEMP': [-20, 50],
506
+ 'AS_PH': [0, 14],
507
+ 'AS_COND': [0, 100000],
508
+ 'AS_DO_SAT': [0, 100],
509
+ 'AS_DO': [0, 15]
503
510
  }
504
511
 
505
512
  def __init__(self):
@@ -508,7 +515,6 @@ class Config(object):
508
515
  self.load()
509
516
  self.get_meta_data()
510
517
 
511
-
512
518
  def __getattr__(self, name):
513
519
  try:
514
520
  return self[name]
@@ -8,19 +8,20 @@ from scdata.tools.date import localise_date
8
8
  from scdata.tools.dictmerge import dict_fmerge
9
9
  from scdata.tools.units import get_units_convf
10
10
  from scdata.tools.find import find_by_field
11
+ from scdata.tools.series import count_nas, infer_sampling_rate, mode_ratio, normalize_central, rolling_deltas
11
12
  from scdata._config import config
12
13
  from scdata.io.device_api import *
13
14
  from scdata.models import Blueprint, Metric, Source, APIParams, CSVParams, DeviceOptions, Sensor
14
15
 
15
16
  from os.path import join, basename, exists
16
17
  from urllib.parse import urlparse
17
- from pandas import DataFrame, to_timedelta, Timedelta
18
+ from pandas import DataFrame, Series, to_timedelta, Timedelta
18
19
  from numpy import nan
19
20
  from collections.abc import Iterable
20
21
  from importlib import import_module
21
22
  from pydantic import TypeAdapter, BaseModel, ConfigDict
22
23
  from pydantic_core import ValidationError
23
- from typing import Optional, List
24
+ from typing import Optional, List, Dict
24
25
  from json import dumps
25
26
 
26
27
  import os
@@ -312,7 +313,7 @@ class Device(BaseModel):
312
313
  else:
313
314
  logger.warning(f'Cache file does not exist: {cache}')
314
315
  else:
315
- if cache.endswith('.csv'):
316
+ if cache.endswith('.csv') or cache.endswith('.csv.gz'):
316
317
  cached_data = read_csv_file(
317
318
  path = cache,
318
319
  timezone = timezone,
@@ -554,83 +555,266 @@ class Device(BaseModel):
554
555
 
555
556
  return self.postprocessing_updated
556
557
 
557
- # Check nans per column, return dict
558
- # TODO-DOCUMENT
559
- def get_nan_ratio(self, **kwargs):
558
+ def get_nan_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_nan_ratio", sampling_rates:Dict[str, int]=None) -> DataFrame:
559
+ '''
560
+ Check NaN ratio per column, return pd.DataFrame with the same index as
561
+ self.data with the NaN ratio per rolling window.
562
+ Parameters
563
+ ----------
564
+ period: str
565
+ "1h"
566
+ Rolling window width.
567
+ subset: List[str]
568
+ Columns to apply the stuck ratio calculation to. If None, all columns
569
+ are used.
570
+ sampling_rates: Dict[str, int]
571
+ Dictionary with sampling rates per column in minutes. If None, will get
572
+ sampling rates from config._default_sampling_rate or try to infer them.
573
+
574
+ Returns
575
+ ----------
576
+ result: DataFrame
577
+ DataFrame with rolling nan ratio.
578
+ '''
560
579
  if not self.loaded:
561
580
  logger.error('Need to load first (device.load())')
562
581
  return False
563
582
 
564
- if 'sampling_rate' not in kwargs:
565
- sampling_rate = config._default_sampling_rate
583
+ if subset is not None:
584
+ data = self.data[subset]
566
585
  else:
567
- sampling_rate = kwargs['sampling_rate']
586
+ data = self.data
587
+
568
588
  result = {}
569
589
 
570
- for column in self.data.columns:
571
- if column not in sampling_rate: continue
572
- df = self.data[column].resample(f'{sampling_rate[column]}Min').mean()
573
- minutes = df.groupby(df.index.date).mean().index.to_series().diff()/Timedelta('60s')
574
- result[column] = (1-(minutes-df.isna().groupby(df.index.date).sum())/minutes)*sampling_rate[column]
590
+ for column in data.columns:
591
+ if sampling_rates is not None and column in sampling_rates:
592
+ sampling_rate = sampling_rates[column]
593
+ elif column in config._default_sampling_rate:
594
+ sampling_rate = config._default_sampling_rate[column]
595
+ else:
596
+ sampling_rate = infer_sampling_rate(data[column])
575
597
 
576
- return result
598
+ if sampling_rate is None:
599
+ logger.warning(f'Cannot infer sampling rate for column {column}. Skipping NaN ratio calculation')
600
+ continue
601
+
602
+ sampling_rate_timedelta = Timedelta(f'{sampling_rate}min')
603
+ max_possible_count = Timedelta(period) / sampling_rate_timedelta
604
+ datapoints = data[column].resample(sampling_rate_timedelta).mean() # Need to aggregate somehow
605
+
606
+ nan_count = datapoints.rolling(period).apply(count_nas).fillna(max_possible_count) # If we didn't fillna, we would get nas as count when the whole period is missing.
607
+
608
+ nan_ratio = nan_count / max_possible_count
609
+
610
+ result[column + suffix] = nan_ratio
611
+
612
+ return DataFrame(result)
613
+
614
+ def get_implausible_values(self, column: str, plausible_interval:List[int]=None) -> Series:
615
+ '''Get a boolean series indicating which values are outside the plausible
616
+ interval for the physical magnitude, as defined by plausible_interval
617
+ or config._default_unplausible_values.
618
+
619
+ For example, we consider values of NOISE_A below 20dB or above 99dB to
620
+ be implausible.
621
+
622
+ Parameters
623
+ ----------
624
+ column: str
625
+ Column to apply the plausible values check to.
626
+ plausible_interval: List[int]
627
+ List with two elements defining the plausible interval [min, max].
628
+ If None, will try to get the plausible interval from
629
+ config._default_unplausible_values.'''
630
+
631
+ series = self.data[column]
577
632
 
578
- # Check plausibility per column, return dict. Doesn't take into account nans
579
- # TODO-DOCUMENT
580
- def get_plausible_ratio(self, **kwargs):
581
633
  if not self.loaded:
582
634
  logger.error('Need to load first (device.load())')
583
635
  return False
584
-
585
- if 'unplausible_values' not in kwargs:
586
- unplausible_values = config._default_unplausible_values
636
+ if plausible_interval is not None:
637
+ left, right = plausible_interval
638
+ elif column in config._default_unplausible_values:
639
+ left, right = config._default_unplausible_values[series.name] # I don't like this name
587
640
  else:
588
- unplausible_values = kwargs['unplausible_values']
641
+ copy = series.copy()
642
+ copy[:] = nan
589
643
 
590
- if 'sampling_rate' not in kwargs:
591
- sampling_rate = config._default_sampling_rate
592
- else:
593
- sampling_rate = kwargs['sampling_rate']
644
+ return copy
645
+
646
+ implausible = (series < left) | (series > right)
647
+ nullable = implausible.convert_dtypes()
648
+ nullable[series.isna()] = None # Propagate NaNs
649
+
650
+ return nullable
594
651
 
595
- return {column: self.data[column].between(left=unplausible_values[column][0], right=unplausible_values[column][1]).groupby(self.data[column].index.date).sum()/self.data.groupby(self.data.index.date).count()[column] for column in self.data.columns if column in unplausible_values}
596
652
 
597
- # Check plausibility per column, return dict. Doesn't take into account nans
598
- def get_outlier_ratio(self, **kwargs):
653
+ def get_implausible_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_implausible_ratio", plausible_intervals:Dict[str, List[int]]=None) -> DataFrame:
654
+ '''Scan the series for values outside the plausible interval for the
655
+ physical magnitude, as defined by plausible_interval or
656
+ config._default_unplausible_values. For example, we find values of
657
+ NOISE_A below 20dB or above 99dB to be implausible.
658
+
659
+ Parameters
660
+ ----------
661
+ period: str
662
+ Rolling window width.
663
+ subset: List[str]
664
+ Columns to apply the plausible ratio calculation to. If None, all columns
665
+ are used.
666
+ plausible_intervals: Dict[str, List[int]]
667
+ Dictionary with plausible intervals per column. Optional.
668
+
669
+ Returns
670
+ ----------
671
+ result: DataFrame
672
+ DataFrame with rolling calculation of plausible ratio.
673
+ '''
674
+
599
675
  if not self.loaded:
600
676
  logger.error('Need to load first (device.load())')
601
677
  return False
678
+
679
+ if subset is not None:
680
+ data = self.data[subset]
681
+ else:
682
+ data = self.data
602
683
  result = {}
603
- resample = '360h'
604
684
 
605
- for column in self.data.columns:
606
- Q1 = self.data[column].resample(resample).mean().quantile(0.25)
607
- Q3 = self.data[column].resample(resample).mean().quantile(0.75)
608
- IQR = Q3 - Q1
685
+ for column in data.columns:
686
+ if plausible_intervals is not None and column in plausible_intervals:
687
+ interval = plausible_intervals[column]
688
+ else:
689
+ interval = None
609
690
 
610
- mask = (self.data[column] < (Q1 - 1.5 * IQR)) | (self.data[column] > (Q3 + 1.5 * IQR))
611
- result[column] = mask.groupby(mask.index.date).mean()
691
+ implausible_values = self.get_implausible_values(column, plausible_interval=interval)
612
692
 
613
- return result
693
+ result[column + suffix] = implausible_values.rolling(period).mean() # Only consider actual observed data
694
+
695
+ return DataFrame(result)
696
+
697
+ def get_outlier_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_outlier_ratio", sigma=5, pct=0.05) -> DataFrame:
698
+ '''Get the percentage of outlier values based on the rate of increase. When sensors
699
+ report sudden jumps, these are likely to be erroneous values.
700
+
701
+ Parameters
702
+ ----------
703
+ period: str
704
+ "1h"
705
+ Rolling window width.
706
+ subset: List[str]
707
+ Columns to apply the outlier ratio calculation to. If None, all columns
708
+ are used.
709
+ sigma: int
710
+ 5
711
+ Number of standard deviations to consider a point an outlier. A higher
712
+ value would be more restrictive, detecting only worse malfunctions.
713
+ pct: float
714
+ 0.05
715
+ Percentage of top and bottom values to ignore when normalizing. We
716
+ assume the outliers will always be a minority of the data, so ignoring
717
+ a small percentage of extreme values should help get a better estimate.
718
+
719
+ Returns
720
+ ----------
721
+ result: DataFrame
722
+ DataFrame with rolling outlier ratio.
723
+ '''
614
724
 
615
- # Check plausibility per column, return dict. Doesn't take into account nans
616
- def get_outliers(self, **kwargs):
617
725
  if not self.loaded:
618
726
  logger.error('Need to load first (device.load())')
619
727
  return False
728
+
729
+ if subset is not None:
730
+ data = self.data[subset]
731
+ else:
732
+ data = self.data
733
+
620
734
  result = {}
621
- resample = '360h'
622
735
 
623
- for column in self.data.columns:
624
- Q1 = self.data[column].resample(resample).mean().quantile(0.25)
625
- Q3 = self.data[column].resample(resample).mean().quantile(0.75)
626
- IQR = Q3 - Q1
736
+ for column in data.columns:
737
+ outlier_values = self.get_outlier_values(column, sigma=sigma, pct=pct)
738
+
739
+ result[column + suffix] = outlier_values.rolling(period).mean()
740
+
741
+ return DataFrame(result)
742
+
743
+ def get_outlier_values(self, column: str, sigma=5, pct=0.05) -> Series:
744
+ '''
745
+ Get a boolean series indicating which values are outliers based on
746
+ the rate of increase. We calculate the first derivative implied by
747
+ each datapoint. Then we normalize the derivative series after removing
748
+ the top and bottom pct% of values to avoid outliers impacting the mean
749
+ too much. Finally, we consider outliers those points where the
750
+ normalized derivative is above `sigma` standard deviations.
751
+
752
+ Parameters
753
+ ----------
754
+ column: str
755
+ Column to apply the outlier values check to.
756
+ sigma: int
757
+ Number of standard deviations to consider a point an outlier.
758
+ pct: float
759
+ Percentage of top and bottom values to ignore when normalizing.
760
+
761
+ Returns
762
+ ----------
763
+ Series
764
+ Boolean series indicating outlier values.
765
+ '''
766
+
767
+ if not self.loaded:
768
+ logger.error('Need to load first (device.load())')
769
+ return False
770
+
771
+ series = self.data[column]
772
+
773
+ deltas = rolling_deltas(series)
774
+ normalized_deltas = normalize_central(deltas, pct=pct)
775
+ outliers = normalized_deltas.abs() > sigma
627
776
 
628
- mask = (self.data[column] < (Q1 - 1.5 * IQR)) | (self.data[column] > (Q3 + 1.5 * IQR))
629
- result[column] = mask
777
+ return outliers
778
+
779
+ def get_top_value_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_top_value_ratio", ignore_zeroes=True) -> DataFrame:
780
+ '''
781
+ Check the frequency of the mode (top value) per column, return
782
+ pd.DataFrame with the same index as self.data with the mode ratio
783
+ per rolling window.
784
+
785
+ Parameters
786
+ ----------
787
+ period: str
788
+ "1h"
789
+ Rolling window width.
790
+ subset: List[str]
791
+ Columns to apply the stuck ratio calculation to. If None, all columns
792
+ are used.
793
+ ignore_zeroes: boolean
794
+ True
795
+ Ignore zeroes when checking for stuck values. Passed through to mode_ratio()
796
+ Returns
797
+ ----------
798
+ result: DataFrame
799
+ DataFrame with rolling mode ratio.
800
+ '''
801
+ if not self.loaded:
802
+ logger.error('Need to load first (device.load())')
803
+ return False
804
+
805
+ if subset is not None:
806
+ data = self.data[subset]
807
+ else:
808
+ data = self.data
809
+
810
+ rolling = data.rolling(period)
811
+ result = rolling.apply(lambda w: mode_ratio(w, ignore_zeroes), raw=False)
812
+ result.columns = [col + suffix for col in result.columns]
630
813
 
631
814
  return result
632
815
 
633
- def export(self, path, forced_overwrite = False, file_format = 'csv'):
816
+
817
+ def export(self, path, forced_overwrite = False, file_format = 'csv', gzip=False):
634
818
  '''
635
819
  Exports Device.data to file
636
820
  Parameters
@@ -653,7 +837,7 @@ class Device(BaseModel):
653
837
  logger.error('Cannot export null data')
654
838
  return False
655
839
  if file_format == 'csv':
656
- return export_csv_file(path, str(self.paramsParsed.id), self.data, forced_overwrite = forced_overwrite)
840
+ return export_csv_file(path, str(self.paramsParsed.id), self.data, forced_overwrite = forced_overwrite, gzip=gzip)
657
841
  else:
658
842
  # TODO Make a list of supported formats
659
843
  return NotImplementedError (f'Not supported format. Formats: [csv]')
@@ -64,7 +64,7 @@ class CSVHandler:
64
64
 
65
65
  return self.data
66
66
 
67
- def export_csv_file(path, file_name, df, forced_overwrite=False):
67
+ def export_csv_file(path, file_name, df, forced_overwrite=False, gzip=False):
68
68
  '''
69
69
  Exports pandas dataframe to a csv file
70
70
  Parameters
@@ -87,10 +87,13 @@ def export_csv_file(path, file_name, df, forced_overwrite=False):
87
87
  if not exists(path):
88
88
  makedirs(path)
89
89
 
90
+ full_path = path + '/' + str(file_name) + '.csv'
91
+ full_path += '.gz' if gzip else ''
92
+
90
93
  # If file does not exist
91
- if not exists(path + '/' + str(file_name) + '.csv') or forced_overwrite:
92
- df.to_csv(path + '/' + str(file_name) + '.csv', sep=",")
93
- logger.info('File saved to: \n' + path + '/' + str(file_name) + '.csv')
94
+ if not exists(full_path) or forced_overwrite:
95
+ df.to_csv(full_path, sep=",")
96
+ logger.info('File saved to: \n' + full_path)
94
97
  else:
95
98
  logger.error("File Already exists - delete it first, I was not asked to overwrite anything!")
96
99
  return False
@@ -0,0 +1,61 @@
1
+ from __future__ import annotations
2
+
3
+ from scdata.tools.custom_logger import logger
4
+
5
+ from pandas import Series, Timedelta
6
+ import numpy as np
7
+
8
+ def infer_sampling_rate(series: Series) -> int | None:
9
+ '''Infer the sampling rate of the given timeseries, rounded to the
10
+ closest minute.
11
+ '''
12
+
13
+ time_differences = series.index.diff().value_counts()
14
+ most_common = time_differences.index[0]
15
+
16
+ minutes = most_common / Timedelta("1min")
17
+ integer_minutes = round(minutes)
18
+
19
+ if abs(integer_minutes - minutes) > 0.05:
20
+ logger.warning('Rounded a time difference with more than 5% error')
21
+ return None
22
+
23
+ return integer_minutes
24
+
25
+
26
+ def mode_ratio(series: Series, ignore_zeroes=True) -> int:
27
+ '''Count the percentage of times the most common value appears in the series,
28
+ ignoring zeroes and NaNs.'''
29
+
30
+ if ignore_zeroes:
31
+ # Replace zeroes with random so that they don't impact value count
32
+ series = series.where(series!=0.0, np.random.random(size=series.size))
33
+
34
+ mode_count = series.value_counts().iloc[0]
35
+
36
+ return mode_count / series.count()
37
+
38
+
39
+ def count_nas(series: Series) -> int:
40
+ '''Count the number of NaN values in the series.'''
41
+ return series.isna().sum()
42
+
43
+
44
+ def rolling_deltas(series: Series) -> Series:
45
+ '''Compute the first derivative of the series at each datapoint.'''
46
+
47
+ dys = series.rolling(window=2).apply(lambda ys: ys.iloc[1] - ys.iloc[0])
48
+ dxs = series.index.diff().total_seconds()
49
+
50
+ return dys / dxs
51
+
52
+
53
+ def normalize_central(series: Series, pct=0.05) -> Series:
54
+ '''Normalize the series by removing the mean and scaling to unit variance,
55
+ ignroring the top and bottom `pct` percent of values. This should be more
56
+ robust to outliers than standard normalization.'''
57
+
58
+ central = series[((series > series.quantile(pct)) | (series > series.quantile(1 - pct)))]
59
+ normalized = (series - central.mean()) / central.std()
60
+
61
+ return normalized
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scdata
3
- Version: 1.3.0
3
+ Version: 1.3.2
4
4
  Summary: Analysis of sensors and time series data
5
5
  Home-page: https://github.com/fablabbcn/smartcitizen-data
6
6
  Author: oscgonfer
@@ -72,6 +72,7 @@ scdata/tools/gets.py
72
72
  scdata/tools/lazy.py
73
73
  scdata/tools/location.py
74
74
  scdata/tools/report.py
75
+ scdata/tools/series.py
75
76
  scdata/tools/stats.py
76
77
  scdata/tools/units.py
77
78
  scdata/tools/url_check.py
@@ -19,7 +19,7 @@ REQUIREMENTS = [i.strip() for i in open("requirements.txt").readlines()]
19
19
 
20
20
  setup(
21
21
  name='scdata',
22
- version='1.3.0',
22
+ version='1.3.2',
23
23
  description='Analysis of sensors and time series data',
24
24
  author='oscgonfer',
25
25
  license='GNU-GPL3.0',
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes