scdata 1.2.6__tar.gz → 1.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. {scdata-1.2.6/scdata.egg-info → scdata-1.3.2}/PKG-INFO +3 -2
  2. {scdata-1.2.6 → scdata-1.3.2}/requirements.txt +3 -2
  3. {scdata-1.2.6 → scdata-1.3.2}/scdata/__init__.py +1 -1
  4. {scdata-1.2.6 → scdata-1.3.2}/scdata/_config/config.py +13 -7
  5. {scdata-1.2.6 → scdata-1.3.2}/scdata/device/device.py +274 -49
  6. {scdata-1.2.6 → scdata-1.3.2}/scdata/io/device_file.py +14 -9
  7. {scdata-1.2.6 → scdata-1.3.2}/scdata/models/models.py +5 -3
  8. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/date.py +7 -7
  9. scdata-1.3.2/scdata/tools/series.py +61 -0
  10. {scdata-1.2.6 → scdata-1.3.2/scdata.egg-info}/PKG-INFO +3 -2
  11. {scdata-1.2.6 → scdata-1.3.2}/scdata.egg-info/SOURCES.txt +1 -0
  12. {scdata-1.2.6 → scdata-1.3.2}/scdata.egg-info/requires.txt +2 -1
  13. {scdata-1.2.6 → scdata-1.3.2}/setup.py +1 -1
  14. {scdata-1.2.6 → scdata-1.3.2}/LICENSE +0 -0
  15. {scdata-1.2.6 → scdata-1.3.2}/MANIFEST.in +0 -0
  16. {scdata-1.2.6 → scdata-1.3.2}/README.md +0 -0
  17. {scdata-1.2.6 → scdata-1.3.2}/scdata/_config/__init__.py +0 -0
  18. {scdata-1.2.6 → scdata-1.3.2}/scdata/device/__init__.py +0 -0
  19. {scdata-1.2.6 → scdata-1.3.2}/scdata/device/plot/__init__.py +0 -0
  20. {scdata-1.2.6 → scdata-1.3.2}/scdata/device/process/__init__.py +0 -0
  21. {scdata-1.2.6 → scdata-1.3.2}/scdata/device/process/alphasense.py +0 -0
  22. {scdata-1.2.6 → scdata-1.3.2}/scdata/device/process/baseline.py +0 -0
  23. {scdata-1.2.6 → scdata-1.3.2}/scdata/device/process/error_codes.py +0 -0
  24. {scdata-1.2.6 → scdata-1.3.2}/scdata/device/process/formulae.py +0 -0
  25. {scdata-1.2.6 → scdata-1.3.2}/scdata/device/process/geoseries.py +0 -0
  26. {scdata-1.2.6 → scdata-1.3.2}/scdata/device/process/params.py +0 -0
  27. {scdata-1.2.6 → scdata-1.3.2}/scdata/device/process/regression.py +0 -0
  28. {scdata-1.2.6 → scdata-1.3.2}/scdata/device/process/timeseries.py +0 -0
  29. {scdata-1.2.6 → scdata-1.3.2}/scdata/io/__init__.py +0 -0
  30. {scdata-1.2.6 → scdata-1.3.2}/scdata/io/device_api.py +0 -0
  31. {scdata-1.2.6 → scdata-1.3.2}/scdata/io/model.py +0 -0
  32. {scdata-1.2.6 → scdata-1.3.2}/scdata/models/__init__.py +0 -0
  33. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/__init__.py +0 -0
  34. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/checks/__init__.py +0 -0
  35. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/checks/checks.py +0 -0
  36. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/dispersion/__init__.py +0 -0
  37. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/dispersion/dispersion.py +0 -0
  38. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/export/__init__.py +0 -0
  39. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/export/templates/sc_template.html +0 -0
  40. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/export/to_file.py +0 -0
  41. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/__init__.py +0 -0
  42. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/box_plot.py +0 -0
  43. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/heatmap_iplot.py +0 -0
  44. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/heatmap_plot.py +0 -0
  45. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/maps.py +0 -0
  46. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/plot_tools.py +0 -0
  47. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/scatter_dispersion_grid.py +0 -0
  48. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/scatter_iplot.py +0 -0
  49. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/scatter_plot.py +0 -0
  50. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/target_diagram.py +0 -0
  51. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/ts_dendrogram.py +0 -0
  52. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/ts_dispersion_grid.py +0 -0
  53. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/ts_dispersion_plot.py +0 -0
  54. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/ts_dispersion_uplot.py +0 -0
  55. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/ts_iplot.py +0 -0
  56. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/ts_plot.py +0 -0
  57. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/ts_scatter.py +0 -0
  58. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/plot/ts_uplot.py +0 -0
  59. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/test.py +0 -0
  60. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/tools/__init__.py +0 -0
  61. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/tools/combine.py +0 -0
  62. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/tools/history.py +0 -0
  63. {scdata-1.2.6 → scdata-1.3.2}/scdata/test/tools/prepare.py +0 -0
  64. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/__init__.py +0 -0
  65. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/cleaning.py +0 -0
  66. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/custom_logger.py +0 -0
  67. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/dictmerge.py +0 -0
  68. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/find.py +0 -0
  69. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/gets.py +0 -0
  70. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/interim/example.csv +0 -0
  71. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/interim/geodata.csv +0 -0
  72. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/lazy.py +0 -0
  73. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/location.py +0 -0
  74. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/report.py +0 -0
  75. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/stats.py +0 -0
  76. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/units.py +0 -0
  77. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/uploads/example_upload_1.json +0 -0
  78. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/uploads/example_zenodo_upload.yaml +0 -0
  79. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/uploads/report.pdf +0 -0
  80. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/url_check.py +0 -0
  81. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/zenodo.py +0 -0
  82. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/zenodo_templates/README.md +0 -0
  83. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/zenodo_templates/template_zenodo_dataset.json +0 -0
  84. {scdata-1.2.6 → scdata-1.3.2}/scdata/tools/zenodo_templates/template_zenodo_publication.json +0 -0
  85. {scdata-1.2.6 → scdata-1.3.2}/scdata.egg-info/dependency_links.txt +0 -0
  86. {scdata-1.2.6 → scdata-1.3.2}/scdata.egg-info/not-zip-safe +0 -0
  87. {scdata-1.2.6 → scdata-1.3.2}/scdata.egg-info/top_level.txt +0 -0
  88. {scdata-1.2.6 → scdata-1.3.2}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scdata
3
- Version: 1.2.6
3
+ Version: 1.3.2
4
4
  Summary: Analysis of sensors and time series data
5
5
  Home-page: https://github.com/fablabbcn/smartcitizen-data
6
6
  Author: oscgonfer
@@ -24,7 +24,6 @@ Requires-Dist: folium~=0.12.1
24
24
  Requires-Dist: geopy~=1.21.0
25
25
  Requires-Dist: Jinja2~=3.1.2
26
26
  Requires-Dist: matplotlib
27
- Requires-Dist: numpy~=1.25.2
28
27
  Requires-Dist: pandas~=2.2.2
29
28
  Requires-Dist: pydantic
30
29
  Requires-Dist: pytest
@@ -38,6 +37,8 @@ Requires-Dist: termcolor==1.1.0
38
37
  Requires-Dist: tqdm~=4.50.2
39
38
  Requires-Dist: timezonefinder~=6.1.9
40
39
  Requires-Dist: urllib3
40
+ Requires-Dist: boto3
41
+ Requires-Dist: awswrangler
41
42
  Dynamic: author
42
43
  Dynamic: classifier
43
44
  Dynamic: description
@@ -9,7 +9,6 @@ geopy~=1.21.0
9
9
  # TODO To be updated?
10
10
  Jinja2~=3.1.2
11
11
  matplotlib
12
- numpy~=1.25.2
13
12
  pandas~=2.2.2
14
13
  pydantic
15
14
  pytest
@@ -23,4 +22,6 @@ smartcitizen-connector
23
22
  termcolor==1.1.0
24
23
  tqdm~=4.50.2
25
24
  timezonefinder~=6.1.9
26
- urllib3
25
+ urllib3
26
+ boto3
27
+ awswrangler
@@ -3,4 +3,4 @@ from .device import Device
3
3
  from .test import Test
4
4
  from .models import Source, TestOptions, DeviceOptions, APIParams, FileParams, CSVParams
5
5
 
6
- __version__ = '1.2.6'
6
+ __version__ = '1.3.2'
@@ -42,9 +42,6 @@ class Config(object):
42
42
  _timeout = 3
43
43
  _max_http_retries = 2
44
44
 
45
- # Max concurrent requests
46
- _max_concurrent_requests = 30
47
-
48
45
  ### ---------------------------------------
49
46
  ### -----------------DATA------------------
50
47
  ### ---------------------------------------
@@ -121,7 +118,6 @@ class Config(object):
121
118
  'https://raw.githubusercontent.com/fablabbcn/smartcitizen-data/master/names/SCDevice.json'
122
119
  ]
123
120
 
124
-
125
121
  ### ---------------------------------------
126
122
  ### -------------METRICS DATA--------------
127
123
  ### ---------------------------------------
@@ -466,7 +462,12 @@ class Config(object):
466
462
  'TEMP': 1,
467
463
  'RSSI': 1,
468
464
  'NO2': 1,
469
- 'O3': 1
465
+ 'O3': 1,
466
+ 'AS_TEMP': 1,
467
+ 'AS_PH': 1,
468
+ 'AS_COND': 1,
469
+ 'AS_DO_SAT': 1,
470
+ 'AS_DO': 1
470
471
  }
471
472
 
472
473
  _default_unplausible_values = {
@@ -474,6 +475,7 @@ class Config(object):
474
475
  'SCD30_CO2': [300, 2000],
475
476
  'SCD30_HUM': [20, 99],
476
477
  'SCD30_TEMP': [-20, 50],
478
+ 'BATT': [0, 100],
477
479
  'ST LPS33 - Barometric Pressure': [50, 110],
478
480
  'PRESS': [50, 110],
479
481
  'PMS5003_PM_1': [0, 500],
@@ -499,7 +501,12 @@ class Config(object):
499
501
  'HUM': [20, 99],
500
502
  'TEMP': [-20, 50],
501
503
  'NO2': [0, 1000],
502
- 'O3': [0, 1000]
504
+ 'O3': [0, 1000],
505
+ 'AS_TEMP': [-20, 50],
506
+ 'AS_PH': [0, 14],
507
+ 'AS_COND': [0, 100000],
508
+ 'AS_DO_SAT': [0, 100],
509
+ 'AS_DO': [0, 15]
503
510
  }
504
511
 
505
512
  def __init__(self):
@@ -508,7 +515,6 @@ class Config(object):
508
515
  self.load()
509
516
  self.get_meta_data()
510
517
 
511
-
512
518
  def __getattr__(self, name):
513
519
  try:
514
520
  return self[name]
@@ -8,21 +8,35 @@ from scdata.tools.date import localise_date
8
8
  from scdata.tools.dictmerge import dict_fmerge
9
9
  from scdata.tools.units import get_units_convf
10
10
  from scdata.tools.find import find_by_field
11
+ from scdata.tools.series import count_nas, infer_sampling_rate, mode_ratio, normalize_central, rolling_deltas
11
12
  from scdata._config import config
12
13
  from scdata.io.device_api import *
13
14
  from scdata.models import Blueprint, Metric, Source, APIParams, CSVParams, DeviceOptions, Sensor
14
15
 
15
16
  from os.path import join, basename, exists
16
17
  from urllib.parse import urlparse
17
- from pandas import DataFrame, to_timedelta, Timedelta
18
+ from pandas import DataFrame, Series, to_timedelta, Timedelta
18
19
  from numpy import nan
19
20
  from collections.abc import Iterable
20
21
  from importlib import import_module
21
22
  from pydantic import TypeAdapter, BaseModel, ConfigDict
22
23
  from pydantic_core import ValidationError
23
- from typing import Optional, List
24
+ from typing import Optional, List, Dict
24
25
  from json import dumps
25
26
 
27
+ import os
28
+ from io import StringIO
29
+
30
+ try:
31
+ import awswrangler as wr
32
+ except ModuleNotFoundError:
33
+ boto_available = False
34
+ pass
35
+ else:
36
+ boto_available = True
37
+
38
+ if boto_available: import boto3
39
+
26
40
  from timezonefinder import TimezoneFinder
27
41
  tf = TimezoneFinder()
28
42
 
@@ -287,6 +301,7 @@ class Device(BaseModel):
287
301
  resample = self.options.resample
288
302
  limit = self.options.limit
289
303
  channels = self.options.channels
304
+ dateformat = self.options.dateformat
290
305
  cached_data = DataFrame()
291
306
 
292
307
  # Only case where cache makes sense
@@ -298,7 +313,7 @@ class Device(BaseModel):
298
313
  else:
299
314
  logger.warning(f'Cache file does not exist: {cache}')
300
315
  else:
301
- if cache.endswith('.csv'):
316
+ if cache.endswith('.csv') or cache.endswith('.csv.gz'):
302
317
  cached_data = read_csv_file(
303
318
  path = cache,
304
319
  timezone = timezone,
@@ -334,7 +349,8 @@ class Device(BaseModel):
334
349
  max_date = max_date,
335
350
  frequency = frequency,
336
351
  clean_na = clean_na,
337
- resample = resample)
352
+ resample = resample,
353
+ dateformat = dateformat)
338
354
 
339
355
  # In principle this links both dataframes as they are unmutable
340
356
  self.data = self.handler.data
@@ -539,83 +555,266 @@ class Device(BaseModel):
539
555
 
540
556
  return self.postprocessing_updated
541
557
 
542
- # Check nans per column, return dict
543
- # TODO-DOCUMENT
544
- def get_nan_ratio(self, **kwargs):
558
+ def get_nan_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_nan_ratio", sampling_rates:Dict[str, int]=None) -> DataFrame:
559
+ '''
560
+ Check NaN ratio per column, return pd.DataFrame with the same index as
561
+ self.data with the NaN ratio per rolling window.
562
+ Parameters
563
+ ----------
564
+ period: str
565
+ "1h"
566
+ Rolling window width.
567
+ subset: List[str]
568
+ Columns to apply the stuck ratio calculation to. If None, all columns
569
+ are used.
570
+ sampling_rates: Dict[str, int]
571
+ Dictionary with sampling rates per column in minutes. If None, will get
572
+ sampling rates from config._default_sampling_rate or try to infer them.
573
+
574
+ Returns
575
+ ----------
576
+ result: DataFrame
577
+ DataFrame with rolling nan ratio.
578
+ '''
545
579
  if not self.loaded:
546
580
  logger.error('Need to load first (device.load())')
547
581
  return False
548
582
 
549
- if 'sampling_rate' not in kwargs:
550
- sampling_rate = config._default_sampling_rate
583
+ if subset is not None:
584
+ data = self.data[subset]
551
585
  else:
552
- sampling_rate = kwargs['sampling_rate']
586
+ data = self.data
587
+
553
588
  result = {}
554
589
 
555
- for column in self.data.columns:
556
- if column not in sampling_rate: continue
557
- df = self.data[column].resample(f'{sampling_rate[column]}Min').mean()
558
- minutes = df.groupby(df.index.date).mean().index.to_series().diff()/Timedelta('60s')
559
- result[column] = (1-(minutes-df.isna().groupby(df.index.date).sum())/minutes)*sampling_rate[column]
590
+ for column in data.columns:
591
+ if sampling_rates is not None and column in sampling_rates:
592
+ sampling_rate = sampling_rates[column]
593
+ elif column in config._default_sampling_rate:
594
+ sampling_rate = config._default_sampling_rate[column]
595
+ else:
596
+ sampling_rate = infer_sampling_rate(data[column])
560
597
 
561
- return result
598
+ if sampling_rate is None:
599
+ logger.warning(f'Cannot infer sampling rate for column {column}. Skipping NaN ratio calculation')
600
+ continue
601
+
602
+ sampling_rate_timedelta = Timedelta(f'{sampling_rate}min')
603
+ max_possible_count = Timedelta(period) / sampling_rate_timedelta
604
+ datapoints = data[column].resample(sampling_rate_timedelta).mean() # Need to aggregate somehow
605
+
606
+ nan_count = datapoints.rolling(period).apply(count_nas).fillna(max_possible_count) # If we didn't fillna, we would get nas as count when the whole period is missing.
607
+
608
+ nan_ratio = nan_count / max_possible_count
609
+
610
+ result[column + suffix] = nan_ratio
611
+
612
+ return DataFrame(result)
613
+
614
+ def get_implausible_values(self, column: str, plausible_interval:List[int]=None) -> Series:
615
+ '''Get a boolean series indicating which values are outside the plausible
616
+ interval for the physical magnitude, as defined by plausible_interval
617
+ or config._default_unplausible_values.
618
+
619
+ For example, we consider values of NOISE_A below 20dB or above 99dB to
620
+ be implausible.
621
+
622
+ Parameters
623
+ ----------
624
+ column: str
625
+ Column to apply the plausible values check to.
626
+ plausible_interval: List[int]
627
+ List with two elements defining the plausible interval [min, max].
628
+ If None, will try to get the plausible interval from
629
+ config._default_unplausible_values.'''
630
+
631
+ series = self.data[column]
562
632
 
563
- # Check plausibility per column, return dict. Doesn't take into account nans
564
- # TODO-DOCUMENT
565
- def get_plausible_ratio(self, **kwargs):
566
633
  if not self.loaded:
567
634
  logger.error('Need to load first (device.load())')
568
635
  return False
569
-
570
- if 'unplausible_values' not in kwargs:
571
- unplausible_values = config._default_unplausible_values
636
+ if plausible_interval is not None:
637
+ left, right = plausible_interval
638
+ elif column in config._default_unplausible_values:
639
+ left, right = config._default_unplausible_values[series.name] # I don't like this name
572
640
  else:
573
- unplausible_values = kwargs['unplausible_values']
641
+ copy = series.copy()
642
+ copy[:] = nan
574
643
 
575
- if 'sampling_rate' not in kwargs:
576
- sampling_rate = config._default_sampling_rate
577
- else:
578
- sampling_rate = kwargs['sampling_rate']
644
+ return copy
645
+
646
+ implausible = (series < left) | (series > right)
647
+ nullable = implausible.convert_dtypes()
648
+ nullable[series.isna()] = None # Propagate NaNs
579
649
 
580
- return {column: self.data[column].between(left=unplausible_values[column][0], right=unplausible_values[column][1]).groupby(self.data[column].index.date).sum()/self.data.groupby(self.data.index.date).count()[column] for column in self.data.columns if column in unplausible_values}
650
+ return nullable
651
+
652
+
653
+ def get_implausible_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_implausible_ratio", plausible_intervals:Dict[str, List[int]]=None) -> DataFrame:
654
+ '''Scan the series for values outside the plausible interval for the
655
+ physical magnitude, as defined by plausible_interval or
656
+ config._default_unplausible_values. For example, we find values of
657
+ NOISE_A below 20dB or above 99dB to be implausible.
658
+
659
+ Parameters
660
+ ----------
661
+ period: str
662
+ Rolling window width.
663
+ subset: List[str]
664
+ Columns to apply the plausible ratio calculation to. If None, all columns
665
+ are used.
666
+ plausible_intervals: Dict[str, List[int]]
667
+ Dictionary with plausible intervals per column. Optional.
668
+
669
+ Returns
670
+ ----------
671
+ result: DataFrame
672
+ DataFrame with rolling calculation of plausible ratio.
673
+ '''
581
674
 
582
- # Check plausibility per column, return dict. Doesn't take into account nans
583
- def get_outlier_ratio(self, **kwargs):
584
675
  if not self.loaded:
585
676
  logger.error('Need to load first (device.load())')
586
677
  return False
678
+
679
+ if subset is not None:
680
+ data = self.data[subset]
681
+ else:
682
+ data = self.data
587
683
  result = {}
588
- resample = '360h'
589
684
 
590
- for column in self.data.columns:
591
- Q1 = self.data[column].resample(resample).mean().quantile(0.25)
592
- Q3 = self.data[column].resample(resample).mean().quantile(0.75)
593
- IQR = Q3 - Q1
685
+ for column in data.columns:
686
+ if plausible_intervals is not None and column in plausible_intervals:
687
+ interval = plausible_intervals[column]
688
+ else:
689
+ interval = None
594
690
 
595
- mask = (self.data[column] < (Q1 - 1.5 * IQR)) | (self.data[column] > (Q3 + 1.5 * IQR))
596
- result[column] = mask.groupby(mask.index.date).mean()
691
+ implausible_values = self.get_implausible_values(column, plausible_interval=interval)
597
692
 
598
- return result
693
+ result[column + suffix] = implausible_values.rolling(period).mean() # Only consider actual observed data
694
+
695
+ return DataFrame(result)
696
+
697
+ def get_outlier_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_outlier_ratio", sigma=5, pct=0.05) -> DataFrame:
698
+ '''Get the percentage of outlier values based on the rate of increase. When sensors
699
+ report sudden jumps, these are likely to be erroneous values.
700
+
701
+ Parameters
702
+ ----------
703
+ period: str
704
+ "1h"
705
+ Rolling window width.
706
+ subset: List[str]
707
+ Columns to apply the outlier ratio calculation to. If None, all columns
708
+ are used.
709
+ sigma: int
710
+ 5
711
+ Number of standard deviations to consider a point an outlier. A higher
712
+ value would be more restrictive, detecting only worse malfunctions.
713
+ pct: float
714
+ 0.05
715
+ Percentage of top and bottom values to ignore when normalizing. We
716
+ assume the outliers will always be a minority of the data, so ignoring
717
+ a small percentage of extreme values should help get a better estimate.
718
+
719
+ Returns
720
+ ----------
721
+ result: DataFrame
722
+ DataFrame with rolling outlier ratio.
723
+ '''
599
724
 
600
- # Check plausibility per column, return dict. Doesn't take into account nans
601
- def get_outliers(self, **kwargs):
602
725
  if not self.loaded:
603
726
  logger.error('Need to load first (device.load())')
604
727
  return False
728
+
729
+ if subset is not None:
730
+ data = self.data[subset]
731
+ else:
732
+ data = self.data
733
+
605
734
  result = {}
606
- resample = '360h'
607
735
 
608
- for column in self.data.columns:
609
- Q1 = self.data[column].resample(resample).mean().quantile(0.25)
610
- Q3 = self.data[column].resample(resample).mean().quantile(0.75)
611
- IQR = Q3 - Q1
736
+ for column in data.columns:
737
+ outlier_values = self.get_outlier_values(column, sigma=sigma, pct=pct)
738
+
739
+ result[column + suffix] = outlier_values.rolling(period).mean()
740
+
741
+ return DataFrame(result)
742
+
743
+ def get_outlier_values(self, column: str, sigma=5, pct=0.05) -> Series:
744
+ '''
745
+ Get a boolean series indicating which values are outliers based on
746
+ the rate of increase. We calculate the first derivative implied by
747
+ each datapoint. Then we normalize the derivative series after removing
748
+ the top and bottom pct% of values to avoid outliers impacting the mean
749
+ too much. Finally, we consider outliers those points where the
750
+ normalized derivative is above `sigma` standard deviations.
612
751
 
613
- mask = (self.data[column] < (Q1 - 1.5 * IQR)) | (self.data[column] > (Q3 + 1.5 * IQR))
614
- result[column] = mask
752
+ Parameters
753
+ ----------
754
+ column: str
755
+ Column to apply the outlier values check to.
756
+ sigma: int
757
+ Number of standard deviations to consider a point an outlier.
758
+ pct: float
759
+ Percentage of top and bottom values to ignore when normalizing.
760
+
761
+ Returns
762
+ ----------
763
+ Series
764
+ Boolean series indicating outlier values.
765
+ '''
766
+
767
+ if not self.loaded:
768
+ logger.error('Need to load first (device.load())')
769
+ return False
770
+
771
+ series = self.data[column]
772
+
773
+ deltas = rolling_deltas(series)
774
+ normalized_deltas = normalize_central(deltas, pct=pct)
775
+ outliers = normalized_deltas.abs() > sigma
776
+
777
+ return outliers
778
+
779
+ def get_top_value_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_top_value_ratio", ignore_zeroes=True) -> DataFrame:
780
+ '''
781
+ Check the frequency of the mode (top value) per column, return
782
+ pd.DataFrame with the same index as self.data with the mode ratio
783
+ per rolling window.
784
+
785
+ Parameters
786
+ ----------
787
+ period: str
788
+ "1h"
789
+ Rolling window width.
790
+ subset: List[str]
791
+ Columns to apply the stuck ratio calculation to. If None, all columns
792
+ are used.
793
+ ignore_zeroes: boolean
794
+ True
795
+ Ignore zeroes when checking for stuck values. Passed through to mode_ratio()
796
+ Returns
797
+ ----------
798
+ result: DataFrame
799
+ DataFrame with rolling mode ratio.
800
+ '''
801
+ if not self.loaded:
802
+ logger.error('Need to load first (device.load())')
803
+ return False
804
+
805
+ if subset is not None:
806
+ data = self.data[subset]
807
+ else:
808
+ data = self.data
809
+
810
+ rolling = data.rolling(period)
811
+ result = rolling.apply(lambda w: mode_ratio(w, ignore_zeroes), raw=False)
812
+ result.columns = [col + suffix for col in result.columns]
615
813
 
616
814
  return result
617
815
 
618
- def export(self, path, forced_overwrite = False, file_format = 'csv'):
816
+
817
+ def export(self, path, forced_overwrite = False, file_format = 'csv', gzip=False):
619
818
  '''
620
819
  Exports Device.data to file
621
820
  Parameters
@@ -638,7 +837,7 @@ class Device(BaseModel):
638
837
  logger.error('Cannot export null data')
639
838
  return False
640
839
  if file_format == 'csv':
641
- return export_csv_file(path, str(self.paramsParsed.id), self.data, forced_overwrite = forced_overwrite)
840
+ return export_csv_file(path, str(self.paramsParsed.id), self.data, forced_overwrite = forced_overwrite, gzip=gzip)
642
841
  else:
643
842
  # TODO Make a list of supported formats
644
843
  return NotImplementedError (f'Not supported format. Formats: [csv]')
@@ -706,3 +905,29 @@ class Device(BaseModel):
706
905
 
707
906
  if post_ok: logger.info(f"Postprocessing posted for device {self.paramsParsed.id}")
708
907
  return post_ok
908
+
909
+ def backup(self, format='parquet', mode='append'):
910
+ if self.data.empty:
911
+ logger.error("Device data empty")
912
+ return False
913
+
914
+ if format == 'parquet':
915
+ if boto_available:
916
+ self.data['TIME']=self.data.index
917
+ target_path = f"s3://{os.environ['S3_DATA_BUCKET']}/devices/{self.id}/data/"
918
+ response = wr.s3.to_parquet(df=self.data, path=target_path, dataset=True, mode=mode)
919
+
920
+ return response
921
+
922
+ def backup_load(self, format='parquet'):
923
+ if format == 'parquet':
924
+ if boto_available:
925
+ session = boto3.Session(aws_access_key_id=os.environ['AWS_ACCESS_KEY_ID'],
926
+ aws_secret_access_key=os.environ['AWS_SECRET_ACCESS_KEY'],
927
+ region_name=os.environ['AWS_REGION'])
928
+ s3_url = f"s3://{os.environ['S3_DATA_BUCKET']}/devices/{self.id}/data/"
929
+ self.data = wr.s3.read_parquet(s3_url, boto3_session=session, dataset=True)
930
+ self.data.set_index('TIME', inplace=True)
931
+ self.data.sort_index(inplace=True)
932
+
933
+ return s3_url
@@ -57,12 +57,14 @@ class CSVHandler:
57
57
  skiprows=self.params.header_skip,
58
58
  sep=self.params.separator,
59
59
  tzaware=self.params.tzaware,
60
- resample=kwargs['resample']
60
+ resample=kwargs['resample'],
61
+ dateformat=kwargs['dateformat']
62
+ # TODO Pandas read_csv kwargs???
61
63
  )
62
64
 
63
65
  return self.data
64
66
 
65
- def export_csv_file(path, file_name, df, forced_overwrite=False):
67
+ def export_csv_file(path, file_name, df, forced_overwrite=False, gzip=False):
66
68
  '''
67
69
  Exports pandas dataframe to a csv file
68
70
  Parameters
@@ -85,16 +87,19 @@ def export_csv_file(path, file_name, df, forced_overwrite=False):
85
87
  if not exists(path):
86
88
  makedirs(path)
87
89
 
90
+ full_path = path + '/' + str(file_name) + '.csv'
91
+ full_path += '.gz' if gzip else ''
92
+
88
93
  # If file does not exist
89
- if not exists(path + '/' + str(file_name) + '.csv') or forced_overwrite:
90
- df.to_csv(path + '/' + str(file_name) + '.csv', sep=",")
91
- logger.info('File saved to: \n' + path + '/' + str(file_name) + '.csv')
94
+ if not exists(full_path) or forced_overwrite:
95
+ df.to_csv(full_path, sep=",")
96
+ logger.info('File saved to: \n' + full_path)
92
97
  else:
93
98
  logger.error("File Already exists - delete it first, I was not asked to overwrite anything!")
94
99
  return False
95
100
  return True
96
101
 
97
- def read_csv_file(path, timezone, frequency=None, clean_na=None, index_name='', skiprows=None, sep=',', encoding='utf-8', tzaware=True, resample=True):
102
+ def read_csv_file(path, timezone, frequency=None, clean_na=None, index_name='', skiprows=None, sep=',', encoding='utf-8', tzaware=True, resample=True, dateformat=None):
98
103
  """
99
104
  Reads a csv file and adds cleaning, localisation and resampling and puts it into a pandas dataframe
100
105
  Parameters
@@ -127,9 +132,9 @@ def read_csv_file(path, timezone, frequency=None, clean_na=None, index_name='',
127
132
  """
128
133
 
129
134
  # Read pandas dataframe
130
-
131
135
  df = read_csv(path, skiprows=skiprows, sep=sep,
132
- encoding=encoding, encoding_errors='ignore')
136
+ encoding=encoding, encoding_errors='ignore',
137
+ date_format=dateformat)
133
138
 
134
139
  flag_found = False
135
140
  if type(index_name) == str:
@@ -156,7 +161,7 @@ def read_csv_file(path, timezone, frequency=None, clean_na=None, index_name='',
156
161
  return None
157
162
 
158
163
  # Set index
159
- df.index = localise_date(df.index, timezone, tzaware=tzaware)
164
+ df.index = localise_date(df.index, timezone, tzaware=tzaware, dateformat=dateformat)
160
165
  # Remove duplicates
161
166
  df = df[~df.index.duplicated(keep='first')]
162
167
 
@@ -1,5 +1,5 @@
1
1
  from pydantic import BaseModel
2
- from typing import Optional, List
2
+ from typing import Optional, List, Any
3
3
  from datetime import datetime
4
4
 
5
5
  class TestOptions(BaseModel):
@@ -28,10 +28,10 @@ class Source(BaseModel):
28
28
  handler: str = 'SCDevice'
29
29
 
30
30
  class APIParams(BaseModel):
31
- id: int
31
+ id: Any
32
32
 
33
33
  class FileParams(BaseModel):
34
- id: int # To be compatible with API id
34
+ id: Any # To be compatible with API id
35
35
  path: str
36
36
 
37
37
  class CSVParams(FileParams):
@@ -40,6 +40,7 @@ class CSVParams(FileParams):
40
40
  separator: Optional[str] = ','
41
41
  tzaware: Optional[bool] = True
42
42
  timezone: Optional[str] = "UTC"
43
+ date_format: Optional[str] = None
43
44
 
44
45
  class DeviceOptions(BaseModel):
45
46
  clean_na: Optional[bool] = None
@@ -51,6 +52,7 @@ class DeviceOptions(BaseModel):
51
52
  channels: Optional[List[str]] = []
52
53
  convert_units: Optional[bool] = True
53
54
  convert_names: Optional[bool] = True
55
+ dateformat: Optional[str] = None
54
56
 
55
57
  class Blueprint(BaseModel):
56
58
  meta: dict = dict()
@@ -1,6 +1,6 @@
1
1
  from pandas import to_datetime
2
2
 
3
- def localise_date(date, timezone, tzaware=True):
3
+ def localise_date(date, timezone, tzaware=True, dateformat = None):
4
4
  """
5
5
  Localises a date if it's tzinfo is None, otherwise converts it to it.
6
6
  If the timestamp is tz-aware, converts it as well
@@ -13,16 +13,16 @@ def localise_date(date, timezone, tzaware=True):
13
13
  Returns
14
14
  -------
15
15
  The date converted to 'UTC' and localised based on the timezone
16
- """
16
+ """
17
17
  if date is not None:
18
18
  # Per default, we consider that timestamps are tz-aware or UTC.
19
19
  # If not, preprocessing should be done to get there
20
- result_date = to_datetime(date, utc = tzaware)
21
- if result_date.tzinfo is not None:
20
+ result_date = to_datetime(date, utc = tzaware, format=dateformat)
21
+ if result_date.tzinfo is not None:
22
22
  result_date = result_date.tz_convert(timezone)
23
23
  else:
24
24
  result_date = result_date.tz_localize(timezone)
25
- else:
25
+ else:
26
26
  result_date = None
27
27
 
28
28
  return result_date
@@ -37,10 +37,10 @@ def find_dates(dataframe):
37
37
  Returns
38
38
  -------
39
39
  Rounded down first day, rounded up last day and number of days between them
40
- """
40
+ """
41
41
 
42
42
  range_days = (dataframe.index.max()-dataframe.index.min()).days
43
43
  min_date_df = dataframe.index.min().floor('D')
44
44
  max_date_df = dataframe.index.max().ceil('D')
45
-
45
+
46
46
  return min_date_df, max_date_df, range_days
@@ -0,0 +1,61 @@
1
+ from __future__ import annotations
2
+
3
+ from scdata.tools.custom_logger import logger
4
+
5
+ from pandas import Series, Timedelta
6
+ import numpy as np
7
+
8
+ def infer_sampling_rate(series: Series) -> int | None:
9
+ '''Infer the sampling rate of the given timeseries, rounded to the
10
+ closest minute.
11
+ '''
12
+
13
+ time_differences = series.index.diff().value_counts()
14
+ most_common = time_differences.index[0]
15
+
16
+ minutes = most_common / Timedelta("1min")
17
+ integer_minutes = round(minutes)
18
+
19
+ if abs(integer_minutes - minutes) > 0.05:
20
+ logger.warning('Rounded a time difference with more than 5% error')
21
+ return None
22
+
23
+ return integer_minutes
24
+
25
+
26
+ def mode_ratio(series: Series, ignore_zeroes=True) -> int:
27
+ '''Count the percentage of times the most common value appears in the series,
28
+ ignoring zeroes and NaNs.'''
29
+
30
+ if ignore_zeroes:
31
+ # Replace zeroes with random so that they don't impact value count
32
+ series = series.where(series!=0.0, np.random.random(size=series.size))
33
+
34
+ mode_count = series.value_counts().iloc[0]
35
+
36
+ return mode_count / series.count()
37
+
38
+
39
+ def count_nas(series: Series) -> int:
40
+ '''Count the number of NaN values in the series.'''
41
+ return series.isna().sum()
42
+
43
+
44
+ def rolling_deltas(series: Series) -> Series:
45
+ '''Compute the first derivative of the series at each datapoint.'''
46
+
47
+ dys = series.rolling(window=2).apply(lambda ys: ys.iloc[1] - ys.iloc[0])
48
+ dxs = series.index.diff().total_seconds()
49
+
50
+ return dys / dxs
51
+
52
+
53
+ def normalize_central(series: Series, pct=0.05) -> Series:
54
+ '''Normalize the series by removing the mean and scaling to unit variance,
55
+ ignroring the top and bottom `pct` percent of values. This should be more
56
+ robust to outliers than standard normalization.'''
57
+
58
+ central = series[((series > series.quantile(pct)) | (series > series.quantile(1 - pct)))]
59
+ normalized = (series - central.mean()) / central.std()
60
+
61
+ return normalized
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scdata
3
- Version: 1.2.6
3
+ Version: 1.3.2
4
4
  Summary: Analysis of sensors and time series data
5
5
  Home-page: https://github.com/fablabbcn/smartcitizen-data
6
6
  Author: oscgonfer
@@ -24,7 +24,6 @@ Requires-Dist: folium~=0.12.1
24
24
  Requires-Dist: geopy~=1.21.0
25
25
  Requires-Dist: Jinja2~=3.1.2
26
26
  Requires-Dist: matplotlib
27
- Requires-Dist: numpy~=1.25.2
28
27
  Requires-Dist: pandas~=2.2.2
29
28
  Requires-Dist: pydantic
30
29
  Requires-Dist: pytest
@@ -38,6 +37,8 @@ Requires-Dist: termcolor==1.1.0
38
37
  Requires-Dist: tqdm~=4.50.2
39
38
  Requires-Dist: timezonefinder~=6.1.9
40
39
  Requires-Dist: urllib3
40
+ Requires-Dist: boto3
41
+ Requires-Dist: awswrangler
41
42
  Dynamic: author
42
43
  Dynamic: classifier
43
44
  Dynamic: description
@@ -72,6 +72,7 @@ scdata/tools/gets.py
72
72
  scdata/tools/lazy.py
73
73
  scdata/tools/location.py
74
74
  scdata/tools/report.py
75
+ scdata/tools/series.py
75
76
  scdata/tools/stats.py
76
77
  scdata/tools/units.py
77
78
  scdata/tools/url_check.py
@@ -4,7 +4,6 @@ folium~=0.12.1
4
4
  geopy~=1.21.0
5
5
  Jinja2~=3.1.2
6
6
  matplotlib
7
- numpy~=1.25.2
8
7
  pandas~=2.2.2
9
8
  pydantic
10
9
  pytest
@@ -18,3 +17,5 @@ termcolor==1.1.0
18
17
  tqdm~=4.50.2
19
18
  timezonefinder~=6.1.9
20
19
  urllib3
20
+ boto3
21
+ awswrangler
@@ -19,7 +19,7 @@ REQUIREMENTS = [i.strip() for i in open("requirements.txt").readlines()]
19
19
 
20
20
  setup(
21
21
  name='scdata',
22
- version='1.2.6',
22
+ version='1.3.2',
23
23
  description='Analysis of sensors and time series data',
24
24
  author='oscgonfer',
25
25
  license='GNU-GPL3.0',
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes