scdata 1.3.2__tar.gz → 1.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. {scdata-1.3.2/scdata.egg-info → scdata-1.4.0}/PKG-INFO +16 -8
  2. {scdata-1.3.2 → scdata-1.4.0}/requirements.txt +1 -10
  3. {scdata-1.3.2 → scdata-1.4.0}/scdata/__init__.py +1 -1
  4. {scdata-1.3.2 → scdata-1.4.0}/scdata/_config/config.py +2 -6
  5. {scdata-1.3.2 → scdata-1.4.0}/scdata/device/device.py +188 -57
  6. {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/__init__.py +2 -3
  7. {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/alphasense.py +144 -3
  8. {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/params.py +3 -2
  9. {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/timeseries.py +67 -2
  10. {scdata-1.3.2 → scdata-1.4.0}/scdata/io/device_api.py +0 -1
  11. {scdata-1.3.2 → scdata-1.4.0}/scdata/io/device_file.py +2 -0
  12. {scdata-1.3.2 → scdata-1.4.0}/scdata/models/models.py +1 -0
  13. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/__init__.py +15 -8
  14. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/box_plot.py +12 -6
  15. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/heatmap_iplot.py +14 -5
  16. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/heatmap_plot.py +13 -6
  17. scdata-1.3.2/scdata/test/plot/ts_uplot.py → scdata-1.4.0/scdata/plot/heatmap_uplot.py +16 -11
  18. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/maps.py +13 -11
  19. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/scatter_dispersion_grid.py +6 -4
  20. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/scatter_iplot.py +8 -5
  21. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/scatter_plot.py +15 -8
  22. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/target_diagram.py +19 -16
  23. scdata-1.3.2/scdata/test/plot/plot_tools.py → scdata-1.4.0/scdata/plot/tools.py +174 -23
  24. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_dendrogram.py +7 -6
  25. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_dispersion_grid.py +6 -4
  26. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_dispersion_plot.py +8 -7
  27. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_dispersion_uplot.py +10 -9
  28. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_iplot.py +15 -8
  29. scdata-1.4.0/scdata/plot/ts_panel.py +301 -0
  30. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_plot.py +16 -11
  31. {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_scatter.py +16 -11
  32. scdata-1.4.0/scdata/plot/ts_uplot.py +326 -0
  33. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/checks/checks.py +4 -6
  34. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/test.py +77 -25
  35. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/series.py +3 -2
  36. scdata-1.4.0/scdata/tools/tree.py +41 -0
  37. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/units.py +2 -0
  38. {scdata-1.3.2 → scdata-1.4.0/scdata.egg-info}/PKG-INFO +16 -8
  39. {scdata-1.3.2 → scdata-1.4.0}/scdata.egg-info/SOURCES.txt +21 -19
  40. {scdata-1.3.2 → scdata-1.4.0}/scdata.egg-info/requires.txt +15 -6
  41. {scdata-1.3.2 → scdata-1.4.0}/setup.py +18 -1
  42. scdata-1.3.2/scdata/device/process/baseline.py +0 -295
  43. {scdata-1.3.2 → scdata-1.4.0}/LICENSE +0 -0
  44. {scdata-1.3.2 → scdata-1.4.0}/MANIFEST.in +0 -0
  45. {scdata-1.3.2 → scdata-1.4.0}/README.md +0 -0
  46. {scdata-1.3.2 → scdata-1.4.0}/scdata/_config/__init__.py +0 -0
  47. {scdata-1.3.2 → scdata-1.4.0}/scdata/device/__init__.py +0 -0
  48. {scdata-1.3.2 → scdata-1.4.0}/scdata/device/plot/__init__.py +0 -0
  49. {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/error_codes.py +0 -0
  50. {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/formulae.py +0 -0
  51. {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/geoseries.py +0 -0
  52. {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/regression.py +0 -0
  53. {scdata-1.3.2 → scdata-1.4.0}/scdata/io/__init__.py +0 -0
  54. {scdata-1.3.2 → scdata-1.4.0}/scdata/io/model.py +0 -0
  55. {scdata-1.3.2 → scdata-1.4.0}/scdata/models/__init__.py +0 -0
  56. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/__init__.py +0 -0
  57. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/checks/__init__.py +0 -0
  58. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/dispersion/__init__.py +0 -0
  59. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/dispersion/dispersion.py +0 -0
  60. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/export/__init__.py +0 -0
  61. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/export/templates/sc_template.html +0 -0
  62. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/export/to_file.py +0 -0
  63. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/tools/__init__.py +0 -0
  64. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/tools/combine.py +0 -0
  65. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/tools/history.py +0 -0
  66. {scdata-1.3.2 → scdata-1.4.0}/scdata/test/tools/prepare.py +0 -0
  67. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/__init__.py +0 -0
  68. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/cleaning.py +0 -0
  69. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/custom_logger.py +0 -0
  70. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/date.py +0 -0
  71. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/dictmerge.py +0 -0
  72. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/find.py +0 -0
  73. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/gets.py +0 -0
  74. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/interim/example.csv +0 -0
  75. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/interim/geodata.csv +0 -0
  76. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/lazy.py +0 -0
  77. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/location.py +0 -0
  78. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/report.py +0 -0
  79. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/stats.py +0 -0
  80. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/uploads/example_upload_1.json +0 -0
  81. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/uploads/example_zenodo_upload.yaml +0 -0
  82. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/uploads/report.pdf +0 -0
  83. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/url_check.py +0 -0
  84. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/zenodo.py +0 -0
  85. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/zenodo_templates/README.md +0 -0
  86. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/zenodo_templates/template_zenodo_dataset.json +0 -0
  87. {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/zenodo_templates/template_zenodo_publication.json +0 -0
  88. {scdata-1.3.2 → scdata-1.4.0}/scdata.egg-info/dependency_links.txt +0 -0
  89. {scdata-1.3.2 → scdata-1.4.0}/scdata.egg-info/not-zip-safe +0 -0
  90. {scdata-1.3.2 → scdata-1.4.0}/scdata.egg-info/top_level.txt +0 -0
  91. {scdata-1.3.2 → scdata-1.4.0}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scdata
3
- Version: 1.3.2
3
+ Version: 1.4.0
4
4
  Summary: Analysis of sensors and time series data
5
5
  Home-page: https://github.com/fablabbcn/smartcitizen-data
6
6
  Author: oscgonfer
@@ -18,15 +18,11 @@ Classifier: Programming Language :: Python :: 3
18
18
  Requires-Python: >=3.9
19
19
  Description-Content-Type: text/markdown
20
20
  License-File: LICENSE
21
- Requires-Dist: branca~=0.4.0
22
- Requires-Dist: Flask~=2.2.2
23
- Requires-Dist: folium~=0.12.1
24
21
  Requires-Dist: geopy~=1.21.0
25
22
  Requires-Dist: Jinja2~=3.1.2
26
23
  Requires-Dist: matplotlib
27
24
  Requires-Dist: pandas~=2.2.2
28
25
  Requires-Dist: pydantic
29
- Requires-Dist: pytest
30
26
  Requires-Dist: PyYAML~=6.0.1
31
27
  Requires-Dist: requests
32
28
  Requires-Dist: scipy
@@ -34,11 +30,22 @@ Requires-Dist: scikit-learn
34
30
  Requires-Dist: seaborn
35
31
  Requires-Dist: smartcitizen-connector
36
32
  Requires-Dist: termcolor==1.1.0
37
- Requires-Dist: tqdm~=4.50.2
38
33
  Requires-Dist: timezonefinder~=6.1.9
39
34
  Requires-Dist: urllib3
40
- Requires-Dist: boto3
41
- Requires-Dist: awswrangler
35
+ Requires-Dist: Flask~=2.2.2
36
+ Provides-Extra: plotting
37
+ Requires-Dist: bokeh; extra == "plotting"
38
+ Requires-Dist: panel; extra == "plotting"
39
+ Requires-Dist: branca~=0.4.0; extra == "plotting"
40
+ Requires-Dist: folium~=0.12.1; extra == "plotting"
41
+ Provides-Extra: dev
42
+ Requires-Dist: pytest; extra == "dev"
43
+ Requires-Dist: bokeh; extra == "dev"
44
+ Requires-Dist: panel; extra == "dev"
45
+ Requires-Dist: branca~=0.4.0; extra == "dev"
46
+ Requires-Dist: folium~=0.12.1; extra == "dev"
47
+ Requires-Dist: awswrangler; extra == "dev"
48
+ Requires-Dist: boto3; extra == "dev"
42
49
  Dynamic: author
43
50
  Dynamic: classifier
44
51
  Dynamic: description
@@ -48,6 +55,7 @@ Dynamic: keywords
48
55
  Dynamic: license
49
56
  Dynamic: license-file
50
57
  Dynamic: project-url
58
+ Dynamic: provides-extra
51
59
  Dynamic: requires-dist
52
60
  Dynamic: requires-python
53
61
  Dynamic: summary
@@ -1,17 +1,10 @@
1
1
  # TODO To be updated?
2
- branca~=0.4.0
3
- # TODO Add once finished with file reports
4
- Flask~=2.2.2
5
- # TODO To be updated?
6
- folium~=0.12.1
7
- # TODO To be updated?
8
2
  geopy~=1.21.0
9
3
  # TODO To be updated?
10
4
  Jinja2~=3.1.2
11
5
  matplotlib
12
6
  pandas~=2.2.2
13
7
  pydantic
14
- pytest
15
8
  # TODO To be updated?
16
9
  PyYAML~=6.0.1
17
10
  requests
@@ -20,8 +13,6 @@ scikit-learn
20
13
  seaborn
21
14
  smartcitizen-connector
22
15
  termcolor==1.1.0
23
- tqdm~=4.50.2
24
16
  timezonefinder~=6.1.9
25
17
  urllib3
26
- boto3
27
- awswrangler
18
+ Flask~=2.2.2
@@ -3,4 +3,4 @@ from .device import Device
3
3
  from .test import Test
4
4
  from .models import Source, TestOptions, DeviceOptions, APIParams, FileParams, CSVParams
5
5
 
6
- __version__ = '1.3.2'
6
+ __version__ = '1.4.0'
@@ -29,10 +29,6 @@ class Config(object):
29
29
 
30
30
  # Framework option
31
31
  # For renderer plots and config files
32
- # Options:
33
- # - 'script': no plots in jupyter, updates config
34
- # - 'jupyterlab': for plots, updates config
35
- # - 'chupiflow': no plots in jupyter, does not update config
36
32
  framework = 'script'
37
33
 
38
34
  if 'IPython' in sys.modules: _ipython_avail = True
@@ -421,7 +417,7 @@ class Config(object):
421
417
  'SCD30_HUM': 1,
422
418
  'SCD30_TEMP': 1,
423
419
  'SD-card': 1,
424
- 'ST LPS33 - Barometric Pressure': 1,
420
+ 'LPS33_PRESS': 1,
425
421
  'PRESS': 1,
426
422
  'PMS5003_PM_1': 5,
427
423
  'PMS5003_PM_25': 5,
@@ -476,7 +472,7 @@ class Config(object):
476
472
  'SCD30_HUM': [20, 99],
477
473
  'SCD30_TEMP': [-20, 50],
478
474
  'BATT': [0, 100],
479
- 'ST LPS33 - Barometric Pressure': [50, 110],
475
+ 'LPS33_PRESS': [50, 110],
480
476
  'PRESS': [50, 110],
481
477
  'PMS5003_PM_1': [0, 500],
482
478
  'PMS5003_PM_25': [0, 500],
@@ -1,50 +1,86 @@
1
1
  ''' Main implementation of class Device '''
2
2
 
3
+ import os
4
+ from collections.abc import Iterable
5
+ from importlib import import_module
6
+ from io import StringIO
7
+ from json import dumps
8
+ from os.path import basename, exists, join
9
+ from typing import Dict, List, Optional
10
+ from urllib.parse import urlparse
11
+
12
+ from numpy import nan
13
+ from pandas import DataFrame, Series, Timedelta, to_timedelta
14
+ from pydantic import BaseModel, ConfigDict, TypeAdapter
15
+ from pydantic_core import ValidationError
16
+
17
+ from scdata._config import config
18
+ from scdata.io import export_csv_file, read_csv_file
19
+ from scdata.io.device_api import *
20
+ from scdata.models import (APIParams, Blueprint, CSVParams, DeviceOptions,
21
+ Metric, Sensor, Source)
3
22
  from scdata.tools.custom_logger import logger
4
- from scdata.io import read_csv_file, export_csv_file
5
- from scdata.tools.lazy import LazyCallable
6
- from scdata.tools.url_check import url_checker
7
23
  from scdata.tools.date import localise_date
8
24
  from scdata.tools.dictmerge import dict_fmerge
9
- from scdata.tools.units import get_units_convf
10
25
  from scdata.tools.find import find_by_field
11
- from scdata.tools.series import count_nas, infer_sampling_rate, mode_ratio, normalize_central, rolling_deltas
12
- from scdata._config import config
13
- from scdata.io.device_api import *
14
- from scdata.models import Blueprint, Metric, Source, APIParams, CSVParams, DeviceOptions, Sensor
26
+ from scdata.tools.lazy import LazyCallable
27
+ from scdata.tools.series import (count_nas, infer_sampling_rate, mode_ratio,
28
+ normalize_central, rolling_deltas)
29
+ from scdata.tools.tree import topological_sort
30
+ from scdata.tools.units import get_units_convf
31
+ from scdata.tools.url_check import url_checker
15
32
 
16
- from os.path import join, basename, exists
17
- from urllib.parse import urlparse
18
- from pandas import DataFrame, Series, to_timedelta, Timedelta
19
- from numpy import nan
20
- from collections.abc import Iterable
21
- from importlib import import_module
22
- from pydantic import TypeAdapter, BaseModel, ConfigDict
23
- from pydantic_core import ValidationError
24
- from typing import Optional, List, Dict
25
- from json import dumps
33
+ try:
34
+ import panel
35
+ import bokeh
36
+ except ModuleNotFoundError:
37
+ bokeh_available = False
38
+ pass
39
+ else:
40
+ bokeh_available = True
26
41
 
27
- import os
28
- from io import StringIO
42
+ if bokeh_available:
43
+ from scdata.plot.ts_panel import TimeSeriesPanel
29
44
 
30
45
  try:
31
46
  import awswrangler as wr
47
+ import boto3
32
48
  except ModuleNotFoundError:
33
49
  boto_available = False
34
50
  pass
35
51
  else:
36
52
  boto_available = True
37
53
 
38
- if boto_available: import boto3
54
+ try:
55
+ from branca import element
56
+ from folium import Circle
57
+ except ModuleNotFoundError:
58
+ map_plotting_available = False
59
+ pass
60
+ else:
61
+ map_plotting_available = True
39
62
 
40
63
  from timezonefinder import TimezoneFinder
64
+
41
65
  tf = TimezoneFinder()
42
66
 
43
67
  class Device(BaseModel):
44
68
  ''' Main implementation of the device class '''
69
+
70
+ from scdata.plot import box_plot # ts_iplot, scatter_iplot, heatmap_iplot,
71
+ from scdata.plot import (heatmap_plot, scatter_dispersion_grid, scatter_plot,
72
+ ts_dendrogram, ts_dispersion_grid, ts_dispersion_plot, ts_plot, ts_scatter)
73
+ #, report_plot, cat_plot, violin_plot)
74
+ if map_plotting_available:
75
+ from scdata.plot import device_metric_map, path_plot
76
+
77
+ if config._ipython_avail:
78
+ from scdata.plot import ts_uplot, ts_dispersion_uplot
79
+
45
80
  model_config = ConfigDict(arbitrary_types_allowed = True)
46
81
 
47
82
  blueprint: str = None
83
+ override_url_blueprint: bool = False
48
84
  source: Source = Source()
49
85
  options: DeviceOptions = DeviceOptions()
50
86
  params: object = None
@@ -89,18 +125,21 @@ class Device(BaseModel):
89
125
 
90
126
  # Set handler
91
127
  self.__set_handler__()
128
+
92
129
  # Set blueprint
93
- if self.blueprint is not None:
94
- if self.blueprint not in config.blueprints:
95
- raise ValueError(f'Specified blueprint {self.blueprint} is not in available blueprints')
96
- self.__set_blueprint_attrs__(config.blueprints[self.blueprint])
97
- else:
130
+ if self.handler.blueprint_url is not None and not self.override_url_blueprint:
131
+ logger.info("Checking blueprint in URL")
98
132
  if url_checker(self.handler.blueprint_url):
99
133
  logger.info(f'Loading postprocessing blueprint from:\n{self.handler.blueprint_url}')
100
134
  self.blueprint = basename(urlparse(self.handler.blueprint_url).path).split('.')[0]
101
135
  self.__set_blueprint_attrs__(self.handler.properties)
102
- else:
103
- raise ValueError(f'Specified blueprint url {self.handler.blueprint_url} is not valid')
136
+ elif self.blueprint is not None:
137
+ logger.info("Using defined blueprint")
138
+ if self.blueprint not in config.blueprints:
139
+ raise ValueError(f'Specified blueprint {self.blueprint} is not in available blueprints')
140
+ self.__set_blueprint_attrs__(config.blueprints[self.blueprint])
141
+ else:
142
+ raise ValueError(f'Specified blueprint url {self.handler.blueprint_url} is not valid')
104
143
 
105
144
  logger.info(f'Device {self.paramsParsed.id} initialised')
106
145
 
@@ -462,7 +501,6 @@ class Device(BaseModel):
462
501
  logger.warning(f'Device {self.paramsParsed.id} has nothing to process. Skipping')
463
502
  return process_ok
464
503
 
465
- logger.info('---------------------------')
466
504
  logger.info(f'Processing device {self.paramsParsed.id}')
467
505
  if lmetrics is None:
468
506
  _lmetrics = [metric.name for metric in self.metrics]
@@ -472,6 +510,10 @@ class Device(BaseModel):
472
510
  logger.warning('Nothing to process')
473
511
  return process_ok
474
512
 
513
+ # Sort metrics
514
+ logger.info('Sorting metrics...')
515
+ self.metrics = topological_sort(self.metrics)
516
+
475
517
  for metric in self.metrics:
476
518
  logger.info('---')
477
519
  if metric.name not in _lmetrics: continue
@@ -518,6 +560,7 @@ class Device(BaseModel):
518
560
  process_ok &= True
519
561
 
520
562
  if process_ok:
563
+ logger.info('---')
521
564
  logger.info(f"Device {self.paramsParsed.id} processed")
522
565
  self.processed = process_ok
523
566
 
@@ -696,7 +739,7 @@ class Device(BaseModel):
696
739
 
697
740
  def get_outlier_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_outlier_ratio", sigma=5, pct=0.05) -> DataFrame:
698
741
  '''Get the percentage of outlier values based on the rate of increase. When sensors
699
- report sudden jumps, these are likely to be erroneous values.
742
+ report sudden jumps, these are likely to be erroneous values.
700
743
 
701
744
  Parameters
702
745
  ----------
@@ -708,15 +751,15 @@ class Device(BaseModel):
708
751
  are used.
709
752
  sigma: int
710
753
  5
711
- Number of standard deviations to consider a point an outlier. A higher
754
+ Number of standard deviations to consider a point an outlier. A higher
712
755
  value would be more restrictive, detecting only worse malfunctions.
713
756
  pct: float
714
757
  0.05
715
- Percentage of top and bottom values to ignore when normalizing. We
758
+ Percentage of top and bottom values to ignore when normalizing. We
716
759
  assume the outliers will always be a minority of the data, so ignoring
717
760
  a small percentage of extreme values should help get a better estimate.
718
761
 
719
- Returns
762
+ Returns
720
763
  ----------
721
764
  result: DataFrame
722
765
  DataFrame with rolling outlier ratio.
@@ -730,13 +773,13 @@ class Device(BaseModel):
730
773
  data = self.data[subset]
731
774
  else:
732
775
  data = self.data
733
-
776
+
734
777
  result = {}
735
778
 
736
779
  for column in data.columns:
737
780
  outlier_values = self.get_outlier_values(column, sigma=sigma, pct=pct)
738
781
 
739
- result[column + suffix] = outlier_values.rolling(period).mean()
782
+ result[column + suffix] = outlier_values.rolling(period).mean()
740
783
 
741
784
  return DataFrame(result)
742
785
 
@@ -756,7 +799,7 @@ class Device(BaseModel):
756
799
  sigma: int
757
800
  Number of standard deviations to consider a point an outlier.
758
801
  pct: float
759
- Percentage of top and bottom values to ignore when normalizing.
802
+ Percentage of top and bottom values to ignore when normalizing.
760
803
 
761
804
  Returns
762
805
  ----------
@@ -906,28 +949,116 @@ class Device(BaseModel):
906
949
  if post_ok: logger.info(f"Postprocessing posted for device {self.paramsParsed.id}")
907
950
  return post_ok
908
951
 
909
- def backup(self, format='parquet', mode='append'):
952
+ def backup_to_storage(self, mode='append', path='devices'):
953
+ """
954
+ Backup device data into S3 storage (requires S3_DATA_BUCKET env
955
+ variable set).
956
+ Parameters
957
+ ----------
958
+ mode: str
959
+ 'append'
960
+ How to handle awswrangler to_parquet() storage
961
+ path: str
962
+ 'devices'
963
+ Path for backup directory
964
+ "s3://{os.environ['S3_DATA_BUCKET']}/{path}/{self.id}/data/"
965
+ Returns
966
+ ----------
967
+ False or response from awswrangler
968
+ """
910
969
  if self.data.empty:
911
970
  logger.error("Device data empty")
912
971
  return False
913
972
 
914
- if format == 'parquet':
915
- if boto_available:
916
- self.data['TIME']=self.data.index
917
- target_path = f"s3://{os.environ['S3_DATA_BUCKET']}/devices/{self.id}/data/"
918
- response = wr.s3.to_parquet(df=self.data, path=target_path, dataset=True, mode=mode)
919
-
920
- return response
921
-
922
- def backup_load(self, format='parquet'):
923
- if format == 'parquet':
924
- if boto_available:
925
- session = boto3.Session(aws_access_key_id=os.environ['AWS_ACCESS_KEY_ID'],
926
- aws_secret_access_key=os.environ['AWS_SECRET_ACCESS_KEY'],
927
- region_name=os.environ['AWS_REGION'])
928
- s3_url = f"s3://{os.environ['S3_DATA_BUCKET']}/devices/{self.id}/data/"
929
- self.data = wr.s3.read_parquet(s3_url, boto3_session=session, dataset=True)
930
- self.data.set_index('TIME', inplace=True)
931
- self.data.sort_index(inplace=True)
932
-
933
- return s3_url
973
+ if 'S3_DATA_BUCKET' not in os.environ:
974
+ logger.error("S3_DATA_BUCKET not set in environment")
975
+ return False
976
+
977
+ # TODO Add more formats
978
+ if boto_available:
979
+ self.data['TIME']=self.data.index
980
+ target_path = f"s3://{os.environ['S3_DATA_BUCKET']}/{path}/{self.id}/data/"
981
+ response = wr.s3.to_parquet(df=self.data, path=target_path, dataset=True, mode=mode)
982
+
983
+ s3 = boto3.resource('s3')
984
+ s3object = s3.Object(f"{os.environ['S3_DATA_BUCKET']}", f"{path}/{self.id}/metadata.json")
985
+ s3object.put(
986
+ Body=(bytes(self.handler.json.model_dump_json().encode('utf-8')))
987
+ )
988
+
989
+ return response
990
+
991
+ def load_from_storage(self, path='devices'):
992
+ """
993
+ Load device data from S3 storage (requires AWS env
994
+ variable set).
995
+ Parameters
996
+ ----------
997
+ path: str
998
+ 'devices'
999
+ Path for backup directory
1000
+ "s3://{os.environ['S3_DATA_BUCKET']}/{path}/{self.id}/data/"
1001
+ Returns
1002
+ ----------
1003
+ S3 bucket url if successful, False otherwise
1004
+ """
1005
+
1006
+ if 'S3_DATA_BUCKET' not in os.environ or \
1007
+ 'AWS_ACCESS_KEY_ID' not in os.environ or \
1008
+ 'AWS_SECRET_ACCESS_KEY' not in os.environ or \
1009
+ 'AWS_REGION' not in os.environ:
1010
+
1011
+ logger.error("Missing environment variables. S3_DATA_BUCKET, AWS_ACCESS_KEY_ID, \
1012
+ AWS_SECRET_ACCESS_KEY and AWS_REGION need to be set.")
1013
+
1014
+ return False
1015
+
1016
+ if boto_available:
1017
+ session = boto3.Session(aws_access_key_id=os.environ['AWS_ACCESS_KEY_ID'],
1018
+ aws_secret_access_key=os.environ['AWS_SECRET_ACCESS_KEY'],
1019
+ region_name=os.environ['AWS_REGION'])
1020
+ s3_url = f"s3://{os.environ['S3_DATA_BUCKET']}/{path}/{self.id}/data/"
1021
+ logger.info(f"Loading data from: {s3_url}")
1022
+
1023
+ self.data = wr.s3.read_parquet(s3_url, boto3_session=session, dataset=True)
1024
+ self.data.set_index('TIME', inplace=True)
1025
+ self.data.sort_index(inplace=True)
1026
+
1027
+ self.loaded = True
1028
+
1029
+ return s3_url
1030
+ else:
1031
+ logger.error("Boto not available. Install awswrangler")
1032
+ return False
1033
+
1034
+ def get_series_dict(self, frequency):
1035
+ df = self.data.copy()
1036
+ df.index = df.index.tz_convert('UTC').tz_localize(None)
1037
+ df = df.resample(frequency).mean()
1038
+ return {
1039
+ f"{self.id}:{col}": df[col]
1040
+ for col in df.columns
1041
+ }
1042
+
1043
+ def ts_panel(self, frequency='10Min', **kwargs):
1044
+ '''
1045
+ Returns a panel for interactive plotting
1046
+ ---
1047
+ frequency: str
1048
+ Default: 10Min
1049
+ Add a resample to the series to reduce
1050
+ width: int
1051
+ Default: 800
1052
+ Max width of each subplot (resizable to max width of window below that)
1053
+ height: int
1054
+ Default: 400
1055
+ Height of each subplot
1056
+ '''
1057
+ if bokeh_available:
1058
+ return TimeSeriesPanel(
1059
+ self.get_series_dict(frequency=frequency),
1060
+ **kwargs
1061
+ ).view()
1062
+ else:
1063
+ logger.error("Bokeh not available. Install with 'pip install scdata[plotting]' or 'pip install bokeh panel'")
1064
+ return False
@@ -4,7 +4,6 @@
4
4
  from scdata.tools.lazy import LazyCallable
5
5
  from .formulae import absolute_humidity, exp_f, fit_exp_f
6
6
  from .geoseries import is_within_circle
7
- from .timeseries import clean_ts, merge_ts, rolling_avg, poly_ts, geo_located, time_derivative, delta_index_ts
8
- from .baseline import find_min_max, baseline_calc, get_delta_baseline, get_als_baseline
9
- from .alphasense import alphasense_803_04, alphasense_pt1000, channel_names, basic_4electrode_alg, baseline_4electrode_alg, deconvolution, ec_sensor_temp
7
+ from .timeseries import clean_ts, merge_ts, rolling_avg, poly_ts, within, time_derivative, delta_index_ts, baseline_als
8
+ from .alphasense import alphasense_803_04, alphasense_als, alphasense_pt1000, channel_names, basic_4electrode_alg, baseline_4electrode_alg, deconvolution, ec_sensor_temp
10
9
  from .regression import apply_regressor
@@ -3,7 +3,7 @@ from scdata.tools.units import get_units_convf
3
3
  from scdata.tools.date import find_dates, localise_date
4
4
  from scdata._config import config
5
5
  from scdata.device.process.params import *
6
- from scdata.device.process import baseline_calc, clean_ts
6
+ from scdata.device.process import clean_ts, baseline_als
7
7
  from scipy.stats import linregress
8
8
  import matplotlib.pyplot as plt
9
9
  from pandas import date_range, DataFrame, Series, isnull
@@ -81,6 +81,16 @@ def alphasense_803_04(dataframe, **kwargs):
81
81
  if kwargs['use_alternative']: algorithm_idx = 1
82
82
  else: algorithm_idx = 0
83
83
 
84
+ # Clip negative values or not
85
+ process_negative_conc = None
86
+ if 'clip_negative_conc' in kwargs:
87
+ if kwargs['clip_negative_conc']:
88
+ process_negative_conc = 'clip_negative_conc'
89
+ # Offset negative values or not
90
+ elif 'offset_negative_conc' in kwargs:
91
+ if kwargs['offset_negative_conc']:
92
+ process_negative_conc = 'offset_negative_conc'
93
+
84
94
  # Get algorithm name
85
95
  algorithm = list(as_sensor_algs[as_type].keys())[algorithm_idx]
86
96
  comp_type = as_sensor_algs[as_type][algorithm][0]
@@ -138,8 +148,139 @@ def alphasense_803_04(dataframe, **kwargs):
138
148
  # Calculate sensor concentration
139
149
  df['conc'] = df['we_c'] / (cal_data['we_sensitivity_mv_ppb'] / 1000.0) # in ppb
140
150
 
141
- if avoid_negative_conc:
142
- df['conc'].clip(lower = 0, inplace = True)
151
+ if process_negative_conc == 'clip_negative_conc':
152
+ df['conc'] = df['conc'].clip(lower = 0)
153
+ elif process_negative_conc == 'offset_negative_conc':
154
+ df['conc'] += abs(df['conc'].min())
155
+
156
+ return ProcessResult(df['conc'], StatusCode.SUCCESS)
157
+
158
+ def alphasense_als(dataframe, **kwargs):
159
+ """
160
+ Calculates pollutant concentration based on 4 electrode sensor readings (mV)
161
+ and calibration ID. It adds a configurable background concentration and correction
162
+ based on AAN803-04
163
+ Parameters
164
+ ----------
165
+ alphasense_id: string
166
+ Alphasense sensor ID (must be in calibrations.json)
167
+ we: string
168
+ Name of working electrode found in dataframe (V)
169
+ ae: string
170
+ Name of auxiliary electrode found in dataframe (V)
171
+ t: string
172
+ Name of reference temperature
173
+ clip_negative_conc: bool
174
+ Clip the negative values after the algorithm
175
+ offset_negative_conc: bool
176
+ Offset the resulting negative values for the signal
177
+ Returns
178
+ -------
179
+ calculation of pollutant in ppb
180
+ """
181
+
182
+ # Check inputs
183
+ flag_error = False
184
+ if 'we' not in kwargs: flag_error = True
185
+ if 'ae' not in kwargs: flag_error = True
186
+ if 'alphasense_id' not in kwargs: flag_error = True
187
+ if 't' not in kwargs: flag_error = True
188
+ if 'clip_negative_conc' not in kwargs:
189
+ kwargs['clip_negative_conc'] = clip_negative_conc
190
+
191
+ if 'offset_negative_conc' not in kwargs:
192
+ kwargs['offset_negative_conc'] = offset_negative_conc
193
+
194
+ if 'lam' in kwargs:
195
+ lam = kwargs['lam']
196
+ else:
197
+ lam = None
198
+
199
+ if 'p' in kwargs:
200
+ p = kwargs['p']
201
+ else:
202
+ p = None
203
+
204
+ if flag_error:
205
+ logger.error('Problem with input data')
206
+ return ProcessResult(None, StatusCode.ERROR_MISSING_INPUTS)
207
+
208
+ if kwargs['alphasense_id'] is None:
209
+ logger.warning(f"Empty ID. Ignoring")
210
+ return ProcessResult(None, StatusCode.WARNING_EMPTY_ID)
211
+
212
+ # Get Sensor data
213
+ if kwargs['alphasense_id'] not in config.calibrations:
214
+ logger.error(f"Sensor {kwargs['alphasense_id']} not in calibration data")
215
+ return ProcessResult(None, StatusCode.ERROR_CALIBRATION_NOT_FOUND)
216
+
217
+ # Make copy
218
+ df = dataframe.copy()
219
+
220
+ # Get sensor type
221
+ as_type = alphasense_sensor_codes[kwargs['alphasense_id'][0:3]]
222
+
223
+ # Retrieve calibration data - verify its all float
224
+ cal_data = config.calibrations[kwargs['alphasense_id']]
225
+
226
+ for item in cal_data:
227
+ try:
228
+ cal_data[item] = float (cal_data[item])
229
+ except:
230
+ logger.error(f"Alphasense calibration data for {kwargs['alphasense_id']} is not correct")
231
+ logger.error(f'Error on {item}: \'{cal_data[item]}\'')
232
+ return ProcessResult(None, StatusCode.ERROR_WRONG_CALIBRATION)
233
+
234
+ # Remove spurious voltages (0V < electrode < 5V)
235
+ for electrode in ['we', 'ae']:
236
+ subkwargs = {'name': kwargs[electrode],
237
+ 'limits': (0, 5), # In V
238
+ 'window_size': None
239
+ }
240
+
241
+ cleaning_result = clean_ts(df, **subkwargs)
242
+ if 'SUCCESS' in cleaning_result.status_code.name:
243
+ df[f'{electrode}_clean'] = cleaning_result.data
244
+ else:
245
+ return ProcessResult(None, StatusCode.ERROR_UNDEFINED)
246
+
247
+ # Compensate electronic zero
248
+ df['we_t'] = df['we_clean'] - (cal_data['we_electronic_zero_mv'] / 1000) # in V
249
+ df['ae_t'] = df['ae_clean'] - (cal_data['ae_electronic_zero_mv'] / 1000) # in V
250
+ # Get requested temperature
251
+ df['t'] = df[kwargs['t']]
252
+
253
+ # Calculate WE baseline
254
+ subkwargs = {'name': 'we_t'}
255
+ if lam is not None: subkwargs['lam'] = lam
256
+ if p is not None: subkwargs['p'] = p
257
+
258
+ baseline_result = baseline_als(df, **subkwargs)
259
+ if 'SUCCESS' in baseline_result.status_code.name:
260
+ df[f'we_baseline'] = baseline_result.data
261
+ else:
262
+ return ProcessResult(None, StatusCode.ERROR_UNDEFINED)
263
+
264
+ # Calculate baseline factor
265
+ df['baseline_t'] = df['we_baseline'] / df['ae_t']
266
+ # Correct Auxiliary electrode based on mean factor
267
+ df['ae_cor'] = df['baseline_t'].mean() * df['ae_t']
268
+ df['we_c'] = df['we_t'] - df['ae_cor']
269
+
270
+ # Verify if it has NO2 cross-sensitivity (in V)
271
+ if cal_data['we_cross_sensitivity_no2_mv_ppb'] != float (0) and 'NO2' not in as_type:
272
+ df['we_no2_eq'] = df['NO2'] * cal_data['we_cross_sensitivity_no2_mv_ppb'] / 1000.0
273
+ df['we_c'] -= df['we_no2_eq'] # in V
274
+
275
+ # Calculate sensor concentration
276
+ df['conc'] = df['we_c'] / (cal_data['we_sensitivity_mv_ppb'] / 1000.0) # in ppb
277
+
278
+ if kwargs['clip_negative_conc']:
279
+ df['conc'] = df['conc'].clip(lower = 0)
280
+
281
+ elif kwargs['offset_negative_conc']:
282
+ if df['conc'].min() < 0:
283
+ df['conc'] = df['conc'] + abs(df['conc'].min())
143
284
 
144
285
  return ProcessResult(df['conc'], StatusCode.SUCCESS)
145
286
 
@@ -1,7 +1,8 @@
1
1
  from numpy import arange
2
2
 
3
- # Avoid negative pollutant concentrations
4
- avoid_negative_conc = False
3
+ # Avoid negative pollutant concentrations by clipping or offseting. These are mutually exclusive
4
+ offset_negative_conc = False
5
+ clip_negative_conc = False
5
6
 
6
7
  # Background concentrations
7
8
  background_conc = {