scdata 1.3.2__tar.gz → 1.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scdata-1.3.2/scdata.egg-info → scdata-1.4.0}/PKG-INFO +16 -8
- {scdata-1.3.2 → scdata-1.4.0}/requirements.txt +1 -10
- {scdata-1.3.2 → scdata-1.4.0}/scdata/__init__.py +1 -1
- {scdata-1.3.2 → scdata-1.4.0}/scdata/_config/config.py +2 -6
- {scdata-1.3.2 → scdata-1.4.0}/scdata/device/device.py +188 -57
- {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/__init__.py +2 -3
- {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/alphasense.py +144 -3
- {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/params.py +3 -2
- {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/timeseries.py +67 -2
- {scdata-1.3.2 → scdata-1.4.0}/scdata/io/device_api.py +0 -1
- {scdata-1.3.2 → scdata-1.4.0}/scdata/io/device_file.py +2 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/models/models.py +1 -0
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/__init__.py +15 -8
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/box_plot.py +12 -6
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/heatmap_iplot.py +14 -5
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/heatmap_plot.py +13 -6
- scdata-1.3.2/scdata/test/plot/ts_uplot.py → scdata-1.4.0/scdata/plot/heatmap_uplot.py +16 -11
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/maps.py +13 -11
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/scatter_dispersion_grid.py +6 -4
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/scatter_iplot.py +8 -5
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/scatter_plot.py +15 -8
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/target_diagram.py +19 -16
- scdata-1.3.2/scdata/test/plot/plot_tools.py → scdata-1.4.0/scdata/plot/tools.py +174 -23
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_dendrogram.py +7 -6
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_dispersion_grid.py +6 -4
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_dispersion_plot.py +8 -7
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_dispersion_uplot.py +10 -9
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_iplot.py +15 -8
- scdata-1.4.0/scdata/plot/ts_panel.py +301 -0
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_plot.py +16 -11
- {scdata-1.3.2/scdata/test → scdata-1.4.0/scdata}/plot/ts_scatter.py +16 -11
- scdata-1.4.0/scdata/plot/ts_uplot.py +326 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/checks/checks.py +4 -6
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/test.py +77 -25
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/series.py +3 -2
- scdata-1.4.0/scdata/tools/tree.py +41 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/units.py +2 -0
- {scdata-1.3.2 → scdata-1.4.0/scdata.egg-info}/PKG-INFO +16 -8
- {scdata-1.3.2 → scdata-1.4.0}/scdata.egg-info/SOURCES.txt +21 -19
- {scdata-1.3.2 → scdata-1.4.0}/scdata.egg-info/requires.txt +15 -6
- {scdata-1.3.2 → scdata-1.4.0}/setup.py +18 -1
- scdata-1.3.2/scdata/device/process/baseline.py +0 -295
- {scdata-1.3.2 → scdata-1.4.0}/LICENSE +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/MANIFEST.in +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/README.md +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/_config/__init__.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/device/__init__.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/device/plot/__init__.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/error_codes.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/formulae.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/geoseries.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/device/process/regression.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/io/__init__.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/io/model.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/models/__init__.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/__init__.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/checks/__init__.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/dispersion/__init__.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/dispersion/dispersion.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/export/__init__.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/export/templates/sc_template.html +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/export/to_file.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/tools/__init__.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/tools/combine.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/tools/history.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/test/tools/prepare.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/__init__.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/cleaning.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/custom_logger.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/date.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/dictmerge.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/find.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/gets.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/interim/example.csv +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/interim/geodata.csv +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/lazy.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/location.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/report.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/stats.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/uploads/example_upload_1.json +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/uploads/example_zenodo_upload.yaml +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/uploads/report.pdf +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/url_check.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/zenodo.py +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/zenodo_templates/README.md +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/zenodo_templates/template_zenodo_dataset.json +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata/tools/zenodo_templates/template_zenodo_publication.json +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata.egg-info/dependency_links.txt +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata.egg-info/not-zip-safe +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/scdata.egg-info/top_level.txt +0 -0
- {scdata-1.3.2 → scdata-1.4.0}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scdata
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.0
|
|
4
4
|
Summary: Analysis of sensors and time series data
|
|
5
5
|
Home-page: https://github.com/fablabbcn/smartcitizen-data
|
|
6
6
|
Author: oscgonfer
|
|
@@ -18,15 +18,11 @@ Classifier: Programming Language :: Python :: 3
|
|
|
18
18
|
Requires-Python: >=3.9
|
|
19
19
|
Description-Content-Type: text/markdown
|
|
20
20
|
License-File: LICENSE
|
|
21
|
-
Requires-Dist: branca~=0.4.0
|
|
22
|
-
Requires-Dist: Flask~=2.2.2
|
|
23
|
-
Requires-Dist: folium~=0.12.1
|
|
24
21
|
Requires-Dist: geopy~=1.21.0
|
|
25
22
|
Requires-Dist: Jinja2~=3.1.2
|
|
26
23
|
Requires-Dist: matplotlib
|
|
27
24
|
Requires-Dist: pandas~=2.2.2
|
|
28
25
|
Requires-Dist: pydantic
|
|
29
|
-
Requires-Dist: pytest
|
|
30
26
|
Requires-Dist: PyYAML~=6.0.1
|
|
31
27
|
Requires-Dist: requests
|
|
32
28
|
Requires-Dist: scipy
|
|
@@ -34,11 +30,22 @@ Requires-Dist: scikit-learn
|
|
|
34
30
|
Requires-Dist: seaborn
|
|
35
31
|
Requires-Dist: smartcitizen-connector
|
|
36
32
|
Requires-Dist: termcolor==1.1.0
|
|
37
|
-
Requires-Dist: tqdm~=4.50.2
|
|
38
33
|
Requires-Dist: timezonefinder~=6.1.9
|
|
39
34
|
Requires-Dist: urllib3
|
|
40
|
-
Requires-Dist:
|
|
41
|
-
|
|
35
|
+
Requires-Dist: Flask~=2.2.2
|
|
36
|
+
Provides-Extra: plotting
|
|
37
|
+
Requires-Dist: bokeh; extra == "plotting"
|
|
38
|
+
Requires-Dist: panel; extra == "plotting"
|
|
39
|
+
Requires-Dist: branca~=0.4.0; extra == "plotting"
|
|
40
|
+
Requires-Dist: folium~=0.12.1; extra == "plotting"
|
|
41
|
+
Provides-Extra: dev
|
|
42
|
+
Requires-Dist: pytest; extra == "dev"
|
|
43
|
+
Requires-Dist: bokeh; extra == "dev"
|
|
44
|
+
Requires-Dist: panel; extra == "dev"
|
|
45
|
+
Requires-Dist: branca~=0.4.0; extra == "dev"
|
|
46
|
+
Requires-Dist: folium~=0.12.1; extra == "dev"
|
|
47
|
+
Requires-Dist: awswrangler; extra == "dev"
|
|
48
|
+
Requires-Dist: boto3; extra == "dev"
|
|
42
49
|
Dynamic: author
|
|
43
50
|
Dynamic: classifier
|
|
44
51
|
Dynamic: description
|
|
@@ -48,6 +55,7 @@ Dynamic: keywords
|
|
|
48
55
|
Dynamic: license
|
|
49
56
|
Dynamic: license-file
|
|
50
57
|
Dynamic: project-url
|
|
58
|
+
Dynamic: provides-extra
|
|
51
59
|
Dynamic: requires-dist
|
|
52
60
|
Dynamic: requires-python
|
|
53
61
|
Dynamic: summary
|
|
@@ -1,17 +1,10 @@
|
|
|
1
1
|
# TODO To be updated?
|
|
2
|
-
branca~=0.4.0
|
|
3
|
-
# TODO Add once finished with file reports
|
|
4
|
-
Flask~=2.2.2
|
|
5
|
-
# TODO To be updated?
|
|
6
|
-
folium~=0.12.1
|
|
7
|
-
# TODO To be updated?
|
|
8
2
|
geopy~=1.21.0
|
|
9
3
|
# TODO To be updated?
|
|
10
4
|
Jinja2~=3.1.2
|
|
11
5
|
matplotlib
|
|
12
6
|
pandas~=2.2.2
|
|
13
7
|
pydantic
|
|
14
|
-
pytest
|
|
15
8
|
# TODO To be updated?
|
|
16
9
|
PyYAML~=6.0.1
|
|
17
10
|
requests
|
|
@@ -20,8 +13,6 @@ scikit-learn
|
|
|
20
13
|
seaborn
|
|
21
14
|
smartcitizen-connector
|
|
22
15
|
termcolor==1.1.0
|
|
23
|
-
tqdm~=4.50.2
|
|
24
16
|
timezonefinder~=6.1.9
|
|
25
17
|
urllib3
|
|
26
|
-
|
|
27
|
-
awswrangler
|
|
18
|
+
Flask~=2.2.2
|
|
@@ -29,10 +29,6 @@ class Config(object):
|
|
|
29
29
|
|
|
30
30
|
# Framework option
|
|
31
31
|
# For renderer plots and config files
|
|
32
|
-
# Options:
|
|
33
|
-
# - 'script': no plots in jupyter, updates config
|
|
34
|
-
# - 'jupyterlab': for plots, updates config
|
|
35
|
-
# - 'chupiflow': no plots in jupyter, does not update config
|
|
36
32
|
framework = 'script'
|
|
37
33
|
|
|
38
34
|
if 'IPython' in sys.modules: _ipython_avail = True
|
|
@@ -421,7 +417,7 @@ class Config(object):
|
|
|
421
417
|
'SCD30_HUM': 1,
|
|
422
418
|
'SCD30_TEMP': 1,
|
|
423
419
|
'SD-card': 1,
|
|
424
|
-
'
|
|
420
|
+
'LPS33_PRESS': 1,
|
|
425
421
|
'PRESS': 1,
|
|
426
422
|
'PMS5003_PM_1': 5,
|
|
427
423
|
'PMS5003_PM_25': 5,
|
|
@@ -476,7 +472,7 @@ class Config(object):
|
|
|
476
472
|
'SCD30_HUM': [20, 99],
|
|
477
473
|
'SCD30_TEMP': [-20, 50],
|
|
478
474
|
'BATT': [0, 100],
|
|
479
|
-
'
|
|
475
|
+
'LPS33_PRESS': [50, 110],
|
|
480
476
|
'PRESS': [50, 110],
|
|
481
477
|
'PMS5003_PM_1': [0, 500],
|
|
482
478
|
'PMS5003_PM_25': [0, 500],
|
|
@@ -1,50 +1,86 @@
|
|
|
1
1
|
''' Main implementation of class Device '''
|
|
2
2
|
|
|
3
|
+
import os
|
|
4
|
+
from collections.abc import Iterable
|
|
5
|
+
from importlib import import_module
|
|
6
|
+
from io import StringIO
|
|
7
|
+
from json import dumps
|
|
8
|
+
from os.path import basename, exists, join
|
|
9
|
+
from typing import Dict, List, Optional
|
|
10
|
+
from urllib.parse import urlparse
|
|
11
|
+
|
|
12
|
+
from numpy import nan
|
|
13
|
+
from pandas import DataFrame, Series, Timedelta, to_timedelta
|
|
14
|
+
from pydantic import BaseModel, ConfigDict, TypeAdapter
|
|
15
|
+
from pydantic_core import ValidationError
|
|
16
|
+
|
|
17
|
+
from scdata._config import config
|
|
18
|
+
from scdata.io import export_csv_file, read_csv_file
|
|
19
|
+
from scdata.io.device_api import *
|
|
20
|
+
from scdata.models import (APIParams, Blueprint, CSVParams, DeviceOptions,
|
|
21
|
+
Metric, Sensor, Source)
|
|
3
22
|
from scdata.tools.custom_logger import logger
|
|
4
|
-
from scdata.io import read_csv_file, export_csv_file
|
|
5
|
-
from scdata.tools.lazy import LazyCallable
|
|
6
|
-
from scdata.tools.url_check import url_checker
|
|
7
23
|
from scdata.tools.date import localise_date
|
|
8
24
|
from scdata.tools.dictmerge import dict_fmerge
|
|
9
|
-
from scdata.tools.units import get_units_convf
|
|
10
25
|
from scdata.tools.find import find_by_field
|
|
11
|
-
from scdata.tools.
|
|
12
|
-
from scdata.
|
|
13
|
-
|
|
14
|
-
from scdata.
|
|
26
|
+
from scdata.tools.lazy import LazyCallable
|
|
27
|
+
from scdata.tools.series import (count_nas, infer_sampling_rate, mode_ratio,
|
|
28
|
+
normalize_central, rolling_deltas)
|
|
29
|
+
from scdata.tools.tree import topological_sort
|
|
30
|
+
from scdata.tools.units import get_units_convf
|
|
31
|
+
from scdata.tools.url_check import url_checker
|
|
15
32
|
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
from typing import Optional, List, Dict
|
|
25
|
-
from json import dumps
|
|
33
|
+
try:
|
|
34
|
+
import panel
|
|
35
|
+
import bokeh
|
|
36
|
+
except ModuleNotFoundError:
|
|
37
|
+
bokeh_available = False
|
|
38
|
+
pass
|
|
39
|
+
else:
|
|
40
|
+
bokeh_available = True
|
|
26
41
|
|
|
27
|
-
|
|
28
|
-
from
|
|
42
|
+
if bokeh_available:
|
|
43
|
+
from scdata.plot.ts_panel import TimeSeriesPanel
|
|
29
44
|
|
|
30
45
|
try:
|
|
31
46
|
import awswrangler as wr
|
|
47
|
+
import boto3
|
|
32
48
|
except ModuleNotFoundError:
|
|
33
49
|
boto_available = False
|
|
34
50
|
pass
|
|
35
51
|
else:
|
|
36
52
|
boto_available = True
|
|
37
53
|
|
|
38
|
-
|
|
54
|
+
try:
|
|
55
|
+
from branca import element
|
|
56
|
+
from folium import Circle
|
|
57
|
+
except ModuleNotFoundError:
|
|
58
|
+
map_plotting_available = False
|
|
59
|
+
pass
|
|
60
|
+
else:
|
|
61
|
+
map_plotting_available = True
|
|
39
62
|
|
|
40
63
|
from timezonefinder import TimezoneFinder
|
|
64
|
+
|
|
41
65
|
tf = TimezoneFinder()
|
|
42
66
|
|
|
43
67
|
class Device(BaseModel):
|
|
44
68
|
''' Main implementation of the device class '''
|
|
69
|
+
|
|
70
|
+
from scdata.plot import box_plot # ts_iplot, scatter_iplot, heatmap_iplot,
|
|
71
|
+
from scdata.plot import (heatmap_plot, scatter_dispersion_grid, scatter_plot,
|
|
72
|
+
ts_dendrogram, ts_dispersion_grid, ts_dispersion_plot, ts_plot, ts_scatter)
|
|
73
|
+
#, report_plot, cat_plot, violin_plot)
|
|
74
|
+
if map_plotting_available:
|
|
75
|
+
from scdata.plot import device_metric_map, path_plot
|
|
76
|
+
|
|
77
|
+
if config._ipython_avail:
|
|
78
|
+
from scdata.plot import ts_uplot, ts_dispersion_uplot
|
|
79
|
+
|
|
45
80
|
model_config = ConfigDict(arbitrary_types_allowed = True)
|
|
46
81
|
|
|
47
82
|
blueprint: str = None
|
|
83
|
+
override_url_blueprint: bool = False
|
|
48
84
|
source: Source = Source()
|
|
49
85
|
options: DeviceOptions = DeviceOptions()
|
|
50
86
|
params: object = None
|
|
@@ -89,18 +125,21 @@ class Device(BaseModel):
|
|
|
89
125
|
|
|
90
126
|
# Set handler
|
|
91
127
|
self.__set_handler__()
|
|
128
|
+
|
|
92
129
|
# Set blueprint
|
|
93
|
-
if self.
|
|
94
|
-
|
|
95
|
-
raise ValueError(f'Specified blueprint {self.blueprint} is not in available blueprints')
|
|
96
|
-
self.__set_blueprint_attrs__(config.blueprints[self.blueprint])
|
|
97
|
-
else:
|
|
130
|
+
if self.handler.blueprint_url is not None and not self.override_url_blueprint:
|
|
131
|
+
logger.info("Checking blueprint in URL")
|
|
98
132
|
if url_checker(self.handler.blueprint_url):
|
|
99
133
|
logger.info(f'Loading postprocessing blueprint from:\n{self.handler.blueprint_url}')
|
|
100
134
|
self.blueprint = basename(urlparse(self.handler.blueprint_url).path).split('.')[0]
|
|
101
135
|
self.__set_blueprint_attrs__(self.handler.properties)
|
|
102
|
-
|
|
103
|
-
|
|
136
|
+
elif self.blueprint is not None:
|
|
137
|
+
logger.info("Using defined blueprint")
|
|
138
|
+
if self.blueprint not in config.blueprints:
|
|
139
|
+
raise ValueError(f'Specified blueprint {self.blueprint} is not in available blueprints')
|
|
140
|
+
self.__set_blueprint_attrs__(config.blueprints[self.blueprint])
|
|
141
|
+
else:
|
|
142
|
+
raise ValueError(f'Specified blueprint url {self.handler.blueprint_url} is not valid')
|
|
104
143
|
|
|
105
144
|
logger.info(f'Device {self.paramsParsed.id} initialised')
|
|
106
145
|
|
|
@@ -462,7 +501,6 @@ class Device(BaseModel):
|
|
|
462
501
|
logger.warning(f'Device {self.paramsParsed.id} has nothing to process. Skipping')
|
|
463
502
|
return process_ok
|
|
464
503
|
|
|
465
|
-
logger.info('---------------------------')
|
|
466
504
|
logger.info(f'Processing device {self.paramsParsed.id}')
|
|
467
505
|
if lmetrics is None:
|
|
468
506
|
_lmetrics = [metric.name for metric in self.metrics]
|
|
@@ -472,6 +510,10 @@ class Device(BaseModel):
|
|
|
472
510
|
logger.warning('Nothing to process')
|
|
473
511
|
return process_ok
|
|
474
512
|
|
|
513
|
+
# Sort metrics
|
|
514
|
+
logger.info('Sorting metrics...')
|
|
515
|
+
self.metrics = topological_sort(self.metrics)
|
|
516
|
+
|
|
475
517
|
for metric in self.metrics:
|
|
476
518
|
logger.info('---')
|
|
477
519
|
if metric.name not in _lmetrics: continue
|
|
@@ -518,6 +560,7 @@ class Device(BaseModel):
|
|
|
518
560
|
process_ok &= True
|
|
519
561
|
|
|
520
562
|
if process_ok:
|
|
563
|
+
logger.info('---')
|
|
521
564
|
logger.info(f"Device {self.paramsParsed.id} processed")
|
|
522
565
|
self.processed = process_ok
|
|
523
566
|
|
|
@@ -696,7 +739,7 @@ class Device(BaseModel):
|
|
|
696
739
|
|
|
697
740
|
def get_outlier_ratio(self, period:str="1h", subset:List[str]=None, suffix:str="_outlier_ratio", sigma=5, pct=0.05) -> DataFrame:
|
|
698
741
|
'''Get the percentage of outlier values based on the rate of increase. When sensors
|
|
699
|
-
report sudden jumps, these are likely to be erroneous values.
|
|
742
|
+
report sudden jumps, these are likely to be erroneous values.
|
|
700
743
|
|
|
701
744
|
Parameters
|
|
702
745
|
----------
|
|
@@ -708,15 +751,15 @@ class Device(BaseModel):
|
|
|
708
751
|
are used.
|
|
709
752
|
sigma: int
|
|
710
753
|
5
|
|
711
|
-
Number of standard deviations to consider a point an outlier. A higher
|
|
754
|
+
Number of standard deviations to consider a point an outlier. A higher
|
|
712
755
|
value would be more restrictive, detecting only worse malfunctions.
|
|
713
756
|
pct: float
|
|
714
757
|
0.05
|
|
715
|
-
Percentage of top and bottom values to ignore when normalizing. We
|
|
758
|
+
Percentage of top and bottom values to ignore when normalizing. We
|
|
716
759
|
assume the outliers will always be a minority of the data, so ignoring
|
|
717
760
|
a small percentage of extreme values should help get a better estimate.
|
|
718
761
|
|
|
719
|
-
Returns
|
|
762
|
+
Returns
|
|
720
763
|
----------
|
|
721
764
|
result: DataFrame
|
|
722
765
|
DataFrame with rolling outlier ratio.
|
|
@@ -730,13 +773,13 @@ class Device(BaseModel):
|
|
|
730
773
|
data = self.data[subset]
|
|
731
774
|
else:
|
|
732
775
|
data = self.data
|
|
733
|
-
|
|
776
|
+
|
|
734
777
|
result = {}
|
|
735
778
|
|
|
736
779
|
for column in data.columns:
|
|
737
780
|
outlier_values = self.get_outlier_values(column, sigma=sigma, pct=pct)
|
|
738
781
|
|
|
739
|
-
result[column + suffix] = outlier_values.rolling(period).mean()
|
|
782
|
+
result[column + suffix] = outlier_values.rolling(period).mean()
|
|
740
783
|
|
|
741
784
|
return DataFrame(result)
|
|
742
785
|
|
|
@@ -756,7 +799,7 @@ class Device(BaseModel):
|
|
|
756
799
|
sigma: int
|
|
757
800
|
Number of standard deviations to consider a point an outlier.
|
|
758
801
|
pct: float
|
|
759
|
-
Percentage of top and bottom values to ignore when normalizing.
|
|
802
|
+
Percentage of top and bottom values to ignore when normalizing.
|
|
760
803
|
|
|
761
804
|
Returns
|
|
762
805
|
----------
|
|
@@ -906,28 +949,116 @@ class Device(BaseModel):
|
|
|
906
949
|
if post_ok: logger.info(f"Postprocessing posted for device {self.paramsParsed.id}")
|
|
907
950
|
return post_ok
|
|
908
951
|
|
|
909
|
-
def
|
|
952
|
+
def backup_to_storage(self, mode='append', path='devices'):
|
|
953
|
+
"""
|
|
954
|
+
Backup device data into S3 storage (requires S3_DATA_BUCKET env
|
|
955
|
+
variable set).
|
|
956
|
+
Parameters
|
|
957
|
+
----------
|
|
958
|
+
mode: str
|
|
959
|
+
'append'
|
|
960
|
+
How to handle awswrangler to_parquet() storage
|
|
961
|
+
path: str
|
|
962
|
+
'devices'
|
|
963
|
+
Path for backup directory
|
|
964
|
+
"s3://{os.environ['S3_DATA_BUCKET']}/{path}/{self.id}/data/"
|
|
965
|
+
Returns
|
|
966
|
+
----------
|
|
967
|
+
False or response from awswrangler
|
|
968
|
+
"""
|
|
910
969
|
if self.data.empty:
|
|
911
970
|
logger.error("Device data empty")
|
|
912
971
|
return False
|
|
913
972
|
|
|
914
|
-
if
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
973
|
+
if 'S3_DATA_BUCKET' not in os.environ:
|
|
974
|
+
logger.error("S3_DATA_BUCKET not set in environment")
|
|
975
|
+
return False
|
|
976
|
+
|
|
977
|
+
# TODO Add more formats
|
|
978
|
+
if boto_available:
|
|
979
|
+
self.data['TIME']=self.data.index
|
|
980
|
+
target_path = f"s3://{os.environ['S3_DATA_BUCKET']}/{path}/{self.id}/data/"
|
|
981
|
+
response = wr.s3.to_parquet(df=self.data, path=target_path, dataset=True, mode=mode)
|
|
982
|
+
|
|
983
|
+
s3 = boto3.resource('s3')
|
|
984
|
+
s3object = s3.Object(f"{os.environ['S3_DATA_BUCKET']}", f"{path}/{self.id}/metadata.json")
|
|
985
|
+
s3object.put(
|
|
986
|
+
Body=(bytes(self.handler.json.model_dump_json().encode('utf-8')))
|
|
987
|
+
)
|
|
988
|
+
|
|
989
|
+
return response
|
|
990
|
+
|
|
991
|
+
def load_from_storage(self, path='devices'):
|
|
992
|
+
"""
|
|
993
|
+
Load device data from S3 storage (requires AWS env
|
|
994
|
+
variable set).
|
|
995
|
+
Parameters
|
|
996
|
+
----------
|
|
997
|
+
path: str
|
|
998
|
+
'devices'
|
|
999
|
+
Path for backup directory
|
|
1000
|
+
"s3://{os.environ['S3_DATA_BUCKET']}/{path}/{self.id}/data/"
|
|
1001
|
+
Returns
|
|
1002
|
+
----------
|
|
1003
|
+
S3 bucket url if successful, False otherwise
|
|
1004
|
+
"""
|
|
1005
|
+
|
|
1006
|
+
if 'S3_DATA_BUCKET' not in os.environ or \
|
|
1007
|
+
'AWS_ACCESS_KEY_ID' not in os.environ or \
|
|
1008
|
+
'AWS_SECRET_ACCESS_KEY' not in os.environ or \
|
|
1009
|
+
'AWS_REGION' not in os.environ:
|
|
1010
|
+
|
|
1011
|
+
logger.error("Missing environment variables. S3_DATA_BUCKET, AWS_ACCESS_KEY_ID, \
|
|
1012
|
+
AWS_SECRET_ACCESS_KEY and AWS_REGION need to be set.")
|
|
1013
|
+
|
|
1014
|
+
return False
|
|
1015
|
+
|
|
1016
|
+
if boto_available:
|
|
1017
|
+
session = boto3.Session(aws_access_key_id=os.environ['AWS_ACCESS_KEY_ID'],
|
|
1018
|
+
aws_secret_access_key=os.environ['AWS_SECRET_ACCESS_KEY'],
|
|
1019
|
+
region_name=os.environ['AWS_REGION'])
|
|
1020
|
+
s3_url = f"s3://{os.environ['S3_DATA_BUCKET']}/{path}/{self.id}/data/"
|
|
1021
|
+
logger.info(f"Loading data from: {s3_url}")
|
|
1022
|
+
|
|
1023
|
+
self.data = wr.s3.read_parquet(s3_url, boto3_session=session, dataset=True)
|
|
1024
|
+
self.data.set_index('TIME', inplace=True)
|
|
1025
|
+
self.data.sort_index(inplace=True)
|
|
1026
|
+
|
|
1027
|
+
self.loaded = True
|
|
1028
|
+
|
|
1029
|
+
return s3_url
|
|
1030
|
+
else:
|
|
1031
|
+
logger.error("Boto not available. Install awswrangler")
|
|
1032
|
+
return False
|
|
1033
|
+
|
|
1034
|
+
def get_series_dict(self, frequency):
|
|
1035
|
+
df = self.data.copy()
|
|
1036
|
+
df.index = df.index.tz_convert('UTC').tz_localize(None)
|
|
1037
|
+
df = df.resample(frequency).mean()
|
|
1038
|
+
return {
|
|
1039
|
+
f"{self.id}:{col}": df[col]
|
|
1040
|
+
for col in df.columns
|
|
1041
|
+
}
|
|
1042
|
+
|
|
1043
|
+
def ts_panel(self, frequency='10Min', **kwargs):
|
|
1044
|
+
'''
|
|
1045
|
+
Returns a panel for interactive plotting
|
|
1046
|
+
---
|
|
1047
|
+
frequency: str
|
|
1048
|
+
Default: 10Min
|
|
1049
|
+
Add a resample to the series to reduce
|
|
1050
|
+
width: int
|
|
1051
|
+
Default: 800
|
|
1052
|
+
Max width of each subplot (resizable to max width of window below that)
|
|
1053
|
+
height: int
|
|
1054
|
+
Default: 400
|
|
1055
|
+
Height of each subplot
|
|
1056
|
+
'''
|
|
1057
|
+
if bokeh_available:
|
|
1058
|
+
return TimeSeriesPanel(
|
|
1059
|
+
self.get_series_dict(frequency=frequency),
|
|
1060
|
+
**kwargs
|
|
1061
|
+
).view()
|
|
1062
|
+
else:
|
|
1063
|
+
logger.error("Bokeh not available. Install with 'pip install scdata[plotting]' or 'pip install bokeh panel'")
|
|
1064
|
+
return False
|
|
@@ -4,7 +4,6 @@
|
|
|
4
4
|
from scdata.tools.lazy import LazyCallable
|
|
5
5
|
from .formulae import absolute_humidity, exp_f, fit_exp_f
|
|
6
6
|
from .geoseries import is_within_circle
|
|
7
|
-
from .timeseries import clean_ts, merge_ts, rolling_avg, poly_ts,
|
|
8
|
-
from .
|
|
9
|
-
from .alphasense import alphasense_803_04, alphasense_pt1000, channel_names, basic_4electrode_alg, baseline_4electrode_alg, deconvolution, ec_sensor_temp
|
|
7
|
+
from .timeseries import clean_ts, merge_ts, rolling_avg, poly_ts, within, time_derivative, delta_index_ts, baseline_als
|
|
8
|
+
from .alphasense import alphasense_803_04, alphasense_als, alphasense_pt1000, channel_names, basic_4electrode_alg, baseline_4electrode_alg, deconvolution, ec_sensor_temp
|
|
10
9
|
from .regression import apply_regressor
|
|
@@ -3,7 +3,7 @@ from scdata.tools.units import get_units_convf
|
|
|
3
3
|
from scdata.tools.date import find_dates, localise_date
|
|
4
4
|
from scdata._config import config
|
|
5
5
|
from scdata.device.process.params import *
|
|
6
|
-
from scdata.device.process import
|
|
6
|
+
from scdata.device.process import clean_ts, baseline_als
|
|
7
7
|
from scipy.stats import linregress
|
|
8
8
|
import matplotlib.pyplot as plt
|
|
9
9
|
from pandas import date_range, DataFrame, Series, isnull
|
|
@@ -81,6 +81,16 @@ def alphasense_803_04(dataframe, **kwargs):
|
|
|
81
81
|
if kwargs['use_alternative']: algorithm_idx = 1
|
|
82
82
|
else: algorithm_idx = 0
|
|
83
83
|
|
|
84
|
+
# Clip negative values or not
|
|
85
|
+
process_negative_conc = None
|
|
86
|
+
if 'clip_negative_conc' in kwargs:
|
|
87
|
+
if kwargs['clip_negative_conc']:
|
|
88
|
+
process_negative_conc = 'clip_negative_conc'
|
|
89
|
+
# Offset negative values or not
|
|
90
|
+
elif 'offset_negative_conc' in kwargs:
|
|
91
|
+
if kwargs['offset_negative_conc']:
|
|
92
|
+
process_negative_conc = 'offset_negative_conc'
|
|
93
|
+
|
|
84
94
|
# Get algorithm name
|
|
85
95
|
algorithm = list(as_sensor_algs[as_type].keys())[algorithm_idx]
|
|
86
96
|
comp_type = as_sensor_algs[as_type][algorithm][0]
|
|
@@ -138,8 +148,139 @@ def alphasense_803_04(dataframe, **kwargs):
|
|
|
138
148
|
# Calculate sensor concentration
|
|
139
149
|
df['conc'] = df['we_c'] / (cal_data['we_sensitivity_mv_ppb'] / 1000.0) # in ppb
|
|
140
150
|
|
|
141
|
-
if
|
|
142
|
-
df['conc'].clip(lower = 0
|
|
151
|
+
if process_negative_conc == 'clip_negative_conc':
|
|
152
|
+
df['conc'] = df['conc'].clip(lower = 0)
|
|
153
|
+
elif process_negative_conc == 'offset_negative_conc':
|
|
154
|
+
df['conc'] += abs(df['conc'].min())
|
|
155
|
+
|
|
156
|
+
return ProcessResult(df['conc'], StatusCode.SUCCESS)
|
|
157
|
+
|
|
158
|
+
def alphasense_als(dataframe, **kwargs):
|
|
159
|
+
"""
|
|
160
|
+
Calculates pollutant concentration based on 4 electrode sensor readings (mV)
|
|
161
|
+
and calibration ID. It adds a configurable background concentration and correction
|
|
162
|
+
based on AAN803-04
|
|
163
|
+
Parameters
|
|
164
|
+
----------
|
|
165
|
+
alphasense_id: string
|
|
166
|
+
Alphasense sensor ID (must be in calibrations.json)
|
|
167
|
+
we: string
|
|
168
|
+
Name of working electrode found in dataframe (V)
|
|
169
|
+
ae: string
|
|
170
|
+
Name of auxiliary electrode found in dataframe (V)
|
|
171
|
+
t: string
|
|
172
|
+
Name of reference temperature
|
|
173
|
+
clip_negative_conc: bool
|
|
174
|
+
Clip the negative values after the algorithm
|
|
175
|
+
offset_negative_conc: bool
|
|
176
|
+
Offset the resulting negative values for the signal
|
|
177
|
+
Returns
|
|
178
|
+
-------
|
|
179
|
+
calculation of pollutant in ppb
|
|
180
|
+
"""
|
|
181
|
+
|
|
182
|
+
# Check inputs
|
|
183
|
+
flag_error = False
|
|
184
|
+
if 'we' not in kwargs: flag_error = True
|
|
185
|
+
if 'ae' not in kwargs: flag_error = True
|
|
186
|
+
if 'alphasense_id' not in kwargs: flag_error = True
|
|
187
|
+
if 't' not in kwargs: flag_error = True
|
|
188
|
+
if 'clip_negative_conc' not in kwargs:
|
|
189
|
+
kwargs['clip_negative_conc'] = clip_negative_conc
|
|
190
|
+
|
|
191
|
+
if 'offset_negative_conc' not in kwargs:
|
|
192
|
+
kwargs['offset_negative_conc'] = offset_negative_conc
|
|
193
|
+
|
|
194
|
+
if 'lam' in kwargs:
|
|
195
|
+
lam = kwargs['lam']
|
|
196
|
+
else:
|
|
197
|
+
lam = None
|
|
198
|
+
|
|
199
|
+
if 'p' in kwargs:
|
|
200
|
+
p = kwargs['p']
|
|
201
|
+
else:
|
|
202
|
+
p = None
|
|
203
|
+
|
|
204
|
+
if flag_error:
|
|
205
|
+
logger.error('Problem with input data')
|
|
206
|
+
return ProcessResult(None, StatusCode.ERROR_MISSING_INPUTS)
|
|
207
|
+
|
|
208
|
+
if kwargs['alphasense_id'] is None:
|
|
209
|
+
logger.warning(f"Empty ID. Ignoring")
|
|
210
|
+
return ProcessResult(None, StatusCode.WARNING_EMPTY_ID)
|
|
211
|
+
|
|
212
|
+
# Get Sensor data
|
|
213
|
+
if kwargs['alphasense_id'] not in config.calibrations:
|
|
214
|
+
logger.error(f"Sensor {kwargs['alphasense_id']} not in calibration data")
|
|
215
|
+
return ProcessResult(None, StatusCode.ERROR_CALIBRATION_NOT_FOUND)
|
|
216
|
+
|
|
217
|
+
# Make copy
|
|
218
|
+
df = dataframe.copy()
|
|
219
|
+
|
|
220
|
+
# Get sensor type
|
|
221
|
+
as_type = alphasense_sensor_codes[kwargs['alphasense_id'][0:3]]
|
|
222
|
+
|
|
223
|
+
# Retrieve calibration data - verify its all float
|
|
224
|
+
cal_data = config.calibrations[kwargs['alphasense_id']]
|
|
225
|
+
|
|
226
|
+
for item in cal_data:
|
|
227
|
+
try:
|
|
228
|
+
cal_data[item] = float (cal_data[item])
|
|
229
|
+
except:
|
|
230
|
+
logger.error(f"Alphasense calibration data for {kwargs['alphasense_id']} is not correct")
|
|
231
|
+
logger.error(f'Error on {item}: \'{cal_data[item]}\'')
|
|
232
|
+
return ProcessResult(None, StatusCode.ERROR_WRONG_CALIBRATION)
|
|
233
|
+
|
|
234
|
+
# Remove spurious voltages (0V < electrode < 5V)
|
|
235
|
+
for electrode in ['we', 'ae']:
|
|
236
|
+
subkwargs = {'name': kwargs[electrode],
|
|
237
|
+
'limits': (0, 5), # In V
|
|
238
|
+
'window_size': None
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
cleaning_result = clean_ts(df, **subkwargs)
|
|
242
|
+
if 'SUCCESS' in cleaning_result.status_code.name:
|
|
243
|
+
df[f'{electrode}_clean'] = cleaning_result.data
|
|
244
|
+
else:
|
|
245
|
+
return ProcessResult(None, StatusCode.ERROR_UNDEFINED)
|
|
246
|
+
|
|
247
|
+
# Compensate electronic zero
|
|
248
|
+
df['we_t'] = df['we_clean'] - (cal_data['we_electronic_zero_mv'] / 1000) # in V
|
|
249
|
+
df['ae_t'] = df['ae_clean'] - (cal_data['ae_electronic_zero_mv'] / 1000) # in V
|
|
250
|
+
# Get requested temperature
|
|
251
|
+
df['t'] = df[kwargs['t']]
|
|
252
|
+
|
|
253
|
+
# Calculate WE baseline
|
|
254
|
+
subkwargs = {'name': 'we_t'}
|
|
255
|
+
if lam is not None: subkwargs['lam'] = lam
|
|
256
|
+
if p is not None: subkwargs['p'] = p
|
|
257
|
+
|
|
258
|
+
baseline_result = baseline_als(df, **subkwargs)
|
|
259
|
+
if 'SUCCESS' in baseline_result.status_code.name:
|
|
260
|
+
df[f'we_baseline'] = baseline_result.data
|
|
261
|
+
else:
|
|
262
|
+
return ProcessResult(None, StatusCode.ERROR_UNDEFINED)
|
|
263
|
+
|
|
264
|
+
# Calculate baseline factor
|
|
265
|
+
df['baseline_t'] = df['we_baseline'] / df['ae_t']
|
|
266
|
+
# Correct Auxiliary electrode based on mean factor
|
|
267
|
+
df['ae_cor'] = df['baseline_t'].mean() * df['ae_t']
|
|
268
|
+
df['we_c'] = df['we_t'] - df['ae_cor']
|
|
269
|
+
|
|
270
|
+
# Verify if it has NO2 cross-sensitivity (in V)
|
|
271
|
+
if cal_data['we_cross_sensitivity_no2_mv_ppb'] != float (0) and 'NO2' not in as_type:
|
|
272
|
+
df['we_no2_eq'] = df['NO2'] * cal_data['we_cross_sensitivity_no2_mv_ppb'] / 1000.0
|
|
273
|
+
df['we_c'] -= df['we_no2_eq'] # in V
|
|
274
|
+
|
|
275
|
+
# Calculate sensor concentration
|
|
276
|
+
df['conc'] = df['we_c'] / (cal_data['we_sensitivity_mv_ppb'] / 1000.0) # in ppb
|
|
277
|
+
|
|
278
|
+
if kwargs['clip_negative_conc']:
|
|
279
|
+
df['conc'] = df['conc'].clip(lower = 0)
|
|
280
|
+
|
|
281
|
+
elif kwargs['offset_negative_conc']:
|
|
282
|
+
if df['conc'].min() < 0:
|
|
283
|
+
df['conc'] = df['conc'] + abs(df['conc'].min())
|
|
143
284
|
|
|
144
285
|
return ProcessResult(df['conc'], StatusCode.SUCCESS)
|
|
145
286
|
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
from numpy import arange
|
|
2
2
|
|
|
3
|
-
# Avoid negative pollutant concentrations
|
|
4
|
-
|
|
3
|
+
# Avoid negative pollutant concentrations by clipping or offseting. These are mutually exclusive
|
|
4
|
+
offset_negative_conc = False
|
|
5
|
+
clip_negative_conc = False
|
|
5
6
|
|
|
6
7
|
# Background concentrations
|
|
7
8
|
background_conc = {
|