fg-data-profiling 4.19.0__py2.py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- data_profiling/__init__.py +34 -0
- data_profiling/compare_reports.py +359 -0
- data_profiling/config.py +496 -0
- data_profiling/config_default.yaml +223 -0
- data_profiling/config_minimal.yaml +222 -0
- data_profiling/controller/__init__.py +1 -0
- data_profiling/controller/console.py +125 -0
- data_profiling/controller/pandas_decorator.py +21 -0
- data_profiling/expectations_report.py +117 -0
- data_profiling/model/__init__.py +4 -0
- data_profiling/model/alerts.py +780 -0
- data_profiling/model/correlations.py +163 -0
- data_profiling/model/dataframe.py +35 -0
- data_profiling/model/describe.py +210 -0
- data_profiling/model/description.py +108 -0
- data_profiling/model/duplicates.py +14 -0
- data_profiling/model/expectation_algorithms.py +112 -0
- data_profiling/model/handler.py +81 -0
- data_profiling/model/missing.py +146 -0
- data_profiling/model/pairwise.py +33 -0
- data_profiling/model/pandas/__init__.py +55 -0
- data_profiling/model/pandas/correlations_pandas.py +207 -0
- data_profiling/model/pandas/dataframe_pandas.py +26 -0
- data_profiling/model/pandas/describe_boolean_pandas.py +43 -0
- data_profiling/model/pandas/describe_categorical_pandas.py +274 -0
- data_profiling/model/pandas/describe_counts_pandas.py +63 -0
- data_profiling/model/pandas/describe_date_pandas.py +77 -0
- data_profiling/model/pandas/describe_file_pandas.py +56 -0
- data_profiling/model/pandas/describe_generic_pandas.py +36 -0
- data_profiling/model/pandas/describe_image_pandas.py +255 -0
- data_profiling/model/pandas/describe_numeric_pandas.py +175 -0
- data_profiling/model/pandas/describe_path_pandas.py +63 -0
- data_profiling/model/pandas/describe_supported_pandas.py +41 -0
- data_profiling/model/pandas/describe_text_pandas.py +62 -0
- data_profiling/model/pandas/describe_timeseries_pandas.py +222 -0
- data_profiling/model/pandas/describe_url_pandas.py +57 -0
- data_profiling/model/pandas/discretize_pandas.py +81 -0
- data_profiling/model/pandas/duplicates_pandas.py +56 -0
- data_profiling/model/pandas/imbalance_pandas.py +35 -0
- data_profiling/model/pandas/missing_pandas.py +42 -0
- data_profiling/model/pandas/sample_pandas.py +38 -0
- data_profiling/model/pandas/summary_pandas.py +101 -0
- data_profiling/model/pandas/table_pandas.py +56 -0
- data_profiling/model/pandas/timeseries_index_pandas.py +33 -0
- data_profiling/model/pandas/utils_pandas.py +27 -0
- data_profiling/model/sample.py +37 -0
- data_profiling/model/spark/__init__.py +48 -0
- data_profiling/model/spark/correlations_spark.py +152 -0
- data_profiling/model/spark/dataframe_spark.py +34 -0
- data_profiling/model/spark/describe_boolean_spark.py +27 -0
- data_profiling/model/spark/describe_categorical_spark.py +28 -0
- data_profiling/model/spark/describe_counts_spark.py +105 -0
- data_profiling/model/spark/describe_date_spark.py +51 -0
- data_profiling/model/spark/describe_generic_spark.py +30 -0
- data_profiling/model/spark/describe_numeric_spark.py +155 -0
- data_profiling/model/spark/describe_supported_spark.py +33 -0
- data_profiling/model/spark/describe_text_spark.py +25 -0
- data_profiling/model/spark/duplicates_spark.py +54 -0
- data_profiling/model/spark/missing_spark.py +96 -0
- data_profiling/model/spark/sample_spark.py +43 -0
- data_profiling/model/spark/summary_spark.py +95 -0
- data_profiling/model/spark/table_spark.py +58 -0
- data_profiling/model/spark/timeseries_index_spark.py +12 -0
- data_profiling/model/summarizer.py +207 -0
- data_profiling/model/summary.py +66 -0
- data_profiling/model/summary_algorithms.py +276 -0
- data_profiling/model/table.py +10 -0
- data_profiling/model/timeseries_index.py +16 -0
- data_profiling/model/typeset.py +365 -0
- data_profiling/model/typeset_relations.py +143 -0
- data_profiling/profile_report.py +573 -0
- data_profiling/report/__init__.py +4 -0
- data_profiling/report/formatters.py +346 -0
- data_profiling/report/presentation/__init__.py +1 -0
- data_profiling/report/presentation/core/__init__.py +39 -0
- data_profiling/report/presentation/core/alerts.py +18 -0
- data_profiling/report/presentation/core/collapse.py +24 -0
- data_profiling/report/presentation/core/container.py +50 -0
- data_profiling/report/presentation/core/correlation_table.py +21 -0
- data_profiling/report/presentation/core/dropdown.py +44 -0
- data_profiling/report/presentation/core/duplicate.py +16 -0
- data_profiling/report/presentation/core/frequency_table.py +14 -0
- data_profiling/report/presentation/core/frequency_table_small.py +16 -0
- data_profiling/report/presentation/core/html.py +14 -0
- data_profiling/report/presentation/core/image.py +34 -0
- data_profiling/report/presentation/core/item_renderer.py +17 -0
- data_profiling/report/presentation/core/renderable.py +42 -0
- data_profiling/report/presentation/core/root.py +35 -0
- data_profiling/report/presentation/core/sample.py +20 -0
- data_profiling/report/presentation/core/scores.py +32 -0
- data_profiling/report/presentation/core/table.py +26 -0
- data_profiling/report/presentation/core/toggle_button.py +14 -0
- data_profiling/report/presentation/core/variable.py +40 -0
- data_profiling/report/presentation/core/variable_info.py +36 -0
- data_profiling/report/presentation/flavours/__init__.py +9 -0
- data_profiling/report/presentation/flavours/flavour_html.py +64 -0
- data_profiling/report/presentation/flavours/flavour_widget.py +61 -0
- data_profiling/report/presentation/flavours/flavours.py +43 -0
- data_profiling/report/presentation/flavours/html/__init__.py +47 -0
- data_profiling/report/presentation/flavours/html/alerts.py +10 -0
- data_profiling/report/presentation/flavours/html/collapse.py +7 -0
- data_profiling/report/presentation/flavours/html/container.py +58 -0
- data_profiling/report/presentation/flavours/html/correlation_table.py +13 -0
- data_profiling/report/presentation/flavours/html/dropdown.py +7 -0
- data_profiling/report/presentation/flavours/html/duplicate.py +24 -0
- data_profiling/report/presentation/flavours/html/frequency_table.py +20 -0
- data_profiling/report/presentation/flavours/html/frequency_table_small.py +15 -0
- data_profiling/report/presentation/flavours/html/html.py +6 -0
- data_profiling/report/presentation/flavours/html/image.py +7 -0
- data_profiling/report/presentation/flavours/html/root.py +14 -0
- data_profiling/report/presentation/flavours/html/sample.py +12 -0
- data_profiling/report/presentation/flavours/html/scores.py +11 -0
- data_profiling/report/presentation/flavours/html/table.py +7 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant_length.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_dirty_category.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_duplicates.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_empty.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_cardinality.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_correlation.html +4 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_imbalance.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_infinite.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_missing.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_near_duplicates.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_non_stationary.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_seasonal.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_skewed.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_truncated.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_type_date.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_uniform.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_unique.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_unsupported.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_zeros.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts.html +47 -0
- data_profiling/report/presentation/flavours/html/templates/collapse.html +11 -0
- data_profiling/report/presentation/flavours/html/templates/correlation_table.html +5 -0
- data_profiling/report/presentation/flavours/html/templates/diagram.html +11 -0
- data_profiling/report/presentation/flavours/html/templates/dropdown.html +16 -0
- data_profiling/report/presentation/flavours/html/templates/duplicate.html +5 -0
- data_profiling/report/presentation/flavours/html/templates/frequency_table.html +45 -0
- data_profiling/report/presentation/flavours/html/templates/frequency_table_small.html +34 -0
- data_profiling/report/presentation/flavours/html/templates/report.html +26 -0
- data_profiling/report/presentation/flavours/html/templates/sample.html +10 -0
- data_profiling/report/presentation/flavours/html/templates/scores.html +78 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/batch_grid.html +16 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/grid.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/list.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/named_list.html +8 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/overview_tabs.html +30 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/scores.html +3 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/sections.html +13 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/select.html +40 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/tabs.html +30 -0
- data_profiling/report/presentation/flavours/html/templates/table.html +38 -0
- data_profiling/report/presentation/flavours/html/templates/toggle_button.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/variable.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/variable_info.html +49 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.bundle.min.js +7 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.min.css +6 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/cosmo.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/flatly.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/script.js +52 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/simplex.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/style.css +253 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/united.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/footer.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/javascript.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/navigation.html +36 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/style.html +53 -0
- data_profiling/report/presentation/flavours/html/templates.py +76 -0
- data_profiling/report/presentation/flavours/html/toggle_button.py +7 -0
- data_profiling/report/presentation/flavours/html/variable.py +7 -0
- data_profiling/report/presentation/flavours/html/variable_info.py +7 -0
- data_profiling/report/presentation/flavours/widget/__init__.py +49 -0
- data_profiling/report/presentation/flavours/widget/alerts.py +45 -0
- data_profiling/report/presentation/flavours/widget/collapse.py +43 -0
- data_profiling/report/presentation/flavours/widget/container.py +121 -0
- data_profiling/report/presentation/flavours/widget/correlation_table.py +14 -0
- data_profiling/report/presentation/flavours/widget/dropdown.py +31 -0
- data_profiling/report/presentation/flavours/widget/duplicate.py +14 -0
- data_profiling/report/presentation/flavours/widget/frequency_table.py +57 -0
- data_profiling/report/presentation/flavours/widget/frequency_table_small.py +66 -0
- data_profiling/report/presentation/flavours/widget/html.py +11 -0
- data_profiling/report/presentation/flavours/widget/image.py +26 -0
- data_profiling/report/presentation/flavours/widget/notebook.py +81 -0
- data_profiling/report/presentation/flavours/widget/root.py +10 -0
- data_profiling/report/presentation/flavours/widget/sample.py +14 -0
- data_profiling/report/presentation/flavours/widget/table.py +30 -0
- data_profiling/report/presentation/flavours/widget/toggle_button.py +17 -0
- data_profiling/report/presentation/flavours/widget/variable.py +12 -0
- data_profiling/report/presentation/flavours/widget/variable_info.py +11 -0
- data_profiling/report/presentation/frequency_table_utils.py +141 -0
- data_profiling/report/structure/__init__.py +1 -0
- data_profiling/report/structure/correlations.py +123 -0
- data_profiling/report/structure/overview.py +376 -0
- data_profiling/report/structure/report.py +457 -0
- data_profiling/report/structure/variables/__init__.py +35 -0
- data_profiling/report/structure/variables/render_boolean.py +132 -0
- data_profiling/report/structure/variables/render_categorical.py +566 -0
- data_profiling/report/structure/variables/render_common.py +31 -0
- data_profiling/report/structure/variables/render_complex.py +102 -0
- data_profiling/report/structure/variables/render_count.py +172 -0
- data_profiling/report/structure/variables/render_date.py +143 -0
- data_profiling/report/structure/variables/render_file.py +70 -0
- data_profiling/report/structure/variables/render_generic.py +45 -0
- data_profiling/report/structure/variables/render_image.py +204 -0
- data_profiling/report/structure/variables/render_path.py +134 -0
- data_profiling/report/structure/variables/render_real.py +314 -0
- data_profiling/report/structure/variables/render_text.py +189 -0
- data_profiling/report/structure/variables/render_timeseries.py +371 -0
- data_profiling/report/structure/variables/render_url.py +132 -0
- data_profiling/report/utils.py +34 -0
- data_profiling/serialize_report.py +143 -0
- data_profiling/utils/__init__.py +1 -0
- data_profiling/utils/backend.py +9 -0
- data_profiling/utils/cache.py +59 -0
- data_profiling/utils/common.py +142 -0
- data_profiling/utils/compat.py +31 -0
- data_profiling/utils/dataframe.py +238 -0
- data_profiling/utils/logger.py +53 -0
- data_profiling/utils/notebook.py +8 -0
- data_profiling/utils/paths.py +45 -0
- data_profiling/utils/progress_bar.py +15 -0
- data_profiling/utils/styles.py +22 -0
- data_profiling/utils/versions.py +19 -0
- data_profiling/version.py +1 -0
- data_profiling/visualisation/__init__.py +1 -0
- data_profiling/visualisation/context.py +87 -0
- data_profiling/visualisation/missing.py +138 -0
- data_profiling/visualisation/plot.py +1158 -0
- data_profiling/visualisation/utils.py +113 -0
- fg_data_profiling-4.19.0.dist-info/METADATA +362 -0
- fg_data_profiling-4.19.0.dist-info/RECORD +238 -0
- fg_data_profiling-4.19.0.dist-info/WHEEL +6 -0
- fg_data_profiling-4.19.0.dist-info/entry_points.txt +3 -0
- fg_data_profiling-4.19.0.dist-info/licenses/LICENSE +21 -0
- fg_data_profiling-4.19.0.dist-info/top_level.txt +2 -0
- ydata_profiling/__init__.py +43 -0
|
@@ -0,0 +1,780 @@
|
|
|
1
|
+
"""Logic for alerting the user on possibly problematic patterns in the data (e.g. high number of zeros , constant
|
|
2
|
+
values, high correlations)."""
|
|
3
|
+
|
|
4
|
+
from enum import Enum, auto, unique
|
|
5
|
+
from typing import Dict, List, Optional, Set
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
import pandas as pd
|
|
9
|
+
|
|
10
|
+
from data_profiling.config import Settings
|
|
11
|
+
from data_profiling.model.correlations import perform_check_correlation
|
|
12
|
+
from data_profiling.utils.styles import get_alert_styles
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def fmt_percent(value: float, edge_cases: bool = True) -> str:
|
|
16
|
+
"""Format a ratio as a percentage.
|
|
17
|
+
|
|
18
|
+
Args:
|
|
19
|
+
edge_cases: Check for edge cases?
|
|
20
|
+
value: The ratio.
|
|
21
|
+
|
|
22
|
+
Returns:
|
|
23
|
+
The percentage with 1 point precision.
|
|
24
|
+
"""
|
|
25
|
+
if edge_cases and round(value, 3) == 0 and value > 0:
|
|
26
|
+
return "< 0.1%"
|
|
27
|
+
if edge_cases and round(value, 3) == 1 and value < 1:
|
|
28
|
+
return "> 99.9%"
|
|
29
|
+
|
|
30
|
+
return f"{value*100:2.1f}%"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@unique
|
|
34
|
+
class AlertType(Enum):
|
|
35
|
+
"""Alert types"""
|
|
36
|
+
|
|
37
|
+
CONSTANT = auto()
|
|
38
|
+
"""This variable has a constant value."""
|
|
39
|
+
|
|
40
|
+
ZEROS = auto()
|
|
41
|
+
"""This variable contains zeros."""
|
|
42
|
+
|
|
43
|
+
HIGH_CORRELATION = auto()
|
|
44
|
+
"""This variable is highly correlated."""
|
|
45
|
+
|
|
46
|
+
HIGH_CARDINALITY = auto()
|
|
47
|
+
"""This variable has a high cardinality."""
|
|
48
|
+
|
|
49
|
+
UNSUPPORTED = auto()
|
|
50
|
+
"""This variable is unsupported."""
|
|
51
|
+
|
|
52
|
+
DUPLICATES = auto()
|
|
53
|
+
"""This variable contains duplicates."""
|
|
54
|
+
|
|
55
|
+
NEAR_DUPLICATES = auto()
|
|
56
|
+
"""This variable contains duplicates."""
|
|
57
|
+
|
|
58
|
+
SKEWED = auto()
|
|
59
|
+
"""This variable is highly skewed."""
|
|
60
|
+
|
|
61
|
+
IMBALANCE = auto()
|
|
62
|
+
"""This variable is imbalanced."""
|
|
63
|
+
|
|
64
|
+
MISSING = auto()
|
|
65
|
+
"""This variable contains missing values."""
|
|
66
|
+
|
|
67
|
+
INFINITE = auto()
|
|
68
|
+
"""This variable contains infinite values."""
|
|
69
|
+
|
|
70
|
+
TYPE_DATE = auto()
|
|
71
|
+
"""This variable is likely a datetime, but treated as categorical."""
|
|
72
|
+
|
|
73
|
+
UNIQUE = auto()
|
|
74
|
+
"""This variable has unique values."""
|
|
75
|
+
|
|
76
|
+
DIRTY_CATEGORY = auto()
|
|
77
|
+
"""This variable is a categories with potential fuzzy values, and for that reason might incur in consistency issues."""
|
|
78
|
+
|
|
79
|
+
CONSTANT_LENGTH = auto()
|
|
80
|
+
"""This variable has a constant length."""
|
|
81
|
+
|
|
82
|
+
REJECTED = auto()
|
|
83
|
+
"""Variables are rejected if we do not want to consider them for further analysis."""
|
|
84
|
+
|
|
85
|
+
UNIFORM = auto()
|
|
86
|
+
"""The variable is uniformly distributed."""
|
|
87
|
+
|
|
88
|
+
NON_STATIONARY = auto()
|
|
89
|
+
"""The variable is a non-stationary series."""
|
|
90
|
+
|
|
91
|
+
SEASONAL = auto()
|
|
92
|
+
"""The variable is a seasonal time series."""
|
|
93
|
+
|
|
94
|
+
EMPTY = auto()
|
|
95
|
+
"""The DataFrame is empty."""
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class Alert:
|
|
99
|
+
"""An alert object (type, values, column)."""
|
|
100
|
+
|
|
101
|
+
_anchor_id: Optional[str] = None
|
|
102
|
+
|
|
103
|
+
def __init__(
|
|
104
|
+
self,
|
|
105
|
+
alert_type: AlertType,
|
|
106
|
+
values: Optional[Dict] = None,
|
|
107
|
+
column_name: Optional[str] = None,
|
|
108
|
+
fields: Optional[Set] = None,
|
|
109
|
+
is_empty: bool = False,
|
|
110
|
+
):
|
|
111
|
+
self.fields = fields or set()
|
|
112
|
+
self.alert_type = alert_type
|
|
113
|
+
self.values = values or {}
|
|
114
|
+
self.column_name = column_name
|
|
115
|
+
self._is_empty = is_empty
|
|
116
|
+
self._styles = get_alert_styles()
|
|
117
|
+
|
|
118
|
+
@property
|
|
119
|
+
def alert_type_name(self) -> str:
|
|
120
|
+
return self.alert_type.name.replace("_", " ").capitalize()
|
|
121
|
+
|
|
122
|
+
@property
|
|
123
|
+
def anchor_id(self) -> Optional[str]:
|
|
124
|
+
if self._anchor_id is None:
|
|
125
|
+
self._anchor_id = str(hash(self.column_name))
|
|
126
|
+
return self._anchor_id
|
|
127
|
+
|
|
128
|
+
def fmt(self) -> str:
|
|
129
|
+
# TODO: render in template
|
|
130
|
+
style = self._styles.get(self.alert_type.name.lower(), "secondary")
|
|
131
|
+
hint = ""
|
|
132
|
+
|
|
133
|
+
if self.alert_type == AlertType.HIGH_CORRELATION and self.values is not None:
|
|
134
|
+
num = len(self.values["fields"])
|
|
135
|
+
title = ", ".join(self.values["fields"])
|
|
136
|
+
corr = self.values["corr"]
|
|
137
|
+
hint = f'data-bs-toggle="tooltip" data-bs-placement="right" data-bs-title="This variable has a high {corr} correlation with {num} fields: {title}"'
|
|
138
|
+
|
|
139
|
+
return (
|
|
140
|
+
f'<span class="badge text-bg-{style}" {hint}>{self.alert_type_name}</span>'
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
def _get_description(self) -> str:
|
|
144
|
+
"""Return a human level description of the alert.
|
|
145
|
+
|
|
146
|
+
Returns:
|
|
147
|
+
str: alert description
|
|
148
|
+
"""
|
|
149
|
+
alert_type = self.alert_type.name
|
|
150
|
+
column = self.column_name
|
|
151
|
+
return f"[{alert_type}] alert on column {column}"
|
|
152
|
+
|
|
153
|
+
def __repr__(self):
|
|
154
|
+
return self._get_description()
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
class ConstantLengthAlert(Alert):
|
|
158
|
+
def __init__(
|
|
159
|
+
self,
|
|
160
|
+
values: Optional[Dict] = None,
|
|
161
|
+
column_name: Optional[str] = None,
|
|
162
|
+
is_empty: bool = False,
|
|
163
|
+
):
|
|
164
|
+
super().__init__(
|
|
165
|
+
alert_type=AlertType.CONSTANT_LENGTH,
|
|
166
|
+
values=values,
|
|
167
|
+
column_name=column_name,
|
|
168
|
+
fields={"composition_min_length", "composition_max_length"},
|
|
169
|
+
is_empty=is_empty,
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
def _get_description(self) -> str:
|
|
173
|
+
return f"[{self.column_name}] has a constant length"
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
class ConstantAlert(Alert):
|
|
177
|
+
def __init__(
|
|
178
|
+
self,
|
|
179
|
+
values: Optional[Dict] = None,
|
|
180
|
+
column_name: Optional[str] = None,
|
|
181
|
+
is_empty: bool = False,
|
|
182
|
+
):
|
|
183
|
+
super().__init__(
|
|
184
|
+
alert_type=AlertType.CONSTANT,
|
|
185
|
+
values=values,
|
|
186
|
+
column_name=column_name,
|
|
187
|
+
fields={"n_distinct"},
|
|
188
|
+
is_empty=is_empty,
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
def _get_description(self) -> str:
|
|
192
|
+
return f"[{self.column_name}] has a constant value"
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
class DuplicatesAlert(Alert):
|
|
196
|
+
def __init__(
|
|
197
|
+
self,
|
|
198
|
+
values: Optional[Dict] = None,
|
|
199
|
+
column_name: Optional[str] = None,
|
|
200
|
+
is_empty: bool = False,
|
|
201
|
+
):
|
|
202
|
+
super().__init__(
|
|
203
|
+
alert_type=AlertType.DUPLICATES,
|
|
204
|
+
values=values,
|
|
205
|
+
column_name=column_name,
|
|
206
|
+
fields={"n_duplicates"},
|
|
207
|
+
is_empty=is_empty,
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
def _get_description(self) -> str:
|
|
211
|
+
if self.values is not None:
|
|
212
|
+
return f"Dataset has {self.values['n_duplicates']} ({fmt_percent(self.values['p_duplicates'])}) duplicate rows"
|
|
213
|
+
else:
|
|
214
|
+
return "Dataset has no duplicated rows"
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
class NearDuplicatesAlert(Alert):
|
|
218
|
+
def __init__(
|
|
219
|
+
self,
|
|
220
|
+
values: Optional[Dict] = None,
|
|
221
|
+
column_name: Optional[str] = None,
|
|
222
|
+
is_empty: bool = False,
|
|
223
|
+
):
|
|
224
|
+
super().__init__(
|
|
225
|
+
alert_type=AlertType.NEAR_DUPLICATES,
|
|
226
|
+
values=values,
|
|
227
|
+
column_name=column_name,
|
|
228
|
+
fields={"n_near_dups"},
|
|
229
|
+
is_empty=is_empty,
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
def _get_description(self) -> str:
|
|
233
|
+
if self.values is not None:
|
|
234
|
+
return f"Dataset has {self.values['n_near_dups']} ({fmt_percent(self.values['p_near_dups'])}) near duplicate rows"
|
|
235
|
+
else:
|
|
236
|
+
return "Dataset has no near duplicated rows"
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
class EmptyAlert(Alert):
|
|
240
|
+
def __init__(
|
|
241
|
+
self,
|
|
242
|
+
values: Optional[Dict] = None,
|
|
243
|
+
column_name: Optional[str] = None,
|
|
244
|
+
is_empty: bool = False,
|
|
245
|
+
):
|
|
246
|
+
super().__init__(
|
|
247
|
+
alert_type=AlertType.EMPTY,
|
|
248
|
+
values=values,
|
|
249
|
+
column_name=column_name,
|
|
250
|
+
fields={"n"},
|
|
251
|
+
is_empty=is_empty,
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
def _get_description(self) -> str:
|
|
255
|
+
return "Dataset is empty"
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
class HighCardinalityAlert(Alert):
|
|
259
|
+
def __init__(
|
|
260
|
+
self,
|
|
261
|
+
values: Optional[Dict] = None,
|
|
262
|
+
column_name: Optional[str] = None,
|
|
263
|
+
is_empty: bool = False,
|
|
264
|
+
):
|
|
265
|
+
super().__init__(
|
|
266
|
+
alert_type=AlertType.HIGH_CARDINALITY,
|
|
267
|
+
values=values,
|
|
268
|
+
column_name=column_name,
|
|
269
|
+
fields={"n_distinct"},
|
|
270
|
+
is_empty=is_empty,
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
def _get_description(self) -> str:
|
|
274
|
+
if self.values is not None:
|
|
275
|
+
return f"[{self.column_name}] has {self.values['n_distinct']:} ({fmt_percent(self.values['p_distinct'])}) distinct values"
|
|
276
|
+
else:
|
|
277
|
+
return f"[{self.column_name}] has a high cardinality"
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
class DirtyCategoryAlert(Alert):
|
|
281
|
+
def __init__(
|
|
282
|
+
self,
|
|
283
|
+
values: Optional[Dict] = None,
|
|
284
|
+
column_name: Optional[str] = None,
|
|
285
|
+
is_empty: bool = False,
|
|
286
|
+
):
|
|
287
|
+
super().__init__(
|
|
288
|
+
alert_type=AlertType.DIRTY_CATEGORY,
|
|
289
|
+
values=values,
|
|
290
|
+
column_name=column_name,
|
|
291
|
+
fields={"n_fuzzy_vals"},
|
|
292
|
+
is_empty=is_empty,
|
|
293
|
+
)
|
|
294
|
+
|
|
295
|
+
def _get_description(self) -> str:
|
|
296
|
+
if self.values is not None:
|
|
297
|
+
return f"[{self.column_name}] has {self.values['n_fuzzy_vals']} fuzzy values: {fmt_percent(self.values['p_fuzzy_vals'])} per category"
|
|
298
|
+
else:
|
|
299
|
+
return f"[{self.column_name}] no dirty categories values."
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
class HighCorrelationAlert(Alert):
|
|
303
|
+
def __init__(
|
|
304
|
+
self,
|
|
305
|
+
values: Optional[Dict] = None,
|
|
306
|
+
column_name: Optional[str] = None,
|
|
307
|
+
is_empty: bool = False,
|
|
308
|
+
):
|
|
309
|
+
super().__init__(
|
|
310
|
+
alert_type=AlertType.HIGH_CORRELATION,
|
|
311
|
+
values=values,
|
|
312
|
+
column_name=column_name,
|
|
313
|
+
is_empty=is_empty,
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
def _get_description(self) -> str:
|
|
317
|
+
if self.values is not None:
|
|
318
|
+
description = f"[{self.column_name}] is highly {self.values['corr']} correlated with [{self.values['fields'][0]}]"
|
|
319
|
+
if len(self.values["fields"]) > 1:
|
|
320
|
+
description += f" and {len(self.values['fields']) - 1} other fields"
|
|
321
|
+
else:
|
|
322
|
+
return (
|
|
323
|
+
f"[{self.column_name}] has a high correlation with one or more colums"
|
|
324
|
+
)
|
|
325
|
+
return description
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
class ImbalanceAlert(Alert):
|
|
329
|
+
def __init__(
|
|
330
|
+
self,
|
|
331
|
+
values: Optional[Dict] = None,
|
|
332
|
+
column_name: Optional[str] = None,
|
|
333
|
+
is_empty: bool = False,
|
|
334
|
+
):
|
|
335
|
+
super().__init__(
|
|
336
|
+
alert_type=AlertType.IMBALANCE,
|
|
337
|
+
values=values,
|
|
338
|
+
column_name=column_name,
|
|
339
|
+
fields={"imbalance"},
|
|
340
|
+
is_empty=is_empty,
|
|
341
|
+
)
|
|
342
|
+
|
|
343
|
+
def _get_description(self) -> str:
|
|
344
|
+
description = f"[{self.column_name}] is highly imbalanced"
|
|
345
|
+
if self.values is not None:
|
|
346
|
+
return description + f" ({self.values['imbalance']})"
|
|
347
|
+
else:
|
|
348
|
+
return description
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
class InfiniteAlert(Alert):
|
|
352
|
+
def __init__(
|
|
353
|
+
self,
|
|
354
|
+
values: Optional[Dict] = None,
|
|
355
|
+
column_name: Optional[str] = None,
|
|
356
|
+
is_empty: bool = False,
|
|
357
|
+
):
|
|
358
|
+
super().__init__(
|
|
359
|
+
alert_type=AlertType.INFINITE,
|
|
360
|
+
values=values,
|
|
361
|
+
column_name=column_name,
|
|
362
|
+
fields={"p_infinite", "n_infinite"},
|
|
363
|
+
is_empty=is_empty,
|
|
364
|
+
)
|
|
365
|
+
|
|
366
|
+
def _get_description(self) -> str:
|
|
367
|
+
if self.values is not None:
|
|
368
|
+
return f"[{self.column_name}] has {self.values['n_infinite']} ({fmt_percent(self.values['p_infinite'])}) infinite values"
|
|
369
|
+
else:
|
|
370
|
+
return f"[{self.column_name}] has infinite values"
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
class MissingAlert(Alert):
|
|
374
|
+
def __init__(
|
|
375
|
+
self,
|
|
376
|
+
values: Optional[Dict] = None,
|
|
377
|
+
column_name: Optional[str] = None,
|
|
378
|
+
is_empty: bool = False,
|
|
379
|
+
):
|
|
380
|
+
super().__init__(
|
|
381
|
+
alert_type=AlertType.MISSING,
|
|
382
|
+
values=values,
|
|
383
|
+
column_name=column_name,
|
|
384
|
+
fields={"p_missing", "n_missing"},
|
|
385
|
+
is_empty=is_empty,
|
|
386
|
+
)
|
|
387
|
+
|
|
388
|
+
def _get_description(self) -> str:
|
|
389
|
+
if self.values is not None:
|
|
390
|
+
return f"[{self.column_name}] {self.values['n_missing']} ({fmt_percent(self.values['p_missing'])}) missing values"
|
|
391
|
+
else:
|
|
392
|
+
return f"[{self.column_name}] has missing values"
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
class NonStationaryAlert(Alert):
|
|
396
|
+
def __init__(
|
|
397
|
+
self,
|
|
398
|
+
values: Optional[Dict] = None,
|
|
399
|
+
column_name: Optional[str] = None,
|
|
400
|
+
is_empty: bool = False,
|
|
401
|
+
):
|
|
402
|
+
super().__init__(
|
|
403
|
+
alert_type=AlertType.NON_STATIONARY,
|
|
404
|
+
values=values,
|
|
405
|
+
column_name=column_name,
|
|
406
|
+
is_empty=is_empty,
|
|
407
|
+
)
|
|
408
|
+
|
|
409
|
+
def _get_description(self) -> str:
|
|
410
|
+
return f"[{self.column_name}] is non stationary"
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
class SeasonalAlert(Alert):
|
|
414
|
+
def __init__(
|
|
415
|
+
self,
|
|
416
|
+
values: Optional[Dict] = None,
|
|
417
|
+
column_name: Optional[str] = None,
|
|
418
|
+
is_empty: bool = False,
|
|
419
|
+
):
|
|
420
|
+
super().__init__(
|
|
421
|
+
alert_type=AlertType.SEASONAL,
|
|
422
|
+
values=values,
|
|
423
|
+
column_name=column_name,
|
|
424
|
+
is_empty=is_empty,
|
|
425
|
+
)
|
|
426
|
+
|
|
427
|
+
def _get_description(self) -> str:
|
|
428
|
+
return f"[{self.column_name}] is seasonal"
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
class SkewedAlert(Alert):
|
|
432
|
+
def __init__(
|
|
433
|
+
self,
|
|
434
|
+
values: Optional[Dict] = None,
|
|
435
|
+
column_name: Optional[str] = None,
|
|
436
|
+
is_empty: bool = False,
|
|
437
|
+
):
|
|
438
|
+
super().__init__(
|
|
439
|
+
alert_type=AlertType.SKEWED,
|
|
440
|
+
values=values,
|
|
441
|
+
column_name=column_name,
|
|
442
|
+
fields={"skewness"},
|
|
443
|
+
is_empty=is_empty,
|
|
444
|
+
)
|
|
445
|
+
|
|
446
|
+
def _get_description(self) -> str:
|
|
447
|
+
description = f"[{self.column_name}] is highly skewed"
|
|
448
|
+
if self.values is not None:
|
|
449
|
+
return description + f"(\u03b31 = {self.values['skewness']})"
|
|
450
|
+
else:
|
|
451
|
+
return description
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
class TypeDateAlert(Alert):
|
|
455
|
+
def __init__(
|
|
456
|
+
self,
|
|
457
|
+
values: Optional[Dict] = None,
|
|
458
|
+
column_name: Optional[str] = None,
|
|
459
|
+
is_empty: bool = False,
|
|
460
|
+
):
|
|
461
|
+
super().__init__(
|
|
462
|
+
alert_type=AlertType.TYPE_DATE,
|
|
463
|
+
values=values,
|
|
464
|
+
column_name=column_name,
|
|
465
|
+
is_empty=is_empty,
|
|
466
|
+
)
|
|
467
|
+
|
|
468
|
+
def _get_description(self) -> str:
|
|
469
|
+
return f"[{self.column_name}] only contains datetime values, but is categorical. Consider applying `pd.to_datetime()`"
|
|
470
|
+
|
|
471
|
+
|
|
472
|
+
class UniformAlert(Alert):
|
|
473
|
+
def __init__(
|
|
474
|
+
self,
|
|
475
|
+
values: Optional[Dict] = None,
|
|
476
|
+
column_name: Optional[str] = None,
|
|
477
|
+
is_empty: bool = False,
|
|
478
|
+
):
|
|
479
|
+
super().__init__(
|
|
480
|
+
alert_type=AlertType.UNIFORM,
|
|
481
|
+
values=values,
|
|
482
|
+
column_name=column_name,
|
|
483
|
+
is_empty=is_empty,
|
|
484
|
+
)
|
|
485
|
+
|
|
486
|
+
def _get_description(self) -> str:
|
|
487
|
+
return f"[{self.column_name}] is uniformly distributed"
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
class UniqueAlert(Alert):
|
|
491
|
+
def __init__(
|
|
492
|
+
self,
|
|
493
|
+
values: Optional[Dict] = None,
|
|
494
|
+
column_name: Optional[str] = None,
|
|
495
|
+
is_empty: bool = False,
|
|
496
|
+
):
|
|
497
|
+
super().__init__(
|
|
498
|
+
alert_type=AlertType.UNIQUE,
|
|
499
|
+
values=values,
|
|
500
|
+
column_name=column_name,
|
|
501
|
+
fields={"n_distinct", "p_distinct", "n_unique", "p_unique"},
|
|
502
|
+
is_empty=is_empty,
|
|
503
|
+
)
|
|
504
|
+
|
|
505
|
+
def _get_description(self) -> str:
|
|
506
|
+
return f"[{self.column_name}] has unique values"
|
|
507
|
+
|
|
508
|
+
|
|
509
|
+
class UnsupportedAlert(Alert):
|
|
510
|
+
def __init__(
|
|
511
|
+
self,
|
|
512
|
+
values: Optional[Dict] = None,
|
|
513
|
+
column_name: Optional[str] = None,
|
|
514
|
+
is_empty: bool = False,
|
|
515
|
+
):
|
|
516
|
+
super().__init__(
|
|
517
|
+
alert_type=AlertType.UNSUPPORTED,
|
|
518
|
+
values=values,
|
|
519
|
+
column_name=column_name,
|
|
520
|
+
is_empty=is_empty,
|
|
521
|
+
)
|
|
522
|
+
|
|
523
|
+
def _get_description(self) -> str:
|
|
524
|
+
return f"[{self.column_name}] is an unsupported type, check if it needs cleaning or further analysis"
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
class ZerosAlert(Alert):
|
|
528
|
+
def __init__(
|
|
529
|
+
self,
|
|
530
|
+
values: Optional[Dict] = None,
|
|
531
|
+
column_name: Optional[str] = None,
|
|
532
|
+
is_empty: bool = False,
|
|
533
|
+
):
|
|
534
|
+
super().__init__(
|
|
535
|
+
alert_type=AlertType.ZEROS,
|
|
536
|
+
values=values,
|
|
537
|
+
column_name=column_name,
|
|
538
|
+
fields={"n_zeros", "p_zeros"},
|
|
539
|
+
is_empty=is_empty,
|
|
540
|
+
)
|
|
541
|
+
|
|
542
|
+
def _get_description(self) -> str:
|
|
543
|
+
if self.values is not None:
|
|
544
|
+
return f"[{self.column_name}] has {self.values['n_zeros']} ({fmt_percent(self.values['p_zeros'])}) zeros"
|
|
545
|
+
else:
|
|
546
|
+
return f"[{self.column_name}] has predominantly zeros"
|
|
547
|
+
|
|
548
|
+
|
|
549
|
+
class RejectedAlert(Alert):
|
|
550
|
+
def __init__(
|
|
551
|
+
self,
|
|
552
|
+
values: Optional[Dict] = None,
|
|
553
|
+
column_name: Optional[str] = None,
|
|
554
|
+
is_empty: bool = False,
|
|
555
|
+
):
|
|
556
|
+
super().__init__(
|
|
557
|
+
alert_type=AlertType.REJECTED,
|
|
558
|
+
values=values,
|
|
559
|
+
column_name=column_name,
|
|
560
|
+
is_empty=is_empty,
|
|
561
|
+
)
|
|
562
|
+
|
|
563
|
+
def _get_description(self) -> str:
|
|
564
|
+
return f"[{self.column_name}] was rejected"
|
|
565
|
+
|
|
566
|
+
|
|
567
|
+
def check_table_alerts(table: dict) -> List[Alert]:
|
|
568
|
+
"""Checks the overall dataset for alerts.
|
|
569
|
+
|
|
570
|
+
Args:
|
|
571
|
+
table: Overall dataset statistics.
|
|
572
|
+
|
|
573
|
+
Returns:
|
|
574
|
+
A list of alerts.
|
|
575
|
+
"""
|
|
576
|
+
alerts: List[Alert] = []
|
|
577
|
+
if alert_value(table.get("n_duplicates", np.nan)):
|
|
578
|
+
alerts.append(
|
|
579
|
+
DuplicatesAlert(
|
|
580
|
+
values=table,
|
|
581
|
+
)
|
|
582
|
+
)
|
|
583
|
+
if table["n"] == 0:
|
|
584
|
+
alerts.append(
|
|
585
|
+
EmptyAlert(
|
|
586
|
+
values=table,
|
|
587
|
+
)
|
|
588
|
+
)
|
|
589
|
+
return alerts
|
|
590
|
+
|
|
591
|
+
|
|
592
|
+
def numeric_alerts(config: Settings, summary: dict) -> List[Alert]:
|
|
593
|
+
alerts: List[Alert] = []
|
|
594
|
+
|
|
595
|
+
# Skewness
|
|
596
|
+
if skewness_alert(summary["skewness"], config.vars.num.skewness_threshold):
|
|
597
|
+
alerts.append(SkewedAlert(summary))
|
|
598
|
+
|
|
599
|
+
# Infinite values
|
|
600
|
+
if alert_value(summary["p_infinite"]):
|
|
601
|
+
alerts.append(InfiniteAlert(summary))
|
|
602
|
+
|
|
603
|
+
# Zeros
|
|
604
|
+
if alert_value(summary["p_zeros"]):
|
|
605
|
+
alerts.append(ZerosAlert(summary))
|
|
606
|
+
|
|
607
|
+
if (
|
|
608
|
+
"chi_squared" in summary
|
|
609
|
+
and summary["chi_squared"]["pvalue"] > config.vars.num.chi_squared_threshold
|
|
610
|
+
):
|
|
611
|
+
alerts.append(UniformAlert())
|
|
612
|
+
|
|
613
|
+
return alerts
|
|
614
|
+
|
|
615
|
+
|
|
616
|
+
def timeseries_alerts(config: Settings, summary: dict) -> List[Alert]:
|
|
617
|
+
alerts: List[Alert] = numeric_alerts(config, summary)
|
|
618
|
+
|
|
619
|
+
if not summary["stationary"]:
|
|
620
|
+
alerts.append(NonStationaryAlert())
|
|
621
|
+
|
|
622
|
+
if summary["seasonal"]:
|
|
623
|
+
alerts.append(SeasonalAlert())
|
|
624
|
+
|
|
625
|
+
return alerts
|
|
626
|
+
|
|
627
|
+
|
|
628
|
+
def categorical_alerts(config: Settings, summary: dict) -> List[Alert]:
|
|
629
|
+
alerts: List[Alert] = []
|
|
630
|
+
|
|
631
|
+
# High cardinality
|
|
632
|
+
if summary.get("n_distinct", np.nan) > config.vars.cat.cardinality_threshold:
|
|
633
|
+
alerts.append(HighCardinalityAlert(summary))
|
|
634
|
+
|
|
635
|
+
if (
|
|
636
|
+
"chi_squared" in summary
|
|
637
|
+
and summary["chi_squared"]["pvalue"] > config.vars.cat.chi_squared_threshold
|
|
638
|
+
):
|
|
639
|
+
alerts.append(UniformAlert())
|
|
640
|
+
|
|
641
|
+
if summary.get("date_warning"):
|
|
642
|
+
alerts.append(TypeDateAlert())
|
|
643
|
+
|
|
644
|
+
# Constant length
|
|
645
|
+
if "composition" in summary and summary["min_length"] == summary["max_length"]:
|
|
646
|
+
alerts.append(ConstantLengthAlert())
|
|
647
|
+
|
|
648
|
+
# Imbalance
|
|
649
|
+
if (
|
|
650
|
+
"imbalance" in summary
|
|
651
|
+
and summary["imbalance"] > config.vars.cat.imbalance_threshold
|
|
652
|
+
):
|
|
653
|
+
alerts.append(ImbalanceAlert(summary))
|
|
654
|
+
return alerts
|
|
655
|
+
|
|
656
|
+
|
|
657
|
+
def boolean_alerts(config: Settings, summary: dict) -> List[Alert]:
|
|
658
|
+
alerts: List[Alert] = []
|
|
659
|
+
|
|
660
|
+
if (
|
|
661
|
+
"imbalance" in summary
|
|
662
|
+
and summary["imbalance"] > config.vars.bool.imbalance_threshold
|
|
663
|
+
):
|
|
664
|
+
alerts.append(ImbalanceAlert())
|
|
665
|
+
return alerts
|
|
666
|
+
|
|
667
|
+
|
|
668
|
+
def generic_alerts(summary: dict) -> List[Alert]:
|
|
669
|
+
alerts: List[Alert] = []
|
|
670
|
+
|
|
671
|
+
# Missing
|
|
672
|
+
if alert_value(summary["p_missing"]):
|
|
673
|
+
alerts.append(MissingAlert())
|
|
674
|
+
|
|
675
|
+
return alerts
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
def supported_alerts(summary: dict) -> List[Alert]:
|
|
679
|
+
alerts: List[Alert] = []
|
|
680
|
+
|
|
681
|
+
if summary.get("n_distinct", np.nan) == summary["n"]:
|
|
682
|
+
alerts.append(UniqueAlert())
|
|
683
|
+
if summary.get("n_distinct", np.nan) == 1:
|
|
684
|
+
alerts.append(ConstantAlert(summary))
|
|
685
|
+
return alerts
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
def unsupported_alerts() -> List[Alert]:
|
|
689
|
+
alerts: List[Alert] = [
|
|
690
|
+
UnsupportedAlert(),
|
|
691
|
+
RejectedAlert(),
|
|
692
|
+
]
|
|
693
|
+
return alerts
|
|
694
|
+
|
|
695
|
+
|
|
696
|
+
def check_variable_alerts(config: Settings, col: str, description: dict) -> List[Alert]:
|
|
697
|
+
"""Checks individual variables for alerts.
|
|
698
|
+
|
|
699
|
+
Args:
|
|
700
|
+
col: The column name that is checked.
|
|
701
|
+
description: The series description.
|
|
702
|
+
|
|
703
|
+
Returns:
|
|
704
|
+
A list of alerts.
|
|
705
|
+
"""
|
|
706
|
+
alerts: List[Alert] = []
|
|
707
|
+
|
|
708
|
+
alerts += generic_alerts(description)
|
|
709
|
+
|
|
710
|
+
if description["type"] == "Unsupported":
|
|
711
|
+
alerts += unsupported_alerts()
|
|
712
|
+
else:
|
|
713
|
+
alerts += supported_alerts(description)
|
|
714
|
+
|
|
715
|
+
if description["type"] == "Categorical":
|
|
716
|
+
alerts += categorical_alerts(config, description)
|
|
717
|
+
if description["type"] == "Numeric":
|
|
718
|
+
alerts += numeric_alerts(config, description)
|
|
719
|
+
if description["type"] == "TimeSeries":
|
|
720
|
+
alerts += timeseries_alerts(config, description)
|
|
721
|
+
if description["type"] == "Boolean":
|
|
722
|
+
alerts += boolean_alerts(config, description)
|
|
723
|
+
|
|
724
|
+
for idx in range(len(alerts)):
|
|
725
|
+
alerts[idx].column_name = col
|
|
726
|
+
alerts[idx].values = description
|
|
727
|
+
return alerts
|
|
728
|
+
|
|
729
|
+
|
|
730
|
+
def check_correlation_alerts(config: Settings, correlations: dict) -> List[Alert]:
|
|
731
|
+
alerts: List[Alert] = []
|
|
732
|
+
|
|
733
|
+
correlations_consolidated = {}
|
|
734
|
+
for corr, matrix in correlations.items():
|
|
735
|
+
if config.correlations[corr].warn_high_correlations:
|
|
736
|
+
threshold = config.correlations[corr].threshold
|
|
737
|
+
correlated_mapping = perform_check_correlation(matrix, threshold)
|
|
738
|
+
for col, fields in correlated_mapping.items():
|
|
739
|
+
set(fields).update(set(correlated_mapping.get(col, [])))
|
|
740
|
+
correlations_consolidated[col] = fields
|
|
741
|
+
|
|
742
|
+
if len(correlations_consolidated) > 0:
|
|
743
|
+
for col, fields in correlations_consolidated.items():
|
|
744
|
+
alerts.append(
|
|
745
|
+
HighCorrelationAlert(
|
|
746
|
+
column_name=col,
|
|
747
|
+
values={"corr": "overall", "fields": fields},
|
|
748
|
+
)
|
|
749
|
+
)
|
|
750
|
+
return alerts
|
|
751
|
+
|
|
752
|
+
|
|
753
|
+
def get_alerts(
|
|
754
|
+
config: Settings, table_stats: dict, series_description: dict, correlations: dict
|
|
755
|
+
) -> List[Alert]:
|
|
756
|
+
alerts: List[Alert] = check_table_alerts(table_stats)
|
|
757
|
+
for col, description in series_description.items():
|
|
758
|
+
alerts += check_variable_alerts(config, col, description)
|
|
759
|
+
alerts += check_correlation_alerts(config, correlations)
|
|
760
|
+
alerts.sort(key=lambda alert: str(alert.alert_type))
|
|
761
|
+
return alerts
|
|
762
|
+
|
|
763
|
+
|
|
764
|
+
def alert_value(value: float) -> bool:
|
|
765
|
+
return not pd.isna(value) and value > 0.01
|
|
766
|
+
|
|
767
|
+
|
|
768
|
+
def skewness_alert(v: float, threshold: int) -> bool:
|
|
769
|
+
return not pd.isna(v) and (v < (-1 * threshold) or v > threshold)
|
|
770
|
+
|
|
771
|
+
|
|
772
|
+
def type_date_alert(series: pd.Series) -> bool:
|
|
773
|
+
from dateutil.parser import ParserError, parse
|
|
774
|
+
|
|
775
|
+
try:
|
|
776
|
+
series.apply(parse)
|
|
777
|
+
except ParserError:
|
|
778
|
+
return False
|
|
779
|
+
else:
|
|
780
|
+
return True
|