fg-data-profiling 4.19.0__py2.py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- data_profiling/__init__.py +34 -0
- data_profiling/compare_reports.py +359 -0
- data_profiling/config.py +496 -0
- data_profiling/config_default.yaml +223 -0
- data_profiling/config_minimal.yaml +222 -0
- data_profiling/controller/__init__.py +1 -0
- data_profiling/controller/console.py +125 -0
- data_profiling/controller/pandas_decorator.py +21 -0
- data_profiling/expectations_report.py +117 -0
- data_profiling/model/__init__.py +4 -0
- data_profiling/model/alerts.py +780 -0
- data_profiling/model/correlations.py +163 -0
- data_profiling/model/dataframe.py +35 -0
- data_profiling/model/describe.py +210 -0
- data_profiling/model/description.py +108 -0
- data_profiling/model/duplicates.py +14 -0
- data_profiling/model/expectation_algorithms.py +112 -0
- data_profiling/model/handler.py +81 -0
- data_profiling/model/missing.py +146 -0
- data_profiling/model/pairwise.py +33 -0
- data_profiling/model/pandas/__init__.py +55 -0
- data_profiling/model/pandas/correlations_pandas.py +207 -0
- data_profiling/model/pandas/dataframe_pandas.py +26 -0
- data_profiling/model/pandas/describe_boolean_pandas.py +43 -0
- data_profiling/model/pandas/describe_categorical_pandas.py +274 -0
- data_profiling/model/pandas/describe_counts_pandas.py +63 -0
- data_profiling/model/pandas/describe_date_pandas.py +77 -0
- data_profiling/model/pandas/describe_file_pandas.py +56 -0
- data_profiling/model/pandas/describe_generic_pandas.py +36 -0
- data_profiling/model/pandas/describe_image_pandas.py +255 -0
- data_profiling/model/pandas/describe_numeric_pandas.py +175 -0
- data_profiling/model/pandas/describe_path_pandas.py +63 -0
- data_profiling/model/pandas/describe_supported_pandas.py +41 -0
- data_profiling/model/pandas/describe_text_pandas.py +62 -0
- data_profiling/model/pandas/describe_timeseries_pandas.py +222 -0
- data_profiling/model/pandas/describe_url_pandas.py +57 -0
- data_profiling/model/pandas/discretize_pandas.py +81 -0
- data_profiling/model/pandas/duplicates_pandas.py +56 -0
- data_profiling/model/pandas/imbalance_pandas.py +35 -0
- data_profiling/model/pandas/missing_pandas.py +42 -0
- data_profiling/model/pandas/sample_pandas.py +38 -0
- data_profiling/model/pandas/summary_pandas.py +101 -0
- data_profiling/model/pandas/table_pandas.py +56 -0
- data_profiling/model/pandas/timeseries_index_pandas.py +33 -0
- data_profiling/model/pandas/utils_pandas.py +27 -0
- data_profiling/model/sample.py +37 -0
- data_profiling/model/spark/__init__.py +48 -0
- data_profiling/model/spark/correlations_spark.py +152 -0
- data_profiling/model/spark/dataframe_spark.py +34 -0
- data_profiling/model/spark/describe_boolean_spark.py +27 -0
- data_profiling/model/spark/describe_categorical_spark.py +28 -0
- data_profiling/model/spark/describe_counts_spark.py +105 -0
- data_profiling/model/spark/describe_date_spark.py +51 -0
- data_profiling/model/spark/describe_generic_spark.py +30 -0
- data_profiling/model/spark/describe_numeric_spark.py +155 -0
- data_profiling/model/spark/describe_supported_spark.py +33 -0
- data_profiling/model/spark/describe_text_spark.py +25 -0
- data_profiling/model/spark/duplicates_spark.py +54 -0
- data_profiling/model/spark/missing_spark.py +96 -0
- data_profiling/model/spark/sample_spark.py +43 -0
- data_profiling/model/spark/summary_spark.py +95 -0
- data_profiling/model/spark/table_spark.py +58 -0
- data_profiling/model/spark/timeseries_index_spark.py +12 -0
- data_profiling/model/summarizer.py +207 -0
- data_profiling/model/summary.py +66 -0
- data_profiling/model/summary_algorithms.py +276 -0
- data_profiling/model/table.py +10 -0
- data_profiling/model/timeseries_index.py +16 -0
- data_profiling/model/typeset.py +365 -0
- data_profiling/model/typeset_relations.py +143 -0
- data_profiling/profile_report.py +573 -0
- data_profiling/report/__init__.py +4 -0
- data_profiling/report/formatters.py +346 -0
- data_profiling/report/presentation/__init__.py +1 -0
- data_profiling/report/presentation/core/__init__.py +39 -0
- data_profiling/report/presentation/core/alerts.py +18 -0
- data_profiling/report/presentation/core/collapse.py +24 -0
- data_profiling/report/presentation/core/container.py +50 -0
- data_profiling/report/presentation/core/correlation_table.py +21 -0
- data_profiling/report/presentation/core/dropdown.py +44 -0
- data_profiling/report/presentation/core/duplicate.py +16 -0
- data_profiling/report/presentation/core/frequency_table.py +14 -0
- data_profiling/report/presentation/core/frequency_table_small.py +16 -0
- data_profiling/report/presentation/core/html.py +14 -0
- data_profiling/report/presentation/core/image.py +34 -0
- data_profiling/report/presentation/core/item_renderer.py +17 -0
- data_profiling/report/presentation/core/renderable.py +42 -0
- data_profiling/report/presentation/core/root.py +35 -0
- data_profiling/report/presentation/core/sample.py +20 -0
- data_profiling/report/presentation/core/scores.py +32 -0
- data_profiling/report/presentation/core/table.py +26 -0
- data_profiling/report/presentation/core/toggle_button.py +14 -0
- data_profiling/report/presentation/core/variable.py +40 -0
- data_profiling/report/presentation/core/variable_info.py +36 -0
- data_profiling/report/presentation/flavours/__init__.py +9 -0
- data_profiling/report/presentation/flavours/flavour_html.py +64 -0
- data_profiling/report/presentation/flavours/flavour_widget.py +61 -0
- data_profiling/report/presentation/flavours/flavours.py +43 -0
- data_profiling/report/presentation/flavours/html/__init__.py +47 -0
- data_profiling/report/presentation/flavours/html/alerts.py +10 -0
- data_profiling/report/presentation/flavours/html/collapse.py +7 -0
- data_profiling/report/presentation/flavours/html/container.py +58 -0
- data_profiling/report/presentation/flavours/html/correlation_table.py +13 -0
- data_profiling/report/presentation/flavours/html/dropdown.py +7 -0
- data_profiling/report/presentation/flavours/html/duplicate.py +24 -0
- data_profiling/report/presentation/flavours/html/frequency_table.py +20 -0
- data_profiling/report/presentation/flavours/html/frequency_table_small.py +15 -0
- data_profiling/report/presentation/flavours/html/html.py +6 -0
- data_profiling/report/presentation/flavours/html/image.py +7 -0
- data_profiling/report/presentation/flavours/html/root.py +14 -0
- data_profiling/report/presentation/flavours/html/sample.py +12 -0
- data_profiling/report/presentation/flavours/html/scores.py +11 -0
- data_profiling/report/presentation/flavours/html/table.py +7 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant_length.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_dirty_category.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_duplicates.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_empty.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_cardinality.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_correlation.html +4 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_imbalance.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_infinite.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_missing.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_near_duplicates.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_non_stationary.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_seasonal.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_skewed.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_truncated.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_type_date.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_uniform.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_unique.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_unsupported.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_zeros.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts.html +47 -0
- data_profiling/report/presentation/flavours/html/templates/collapse.html +11 -0
- data_profiling/report/presentation/flavours/html/templates/correlation_table.html +5 -0
- data_profiling/report/presentation/flavours/html/templates/diagram.html +11 -0
- data_profiling/report/presentation/flavours/html/templates/dropdown.html +16 -0
- data_profiling/report/presentation/flavours/html/templates/duplicate.html +5 -0
- data_profiling/report/presentation/flavours/html/templates/frequency_table.html +45 -0
- data_profiling/report/presentation/flavours/html/templates/frequency_table_small.html +34 -0
- data_profiling/report/presentation/flavours/html/templates/report.html +26 -0
- data_profiling/report/presentation/flavours/html/templates/sample.html +10 -0
- data_profiling/report/presentation/flavours/html/templates/scores.html +78 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/batch_grid.html +16 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/grid.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/list.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/named_list.html +8 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/overview_tabs.html +30 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/scores.html +3 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/sections.html +13 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/select.html +40 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/tabs.html +30 -0
- data_profiling/report/presentation/flavours/html/templates/table.html +38 -0
- data_profiling/report/presentation/flavours/html/templates/toggle_button.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/variable.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/variable_info.html +49 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.bundle.min.js +7 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.min.css +6 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/cosmo.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/flatly.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/script.js +52 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/simplex.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/style.css +253 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/united.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/footer.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/javascript.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/navigation.html +36 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/style.html +53 -0
- data_profiling/report/presentation/flavours/html/templates.py +76 -0
- data_profiling/report/presentation/flavours/html/toggle_button.py +7 -0
- data_profiling/report/presentation/flavours/html/variable.py +7 -0
- data_profiling/report/presentation/flavours/html/variable_info.py +7 -0
- data_profiling/report/presentation/flavours/widget/__init__.py +49 -0
- data_profiling/report/presentation/flavours/widget/alerts.py +45 -0
- data_profiling/report/presentation/flavours/widget/collapse.py +43 -0
- data_profiling/report/presentation/flavours/widget/container.py +121 -0
- data_profiling/report/presentation/flavours/widget/correlation_table.py +14 -0
- data_profiling/report/presentation/flavours/widget/dropdown.py +31 -0
- data_profiling/report/presentation/flavours/widget/duplicate.py +14 -0
- data_profiling/report/presentation/flavours/widget/frequency_table.py +57 -0
- data_profiling/report/presentation/flavours/widget/frequency_table_small.py +66 -0
- data_profiling/report/presentation/flavours/widget/html.py +11 -0
- data_profiling/report/presentation/flavours/widget/image.py +26 -0
- data_profiling/report/presentation/flavours/widget/notebook.py +81 -0
- data_profiling/report/presentation/flavours/widget/root.py +10 -0
- data_profiling/report/presentation/flavours/widget/sample.py +14 -0
- data_profiling/report/presentation/flavours/widget/table.py +30 -0
- data_profiling/report/presentation/flavours/widget/toggle_button.py +17 -0
- data_profiling/report/presentation/flavours/widget/variable.py +12 -0
- data_profiling/report/presentation/flavours/widget/variable_info.py +11 -0
- data_profiling/report/presentation/frequency_table_utils.py +141 -0
- data_profiling/report/structure/__init__.py +1 -0
- data_profiling/report/structure/correlations.py +123 -0
- data_profiling/report/structure/overview.py +376 -0
- data_profiling/report/structure/report.py +457 -0
- data_profiling/report/structure/variables/__init__.py +35 -0
- data_profiling/report/structure/variables/render_boolean.py +132 -0
- data_profiling/report/structure/variables/render_categorical.py +566 -0
- data_profiling/report/structure/variables/render_common.py +31 -0
- data_profiling/report/structure/variables/render_complex.py +102 -0
- data_profiling/report/structure/variables/render_count.py +172 -0
- data_profiling/report/structure/variables/render_date.py +143 -0
- data_profiling/report/structure/variables/render_file.py +70 -0
- data_profiling/report/structure/variables/render_generic.py +45 -0
- data_profiling/report/structure/variables/render_image.py +204 -0
- data_profiling/report/structure/variables/render_path.py +134 -0
- data_profiling/report/structure/variables/render_real.py +314 -0
- data_profiling/report/structure/variables/render_text.py +189 -0
- data_profiling/report/structure/variables/render_timeseries.py +371 -0
- data_profiling/report/structure/variables/render_url.py +132 -0
- data_profiling/report/utils.py +34 -0
- data_profiling/serialize_report.py +143 -0
- data_profiling/utils/__init__.py +1 -0
- data_profiling/utils/backend.py +9 -0
- data_profiling/utils/cache.py +59 -0
- data_profiling/utils/common.py +142 -0
- data_profiling/utils/compat.py +31 -0
- data_profiling/utils/dataframe.py +238 -0
- data_profiling/utils/logger.py +53 -0
- data_profiling/utils/notebook.py +8 -0
- data_profiling/utils/paths.py +45 -0
- data_profiling/utils/progress_bar.py +15 -0
- data_profiling/utils/styles.py +22 -0
- data_profiling/utils/versions.py +19 -0
- data_profiling/version.py +1 -0
- data_profiling/visualisation/__init__.py +1 -0
- data_profiling/visualisation/context.py +87 -0
- data_profiling/visualisation/missing.py +138 -0
- data_profiling/visualisation/plot.py +1158 -0
- data_profiling/visualisation/utils.py +113 -0
- fg_data_profiling-4.19.0.dist-info/METADATA +362 -0
- fg_data_profiling-4.19.0.dist-info/RECORD +238 -0
- fg_data_profiling-4.19.0.dist-info/WHEEL +6 -0
- fg_data_profiling-4.19.0.dist-info/entry_points.txt +3 -0
- fg_data_profiling-4.19.0.dist-info/licenses/LICENSE +21 -0
- fg_data_profiling-4.19.0.dist-info/top_level.txt +2 -0
- ydata_profiling/__init__.py +43 -0
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
from typing import Any, Dict, Tuple
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from scipy.fft import _pocketfft
|
|
6
|
+
from scipy.signal import find_peaks
|
|
7
|
+
from statsmodels.tsa.stattools import adfuller
|
|
8
|
+
|
|
9
|
+
from data_profiling.config import Settings
|
|
10
|
+
from data_profiling.model.summary_algorithms import (
|
|
11
|
+
describe_numeric_1d,
|
|
12
|
+
describe_timeseries_1d,
|
|
13
|
+
series_handle_nulls,
|
|
14
|
+
series_hashable,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def stationarity_test(config: Settings, series: pd.Series) -> Tuple[bool, float]:
|
|
19
|
+
# make sure the data has no missing values
|
|
20
|
+
adfuller_test = adfuller(
|
|
21
|
+
series.dropna(),
|
|
22
|
+
autolag=config.vars.timeseries.autolag,
|
|
23
|
+
maxlag=config.vars.timeseries.maxlag,
|
|
24
|
+
)
|
|
25
|
+
p_value = adfuller_test[1]
|
|
26
|
+
|
|
27
|
+
significance_threshold = config.vars.timeseries.significance
|
|
28
|
+
return p_value < significance_threshold, p_value
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def fftfreq(n: int, d: float = 1.0) -> np.ndarray:
|
|
32
|
+
"""
|
|
33
|
+
Return the Discrete Fourier Transform sample frequencies.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
n : int
|
|
37
|
+
Window length.
|
|
38
|
+
d : scalar, optional
|
|
39
|
+
Sample spacing (inverse of the sampling rate). Defaults to 1.
|
|
40
|
+
|
|
41
|
+
Returns:
|
|
42
|
+
f : ndarray
|
|
43
|
+
Array of length `n` containing the sample frequencies.
|
|
44
|
+
"""
|
|
45
|
+
val = 1.0 / (n * d)
|
|
46
|
+
results = np.empty(n, int)
|
|
47
|
+
N = (n - 1) // 2 + 1
|
|
48
|
+
p1 = np.arange(0, N, dtype=int)
|
|
49
|
+
results[:N] = p1
|
|
50
|
+
p2 = np.arange(-(n // 2), 0, dtype=int)
|
|
51
|
+
results[N:] = p2
|
|
52
|
+
return results * val
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def seasonality_test(series: pd.Series, mad_threshold: float = 6.0) -> Dict[str, Any]:
|
|
56
|
+
"""Detect seasonality with FFT
|
|
57
|
+
|
|
58
|
+
Source: https://github.com/facebookresearch/Kats/blob/main/kats/detectors/seasonality.py
|
|
59
|
+
|
|
60
|
+
Args:
|
|
61
|
+
mad_threshold: Optional; float; constant for the outlier algorithm for peak
|
|
62
|
+
detector. The larger the value the less sensitive the outlier algorithm
|
|
63
|
+
is.
|
|
64
|
+
|
|
65
|
+
Returns:
|
|
66
|
+
FFT Plot with peaks, selected peaks, and outlier boundary line.
|
|
67
|
+
"""
|
|
68
|
+
|
|
69
|
+
fft = get_fft(series)
|
|
70
|
+
_, _, peaks = get_fft_peaks(fft, mad_threshold)
|
|
71
|
+
seasonality_presence = len(peaks.index) > 0
|
|
72
|
+
selected_seasonalities = []
|
|
73
|
+
if seasonality_presence:
|
|
74
|
+
selected_seasonalities = peaks["freq"].transform(lambda x: 1 / x).tolist()
|
|
75
|
+
|
|
76
|
+
return {
|
|
77
|
+
"seasonality_presence": seasonality_presence,
|
|
78
|
+
"seasonalities": selected_seasonalities,
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def get_fft(series: pd.Series) -> pd.DataFrame:
|
|
83
|
+
"""Computes FFT
|
|
84
|
+
|
|
85
|
+
Args:
|
|
86
|
+
series: pd.Series
|
|
87
|
+
time series
|
|
88
|
+
|
|
89
|
+
Returns:
|
|
90
|
+
DataFrame with columns 'freq' and 'ampl'.
|
|
91
|
+
"""
|
|
92
|
+
data_fft = _pocketfft.fft(series.to_numpy())
|
|
93
|
+
data_psd = np.abs(data_fft) ** 2
|
|
94
|
+
fftfreq_ = fftfreq(len(data_psd), 1.0)
|
|
95
|
+
pos_freq_ix = fftfreq_ > 0
|
|
96
|
+
|
|
97
|
+
freq = fftfreq_[pos_freq_ix]
|
|
98
|
+
ampl = 10 * np.log10(data_psd[pos_freq_ix])
|
|
99
|
+
|
|
100
|
+
return pd.DataFrame({"freq": freq, "ampl": ampl})
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def get_fft_peaks(
|
|
104
|
+
fft: pd.DataFrame, mad_threshold: float = 6.0
|
|
105
|
+
) -> Tuple[float, pd.DataFrame, pd.DataFrame]:
|
|
106
|
+
"""Computes peaks in fft, selects the highest peaks (outliers) and
|
|
107
|
+
removes the harmonics (multiplies of the base harmonics found)
|
|
108
|
+
|
|
109
|
+
Args:
|
|
110
|
+
fft: FFT computed by get_fft
|
|
111
|
+
mad_threshold: Optional; constant for the outlier algorithm for peak detector.
|
|
112
|
+
The larger the value the less sensitive the outlier algorithm is.
|
|
113
|
+
|
|
114
|
+
Returns:
|
|
115
|
+
outlier threshold, peaks, selected peaks.
|
|
116
|
+
"""
|
|
117
|
+
pos_fft = fft.loc[fft["ampl"] > 0]
|
|
118
|
+
median = pos_fft["ampl"].median()
|
|
119
|
+
pos_fft_above_med = pos_fft[pos_fft["ampl"] > median]
|
|
120
|
+
mad = abs(pos_fft_above_med["ampl"] - pos_fft_above_med["ampl"].mean()).mean()
|
|
121
|
+
|
|
122
|
+
threshold = median + mad * mad_threshold
|
|
123
|
+
|
|
124
|
+
peak_indices = find_peaks(fft["ampl"], threshold=0.1)
|
|
125
|
+
peaks = fft.loc[peak_indices[0], :]
|
|
126
|
+
|
|
127
|
+
orig_peaks = peaks.copy()
|
|
128
|
+
|
|
129
|
+
peaks = peaks.loc[peaks["ampl"] > threshold].copy()
|
|
130
|
+
peaks["Remove"] = [False] * len(peaks.index)
|
|
131
|
+
peaks.reset_index(inplace=True)
|
|
132
|
+
|
|
133
|
+
# Filter out harmonics
|
|
134
|
+
for idx1 in range(len(peaks)):
|
|
135
|
+
curr = peaks.loc[idx1, "freq"]
|
|
136
|
+
for idx2 in range(idx1 + 1, len(peaks)):
|
|
137
|
+
if peaks.loc[idx2, "Remove"] is True:
|
|
138
|
+
continue
|
|
139
|
+
fraction = (peaks.loc[idx2, "freq"] / curr) % 1
|
|
140
|
+
if fraction < 0.01 or fraction > 0.99:
|
|
141
|
+
peaks.loc[idx2, "Remove"] = True
|
|
142
|
+
peaks = peaks.loc[~peaks["Remove"]]
|
|
143
|
+
peaks.drop(inplace=True, columns="Remove")
|
|
144
|
+
return threshold, orig_peaks, peaks
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def identify_gaps(
|
|
148
|
+
gap: pd.Series, is_datetime: bool, gap_tolerance: int = 2
|
|
149
|
+
) -> Tuple[pd.Series, list]:
|
|
150
|
+
zero = pd.Timedelta(0) if is_datetime else 0
|
|
151
|
+
diff = gap.diff()
|
|
152
|
+
|
|
153
|
+
non_zero_diff = diff[diff > zero]
|
|
154
|
+
min_gap_size = gap_tolerance * non_zero_diff.mean()
|
|
155
|
+
|
|
156
|
+
gap_stats = non_zero_diff[non_zero_diff > min_gap_size]
|
|
157
|
+
anchors = gap[diff > min_gap_size].index
|
|
158
|
+
|
|
159
|
+
gaps = []
|
|
160
|
+
for i in anchors:
|
|
161
|
+
gaps.append(gap.loc[gap.index[[i - 1, i]]].values)
|
|
162
|
+
|
|
163
|
+
return gap_stats, gaps
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def compute_gap_stats(series: pd.Series) -> pd.Series:
|
|
167
|
+
"""Computes the intertevals in the series normalized by the period.
|
|
168
|
+
|
|
169
|
+
Args:
|
|
170
|
+
series (pd.Series): time series data to analysis.
|
|
171
|
+
|
|
172
|
+
Returns:
|
|
173
|
+
A series with the gaps intervals.
|
|
174
|
+
"""
|
|
175
|
+
|
|
176
|
+
gap = series.dropna()
|
|
177
|
+
index_name = gap.index.name if gap.index.name else "index"
|
|
178
|
+
gap = gap.reset_index()[index_name]
|
|
179
|
+
gap.index.name = None
|
|
180
|
+
|
|
181
|
+
is_datetime = isinstance(series.index, pd.DatetimeIndex)
|
|
182
|
+
gap_stats, gaps = identify_gaps(gap, is_datetime)
|
|
183
|
+
has_gaps = len(gap_stats) > 0
|
|
184
|
+
|
|
185
|
+
stats = {
|
|
186
|
+
"min": gap_stats.min() if has_gaps else 0,
|
|
187
|
+
"max": gap_stats.max() if has_gaps else 0,
|
|
188
|
+
"mean": gap_stats.mean() if has_gaps else 0,
|
|
189
|
+
"std": gap_stats.std() if len(gap_stats) > 1 else 0,
|
|
190
|
+
"series": series,
|
|
191
|
+
"gaps": gaps,
|
|
192
|
+
"n_gaps": len(gaps),
|
|
193
|
+
}
|
|
194
|
+
return stats
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
@describe_timeseries_1d.register
|
|
198
|
+
@series_hashable
|
|
199
|
+
@series_handle_nulls
|
|
200
|
+
def pandas_describe_timeseries_1d(
|
|
201
|
+
config: Settings, series: pd.Series, summary: dict
|
|
202
|
+
) -> Tuple[Settings, pd.Series, dict]:
|
|
203
|
+
"""Describe a timeseries.
|
|
204
|
+
|
|
205
|
+
Args:
|
|
206
|
+
config: report Settings object
|
|
207
|
+
series: The Series to describe.
|
|
208
|
+
summary: The dict containing the series description so far.
|
|
209
|
+
|
|
210
|
+
Returns:
|
|
211
|
+
A dict containing calculated series description values.
|
|
212
|
+
"""
|
|
213
|
+
config, series, stats = describe_numeric_1d(config, series, summary)
|
|
214
|
+
|
|
215
|
+
stats["seasonal"] = seasonality_test(series)["seasonality_presence"]
|
|
216
|
+
is_stationary, p_value = stationarity_test(config, series)
|
|
217
|
+
stats["stationary"] = is_stationary and not stats["seasonal"]
|
|
218
|
+
stats["addfuller"] = p_value
|
|
219
|
+
stats["series"] = series
|
|
220
|
+
stats["gap_stats"] = compute_gap_stats(series)
|
|
221
|
+
|
|
222
|
+
return config, series, stats
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
from typing import Tuple
|
|
2
|
+
from urllib.parse import urlsplit
|
|
3
|
+
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
from data_profiling.config import Settings
|
|
7
|
+
from data_profiling.model.summary_algorithms import describe_url_1d
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def url_summary(series: pd.Series) -> dict:
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
Args:
|
|
14
|
+
series: series to summarize
|
|
15
|
+
|
|
16
|
+
Returns:
|
|
17
|
+
|
|
18
|
+
"""
|
|
19
|
+
summary = {
|
|
20
|
+
"scheme_counts": series.map(lambda x: x.scheme).value_counts(),
|
|
21
|
+
"netloc_counts": series.map(lambda x: x.netloc).value_counts(),
|
|
22
|
+
"path_counts": series.map(lambda x: x.path).value_counts(),
|
|
23
|
+
"query_counts": series.map(lambda x: x.query).value_counts(),
|
|
24
|
+
"fragment_counts": series.map(lambda x: x.fragment).value_counts(),
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
return summary
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@describe_url_1d.register
|
|
31
|
+
def pandas_describe_url_1d(
|
|
32
|
+
config: Settings, series: pd.Series, summary: dict
|
|
33
|
+
) -> Tuple[Settings, pd.Series, dict]:
|
|
34
|
+
"""Describe a url series.
|
|
35
|
+
|
|
36
|
+
Args:
|
|
37
|
+
config: report Settings object
|
|
38
|
+
series: The Series to describe.
|
|
39
|
+
summary: The dict containing the series description so far.
|
|
40
|
+
|
|
41
|
+
Returns:
|
|
42
|
+
A dict containing calculated series description values.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
# Make sure we deal with strings (Issue #100)
|
|
46
|
+
if series.hasnans:
|
|
47
|
+
raise ValueError("May not contain NaNs")
|
|
48
|
+
if not hasattr(series, "str"):
|
|
49
|
+
raise ValueError("series should have .str accessor")
|
|
50
|
+
|
|
51
|
+
# Transform
|
|
52
|
+
series = series.apply(urlsplit)
|
|
53
|
+
|
|
54
|
+
# Update
|
|
55
|
+
summary.update(url_summary(series))
|
|
56
|
+
|
|
57
|
+
return config, series, summary
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
from enum import Enum
|
|
2
|
+
from typing import List
|
|
3
|
+
|
|
4
|
+
import numpy as np
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class DiscretizationType(Enum):
|
|
9
|
+
UNIFORM = "uniform"
|
|
10
|
+
QUANTILE = "quantile"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class Discretizer:
|
|
14
|
+
"""
|
|
15
|
+
A class which enables the discretization of a pandas dataframe.
|
|
16
|
+
Perform this action when you want to convert a continuous variable
|
|
17
|
+
into a categorical variable.
|
|
18
|
+
|
|
19
|
+
Attributes:
|
|
20
|
+
|
|
21
|
+
method (DiscretizationType): this attribute controls how the buckets
|
|
22
|
+
of your discretization are formed. A uniform discretization type forms
|
|
23
|
+
the bins to be of equal width whereas a quantile discretization type
|
|
24
|
+
forms the bins to be of equal size.
|
|
25
|
+
|
|
26
|
+
n_bins (int): number of bins
|
|
27
|
+
reset_index (bool): instruction to reset the index of
|
|
28
|
+
the dataframe after the discretization
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
def __init__(
|
|
32
|
+
self, method: DiscretizationType, n_bins: int = 10, reset_index: bool = False
|
|
33
|
+
) -> None:
|
|
34
|
+
self.discretization_type = method
|
|
35
|
+
self.n_bins = n_bins
|
|
36
|
+
self.reset_index = reset_index
|
|
37
|
+
|
|
38
|
+
def discretize_dataframe(self, dataframe: pd.DataFrame) -> pd.DataFrame:
|
|
39
|
+
"""_summary_
|
|
40
|
+
|
|
41
|
+
Args:
|
|
42
|
+
dataframe (pd.DataFrame): pandas dataframe
|
|
43
|
+
|
|
44
|
+
Returns:
|
|
45
|
+
pd.DataFrame: discretized dataframe
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
discretized_df = dataframe.copy()
|
|
49
|
+
all_columns = dataframe.columns
|
|
50
|
+
num_columns = self._get_numerical_columns(dataframe)
|
|
51
|
+
for column in num_columns:
|
|
52
|
+
discretized_df.loc[:, column] = self._discretize_column(
|
|
53
|
+
discretized_df[column]
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
discretized_df = discretized_df[all_columns]
|
|
57
|
+
return (
|
|
58
|
+
discretized_df.reset_index(drop=True)
|
|
59
|
+
if self.reset_index
|
|
60
|
+
else discretized_df
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
def _discretize_column(self, column: pd.Series) -> pd.Series:
|
|
64
|
+
if self.discretization_type == DiscretizationType.QUANTILE:
|
|
65
|
+
return self._descritize_quantile(column)
|
|
66
|
+
|
|
67
|
+
elif self.discretization_type == DiscretizationType.UNIFORM:
|
|
68
|
+
return self._descritize_uniform(column)
|
|
69
|
+
|
|
70
|
+
def _descritize_quantile(self, column: pd.Series) -> pd.Series:
|
|
71
|
+
return pd.qcut(
|
|
72
|
+
column, q=self.n_bins, labels=False, retbins=False, duplicates="drop"
|
|
73
|
+
).values
|
|
74
|
+
|
|
75
|
+
def _descritize_uniform(self, column: pd.Series) -> pd.Series:
|
|
76
|
+
return pd.cut(
|
|
77
|
+
column, bins=self.n_bins, labels=False, retbins=True, duplicates="drop"
|
|
78
|
+
)[0].values
|
|
79
|
+
|
|
80
|
+
def _get_numerical_columns(self, dataframe: pd.DataFrame) -> List[str]:
|
|
81
|
+
return dataframe.select_dtypes(include=np.number).columns.tolist()
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
from typing import Any, Dict, Optional, Sequence, Tuple
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from data_profiling.config import Settings
|
|
6
|
+
from data_profiling.model.duplicates import get_duplicates
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@get_duplicates.register(Settings, pd.DataFrame, Sequence)
|
|
10
|
+
def pandas_get_duplicates(
|
|
11
|
+
config: Settings, df: pd.DataFrame, supported_columns: Sequence
|
|
12
|
+
) -> Tuple[Dict[str, Any], Optional[pd.DataFrame]]:
|
|
13
|
+
"""Obtain the most occurring duplicate rows in the DataFrame.
|
|
14
|
+
|
|
15
|
+
Args:
|
|
16
|
+
config: report Settings object
|
|
17
|
+
df: the Pandas DataFrame.
|
|
18
|
+
supported_columns: the columns to consider
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
A subset of the DataFrame, ordered by occurrence.
|
|
22
|
+
"""
|
|
23
|
+
n_head = config.duplicates.head
|
|
24
|
+
|
|
25
|
+
metrics: Dict[str, Any] = {}
|
|
26
|
+
if n_head > 0:
|
|
27
|
+
if supported_columns and len(df) > 0:
|
|
28
|
+
duplicates_key = config.duplicates.key
|
|
29
|
+
if duplicates_key in df.columns:
|
|
30
|
+
raise ValueError(
|
|
31
|
+
f"Duplicates key ({duplicates_key}) may not be part of the DataFrame. Either change the "
|
|
32
|
+
f" column name in the DataFrame or change the 'duplicates.key' parameter."
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
duplicated_rows = df.duplicated(subset=supported_columns, keep=False)
|
|
36
|
+
duplicated_rows = (
|
|
37
|
+
df[duplicated_rows]
|
|
38
|
+
.rename_axis(index=lambda _: None)
|
|
39
|
+
.groupby(supported_columns, dropna=False, observed=True)
|
|
40
|
+
.size()
|
|
41
|
+
.reset_index(name=duplicates_key)
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
metrics["n_duplicates"] = len(duplicated_rows[duplicates_key])
|
|
45
|
+
metrics["p_duplicates"] = metrics["n_duplicates"] / len(df)
|
|
46
|
+
|
|
47
|
+
return (
|
|
48
|
+
metrics,
|
|
49
|
+
duplicated_rows.nlargest(n_head, duplicates_key),
|
|
50
|
+
)
|
|
51
|
+
else:
|
|
52
|
+
metrics["n_duplicates"] = 0
|
|
53
|
+
metrics["p_duplicates"] = 0.0
|
|
54
|
+
return metrics, None
|
|
55
|
+
else:
|
|
56
|
+
return metrics, None
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
from typing import Union
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from numpy import log2
|
|
5
|
+
from scipy.stats import entropy
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def column_imbalance_score(
|
|
9
|
+
value_counts: pd.Series, n_classes: int
|
|
10
|
+
) -> Union[float, int]:
|
|
11
|
+
"""column_imbalance_score
|
|
12
|
+
|
|
13
|
+
The class balance score for categorical and boolean variables uses entropy to calculate a bounded score between 0 and 1.
|
|
14
|
+
A perfectly uniform distribution would return a score of 0, and a perfectly imbalanced distribution would return a score of 1.
|
|
15
|
+
|
|
16
|
+
When dealing with probabilities with finite values (e.g categorical), entropy is maximised the ‘flatter’ the distribution is. (Jaynes: Probability Theory, The Logic of Science)
|
|
17
|
+
To calculate the class imbalance, we calculate the entropy of that distribution and the maximum possible entropy for that number of classes.
|
|
18
|
+
To calculate the entropy of the 'distribution' we use value counts (e.g frequency of classes) and we can determine the maximum entropy as log2(number of classes).
|
|
19
|
+
We then divide the entropy by the maximum possible entropy to get a value between 0 and 1 which we then subtract from 1.
|
|
20
|
+
|
|
21
|
+
Args:
|
|
22
|
+
value_counts (pd.Series): frequency of each category
|
|
23
|
+
n_classes (int): number of classes
|
|
24
|
+
|
|
25
|
+
Returns:
|
|
26
|
+
Union[float, int]: float or integer bounded between 0 and 1 inclusively
|
|
27
|
+
"""
|
|
28
|
+
# return 0 if there is only one class (when entropy =0) as it is balanced.
|
|
29
|
+
# note that this also prevents a zero division error with log2(n_classes)
|
|
30
|
+
if n_classes > 1:
|
|
31
|
+
# casting to numpy array to ensure correct dtype when a categorical integer
|
|
32
|
+
# variable is evaluated
|
|
33
|
+
value_counts = value_counts.to_numpy(dtype=float)
|
|
34
|
+
return 1 - (entropy(value_counts, base=2) / log2(n_classes))
|
|
35
|
+
return 0
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
import pandas as pd
|
|
3
|
+
|
|
4
|
+
from data_profiling.config import Settings
|
|
5
|
+
from data_profiling.visualisation.missing import (
|
|
6
|
+
plot_missing_bar,
|
|
7
|
+
plot_missing_heatmap,
|
|
8
|
+
plot_missing_matrix,
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def missing_bar(config: Settings, df: pd.DataFrame) -> str:
|
|
13
|
+
notnull_counts = len(df) - df.isnull().sum()
|
|
14
|
+
return plot_missing_bar(
|
|
15
|
+
config,
|
|
16
|
+
notnull_counts=notnull_counts,
|
|
17
|
+
nrows=len(df),
|
|
18
|
+
columns=list(df.columns),
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def missing_matrix(config: Settings, df: pd.DataFrame) -> str:
|
|
23
|
+
return plot_missing_matrix(
|
|
24
|
+
config,
|
|
25
|
+
columns=list(df.columns),
|
|
26
|
+
notnull=df.notnull().values,
|
|
27
|
+
nrows=len(df),
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def missing_heatmap(config: Settings, df: pd.DataFrame) -> str:
|
|
32
|
+
# Remove completely filled or completely empty variables.
|
|
33
|
+
columns = [i for i, n in enumerate(np.var(df.isnull(), axis="rows")) if n > 0]
|
|
34
|
+
df = df.iloc[:, columns]
|
|
35
|
+
|
|
36
|
+
# Create and mask the correlation matrix. Construct the base heatmap.
|
|
37
|
+
corr_mat = df.isnull().corr()
|
|
38
|
+
mask = np.zeros_like(corr_mat)
|
|
39
|
+
mask[np.triu_indices_from(mask)] = True
|
|
40
|
+
return plot_missing_heatmap(
|
|
41
|
+
config, corr_mat=corr_mat, mask=mask, columns=list(df.columns)
|
|
42
|
+
)
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
from typing import List
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from data_profiling.config import Settings
|
|
6
|
+
from data_profiling.model.sample import Sample, get_sample
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@get_sample.register(Settings, pd.DataFrame)
|
|
10
|
+
def pandas_get_sample(config: Settings, df: pd.DataFrame) -> List[Sample]:
|
|
11
|
+
"""Obtains a sample from head and tail of the DataFrame
|
|
12
|
+
|
|
13
|
+
Args:
|
|
14
|
+
config: Settings object
|
|
15
|
+
df: the pandas DataFrame
|
|
16
|
+
|
|
17
|
+
Returns:
|
|
18
|
+
a list of Sample objects
|
|
19
|
+
"""
|
|
20
|
+
samples: List[Sample] = []
|
|
21
|
+
if len(df) == 0:
|
|
22
|
+
return samples
|
|
23
|
+
|
|
24
|
+
n_head = config.samples.head
|
|
25
|
+
if n_head > 0:
|
|
26
|
+
samples.append(Sample(id="head", data=df.head(n=n_head), name="First rows"))
|
|
27
|
+
|
|
28
|
+
n_tail = config.samples.tail
|
|
29
|
+
if n_tail > 0:
|
|
30
|
+
samples.append(Sample(id="tail", data=df.tail(n=n_tail), name="Last rows"))
|
|
31
|
+
|
|
32
|
+
n_random = config.samples.random
|
|
33
|
+
if n_random > 0:
|
|
34
|
+
samples.append(
|
|
35
|
+
Sample(id="random", data=df.sample(n=n_random), name="Random sample")
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
return samples
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""Compute statistical description of datasets."""
|
|
2
|
+
import multiprocessing
|
|
3
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
4
|
+
from typing import Any, Tuple
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
from tqdm import tqdm
|
|
9
|
+
from visions import VisionsTypeset
|
|
10
|
+
|
|
11
|
+
from data_profiling.config import Settings
|
|
12
|
+
from data_profiling.model.typeset import ProfilingTypeSet
|
|
13
|
+
from data_profiling.utils.compat import optional_option_context
|
|
14
|
+
from data_profiling.utils.dataframe import sort_column_names
|
|
15
|
+
|
|
16
|
+
BaseSummarizer: Any = "BaseSummarizer" # type: ignore
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _is_cast_type_defined(typeset: VisionsTypeset, series: str) -> bool:
|
|
20
|
+
return isinstance(typeset, ProfilingTypeSet) and series in typeset.type_schema
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def pandas_describe_1d(
|
|
24
|
+
config: Settings,
|
|
25
|
+
series: pd.Series,
|
|
26
|
+
summarizer: BaseSummarizer,
|
|
27
|
+
typeset: VisionsTypeset,
|
|
28
|
+
) -> dict:
|
|
29
|
+
"""Describe a series (infer the variable type, then calculate type-specific values).
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
config: report Settings object
|
|
33
|
+
series: The Series to describe.
|
|
34
|
+
summarizer: Summarizer object
|
|
35
|
+
typeset: Typeset
|
|
36
|
+
|
|
37
|
+
Returns:
|
|
38
|
+
A Series containing calculated series description values.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
# Make sure pd.NA is not in the series
|
|
42
|
+
with optional_option_context("future.no_silent_downcasting", True):
|
|
43
|
+
series = series.fillna(np.nan).infer_objects(copy=False)
|
|
44
|
+
|
|
45
|
+
has_cast_type = _is_cast_type_defined(typeset, series.name) # type:ignore
|
|
46
|
+
cast_type = (
|
|
47
|
+
str(typeset.type_schema[series.name]) if has_cast_type else None
|
|
48
|
+
) # type:ignore
|
|
49
|
+
|
|
50
|
+
if has_cast_type and not series.isna().all():
|
|
51
|
+
vtype = typeset.type_schema[series.name] # type:ignore
|
|
52
|
+
|
|
53
|
+
elif config.infer_dtypes:
|
|
54
|
+
# Infer variable types
|
|
55
|
+
vtype = typeset.infer_type(series)
|
|
56
|
+
series = typeset.cast_to_inferred(series)
|
|
57
|
+
else:
|
|
58
|
+
# Detect variable types from pandas dataframe (df.dtypes).
|
|
59
|
+
# [new dtypes, changed using `astype` function are now considered]
|
|
60
|
+
vtype = typeset.detect_type(series)
|
|
61
|
+
|
|
62
|
+
typeset.type_schema[series.name] = vtype # type:ignore
|
|
63
|
+
summary = summarizer.summarize(config, series, dtype=vtype)
|
|
64
|
+
# Cast type is only used on unsupported columns rendering pipeline
|
|
65
|
+
# to indicate the correct variable type when inference is not possible
|
|
66
|
+
summary["cast_type"] = cast_type
|
|
67
|
+
|
|
68
|
+
return summary
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def pandas_get_series_descriptions(
|
|
72
|
+
config: Settings,
|
|
73
|
+
df: pd.DataFrame,
|
|
74
|
+
summarizer: BaseSummarizer,
|
|
75
|
+
typeset: VisionsTypeset,
|
|
76
|
+
pbar: tqdm,
|
|
77
|
+
) -> dict:
|
|
78
|
+
def describe_column(name: str, series: pd.Series) -> Tuple[str, dict]:
|
|
79
|
+
"""Process a single series to get the column description."""
|
|
80
|
+
pbar.set_postfix_str(f"Describe variable: {name}")
|
|
81
|
+
description = pandas_describe_1d(config, series, summarizer, typeset)
|
|
82
|
+
pbar.update()
|
|
83
|
+
return name, description
|
|
84
|
+
|
|
85
|
+
pool_size = (
|
|
86
|
+
config.pool_size if config.pool_size > 0 else multiprocessing.cpu_count()
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
series_description = {}
|
|
90
|
+
|
|
91
|
+
with ThreadPoolExecutor(max_workers=pool_size) as executor:
|
|
92
|
+
future_to_col = {
|
|
93
|
+
executor.submit(describe_column, name, series): name # type:ignore
|
|
94
|
+
for name, series in df.items()
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
for future in tqdm(future_to_col.keys(), total=len(future_to_col)):
|
|
98
|
+
name, description = future.result()
|
|
99
|
+
series_description[name] = description
|
|
100
|
+
|
|
101
|
+
return sort_column_names(series_description, config.sort)
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
from collections import Counter
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from data_profiling.config import Settings
|
|
6
|
+
from data_profiling.model.table import get_table_stats
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@get_table_stats.register
|
|
10
|
+
def pandas_get_table_stats(
|
|
11
|
+
config: Settings, df: pd.DataFrame, variable_stats: dict
|
|
12
|
+
) -> dict:
|
|
13
|
+
"""General statistics for the DataFrame.
|
|
14
|
+
|
|
15
|
+
Args:
|
|
16
|
+
config: report Settings object
|
|
17
|
+
df: The DataFrame to describe.
|
|
18
|
+
variable_stats: Previously calculated statistic on the DataFrame.
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
A dictionary that contains the table statistics.
|
|
22
|
+
"""
|
|
23
|
+
n = len(df) if not df.empty else 0
|
|
24
|
+
|
|
25
|
+
memory_size = df.memory_usage(deep=config.memory_deep).sum()
|
|
26
|
+
record_size = float(memory_size) / n if n > 0 else 0
|
|
27
|
+
|
|
28
|
+
table_stats = {
|
|
29
|
+
"n": n,
|
|
30
|
+
"n_var": len(df.columns),
|
|
31
|
+
"memory_size": memory_size,
|
|
32
|
+
"record_size": record_size,
|
|
33
|
+
"n_cells_missing": 0,
|
|
34
|
+
"n_vars_with_missing": 0,
|
|
35
|
+
"n_vars_all_missing": 0,
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
for series_summary in variable_stats.values():
|
|
39
|
+
if "n_missing" in series_summary and series_summary["n_missing"] > 0:
|
|
40
|
+
table_stats["n_vars_with_missing"] += 1
|
|
41
|
+
table_stats["n_cells_missing"] += series_summary["n_missing"]
|
|
42
|
+
if series_summary["n_missing"] == n:
|
|
43
|
+
table_stats["n_vars_all_missing"] += 1
|
|
44
|
+
|
|
45
|
+
table_stats["p_cells_missing"] = (
|
|
46
|
+
table_stats["n_cells_missing"] / (table_stats["n"] * table_stats["n_var"])
|
|
47
|
+
if table_stats["n"] > 0 and table_stats["n_var"] > 0
|
|
48
|
+
else 0
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
# Variable type counts
|
|
52
|
+
table_stats.update(
|
|
53
|
+
{"types": dict(Counter([v["type"] for v in variable_stats.values()]))}
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
return table_stats
|