fg-data-profiling 4.19.0__py2.py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (238) hide show
  1. data_profiling/__init__.py +34 -0
  2. data_profiling/compare_reports.py +359 -0
  3. data_profiling/config.py +496 -0
  4. data_profiling/config_default.yaml +223 -0
  5. data_profiling/config_minimal.yaml +222 -0
  6. data_profiling/controller/__init__.py +1 -0
  7. data_profiling/controller/console.py +125 -0
  8. data_profiling/controller/pandas_decorator.py +21 -0
  9. data_profiling/expectations_report.py +117 -0
  10. data_profiling/model/__init__.py +4 -0
  11. data_profiling/model/alerts.py +780 -0
  12. data_profiling/model/correlations.py +163 -0
  13. data_profiling/model/dataframe.py +35 -0
  14. data_profiling/model/describe.py +210 -0
  15. data_profiling/model/description.py +108 -0
  16. data_profiling/model/duplicates.py +14 -0
  17. data_profiling/model/expectation_algorithms.py +112 -0
  18. data_profiling/model/handler.py +81 -0
  19. data_profiling/model/missing.py +146 -0
  20. data_profiling/model/pairwise.py +33 -0
  21. data_profiling/model/pandas/__init__.py +55 -0
  22. data_profiling/model/pandas/correlations_pandas.py +207 -0
  23. data_profiling/model/pandas/dataframe_pandas.py +26 -0
  24. data_profiling/model/pandas/describe_boolean_pandas.py +43 -0
  25. data_profiling/model/pandas/describe_categorical_pandas.py +274 -0
  26. data_profiling/model/pandas/describe_counts_pandas.py +63 -0
  27. data_profiling/model/pandas/describe_date_pandas.py +77 -0
  28. data_profiling/model/pandas/describe_file_pandas.py +56 -0
  29. data_profiling/model/pandas/describe_generic_pandas.py +36 -0
  30. data_profiling/model/pandas/describe_image_pandas.py +255 -0
  31. data_profiling/model/pandas/describe_numeric_pandas.py +175 -0
  32. data_profiling/model/pandas/describe_path_pandas.py +63 -0
  33. data_profiling/model/pandas/describe_supported_pandas.py +41 -0
  34. data_profiling/model/pandas/describe_text_pandas.py +62 -0
  35. data_profiling/model/pandas/describe_timeseries_pandas.py +222 -0
  36. data_profiling/model/pandas/describe_url_pandas.py +57 -0
  37. data_profiling/model/pandas/discretize_pandas.py +81 -0
  38. data_profiling/model/pandas/duplicates_pandas.py +56 -0
  39. data_profiling/model/pandas/imbalance_pandas.py +35 -0
  40. data_profiling/model/pandas/missing_pandas.py +42 -0
  41. data_profiling/model/pandas/sample_pandas.py +38 -0
  42. data_profiling/model/pandas/summary_pandas.py +101 -0
  43. data_profiling/model/pandas/table_pandas.py +56 -0
  44. data_profiling/model/pandas/timeseries_index_pandas.py +33 -0
  45. data_profiling/model/pandas/utils_pandas.py +27 -0
  46. data_profiling/model/sample.py +37 -0
  47. data_profiling/model/spark/__init__.py +48 -0
  48. data_profiling/model/spark/correlations_spark.py +152 -0
  49. data_profiling/model/spark/dataframe_spark.py +34 -0
  50. data_profiling/model/spark/describe_boolean_spark.py +27 -0
  51. data_profiling/model/spark/describe_categorical_spark.py +28 -0
  52. data_profiling/model/spark/describe_counts_spark.py +105 -0
  53. data_profiling/model/spark/describe_date_spark.py +51 -0
  54. data_profiling/model/spark/describe_generic_spark.py +30 -0
  55. data_profiling/model/spark/describe_numeric_spark.py +155 -0
  56. data_profiling/model/spark/describe_supported_spark.py +33 -0
  57. data_profiling/model/spark/describe_text_spark.py +25 -0
  58. data_profiling/model/spark/duplicates_spark.py +54 -0
  59. data_profiling/model/spark/missing_spark.py +96 -0
  60. data_profiling/model/spark/sample_spark.py +43 -0
  61. data_profiling/model/spark/summary_spark.py +95 -0
  62. data_profiling/model/spark/table_spark.py +58 -0
  63. data_profiling/model/spark/timeseries_index_spark.py +12 -0
  64. data_profiling/model/summarizer.py +207 -0
  65. data_profiling/model/summary.py +66 -0
  66. data_profiling/model/summary_algorithms.py +276 -0
  67. data_profiling/model/table.py +10 -0
  68. data_profiling/model/timeseries_index.py +16 -0
  69. data_profiling/model/typeset.py +365 -0
  70. data_profiling/model/typeset_relations.py +143 -0
  71. data_profiling/profile_report.py +573 -0
  72. data_profiling/report/__init__.py +4 -0
  73. data_profiling/report/formatters.py +346 -0
  74. data_profiling/report/presentation/__init__.py +1 -0
  75. data_profiling/report/presentation/core/__init__.py +39 -0
  76. data_profiling/report/presentation/core/alerts.py +18 -0
  77. data_profiling/report/presentation/core/collapse.py +24 -0
  78. data_profiling/report/presentation/core/container.py +50 -0
  79. data_profiling/report/presentation/core/correlation_table.py +21 -0
  80. data_profiling/report/presentation/core/dropdown.py +44 -0
  81. data_profiling/report/presentation/core/duplicate.py +16 -0
  82. data_profiling/report/presentation/core/frequency_table.py +14 -0
  83. data_profiling/report/presentation/core/frequency_table_small.py +16 -0
  84. data_profiling/report/presentation/core/html.py +14 -0
  85. data_profiling/report/presentation/core/image.py +34 -0
  86. data_profiling/report/presentation/core/item_renderer.py +17 -0
  87. data_profiling/report/presentation/core/renderable.py +42 -0
  88. data_profiling/report/presentation/core/root.py +35 -0
  89. data_profiling/report/presentation/core/sample.py +20 -0
  90. data_profiling/report/presentation/core/scores.py +32 -0
  91. data_profiling/report/presentation/core/table.py +26 -0
  92. data_profiling/report/presentation/core/toggle_button.py +14 -0
  93. data_profiling/report/presentation/core/variable.py +40 -0
  94. data_profiling/report/presentation/core/variable_info.py +36 -0
  95. data_profiling/report/presentation/flavours/__init__.py +9 -0
  96. data_profiling/report/presentation/flavours/flavour_html.py +64 -0
  97. data_profiling/report/presentation/flavours/flavour_widget.py +61 -0
  98. data_profiling/report/presentation/flavours/flavours.py +43 -0
  99. data_profiling/report/presentation/flavours/html/__init__.py +47 -0
  100. data_profiling/report/presentation/flavours/html/alerts.py +10 -0
  101. data_profiling/report/presentation/flavours/html/collapse.py +7 -0
  102. data_profiling/report/presentation/flavours/html/container.py +58 -0
  103. data_profiling/report/presentation/flavours/html/correlation_table.py +13 -0
  104. data_profiling/report/presentation/flavours/html/dropdown.py +7 -0
  105. data_profiling/report/presentation/flavours/html/duplicate.py +24 -0
  106. data_profiling/report/presentation/flavours/html/frequency_table.py +20 -0
  107. data_profiling/report/presentation/flavours/html/frequency_table_small.py +15 -0
  108. data_profiling/report/presentation/flavours/html/html.py +6 -0
  109. data_profiling/report/presentation/flavours/html/image.py +7 -0
  110. data_profiling/report/presentation/flavours/html/root.py +14 -0
  111. data_profiling/report/presentation/flavours/html/sample.py +12 -0
  112. data_profiling/report/presentation/flavours/html/scores.py +11 -0
  113. data_profiling/report/presentation/flavours/html/table.py +7 -0
  114. data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant.html +1 -0
  115. data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant_length.html +1 -0
  116. data_profiling/report/presentation/flavours/html/templates/alerts/alert_dirty_category.html +1 -0
  117. data_profiling/report/presentation/flavours/html/templates/alerts/alert_duplicates.html +1 -0
  118. data_profiling/report/presentation/flavours/html/templates/alerts/alert_empty.html +1 -0
  119. data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_cardinality.html +1 -0
  120. data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_correlation.html +4 -0
  121. data_profiling/report/presentation/flavours/html/templates/alerts/alert_imbalance.html +1 -0
  122. data_profiling/report/presentation/flavours/html/templates/alerts/alert_infinite.html +1 -0
  123. data_profiling/report/presentation/flavours/html/templates/alerts/alert_missing.html +1 -0
  124. data_profiling/report/presentation/flavours/html/templates/alerts/alert_near_duplicates.html +1 -0
  125. data_profiling/report/presentation/flavours/html/templates/alerts/alert_non_stationary.html +1 -0
  126. data_profiling/report/presentation/flavours/html/templates/alerts/alert_seasonal.html +1 -0
  127. data_profiling/report/presentation/flavours/html/templates/alerts/alert_skewed.html +1 -0
  128. data_profiling/report/presentation/flavours/html/templates/alerts/alert_truncated.html +1 -0
  129. data_profiling/report/presentation/flavours/html/templates/alerts/alert_type_date.html +1 -0
  130. data_profiling/report/presentation/flavours/html/templates/alerts/alert_uniform.html +1 -0
  131. data_profiling/report/presentation/flavours/html/templates/alerts/alert_unique.html +1 -0
  132. data_profiling/report/presentation/flavours/html/templates/alerts/alert_unsupported.html +1 -0
  133. data_profiling/report/presentation/flavours/html/templates/alerts/alert_zeros.html +1 -0
  134. data_profiling/report/presentation/flavours/html/templates/alerts.html +47 -0
  135. data_profiling/report/presentation/flavours/html/templates/collapse.html +11 -0
  136. data_profiling/report/presentation/flavours/html/templates/correlation_table.html +5 -0
  137. data_profiling/report/presentation/flavours/html/templates/diagram.html +11 -0
  138. data_profiling/report/presentation/flavours/html/templates/dropdown.html +16 -0
  139. data_profiling/report/presentation/flavours/html/templates/duplicate.html +5 -0
  140. data_profiling/report/presentation/flavours/html/templates/frequency_table.html +45 -0
  141. data_profiling/report/presentation/flavours/html/templates/frequency_table_small.html +34 -0
  142. data_profiling/report/presentation/flavours/html/templates/report.html +26 -0
  143. data_profiling/report/presentation/flavours/html/templates/sample.html +10 -0
  144. data_profiling/report/presentation/flavours/html/templates/scores.html +78 -0
  145. data_profiling/report/presentation/flavours/html/templates/sequence/batch_grid.html +16 -0
  146. data_profiling/report/presentation/flavours/html/templates/sequence/grid.html +18 -0
  147. data_profiling/report/presentation/flavours/html/templates/sequence/list.html +7 -0
  148. data_profiling/report/presentation/flavours/html/templates/sequence/named_list.html +8 -0
  149. data_profiling/report/presentation/flavours/html/templates/sequence/overview_tabs.html +30 -0
  150. data_profiling/report/presentation/flavours/html/templates/sequence/scores.html +3 -0
  151. data_profiling/report/presentation/flavours/html/templates/sequence/sections.html +13 -0
  152. data_profiling/report/presentation/flavours/html/templates/sequence/select.html +40 -0
  153. data_profiling/report/presentation/flavours/html/templates/sequence/tabs.html +30 -0
  154. data_profiling/report/presentation/flavours/html/templates/table.html +38 -0
  155. data_profiling/report/presentation/flavours/html/templates/toggle_button.html +18 -0
  156. data_profiling/report/presentation/flavours/html/templates/variable.html +7 -0
  157. data_profiling/report/presentation/flavours/html/templates/variable_info.html +49 -0
  158. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.bundle.min.js +7 -0
  159. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.min.css +6 -0
  160. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/cosmo.bootstrap.min.css +12 -0
  161. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/flatly.bootstrap.min.css +12 -0
  162. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/script.js +52 -0
  163. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/simplex.bootstrap.min.css +12 -0
  164. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/style.css +253 -0
  165. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/united.bootstrap.min.css +12 -0
  166. data_profiling/report/presentation/flavours/html/templates/wrapper/footer.html +7 -0
  167. data_profiling/report/presentation/flavours/html/templates/wrapper/javascript.html +18 -0
  168. data_profiling/report/presentation/flavours/html/templates/wrapper/navigation.html +36 -0
  169. data_profiling/report/presentation/flavours/html/templates/wrapper/style.html +53 -0
  170. data_profiling/report/presentation/flavours/html/templates.py +76 -0
  171. data_profiling/report/presentation/flavours/html/toggle_button.py +7 -0
  172. data_profiling/report/presentation/flavours/html/variable.py +7 -0
  173. data_profiling/report/presentation/flavours/html/variable_info.py +7 -0
  174. data_profiling/report/presentation/flavours/widget/__init__.py +49 -0
  175. data_profiling/report/presentation/flavours/widget/alerts.py +45 -0
  176. data_profiling/report/presentation/flavours/widget/collapse.py +43 -0
  177. data_profiling/report/presentation/flavours/widget/container.py +121 -0
  178. data_profiling/report/presentation/flavours/widget/correlation_table.py +14 -0
  179. data_profiling/report/presentation/flavours/widget/dropdown.py +31 -0
  180. data_profiling/report/presentation/flavours/widget/duplicate.py +14 -0
  181. data_profiling/report/presentation/flavours/widget/frequency_table.py +57 -0
  182. data_profiling/report/presentation/flavours/widget/frequency_table_small.py +66 -0
  183. data_profiling/report/presentation/flavours/widget/html.py +11 -0
  184. data_profiling/report/presentation/flavours/widget/image.py +26 -0
  185. data_profiling/report/presentation/flavours/widget/notebook.py +81 -0
  186. data_profiling/report/presentation/flavours/widget/root.py +10 -0
  187. data_profiling/report/presentation/flavours/widget/sample.py +14 -0
  188. data_profiling/report/presentation/flavours/widget/table.py +30 -0
  189. data_profiling/report/presentation/flavours/widget/toggle_button.py +17 -0
  190. data_profiling/report/presentation/flavours/widget/variable.py +12 -0
  191. data_profiling/report/presentation/flavours/widget/variable_info.py +11 -0
  192. data_profiling/report/presentation/frequency_table_utils.py +141 -0
  193. data_profiling/report/structure/__init__.py +1 -0
  194. data_profiling/report/structure/correlations.py +123 -0
  195. data_profiling/report/structure/overview.py +376 -0
  196. data_profiling/report/structure/report.py +457 -0
  197. data_profiling/report/structure/variables/__init__.py +35 -0
  198. data_profiling/report/structure/variables/render_boolean.py +132 -0
  199. data_profiling/report/structure/variables/render_categorical.py +566 -0
  200. data_profiling/report/structure/variables/render_common.py +31 -0
  201. data_profiling/report/structure/variables/render_complex.py +102 -0
  202. data_profiling/report/structure/variables/render_count.py +172 -0
  203. data_profiling/report/structure/variables/render_date.py +143 -0
  204. data_profiling/report/structure/variables/render_file.py +70 -0
  205. data_profiling/report/structure/variables/render_generic.py +45 -0
  206. data_profiling/report/structure/variables/render_image.py +204 -0
  207. data_profiling/report/structure/variables/render_path.py +134 -0
  208. data_profiling/report/structure/variables/render_real.py +314 -0
  209. data_profiling/report/structure/variables/render_text.py +189 -0
  210. data_profiling/report/structure/variables/render_timeseries.py +371 -0
  211. data_profiling/report/structure/variables/render_url.py +132 -0
  212. data_profiling/report/utils.py +34 -0
  213. data_profiling/serialize_report.py +143 -0
  214. data_profiling/utils/__init__.py +1 -0
  215. data_profiling/utils/backend.py +9 -0
  216. data_profiling/utils/cache.py +59 -0
  217. data_profiling/utils/common.py +142 -0
  218. data_profiling/utils/compat.py +31 -0
  219. data_profiling/utils/dataframe.py +238 -0
  220. data_profiling/utils/logger.py +53 -0
  221. data_profiling/utils/notebook.py +8 -0
  222. data_profiling/utils/paths.py +45 -0
  223. data_profiling/utils/progress_bar.py +15 -0
  224. data_profiling/utils/styles.py +22 -0
  225. data_profiling/utils/versions.py +19 -0
  226. data_profiling/version.py +1 -0
  227. data_profiling/visualisation/__init__.py +1 -0
  228. data_profiling/visualisation/context.py +87 -0
  229. data_profiling/visualisation/missing.py +138 -0
  230. data_profiling/visualisation/plot.py +1158 -0
  231. data_profiling/visualisation/utils.py +113 -0
  232. fg_data_profiling-4.19.0.dist-info/METADATA +362 -0
  233. fg_data_profiling-4.19.0.dist-info/RECORD +238 -0
  234. fg_data_profiling-4.19.0.dist-info/WHEEL +6 -0
  235. fg_data_profiling-4.19.0.dist-info/entry_points.txt +3 -0
  236. fg_data_profiling-4.19.0.dist-info/licenses/LICENSE +21 -0
  237. fg_data_profiling-4.19.0.dist-info/top_level.txt +2 -0
  238. ydata_profiling/__init__.py +43 -0
@@ -0,0 +1,222 @@
1
+ from typing import Any, Dict, Tuple
2
+
3
+ import numpy as np
4
+ import pandas as pd
5
+ from scipy.fft import _pocketfft
6
+ from scipy.signal import find_peaks
7
+ from statsmodels.tsa.stattools import adfuller
8
+
9
+ from data_profiling.config import Settings
10
+ from data_profiling.model.summary_algorithms import (
11
+ describe_numeric_1d,
12
+ describe_timeseries_1d,
13
+ series_handle_nulls,
14
+ series_hashable,
15
+ )
16
+
17
+
18
+ def stationarity_test(config: Settings, series: pd.Series) -> Tuple[bool, float]:
19
+ # make sure the data has no missing values
20
+ adfuller_test = adfuller(
21
+ series.dropna(),
22
+ autolag=config.vars.timeseries.autolag,
23
+ maxlag=config.vars.timeseries.maxlag,
24
+ )
25
+ p_value = adfuller_test[1]
26
+
27
+ significance_threshold = config.vars.timeseries.significance
28
+ return p_value < significance_threshold, p_value
29
+
30
+
31
+ def fftfreq(n: int, d: float = 1.0) -> np.ndarray:
32
+ """
33
+ Return the Discrete Fourier Transform sample frequencies.
34
+
35
+ Args:
36
+ n : int
37
+ Window length.
38
+ d : scalar, optional
39
+ Sample spacing (inverse of the sampling rate). Defaults to 1.
40
+
41
+ Returns:
42
+ f : ndarray
43
+ Array of length `n` containing the sample frequencies.
44
+ """
45
+ val = 1.0 / (n * d)
46
+ results = np.empty(n, int)
47
+ N = (n - 1) // 2 + 1
48
+ p1 = np.arange(0, N, dtype=int)
49
+ results[:N] = p1
50
+ p2 = np.arange(-(n // 2), 0, dtype=int)
51
+ results[N:] = p2
52
+ return results * val
53
+
54
+
55
+ def seasonality_test(series: pd.Series, mad_threshold: float = 6.0) -> Dict[str, Any]:
56
+ """Detect seasonality with FFT
57
+
58
+ Source: https://github.com/facebookresearch/Kats/blob/main/kats/detectors/seasonality.py
59
+
60
+ Args:
61
+ mad_threshold: Optional; float; constant for the outlier algorithm for peak
62
+ detector. The larger the value the less sensitive the outlier algorithm
63
+ is.
64
+
65
+ Returns:
66
+ FFT Plot with peaks, selected peaks, and outlier boundary line.
67
+ """
68
+
69
+ fft = get_fft(series)
70
+ _, _, peaks = get_fft_peaks(fft, mad_threshold)
71
+ seasonality_presence = len(peaks.index) > 0
72
+ selected_seasonalities = []
73
+ if seasonality_presence:
74
+ selected_seasonalities = peaks["freq"].transform(lambda x: 1 / x).tolist()
75
+
76
+ return {
77
+ "seasonality_presence": seasonality_presence,
78
+ "seasonalities": selected_seasonalities,
79
+ }
80
+
81
+
82
+ def get_fft(series: pd.Series) -> pd.DataFrame:
83
+ """Computes FFT
84
+
85
+ Args:
86
+ series: pd.Series
87
+ time series
88
+
89
+ Returns:
90
+ DataFrame with columns 'freq' and 'ampl'.
91
+ """
92
+ data_fft = _pocketfft.fft(series.to_numpy())
93
+ data_psd = np.abs(data_fft) ** 2
94
+ fftfreq_ = fftfreq(len(data_psd), 1.0)
95
+ pos_freq_ix = fftfreq_ > 0
96
+
97
+ freq = fftfreq_[pos_freq_ix]
98
+ ampl = 10 * np.log10(data_psd[pos_freq_ix])
99
+
100
+ return pd.DataFrame({"freq": freq, "ampl": ampl})
101
+
102
+
103
+ def get_fft_peaks(
104
+ fft: pd.DataFrame, mad_threshold: float = 6.0
105
+ ) -> Tuple[float, pd.DataFrame, pd.DataFrame]:
106
+ """Computes peaks in fft, selects the highest peaks (outliers) and
107
+ removes the harmonics (multiplies of the base harmonics found)
108
+
109
+ Args:
110
+ fft: FFT computed by get_fft
111
+ mad_threshold: Optional; constant for the outlier algorithm for peak detector.
112
+ The larger the value the less sensitive the outlier algorithm is.
113
+
114
+ Returns:
115
+ outlier threshold, peaks, selected peaks.
116
+ """
117
+ pos_fft = fft.loc[fft["ampl"] > 0]
118
+ median = pos_fft["ampl"].median()
119
+ pos_fft_above_med = pos_fft[pos_fft["ampl"] > median]
120
+ mad = abs(pos_fft_above_med["ampl"] - pos_fft_above_med["ampl"].mean()).mean()
121
+
122
+ threshold = median + mad * mad_threshold
123
+
124
+ peak_indices = find_peaks(fft["ampl"], threshold=0.1)
125
+ peaks = fft.loc[peak_indices[0], :]
126
+
127
+ orig_peaks = peaks.copy()
128
+
129
+ peaks = peaks.loc[peaks["ampl"] > threshold].copy()
130
+ peaks["Remove"] = [False] * len(peaks.index)
131
+ peaks.reset_index(inplace=True)
132
+
133
+ # Filter out harmonics
134
+ for idx1 in range(len(peaks)):
135
+ curr = peaks.loc[idx1, "freq"]
136
+ for idx2 in range(idx1 + 1, len(peaks)):
137
+ if peaks.loc[idx2, "Remove"] is True:
138
+ continue
139
+ fraction = (peaks.loc[idx2, "freq"] / curr) % 1
140
+ if fraction < 0.01 or fraction > 0.99:
141
+ peaks.loc[idx2, "Remove"] = True
142
+ peaks = peaks.loc[~peaks["Remove"]]
143
+ peaks.drop(inplace=True, columns="Remove")
144
+ return threshold, orig_peaks, peaks
145
+
146
+
147
+ def identify_gaps(
148
+ gap: pd.Series, is_datetime: bool, gap_tolerance: int = 2
149
+ ) -> Tuple[pd.Series, list]:
150
+ zero = pd.Timedelta(0) if is_datetime else 0
151
+ diff = gap.diff()
152
+
153
+ non_zero_diff = diff[diff > zero]
154
+ min_gap_size = gap_tolerance * non_zero_diff.mean()
155
+
156
+ gap_stats = non_zero_diff[non_zero_diff > min_gap_size]
157
+ anchors = gap[diff > min_gap_size].index
158
+
159
+ gaps = []
160
+ for i in anchors:
161
+ gaps.append(gap.loc[gap.index[[i - 1, i]]].values)
162
+
163
+ return gap_stats, gaps
164
+
165
+
166
+ def compute_gap_stats(series: pd.Series) -> pd.Series:
167
+ """Computes the intertevals in the series normalized by the period.
168
+
169
+ Args:
170
+ series (pd.Series): time series data to analysis.
171
+
172
+ Returns:
173
+ A series with the gaps intervals.
174
+ """
175
+
176
+ gap = series.dropna()
177
+ index_name = gap.index.name if gap.index.name else "index"
178
+ gap = gap.reset_index()[index_name]
179
+ gap.index.name = None
180
+
181
+ is_datetime = isinstance(series.index, pd.DatetimeIndex)
182
+ gap_stats, gaps = identify_gaps(gap, is_datetime)
183
+ has_gaps = len(gap_stats) > 0
184
+
185
+ stats = {
186
+ "min": gap_stats.min() if has_gaps else 0,
187
+ "max": gap_stats.max() if has_gaps else 0,
188
+ "mean": gap_stats.mean() if has_gaps else 0,
189
+ "std": gap_stats.std() if len(gap_stats) > 1 else 0,
190
+ "series": series,
191
+ "gaps": gaps,
192
+ "n_gaps": len(gaps),
193
+ }
194
+ return stats
195
+
196
+
197
+ @describe_timeseries_1d.register
198
+ @series_hashable
199
+ @series_handle_nulls
200
+ def pandas_describe_timeseries_1d(
201
+ config: Settings, series: pd.Series, summary: dict
202
+ ) -> Tuple[Settings, pd.Series, dict]:
203
+ """Describe a timeseries.
204
+
205
+ Args:
206
+ config: report Settings object
207
+ series: The Series to describe.
208
+ summary: The dict containing the series description so far.
209
+
210
+ Returns:
211
+ A dict containing calculated series description values.
212
+ """
213
+ config, series, stats = describe_numeric_1d(config, series, summary)
214
+
215
+ stats["seasonal"] = seasonality_test(series)["seasonality_presence"]
216
+ is_stationary, p_value = stationarity_test(config, series)
217
+ stats["stationary"] = is_stationary and not stats["seasonal"]
218
+ stats["addfuller"] = p_value
219
+ stats["series"] = series
220
+ stats["gap_stats"] = compute_gap_stats(series)
221
+
222
+ return config, series, stats
@@ -0,0 +1,57 @@
1
+ from typing import Tuple
2
+ from urllib.parse import urlsplit
3
+
4
+ import pandas as pd
5
+
6
+ from data_profiling.config import Settings
7
+ from data_profiling.model.summary_algorithms import describe_url_1d
8
+
9
+
10
+ def url_summary(series: pd.Series) -> dict:
11
+ """
12
+
13
+ Args:
14
+ series: series to summarize
15
+
16
+ Returns:
17
+
18
+ """
19
+ summary = {
20
+ "scheme_counts": series.map(lambda x: x.scheme).value_counts(),
21
+ "netloc_counts": series.map(lambda x: x.netloc).value_counts(),
22
+ "path_counts": series.map(lambda x: x.path).value_counts(),
23
+ "query_counts": series.map(lambda x: x.query).value_counts(),
24
+ "fragment_counts": series.map(lambda x: x.fragment).value_counts(),
25
+ }
26
+
27
+ return summary
28
+
29
+
30
+ @describe_url_1d.register
31
+ def pandas_describe_url_1d(
32
+ config: Settings, series: pd.Series, summary: dict
33
+ ) -> Tuple[Settings, pd.Series, dict]:
34
+ """Describe a url series.
35
+
36
+ Args:
37
+ config: report Settings object
38
+ series: The Series to describe.
39
+ summary: The dict containing the series description so far.
40
+
41
+ Returns:
42
+ A dict containing calculated series description values.
43
+ """
44
+
45
+ # Make sure we deal with strings (Issue #100)
46
+ if series.hasnans:
47
+ raise ValueError("May not contain NaNs")
48
+ if not hasattr(series, "str"):
49
+ raise ValueError("series should have .str accessor")
50
+
51
+ # Transform
52
+ series = series.apply(urlsplit)
53
+
54
+ # Update
55
+ summary.update(url_summary(series))
56
+
57
+ return config, series, summary
@@ -0,0 +1,81 @@
1
+ from enum import Enum
2
+ from typing import List
3
+
4
+ import numpy as np
5
+ import pandas as pd
6
+
7
+
8
+ class DiscretizationType(Enum):
9
+ UNIFORM = "uniform"
10
+ QUANTILE = "quantile"
11
+
12
+
13
+ class Discretizer:
14
+ """
15
+ A class which enables the discretization of a pandas dataframe.
16
+ Perform this action when you want to convert a continuous variable
17
+ into a categorical variable.
18
+
19
+ Attributes:
20
+
21
+ method (DiscretizationType): this attribute controls how the buckets
22
+ of your discretization are formed. A uniform discretization type forms
23
+ the bins to be of equal width whereas a quantile discretization type
24
+ forms the bins to be of equal size.
25
+
26
+ n_bins (int): number of bins
27
+ reset_index (bool): instruction to reset the index of
28
+ the dataframe after the discretization
29
+ """
30
+
31
+ def __init__(
32
+ self, method: DiscretizationType, n_bins: int = 10, reset_index: bool = False
33
+ ) -> None:
34
+ self.discretization_type = method
35
+ self.n_bins = n_bins
36
+ self.reset_index = reset_index
37
+
38
+ def discretize_dataframe(self, dataframe: pd.DataFrame) -> pd.DataFrame:
39
+ """_summary_
40
+
41
+ Args:
42
+ dataframe (pd.DataFrame): pandas dataframe
43
+
44
+ Returns:
45
+ pd.DataFrame: discretized dataframe
46
+ """
47
+
48
+ discretized_df = dataframe.copy()
49
+ all_columns = dataframe.columns
50
+ num_columns = self._get_numerical_columns(dataframe)
51
+ for column in num_columns:
52
+ discretized_df.loc[:, column] = self._discretize_column(
53
+ discretized_df[column]
54
+ )
55
+
56
+ discretized_df = discretized_df[all_columns]
57
+ return (
58
+ discretized_df.reset_index(drop=True)
59
+ if self.reset_index
60
+ else discretized_df
61
+ )
62
+
63
+ def _discretize_column(self, column: pd.Series) -> pd.Series:
64
+ if self.discretization_type == DiscretizationType.QUANTILE:
65
+ return self._descritize_quantile(column)
66
+
67
+ elif self.discretization_type == DiscretizationType.UNIFORM:
68
+ return self._descritize_uniform(column)
69
+
70
+ def _descritize_quantile(self, column: pd.Series) -> pd.Series:
71
+ return pd.qcut(
72
+ column, q=self.n_bins, labels=False, retbins=False, duplicates="drop"
73
+ ).values
74
+
75
+ def _descritize_uniform(self, column: pd.Series) -> pd.Series:
76
+ return pd.cut(
77
+ column, bins=self.n_bins, labels=False, retbins=True, duplicates="drop"
78
+ )[0].values
79
+
80
+ def _get_numerical_columns(self, dataframe: pd.DataFrame) -> List[str]:
81
+ return dataframe.select_dtypes(include=np.number).columns.tolist()
@@ -0,0 +1,56 @@
1
+ from typing import Any, Dict, Optional, Sequence, Tuple
2
+
3
+ import pandas as pd
4
+
5
+ from data_profiling.config import Settings
6
+ from data_profiling.model.duplicates import get_duplicates
7
+
8
+
9
+ @get_duplicates.register(Settings, pd.DataFrame, Sequence)
10
+ def pandas_get_duplicates(
11
+ config: Settings, df: pd.DataFrame, supported_columns: Sequence
12
+ ) -> Tuple[Dict[str, Any], Optional[pd.DataFrame]]:
13
+ """Obtain the most occurring duplicate rows in the DataFrame.
14
+
15
+ Args:
16
+ config: report Settings object
17
+ df: the Pandas DataFrame.
18
+ supported_columns: the columns to consider
19
+
20
+ Returns:
21
+ A subset of the DataFrame, ordered by occurrence.
22
+ """
23
+ n_head = config.duplicates.head
24
+
25
+ metrics: Dict[str, Any] = {}
26
+ if n_head > 0:
27
+ if supported_columns and len(df) > 0:
28
+ duplicates_key = config.duplicates.key
29
+ if duplicates_key in df.columns:
30
+ raise ValueError(
31
+ f"Duplicates key ({duplicates_key}) may not be part of the DataFrame. Either change the "
32
+ f" column name in the DataFrame or change the 'duplicates.key' parameter."
33
+ )
34
+
35
+ duplicated_rows = df.duplicated(subset=supported_columns, keep=False)
36
+ duplicated_rows = (
37
+ df[duplicated_rows]
38
+ .rename_axis(index=lambda _: None)
39
+ .groupby(supported_columns, dropna=False, observed=True)
40
+ .size()
41
+ .reset_index(name=duplicates_key)
42
+ )
43
+
44
+ metrics["n_duplicates"] = len(duplicated_rows[duplicates_key])
45
+ metrics["p_duplicates"] = metrics["n_duplicates"] / len(df)
46
+
47
+ return (
48
+ metrics,
49
+ duplicated_rows.nlargest(n_head, duplicates_key),
50
+ )
51
+ else:
52
+ metrics["n_duplicates"] = 0
53
+ metrics["p_duplicates"] = 0.0
54
+ return metrics, None
55
+ else:
56
+ return metrics, None
@@ -0,0 +1,35 @@
1
+ from typing import Union
2
+
3
+ import pandas as pd
4
+ from numpy import log2
5
+ from scipy.stats import entropy
6
+
7
+
8
+ def column_imbalance_score(
9
+ value_counts: pd.Series, n_classes: int
10
+ ) -> Union[float, int]:
11
+ """column_imbalance_score
12
+
13
+ The class balance score for categorical and boolean variables uses entropy to calculate a bounded score between 0 and 1.
14
+ A perfectly uniform distribution would return a score of 0, and a perfectly imbalanced distribution would return a score of 1.
15
+
16
+ When dealing with probabilities with finite values (e.g categorical), entropy is maximised the ‘flatter’ the distribution is. (Jaynes: Probability Theory, The Logic of Science)
17
+ To calculate the class imbalance, we calculate the entropy of that distribution and the maximum possible entropy for that number of classes.
18
+ To calculate the entropy of the 'distribution' we use value counts (e.g frequency of classes) and we can determine the maximum entropy as log2(number of classes).
19
+ We then divide the entropy by the maximum possible entropy to get a value between 0 and 1 which we then subtract from 1.
20
+
21
+ Args:
22
+ value_counts (pd.Series): frequency of each category
23
+ n_classes (int): number of classes
24
+
25
+ Returns:
26
+ Union[float, int]: float or integer bounded between 0 and 1 inclusively
27
+ """
28
+ # return 0 if there is only one class (when entropy =0) as it is balanced.
29
+ # note that this also prevents a zero division error with log2(n_classes)
30
+ if n_classes > 1:
31
+ # casting to numpy array to ensure correct dtype when a categorical integer
32
+ # variable is evaluated
33
+ value_counts = value_counts.to_numpy(dtype=float)
34
+ return 1 - (entropy(value_counts, base=2) / log2(n_classes))
35
+ return 0
@@ -0,0 +1,42 @@
1
+ import numpy as np
2
+ import pandas as pd
3
+
4
+ from data_profiling.config import Settings
5
+ from data_profiling.visualisation.missing import (
6
+ plot_missing_bar,
7
+ plot_missing_heatmap,
8
+ plot_missing_matrix,
9
+ )
10
+
11
+
12
+ def missing_bar(config: Settings, df: pd.DataFrame) -> str:
13
+ notnull_counts = len(df) - df.isnull().sum()
14
+ return plot_missing_bar(
15
+ config,
16
+ notnull_counts=notnull_counts,
17
+ nrows=len(df),
18
+ columns=list(df.columns),
19
+ )
20
+
21
+
22
+ def missing_matrix(config: Settings, df: pd.DataFrame) -> str:
23
+ return plot_missing_matrix(
24
+ config,
25
+ columns=list(df.columns),
26
+ notnull=df.notnull().values,
27
+ nrows=len(df),
28
+ )
29
+
30
+
31
+ def missing_heatmap(config: Settings, df: pd.DataFrame) -> str:
32
+ # Remove completely filled or completely empty variables.
33
+ columns = [i for i, n in enumerate(np.var(df.isnull(), axis="rows")) if n > 0]
34
+ df = df.iloc[:, columns]
35
+
36
+ # Create and mask the correlation matrix. Construct the base heatmap.
37
+ corr_mat = df.isnull().corr()
38
+ mask = np.zeros_like(corr_mat)
39
+ mask[np.triu_indices_from(mask)] = True
40
+ return plot_missing_heatmap(
41
+ config, corr_mat=corr_mat, mask=mask, columns=list(df.columns)
42
+ )
@@ -0,0 +1,38 @@
1
+ from typing import List
2
+
3
+ import pandas as pd
4
+
5
+ from data_profiling.config import Settings
6
+ from data_profiling.model.sample import Sample, get_sample
7
+
8
+
9
+ @get_sample.register(Settings, pd.DataFrame)
10
+ def pandas_get_sample(config: Settings, df: pd.DataFrame) -> List[Sample]:
11
+ """Obtains a sample from head and tail of the DataFrame
12
+
13
+ Args:
14
+ config: Settings object
15
+ df: the pandas DataFrame
16
+
17
+ Returns:
18
+ a list of Sample objects
19
+ """
20
+ samples: List[Sample] = []
21
+ if len(df) == 0:
22
+ return samples
23
+
24
+ n_head = config.samples.head
25
+ if n_head > 0:
26
+ samples.append(Sample(id="head", data=df.head(n=n_head), name="First rows"))
27
+
28
+ n_tail = config.samples.tail
29
+ if n_tail > 0:
30
+ samples.append(Sample(id="tail", data=df.tail(n=n_tail), name="Last rows"))
31
+
32
+ n_random = config.samples.random
33
+ if n_random > 0:
34
+ samples.append(
35
+ Sample(id="random", data=df.sample(n=n_random), name="Random sample")
36
+ )
37
+
38
+ return samples
@@ -0,0 +1,101 @@
1
+ """Compute statistical description of datasets."""
2
+ import multiprocessing
3
+ from concurrent.futures import ThreadPoolExecutor
4
+ from typing import Any, Tuple
5
+
6
+ import numpy as np
7
+ import pandas as pd
8
+ from tqdm import tqdm
9
+ from visions import VisionsTypeset
10
+
11
+ from data_profiling.config import Settings
12
+ from data_profiling.model.typeset import ProfilingTypeSet
13
+ from data_profiling.utils.compat import optional_option_context
14
+ from data_profiling.utils.dataframe import sort_column_names
15
+
16
+ BaseSummarizer: Any = "BaseSummarizer" # type: ignore
17
+
18
+
19
+ def _is_cast_type_defined(typeset: VisionsTypeset, series: str) -> bool:
20
+ return isinstance(typeset, ProfilingTypeSet) and series in typeset.type_schema
21
+
22
+
23
+ def pandas_describe_1d(
24
+ config: Settings,
25
+ series: pd.Series,
26
+ summarizer: BaseSummarizer,
27
+ typeset: VisionsTypeset,
28
+ ) -> dict:
29
+ """Describe a series (infer the variable type, then calculate type-specific values).
30
+
31
+ Args:
32
+ config: report Settings object
33
+ series: The Series to describe.
34
+ summarizer: Summarizer object
35
+ typeset: Typeset
36
+
37
+ Returns:
38
+ A Series containing calculated series description values.
39
+ """
40
+
41
+ # Make sure pd.NA is not in the series
42
+ with optional_option_context("future.no_silent_downcasting", True):
43
+ series = series.fillna(np.nan).infer_objects(copy=False)
44
+
45
+ has_cast_type = _is_cast_type_defined(typeset, series.name) # type:ignore
46
+ cast_type = (
47
+ str(typeset.type_schema[series.name]) if has_cast_type else None
48
+ ) # type:ignore
49
+
50
+ if has_cast_type and not series.isna().all():
51
+ vtype = typeset.type_schema[series.name] # type:ignore
52
+
53
+ elif config.infer_dtypes:
54
+ # Infer variable types
55
+ vtype = typeset.infer_type(series)
56
+ series = typeset.cast_to_inferred(series)
57
+ else:
58
+ # Detect variable types from pandas dataframe (df.dtypes).
59
+ # [new dtypes, changed using `astype` function are now considered]
60
+ vtype = typeset.detect_type(series)
61
+
62
+ typeset.type_schema[series.name] = vtype # type:ignore
63
+ summary = summarizer.summarize(config, series, dtype=vtype)
64
+ # Cast type is only used on unsupported columns rendering pipeline
65
+ # to indicate the correct variable type when inference is not possible
66
+ summary["cast_type"] = cast_type
67
+
68
+ return summary
69
+
70
+
71
+ def pandas_get_series_descriptions(
72
+ config: Settings,
73
+ df: pd.DataFrame,
74
+ summarizer: BaseSummarizer,
75
+ typeset: VisionsTypeset,
76
+ pbar: tqdm,
77
+ ) -> dict:
78
+ def describe_column(name: str, series: pd.Series) -> Tuple[str, dict]:
79
+ """Process a single series to get the column description."""
80
+ pbar.set_postfix_str(f"Describe variable: {name}")
81
+ description = pandas_describe_1d(config, series, summarizer, typeset)
82
+ pbar.update()
83
+ return name, description
84
+
85
+ pool_size = (
86
+ config.pool_size if config.pool_size > 0 else multiprocessing.cpu_count()
87
+ )
88
+
89
+ series_description = {}
90
+
91
+ with ThreadPoolExecutor(max_workers=pool_size) as executor:
92
+ future_to_col = {
93
+ executor.submit(describe_column, name, series): name # type:ignore
94
+ for name, series in df.items()
95
+ }
96
+
97
+ for future in tqdm(future_to_col.keys(), total=len(future_to_col)):
98
+ name, description = future.result()
99
+ series_description[name] = description
100
+
101
+ return sort_column_names(series_description, config.sort)
@@ -0,0 +1,56 @@
1
+ from collections import Counter
2
+
3
+ import pandas as pd
4
+
5
+ from data_profiling.config import Settings
6
+ from data_profiling.model.table import get_table_stats
7
+
8
+
9
+ @get_table_stats.register
10
+ def pandas_get_table_stats(
11
+ config: Settings, df: pd.DataFrame, variable_stats: dict
12
+ ) -> dict:
13
+ """General statistics for the DataFrame.
14
+
15
+ Args:
16
+ config: report Settings object
17
+ df: The DataFrame to describe.
18
+ variable_stats: Previously calculated statistic on the DataFrame.
19
+
20
+ Returns:
21
+ A dictionary that contains the table statistics.
22
+ """
23
+ n = len(df) if not df.empty else 0
24
+
25
+ memory_size = df.memory_usage(deep=config.memory_deep).sum()
26
+ record_size = float(memory_size) / n if n > 0 else 0
27
+
28
+ table_stats = {
29
+ "n": n,
30
+ "n_var": len(df.columns),
31
+ "memory_size": memory_size,
32
+ "record_size": record_size,
33
+ "n_cells_missing": 0,
34
+ "n_vars_with_missing": 0,
35
+ "n_vars_all_missing": 0,
36
+ }
37
+
38
+ for series_summary in variable_stats.values():
39
+ if "n_missing" in series_summary and series_summary["n_missing"] > 0:
40
+ table_stats["n_vars_with_missing"] += 1
41
+ table_stats["n_cells_missing"] += series_summary["n_missing"]
42
+ if series_summary["n_missing"] == n:
43
+ table_stats["n_vars_all_missing"] += 1
44
+
45
+ table_stats["p_cells_missing"] = (
46
+ table_stats["n_cells_missing"] / (table_stats["n"] * table_stats["n_var"])
47
+ if table_stats["n"] > 0 and table_stats["n_var"] > 0
48
+ else 0
49
+ )
50
+
51
+ # Variable type counts
52
+ table_stats.update(
53
+ {"types": dict(Counter([v["type"] for v in variable_stats.values()]))}
54
+ )
55
+
56
+ return table_stats