fg-data-profiling 4.19.0__py2.py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (238) hide show
  1. data_profiling/__init__.py +34 -0
  2. data_profiling/compare_reports.py +359 -0
  3. data_profiling/config.py +496 -0
  4. data_profiling/config_default.yaml +223 -0
  5. data_profiling/config_minimal.yaml +222 -0
  6. data_profiling/controller/__init__.py +1 -0
  7. data_profiling/controller/console.py +125 -0
  8. data_profiling/controller/pandas_decorator.py +21 -0
  9. data_profiling/expectations_report.py +117 -0
  10. data_profiling/model/__init__.py +4 -0
  11. data_profiling/model/alerts.py +780 -0
  12. data_profiling/model/correlations.py +163 -0
  13. data_profiling/model/dataframe.py +35 -0
  14. data_profiling/model/describe.py +210 -0
  15. data_profiling/model/description.py +108 -0
  16. data_profiling/model/duplicates.py +14 -0
  17. data_profiling/model/expectation_algorithms.py +112 -0
  18. data_profiling/model/handler.py +81 -0
  19. data_profiling/model/missing.py +146 -0
  20. data_profiling/model/pairwise.py +33 -0
  21. data_profiling/model/pandas/__init__.py +55 -0
  22. data_profiling/model/pandas/correlations_pandas.py +207 -0
  23. data_profiling/model/pandas/dataframe_pandas.py +26 -0
  24. data_profiling/model/pandas/describe_boolean_pandas.py +43 -0
  25. data_profiling/model/pandas/describe_categorical_pandas.py +274 -0
  26. data_profiling/model/pandas/describe_counts_pandas.py +63 -0
  27. data_profiling/model/pandas/describe_date_pandas.py +77 -0
  28. data_profiling/model/pandas/describe_file_pandas.py +56 -0
  29. data_profiling/model/pandas/describe_generic_pandas.py +36 -0
  30. data_profiling/model/pandas/describe_image_pandas.py +255 -0
  31. data_profiling/model/pandas/describe_numeric_pandas.py +175 -0
  32. data_profiling/model/pandas/describe_path_pandas.py +63 -0
  33. data_profiling/model/pandas/describe_supported_pandas.py +41 -0
  34. data_profiling/model/pandas/describe_text_pandas.py +62 -0
  35. data_profiling/model/pandas/describe_timeseries_pandas.py +222 -0
  36. data_profiling/model/pandas/describe_url_pandas.py +57 -0
  37. data_profiling/model/pandas/discretize_pandas.py +81 -0
  38. data_profiling/model/pandas/duplicates_pandas.py +56 -0
  39. data_profiling/model/pandas/imbalance_pandas.py +35 -0
  40. data_profiling/model/pandas/missing_pandas.py +42 -0
  41. data_profiling/model/pandas/sample_pandas.py +38 -0
  42. data_profiling/model/pandas/summary_pandas.py +101 -0
  43. data_profiling/model/pandas/table_pandas.py +56 -0
  44. data_profiling/model/pandas/timeseries_index_pandas.py +33 -0
  45. data_profiling/model/pandas/utils_pandas.py +27 -0
  46. data_profiling/model/sample.py +37 -0
  47. data_profiling/model/spark/__init__.py +48 -0
  48. data_profiling/model/spark/correlations_spark.py +152 -0
  49. data_profiling/model/spark/dataframe_spark.py +34 -0
  50. data_profiling/model/spark/describe_boolean_spark.py +27 -0
  51. data_profiling/model/spark/describe_categorical_spark.py +28 -0
  52. data_profiling/model/spark/describe_counts_spark.py +105 -0
  53. data_profiling/model/spark/describe_date_spark.py +51 -0
  54. data_profiling/model/spark/describe_generic_spark.py +30 -0
  55. data_profiling/model/spark/describe_numeric_spark.py +155 -0
  56. data_profiling/model/spark/describe_supported_spark.py +33 -0
  57. data_profiling/model/spark/describe_text_spark.py +25 -0
  58. data_profiling/model/spark/duplicates_spark.py +54 -0
  59. data_profiling/model/spark/missing_spark.py +96 -0
  60. data_profiling/model/spark/sample_spark.py +43 -0
  61. data_profiling/model/spark/summary_spark.py +95 -0
  62. data_profiling/model/spark/table_spark.py +58 -0
  63. data_profiling/model/spark/timeseries_index_spark.py +12 -0
  64. data_profiling/model/summarizer.py +207 -0
  65. data_profiling/model/summary.py +66 -0
  66. data_profiling/model/summary_algorithms.py +276 -0
  67. data_profiling/model/table.py +10 -0
  68. data_profiling/model/timeseries_index.py +16 -0
  69. data_profiling/model/typeset.py +365 -0
  70. data_profiling/model/typeset_relations.py +143 -0
  71. data_profiling/profile_report.py +573 -0
  72. data_profiling/report/__init__.py +4 -0
  73. data_profiling/report/formatters.py +346 -0
  74. data_profiling/report/presentation/__init__.py +1 -0
  75. data_profiling/report/presentation/core/__init__.py +39 -0
  76. data_profiling/report/presentation/core/alerts.py +18 -0
  77. data_profiling/report/presentation/core/collapse.py +24 -0
  78. data_profiling/report/presentation/core/container.py +50 -0
  79. data_profiling/report/presentation/core/correlation_table.py +21 -0
  80. data_profiling/report/presentation/core/dropdown.py +44 -0
  81. data_profiling/report/presentation/core/duplicate.py +16 -0
  82. data_profiling/report/presentation/core/frequency_table.py +14 -0
  83. data_profiling/report/presentation/core/frequency_table_small.py +16 -0
  84. data_profiling/report/presentation/core/html.py +14 -0
  85. data_profiling/report/presentation/core/image.py +34 -0
  86. data_profiling/report/presentation/core/item_renderer.py +17 -0
  87. data_profiling/report/presentation/core/renderable.py +42 -0
  88. data_profiling/report/presentation/core/root.py +35 -0
  89. data_profiling/report/presentation/core/sample.py +20 -0
  90. data_profiling/report/presentation/core/scores.py +32 -0
  91. data_profiling/report/presentation/core/table.py +26 -0
  92. data_profiling/report/presentation/core/toggle_button.py +14 -0
  93. data_profiling/report/presentation/core/variable.py +40 -0
  94. data_profiling/report/presentation/core/variable_info.py +36 -0
  95. data_profiling/report/presentation/flavours/__init__.py +9 -0
  96. data_profiling/report/presentation/flavours/flavour_html.py +64 -0
  97. data_profiling/report/presentation/flavours/flavour_widget.py +61 -0
  98. data_profiling/report/presentation/flavours/flavours.py +43 -0
  99. data_profiling/report/presentation/flavours/html/__init__.py +47 -0
  100. data_profiling/report/presentation/flavours/html/alerts.py +10 -0
  101. data_profiling/report/presentation/flavours/html/collapse.py +7 -0
  102. data_profiling/report/presentation/flavours/html/container.py +58 -0
  103. data_profiling/report/presentation/flavours/html/correlation_table.py +13 -0
  104. data_profiling/report/presentation/flavours/html/dropdown.py +7 -0
  105. data_profiling/report/presentation/flavours/html/duplicate.py +24 -0
  106. data_profiling/report/presentation/flavours/html/frequency_table.py +20 -0
  107. data_profiling/report/presentation/flavours/html/frequency_table_small.py +15 -0
  108. data_profiling/report/presentation/flavours/html/html.py +6 -0
  109. data_profiling/report/presentation/flavours/html/image.py +7 -0
  110. data_profiling/report/presentation/flavours/html/root.py +14 -0
  111. data_profiling/report/presentation/flavours/html/sample.py +12 -0
  112. data_profiling/report/presentation/flavours/html/scores.py +11 -0
  113. data_profiling/report/presentation/flavours/html/table.py +7 -0
  114. data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant.html +1 -0
  115. data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant_length.html +1 -0
  116. data_profiling/report/presentation/flavours/html/templates/alerts/alert_dirty_category.html +1 -0
  117. data_profiling/report/presentation/flavours/html/templates/alerts/alert_duplicates.html +1 -0
  118. data_profiling/report/presentation/flavours/html/templates/alerts/alert_empty.html +1 -0
  119. data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_cardinality.html +1 -0
  120. data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_correlation.html +4 -0
  121. data_profiling/report/presentation/flavours/html/templates/alerts/alert_imbalance.html +1 -0
  122. data_profiling/report/presentation/flavours/html/templates/alerts/alert_infinite.html +1 -0
  123. data_profiling/report/presentation/flavours/html/templates/alerts/alert_missing.html +1 -0
  124. data_profiling/report/presentation/flavours/html/templates/alerts/alert_near_duplicates.html +1 -0
  125. data_profiling/report/presentation/flavours/html/templates/alerts/alert_non_stationary.html +1 -0
  126. data_profiling/report/presentation/flavours/html/templates/alerts/alert_seasonal.html +1 -0
  127. data_profiling/report/presentation/flavours/html/templates/alerts/alert_skewed.html +1 -0
  128. data_profiling/report/presentation/flavours/html/templates/alerts/alert_truncated.html +1 -0
  129. data_profiling/report/presentation/flavours/html/templates/alerts/alert_type_date.html +1 -0
  130. data_profiling/report/presentation/flavours/html/templates/alerts/alert_uniform.html +1 -0
  131. data_profiling/report/presentation/flavours/html/templates/alerts/alert_unique.html +1 -0
  132. data_profiling/report/presentation/flavours/html/templates/alerts/alert_unsupported.html +1 -0
  133. data_profiling/report/presentation/flavours/html/templates/alerts/alert_zeros.html +1 -0
  134. data_profiling/report/presentation/flavours/html/templates/alerts.html +47 -0
  135. data_profiling/report/presentation/flavours/html/templates/collapse.html +11 -0
  136. data_profiling/report/presentation/flavours/html/templates/correlation_table.html +5 -0
  137. data_profiling/report/presentation/flavours/html/templates/diagram.html +11 -0
  138. data_profiling/report/presentation/flavours/html/templates/dropdown.html +16 -0
  139. data_profiling/report/presentation/flavours/html/templates/duplicate.html +5 -0
  140. data_profiling/report/presentation/flavours/html/templates/frequency_table.html +45 -0
  141. data_profiling/report/presentation/flavours/html/templates/frequency_table_small.html +34 -0
  142. data_profiling/report/presentation/flavours/html/templates/report.html +26 -0
  143. data_profiling/report/presentation/flavours/html/templates/sample.html +10 -0
  144. data_profiling/report/presentation/flavours/html/templates/scores.html +78 -0
  145. data_profiling/report/presentation/flavours/html/templates/sequence/batch_grid.html +16 -0
  146. data_profiling/report/presentation/flavours/html/templates/sequence/grid.html +18 -0
  147. data_profiling/report/presentation/flavours/html/templates/sequence/list.html +7 -0
  148. data_profiling/report/presentation/flavours/html/templates/sequence/named_list.html +8 -0
  149. data_profiling/report/presentation/flavours/html/templates/sequence/overview_tabs.html +30 -0
  150. data_profiling/report/presentation/flavours/html/templates/sequence/scores.html +3 -0
  151. data_profiling/report/presentation/flavours/html/templates/sequence/sections.html +13 -0
  152. data_profiling/report/presentation/flavours/html/templates/sequence/select.html +40 -0
  153. data_profiling/report/presentation/flavours/html/templates/sequence/tabs.html +30 -0
  154. data_profiling/report/presentation/flavours/html/templates/table.html +38 -0
  155. data_profiling/report/presentation/flavours/html/templates/toggle_button.html +18 -0
  156. data_profiling/report/presentation/flavours/html/templates/variable.html +7 -0
  157. data_profiling/report/presentation/flavours/html/templates/variable_info.html +49 -0
  158. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.bundle.min.js +7 -0
  159. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.min.css +6 -0
  160. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/cosmo.bootstrap.min.css +12 -0
  161. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/flatly.bootstrap.min.css +12 -0
  162. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/script.js +52 -0
  163. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/simplex.bootstrap.min.css +12 -0
  164. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/style.css +253 -0
  165. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/united.bootstrap.min.css +12 -0
  166. data_profiling/report/presentation/flavours/html/templates/wrapper/footer.html +7 -0
  167. data_profiling/report/presentation/flavours/html/templates/wrapper/javascript.html +18 -0
  168. data_profiling/report/presentation/flavours/html/templates/wrapper/navigation.html +36 -0
  169. data_profiling/report/presentation/flavours/html/templates/wrapper/style.html +53 -0
  170. data_profiling/report/presentation/flavours/html/templates.py +76 -0
  171. data_profiling/report/presentation/flavours/html/toggle_button.py +7 -0
  172. data_profiling/report/presentation/flavours/html/variable.py +7 -0
  173. data_profiling/report/presentation/flavours/html/variable_info.py +7 -0
  174. data_profiling/report/presentation/flavours/widget/__init__.py +49 -0
  175. data_profiling/report/presentation/flavours/widget/alerts.py +45 -0
  176. data_profiling/report/presentation/flavours/widget/collapse.py +43 -0
  177. data_profiling/report/presentation/flavours/widget/container.py +121 -0
  178. data_profiling/report/presentation/flavours/widget/correlation_table.py +14 -0
  179. data_profiling/report/presentation/flavours/widget/dropdown.py +31 -0
  180. data_profiling/report/presentation/flavours/widget/duplicate.py +14 -0
  181. data_profiling/report/presentation/flavours/widget/frequency_table.py +57 -0
  182. data_profiling/report/presentation/flavours/widget/frequency_table_small.py +66 -0
  183. data_profiling/report/presentation/flavours/widget/html.py +11 -0
  184. data_profiling/report/presentation/flavours/widget/image.py +26 -0
  185. data_profiling/report/presentation/flavours/widget/notebook.py +81 -0
  186. data_profiling/report/presentation/flavours/widget/root.py +10 -0
  187. data_profiling/report/presentation/flavours/widget/sample.py +14 -0
  188. data_profiling/report/presentation/flavours/widget/table.py +30 -0
  189. data_profiling/report/presentation/flavours/widget/toggle_button.py +17 -0
  190. data_profiling/report/presentation/flavours/widget/variable.py +12 -0
  191. data_profiling/report/presentation/flavours/widget/variable_info.py +11 -0
  192. data_profiling/report/presentation/frequency_table_utils.py +141 -0
  193. data_profiling/report/structure/__init__.py +1 -0
  194. data_profiling/report/structure/correlations.py +123 -0
  195. data_profiling/report/structure/overview.py +376 -0
  196. data_profiling/report/structure/report.py +457 -0
  197. data_profiling/report/structure/variables/__init__.py +35 -0
  198. data_profiling/report/structure/variables/render_boolean.py +132 -0
  199. data_profiling/report/structure/variables/render_categorical.py +566 -0
  200. data_profiling/report/structure/variables/render_common.py +31 -0
  201. data_profiling/report/structure/variables/render_complex.py +102 -0
  202. data_profiling/report/structure/variables/render_count.py +172 -0
  203. data_profiling/report/structure/variables/render_date.py +143 -0
  204. data_profiling/report/structure/variables/render_file.py +70 -0
  205. data_profiling/report/structure/variables/render_generic.py +45 -0
  206. data_profiling/report/structure/variables/render_image.py +204 -0
  207. data_profiling/report/structure/variables/render_path.py +134 -0
  208. data_profiling/report/structure/variables/render_real.py +314 -0
  209. data_profiling/report/structure/variables/render_text.py +189 -0
  210. data_profiling/report/structure/variables/render_timeseries.py +371 -0
  211. data_profiling/report/structure/variables/render_url.py +132 -0
  212. data_profiling/report/utils.py +34 -0
  213. data_profiling/serialize_report.py +143 -0
  214. data_profiling/utils/__init__.py +1 -0
  215. data_profiling/utils/backend.py +9 -0
  216. data_profiling/utils/cache.py +59 -0
  217. data_profiling/utils/common.py +142 -0
  218. data_profiling/utils/compat.py +31 -0
  219. data_profiling/utils/dataframe.py +238 -0
  220. data_profiling/utils/logger.py +53 -0
  221. data_profiling/utils/notebook.py +8 -0
  222. data_profiling/utils/paths.py +45 -0
  223. data_profiling/utils/progress_bar.py +15 -0
  224. data_profiling/utils/styles.py +22 -0
  225. data_profiling/utils/versions.py +19 -0
  226. data_profiling/version.py +1 -0
  227. data_profiling/visualisation/__init__.py +1 -0
  228. data_profiling/visualisation/context.py +87 -0
  229. data_profiling/visualisation/missing.py +138 -0
  230. data_profiling/visualisation/plot.py +1158 -0
  231. data_profiling/visualisation/utils.py +113 -0
  232. fg_data_profiling-4.19.0.dist-info/METADATA +362 -0
  233. fg_data_profiling-4.19.0.dist-info/RECORD +238 -0
  234. fg_data_profiling-4.19.0.dist-info/WHEEL +6 -0
  235. fg_data_profiling-4.19.0.dist-info/entry_points.txt +3 -0
  236. fg_data_profiling-4.19.0.dist-info/licenses/LICENSE +21 -0
  237. fg_data_profiling-4.19.0.dist-info/top_level.txt +2 -0
  238. ydata_profiling/__init__.py +43 -0
@@ -0,0 +1,163 @@
1
+ # mypy: ignore-errors
2
+
3
+ """Correlations between variables."""
4
+
5
+ import warnings
6
+ from typing import Dict, List, Optional, Sized, no_type_check
7
+
8
+ import numpy as np
9
+ import pandas as pd
10
+
11
+ from data_profiling.config import Settings
12
+
13
+ try:
14
+ from pandas.core.base import DataError
15
+ except ImportError:
16
+ from pandas.errors import DataError
17
+
18
+
19
+ class CorrelationBackend:
20
+ """Helper class to select and cache the appropriate correlation backend (Pandas or Spark)."""
21
+
22
+ @no_type_check
23
+ def __init__(self, df: Sized):
24
+ """Determine backend once and store it for all correlation computations."""
25
+ if isinstance(df, pd.DataFrame):
26
+ from data_profiling.model.pandas import (
27
+ correlations_pandas as correlation_backend, # type: ignore
28
+ )
29
+ else:
30
+ from data_profiling.model.spark import (
31
+ correlations_spark as correlation_backend, # type: ignore
32
+ )
33
+
34
+ self.backend = correlation_backend
35
+
36
+ def get_method(self, method_name: str): # noqa: ANN201
37
+ """Retrieve the appropriate correlation method class from the backend."""
38
+ if hasattr(self.backend, method_name):
39
+ return getattr(self.backend, method_name)
40
+ raise AttributeError(
41
+ f"Correlation method '{method_name}' is not available in the backend."
42
+ )
43
+
44
+
45
+ class Correlation:
46
+ _method_name: str = ""
47
+
48
+ def compute(
49
+ self, config: Settings, df: Sized, summary: dict, backend: CorrelationBackend
50
+ ) -> Optional[Sized]:
51
+ """Computes correlation using the correct backend (Pandas or Spark)."""
52
+ try:
53
+ method = backend.get_method(self._method_name)
54
+ except AttributeError as ex:
55
+ raise NotImplementedError() from ex
56
+ else:
57
+ return method(config, df, summary)
58
+
59
+
60
+ class Auto(Correlation):
61
+ """Automatically selects the appropriate correlation method based on the DataFrame type."""
62
+
63
+ _method_name = "auto_compute"
64
+
65
+
66
+ class Spearman(Correlation):
67
+ _method_name = "spearman_compute"
68
+
69
+
70
+ class Pearson(Correlation):
71
+ _method_name = "pearson_compute"
72
+
73
+
74
+ class Kendall(Correlation):
75
+ _method_name = "kendall_compute"
76
+
77
+
78
+ class Cramers(Correlation):
79
+ _method_name = "cramers_compute"
80
+
81
+
82
+ class PhiK(Correlation):
83
+ _method_name = "phik_compute"
84
+
85
+
86
+ def warn_correlation(correlation_name: str, error: str) -> None:
87
+ warnings.warn(
88
+ f"""There was an attempt to calculate the {correlation_name} correlation, but this failed.
89
+ To hide this warning, disable the calculation
90
+ (using `df.profile_report(correlations={{\"{correlation_name}\": {{\"calculate\": False}}}})`
91
+ If this is problematic for your use case, please report this as an issue:
92
+ https://github.com/Data-Centric-AI-Community/data-profiling/issues
93
+ (include the error message: '{error}')"""
94
+ )
95
+
96
+
97
+ def calculate_correlation(
98
+ config: Settings, df: Sized, correlation_name: str, summary: dict
99
+ ) -> Optional[Sized]:
100
+ """Calculate the correlation coefficients between variables for the correlation types selected in the config
101
+ (auto, pearson, spearman, kendall, phi_k, cramers).
102
+
103
+ Args:
104
+ config: report Settings object
105
+ df: The DataFrame with variables.
106
+ correlation_name:
107
+ summary: summary dictionary
108
+
109
+ Returns:
110
+ The correlation matrices for the given correlation measures. Return None if correlation is empty.
111
+ """
112
+ backend = CorrelationBackend(df)
113
+
114
+ correlation_measures = {
115
+ "auto": Auto,
116
+ "pearson": Pearson,
117
+ "spearman": Spearman,
118
+ "kendall": Kendall,
119
+ "cramers": Cramers,
120
+ "phi_k": PhiK,
121
+ }
122
+
123
+ correlation = None
124
+ try:
125
+ correlation = correlation_measures[correlation_name]().compute(
126
+ config, df, summary, backend
127
+ )
128
+ except (ValueError, AssertionError, TypeError, DataError, IndexError) as e:
129
+ warn_correlation(correlation_name, str(e))
130
+
131
+ return correlation if correlation is not None and len(correlation) > 0 else None
132
+
133
+
134
+ def perform_check_correlation(
135
+ correlation_matrix: pd.DataFrame, threshold: float
136
+ ) -> Dict[str, List[str]]:
137
+ """Check whether selected variables are highly correlated values in the correlation matrix.
138
+
139
+ Args:
140
+ correlation_matrix: The correlation matrix for the DataFrame.
141
+ threshold:.
142
+
143
+ Returns:
144
+ The variables that are highly correlated.
145
+ """
146
+
147
+ cols = correlation_matrix.columns
148
+ bool_index = abs(correlation_matrix.values) >= threshold
149
+ np.fill_diagonal(bool_index, False)
150
+ return {
151
+ col: cols[bool_index[i]].values.tolist()
152
+ for i, col in enumerate(cols)
153
+ if any(bool_index[i])
154
+ }
155
+
156
+
157
+ def get_active_correlations(config: Settings) -> List[str]:
158
+ correlation_names = [
159
+ correlation_name
160
+ for correlation_name in config.correlations.keys()
161
+ if config.correlations[correlation_name].calculate
162
+ ]
163
+ return correlation_names
@@ -0,0 +1,35 @@
1
+ import importlib
2
+ from typing import Any
3
+
4
+ import pandas as pd
5
+
6
+ from data_profiling.config import Settings
7
+ from data_profiling.model.pandas.dataframe_pandas import pandas_preprocess
8
+
9
+ spec = importlib.util.find_spec("pyspark")
10
+ if spec is None:
11
+ from typing import TypeVar
12
+
13
+ sparkDataFrame = TypeVar("sparkDataFrame")
14
+ else:
15
+ from pyspark.sql import DataFrame as sparkDataFrame # type: ignore
16
+
17
+ from data_profiling.model.spark.dataframe_spark import spark_preprocess
18
+
19
+
20
+ def preprocess(config: Settings, df: Any) -> Any:
21
+ """
22
+ Search for invalid columns datatypes as well as ensures column names follow the expected rules
23
+ Args:
24
+ config: ydataprofiling Settings class
25
+ df: a pandas or spark dataframe
26
+
27
+ Returns: a pandas or spark dataframe
28
+ """
29
+ if isinstance(df, pd.DataFrame):
30
+ df = pandas_preprocess(config=config, df=df)
31
+ elif isinstance(df, sparkDataFrame): # type: ignore
32
+ df = spark_preprocess(config=config, df=df)
33
+ else:
34
+ return NotImplementedError()
35
+ return df
@@ -0,0 +1,210 @@
1
+ """Organize the calculation of statistics for each series in this DataFrame."""
2
+ from datetime import datetime
3
+ from typing import Any, Dict, Optional, Union
4
+
5
+ import pandas as pd
6
+ from tqdm.auto import tqdm
7
+ from visions import VisionsTypeset
8
+
9
+ from data_profiling.config import Settings
10
+ from data_profiling.model import BaseAnalysis, BaseDescription
11
+ from data_profiling.model.alerts import get_alerts
12
+ from data_profiling.model.correlations import (
13
+ calculate_correlation,
14
+ get_active_correlations,
15
+ )
16
+ from data_profiling.model.dataframe import preprocess
17
+ from data_profiling.model.description import TimeIndexAnalysis
18
+ from data_profiling.model.duplicates import get_duplicates
19
+ from data_profiling.model.missing import get_missing_active, get_missing_diagram
20
+ from data_profiling.model.pairwise import get_scatter_plot, get_scatter_tasks
21
+ from data_profiling.model.sample import get_custom_sample, get_sample
22
+ from data_profiling.model.summarizer import BaseSummarizer
23
+ from data_profiling.model.summary import get_series_descriptions
24
+ from data_profiling.model.table import get_table_stats
25
+ from data_profiling.model.timeseries_index import get_time_index_description
26
+ from data_profiling.utils.progress_bar import progress
27
+ from data_profiling.version import __version__
28
+
29
+
30
+ def describe(
31
+ config: Settings,
32
+ df: Union[pd.DataFrame, "pyspark.sql.DataFrame"], # type: ignore[name-defined] # noqa: F821
33
+ summarizer: BaseSummarizer,
34
+ typeset: VisionsTypeset,
35
+ sample: Optional[dict] = None,
36
+ ) -> BaseDescription: # noqa: TC301
37
+ """Calculate the statistics for each series in this DataFrame.
38
+
39
+ Args:
40
+ config: report Settings object
41
+ df: DataFrame.
42
+ summarizer: summarizer object
43
+ typeset: visions typeset
44
+ sample: optional, dict with custom sample
45
+
46
+ Returns:
47
+ This function returns a dictionary containing:
48
+ - table: overall statistics.
49
+ - variables: descriptions per series.
50
+ - correlations: correlation matrices.
51
+ - missing: missing value diagrams.
52
+ - alerts: direct special attention to these patterns in your data.
53
+ - package: package details.
54
+ """
55
+ # ** Validate Input types **
56
+ if not isinstance(config, Settings):
57
+ raise TypeError(f"`config` must be of type `Settings`, got {type(config)}")
58
+
59
+ # Validate df input type
60
+
61
+ if not isinstance(df, pd.DataFrame):
62
+ try:
63
+ from pyspark.sql import DataFrame as SparkDataFrame # type: ignore
64
+
65
+ if not isinstance(df, SparkDataFrame): # noqa: TC301
66
+ raise TypeError( # noqa: TC301
67
+ f"`df` must be either a `pandas.DataFrame` or a `pyspark.sql.DataFrame`, but got {type(df)}."
68
+ )
69
+ except ImportError as ex:
70
+ raise TypeError(
71
+ f"`df must be either a `pandas.DataFrame` or a `pyspark.sql.DataFrame`, but got {type(df)}."
72
+ f"If using Spark, make sure PySpark is installed."
73
+ ) from ex
74
+
75
+ df = preprocess(config, df)
76
+
77
+ number_of_tasks = 5
78
+
79
+ with tqdm(
80
+ total=number_of_tasks,
81
+ desc="Summarize dataset",
82
+ disable=not config.progress_bar,
83
+ position=0,
84
+ ) as pbar:
85
+ date_start = datetime.utcnow()
86
+
87
+ # Variable-specific
88
+ pbar.total += len(df.columns)
89
+ series_description = get_series_descriptions(
90
+ config, df, summarizer, typeset, pbar
91
+ )
92
+
93
+ pbar.set_postfix_str("Get variable types")
94
+ pbar.total += 1
95
+ variables = {
96
+ column: description["type"]
97
+ for column, description in series_description.items()
98
+ }
99
+ supported_columns = [
100
+ column
101
+ for column, type_name in variables.items()
102
+ if type_name != "Unsupported"
103
+ ]
104
+ interval_columns = [
105
+ column
106
+ for column, type_name in variables.items()
107
+ if type_name in {"Numeric", "TimeSeries"}
108
+ ]
109
+ pbar.update()
110
+
111
+ # Table statistics
112
+ table_stats = progress(get_table_stats, pbar, "Get dataframe statistics")(
113
+ config, df, series_description
114
+ )
115
+
116
+ # Get correlations
117
+ if table_stats["n"] != 0:
118
+ correlation_names = get_active_correlations(config)
119
+ pbar.total += len(correlation_names)
120
+
121
+ correlations = {
122
+ correlation_name: progress(
123
+ calculate_correlation,
124
+ pbar,
125
+ f"Calculate {correlation_name} correlation",
126
+ )(config, df, correlation_name, series_description)
127
+ for correlation_name in correlation_names
128
+ }
129
+
130
+ # make sure correlations is not None
131
+ correlations = {
132
+ key: value for key, value in correlations.items() if value is not None
133
+ }
134
+ else:
135
+ correlations = {}
136
+
137
+ # Scatter matrix
138
+ pbar.set_postfix_str("Get scatter matrix")
139
+ scatter_tasks = get_scatter_tasks(config, interval_columns)
140
+ pbar.total += len(scatter_tasks)
141
+ scatter_matrix: Dict[Any, Dict[Any, Any]] = {
142
+ x: {y: None} for x, y in scatter_tasks
143
+ }
144
+ for x, y in scatter_tasks:
145
+ scatter_matrix[x][y] = progress(
146
+ get_scatter_plot, pbar, f"scatter {x}, {y}"
147
+ )(config, df, x, y, interval_columns)
148
+
149
+ # missing diagrams
150
+ missing_map = get_missing_active(config, table_stats)
151
+ pbar.total += len(missing_map)
152
+ missing = {
153
+ name: progress(get_missing_diagram, pbar, f"Missing diagram {name}")(
154
+ config, df, settings
155
+ )
156
+ for name, settings in missing_map.items()
157
+ }
158
+ missing = {name: value for name, value in missing.items() if value is not None}
159
+
160
+ # Sample
161
+ pbar.set_postfix_str("Take sample")
162
+ if sample is None:
163
+ samples = get_sample(config, df)
164
+ else:
165
+ samples = get_custom_sample(sample)
166
+ pbar.update()
167
+
168
+ # Duplicates
169
+ metrics, duplicates = progress(get_duplicates, pbar, "Detecting duplicates")(
170
+ config, df, supported_columns
171
+ )
172
+ table_stats.update(metrics)
173
+
174
+ alerts = progress(get_alerts, pbar, "Get alerts")(
175
+ config, table_stats, series_description, correlations
176
+ )
177
+
178
+ if config.vars.timeseries.active:
179
+ tsindex_description = get_time_index_description(config, df, table_stats)
180
+
181
+ pbar.set_postfix_str("Get reproduction details")
182
+ package = {
183
+ "data_profiling_version": __version__,
184
+ "data_profiling_config": config.json(),
185
+ }
186
+ pbar.update()
187
+
188
+ pbar.set_postfix_str("Completed")
189
+
190
+ date_end = datetime.utcnow()
191
+
192
+ analysis = BaseAnalysis(config.title, date_start, date_end)
193
+ time_index_analysis = None
194
+ if config.vars.timeseries.active and tsindex_description:
195
+ time_index_analysis = TimeIndexAnalysis(**tsindex_description)
196
+
197
+ description = BaseDescription(
198
+ analysis=analysis,
199
+ time_index_analysis=time_index_analysis,
200
+ table=table_stats,
201
+ variables=series_description,
202
+ scatter=scatter_matrix,
203
+ correlations=correlations,
204
+ missing=missing,
205
+ alerts=alerts,
206
+ package=package,
207
+ sample=samples,
208
+ duplicates=duplicates,
209
+ )
210
+ return description
@@ -0,0 +1,108 @@
1
+ from dataclasses import dataclass
2
+ from datetime import datetime, timedelta
3
+ from typing import Any, Dict, List, Optional, Union
4
+
5
+ from pandas import Timedelta
6
+
7
+
8
+ @dataclass
9
+ class BaseAnalysis:
10
+ """Description of base analysis module of report.
11
+ Overall info about report.
12
+
13
+ Attributes
14
+ title (str): Title of report.
15
+ date_start (Union[datetime, List[datetime]]): Start of generating description.
16
+ date_end (Union[datetime, List[datetime]]): End of generating description.
17
+ """
18
+
19
+ title: str
20
+ date_start: Union[datetime, List[datetime]]
21
+ date_end: Union[datetime, List[datetime]]
22
+
23
+ def __init__(self, title: str, date_start: datetime, date_end: datetime) -> None:
24
+ self.title = title
25
+ self.date_start = date_start
26
+ self.date_end = date_end
27
+
28
+ @property
29
+ def duration(self) -> Union[timedelta, List[timedelta]]:
30
+ if isinstance(self.date_start, datetime) and isinstance(
31
+ self.date_end, datetime
32
+ ):
33
+ return self.date_end - self.date_start
34
+ if isinstance(self.date_start, list) and isinstance(self.date_end, list):
35
+ return [
36
+ self.date_end[i] - self.date_start[i]
37
+ for i in range(len(self.date_start))
38
+ ]
39
+ else:
40
+ raise TypeError()
41
+
42
+
43
+ @dataclass
44
+ class TimeIndexAnalysis:
45
+ """Description of timeseries index analysis module of report.
46
+
47
+ Attributes:
48
+ n_series (Union[int, List[int]): Number of time series identified in the dataset.
49
+ length (Union[int, List[int]): Number of data points in the time series.
50
+ start (Any): Starting point of the time series.
51
+ end (Any): Ending point of the time series.
52
+ period (Union[float, List[float]): Average interval between data points in the time series.
53
+ frequency (Union[Optional[str], List[Optional[str]]): A string alias given to useful common time series frequencies, e.g. H - hours.
54
+ """
55
+
56
+ n_series: Union[int, List[int]]
57
+ length: Union[int, List[int]]
58
+ start: Any
59
+ end: Any
60
+ period: Union[float, List[float], Timedelta, List[Timedelta]]
61
+ frequency: Union[Optional[str], List[Optional[str]]]
62
+
63
+ def __init__(
64
+ self,
65
+ n_series: int,
66
+ length: int,
67
+ start: Any,
68
+ end: Any,
69
+ period: float,
70
+ frequency: Optional[str] = None,
71
+ ) -> None:
72
+ self.n_series = n_series
73
+ self.length = length
74
+ self.start = start
75
+ self.end = end
76
+ self.period = period
77
+ self.frequency = frequency
78
+
79
+
80
+ @dataclass
81
+ class BaseDescription:
82
+ """Description of DataFrame.
83
+
84
+ Attributes:
85
+ analysis (BaseAnalysis): Base info about report. Title, start time and end time of description generating.
86
+ time_index_analysis (Optional[TimeIndexAnalysis]): Description of timeseries index analysis module of report.
87
+ table (Any): DataFrame statistic. Base information about DataFrame.
88
+ variables (Dict[str, Any]): Description of variables (columns) of DataFrame. Key is column name, value is description dictionary.
89
+ scatter (Any): Pairwise scatter for all variables. Plot interactions between variables.
90
+ correlations (Dict[str, Any]): Prepare correlation matrix for DataFrame
91
+ missing (Dict[str, Any]): Describe missing values.
92
+ alerts (Any): Take alerts from all modules (variables, scatter, correlations), and group them.
93
+ package (Dict[str, Any]): Contains version of data-profiling and config.
94
+ sample (Any): Sample of data.
95
+ duplicates (Any): Description of duplicates.
96
+ """
97
+
98
+ analysis: BaseAnalysis
99
+ time_index_analysis: Optional[TimeIndexAnalysis]
100
+ table: Any
101
+ variables: Dict[str, Any]
102
+ scatter: Any
103
+ correlations: Dict[str, Any]
104
+ missing: Dict[str, Any]
105
+ alerts: Any
106
+ package: Dict[str, Any]
107
+ sample: Any
108
+ duplicates: Any
@@ -0,0 +1,14 @@
1
+ from typing import Any, Dict, Optional, Sequence, Tuple, TypeVar
2
+
3
+ from multimethod import multimethod
4
+
5
+ from data_profiling.config import Settings
6
+
7
+ T = TypeVar("T")
8
+
9
+
10
+ @multimethod
11
+ def get_duplicates(
12
+ config: Settings, df: T, supported_columns: Sequence
13
+ ) -> Tuple[Dict[str, Any], Optional[T]]:
14
+ raise NotImplementedError()
@@ -0,0 +1,112 @@
1
+ from typing import Any, Tuple
2
+
3
+
4
+ def generic_expectations(
5
+ name: str, summary: dict, batch: Any, *args
6
+ ) -> Tuple[str, dict, Any]:
7
+ batch.expect_column_to_exist(name)
8
+
9
+ if summary["n_missing"] == 0:
10
+ batch.expect_column_values_to_not_be_null(name)
11
+
12
+ if summary["p_unique"] == 1.0:
13
+ batch.expect_column_values_to_be_unique(name)
14
+
15
+ return name, summary, batch
16
+
17
+
18
+ def numeric_expectations(
19
+ name: str, summary: dict, batch: Any, *args
20
+ ) -> Tuple[str, dict, Any]:
21
+ from great_expectations.profile.base import ProfilerTypeMapping
22
+
23
+ numeric_type_names = (
24
+ ProfilerTypeMapping.INT_TYPE_NAMES + ProfilerTypeMapping.FLOAT_TYPE_NAMES
25
+ )
26
+
27
+ batch.expect_column_values_to_be_in_type_list(
28
+ name,
29
+ numeric_type_names,
30
+ meta={
31
+ "notes": {
32
+ "format": "markdown",
33
+ "content": [
34
+ "The column values should be stored in one of these types."
35
+ ],
36
+ }
37
+ },
38
+ )
39
+
40
+ if summary["monotonic_increase"]:
41
+ batch.expect_column_values_to_be_increasing(
42
+ name, strictly=summary["monotonic_increase_strict"]
43
+ )
44
+
45
+ if summary["monotonic_decrease"]:
46
+ batch.expect_column_values_to_be_decreasing(
47
+ name, strictly=summary["monotonic_decrease_strict"]
48
+ )
49
+
50
+ if any(k in summary for k in ["min", "max"]):
51
+ batch.expect_column_values_to_be_between(
52
+ name, min_value=summary.get("min"), max_value=summary.get("max")
53
+ )
54
+
55
+ return name, summary, batch
56
+
57
+
58
+ def categorical_expectations(
59
+ name: str, summary: dict, batch: Any, *args
60
+ ) -> Tuple[str, dict, Any]:
61
+ # Use for both categorical and special case (boolean)
62
+ absolute_threshold = 10
63
+ relative_threshold = 0.2
64
+ if (
65
+ summary["n_distinct"] < absolute_threshold
66
+ or summary["p_distinct"] < relative_threshold
67
+ ):
68
+ batch.expect_column_values_to_be_in_set(
69
+ name, set(summary["value_counts_without_nan"].keys())
70
+ )
71
+ return name, summary, batch
72
+
73
+
74
+ def path_expectations(
75
+ name: str, summary: dict, batch: Any, *args
76
+ ) -> Tuple[str, dict, Any]:
77
+ return name, summary, batch
78
+
79
+
80
+ def datetime_expectations(
81
+ name: str, summary: dict, batch: Any, *args
82
+ ) -> Tuple[str, dict, Any]:
83
+ if any(k in summary for k in ["min", "max"]):
84
+ batch.expect_column_values_to_be_between(
85
+ name,
86
+ min_value=summary.get("min"),
87
+ max_value=summary.get("max"),
88
+ parse_strings_as_datetimes=True,
89
+ )
90
+
91
+ return name, summary, batch
92
+
93
+
94
+ def image_expectations(
95
+ name: str, summary: dict, batch: Any, *args
96
+ ) -> Tuple[str, dict, Any]:
97
+ return name, summary, batch
98
+
99
+
100
+ def url_expectations(
101
+ name: str, summary: dict, batch: Any, *args
102
+ ) -> Tuple[str, dict, Any]:
103
+ return name, summary, batch
104
+
105
+
106
+ def file_expectations(
107
+ name: str, summary: dict, batch: Any, *args
108
+ ) -> Tuple[str, dict, Any]:
109
+ # By definition within our type logic, a file exists (as it's a path that also exists)
110
+ batch.expect_file_to_exist(name)
111
+
112
+ return name, summary, batch