fg-data-profiling 4.19.0__py2.py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- data_profiling/__init__.py +34 -0
- data_profiling/compare_reports.py +359 -0
- data_profiling/config.py +496 -0
- data_profiling/config_default.yaml +223 -0
- data_profiling/config_minimal.yaml +222 -0
- data_profiling/controller/__init__.py +1 -0
- data_profiling/controller/console.py +125 -0
- data_profiling/controller/pandas_decorator.py +21 -0
- data_profiling/expectations_report.py +117 -0
- data_profiling/model/__init__.py +4 -0
- data_profiling/model/alerts.py +780 -0
- data_profiling/model/correlations.py +163 -0
- data_profiling/model/dataframe.py +35 -0
- data_profiling/model/describe.py +210 -0
- data_profiling/model/description.py +108 -0
- data_profiling/model/duplicates.py +14 -0
- data_profiling/model/expectation_algorithms.py +112 -0
- data_profiling/model/handler.py +81 -0
- data_profiling/model/missing.py +146 -0
- data_profiling/model/pairwise.py +33 -0
- data_profiling/model/pandas/__init__.py +55 -0
- data_profiling/model/pandas/correlations_pandas.py +207 -0
- data_profiling/model/pandas/dataframe_pandas.py +26 -0
- data_profiling/model/pandas/describe_boolean_pandas.py +43 -0
- data_profiling/model/pandas/describe_categorical_pandas.py +274 -0
- data_profiling/model/pandas/describe_counts_pandas.py +63 -0
- data_profiling/model/pandas/describe_date_pandas.py +77 -0
- data_profiling/model/pandas/describe_file_pandas.py +56 -0
- data_profiling/model/pandas/describe_generic_pandas.py +36 -0
- data_profiling/model/pandas/describe_image_pandas.py +255 -0
- data_profiling/model/pandas/describe_numeric_pandas.py +175 -0
- data_profiling/model/pandas/describe_path_pandas.py +63 -0
- data_profiling/model/pandas/describe_supported_pandas.py +41 -0
- data_profiling/model/pandas/describe_text_pandas.py +62 -0
- data_profiling/model/pandas/describe_timeseries_pandas.py +222 -0
- data_profiling/model/pandas/describe_url_pandas.py +57 -0
- data_profiling/model/pandas/discretize_pandas.py +81 -0
- data_profiling/model/pandas/duplicates_pandas.py +56 -0
- data_profiling/model/pandas/imbalance_pandas.py +35 -0
- data_profiling/model/pandas/missing_pandas.py +42 -0
- data_profiling/model/pandas/sample_pandas.py +38 -0
- data_profiling/model/pandas/summary_pandas.py +101 -0
- data_profiling/model/pandas/table_pandas.py +56 -0
- data_profiling/model/pandas/timeseries_index_pandas.py +33 -0
- data_profiling/model/pandas/utils_pandas.py +27 -0
- data_profiling/model/sample.py +37 -0
- data_profiling/model/spark/__init__.py +48 -0
- data_profiling/model/spark/correlations_spark.py +152 -0
- data_profiling/model/spark/dataframe_spark.py +34 -0
- data_profiling/model/spark/describe_boolean_spark.py +27 -0
- data_profiling/model/spark/describe_categorical_spark.py +28 -0
- data_profiling/model/spark/describe_counts_spark.py +105 -0
- data_profiling/model/spark/describe_date_spark.py +51 -0
- data_profiling/model/spark/describe_generic_spark.py +30 -0
- data_profiling/model/spark/describe_numeric_spark.py +155 -0
- data_profiling/model/spark/describe_supported_spark.py +33 -0
- data_profiling/model/spark/describe_text_spark.py +25 -0
- data_profiling/model/spark/duplicates_spark.py +54 -0
- data_profiling/model/spark/missing_spark.py +96 -0
- data_profiling/model/spark/sample_spark.py +43 -0
- data_profiling/model/spark/summary_spark.py +95 -0
- data_profiling/model/spark/table_spark.py +58 -0
- data_profiling/model/spark/timeseries_index_spark.py +12 -0
- data_profiling/model/summarizer.py +207 -0
- data_profiling/model/summary.py +66 -0
- data_profiling/model/summary_algorithms.py +276 -0
- data_profiling/model/table.py +10 -0
- data_profiling/model/timeseries_index.py +16 -0
- data_profiling/model/typeset.py +365 -0
- data_profiling/model/typeset_relations.py +143 -0
- data_profiling/profile_report.py +573 -0
- data_profiling/report/__init__.py +4 -0
- data_profiling/report/formatters.py +346 -0
- data_profiling/report/presentation/__init__.py +1 -0
- data_profiling/report/presentation/core/__init__.py +39 -0
- data_profiling/report/presentation/core/alerts.py +18 -0
- data_profiling/report/presentation/core/collapse.py +24 -0
- data_profiling/report/presentation/core/container.py +50 -0
- data_profiling/report/presentation/core/correlation_table.py +21 -0
- data_profiling/report/presentation/core/dropdown.py +44 -0
- data_profiling/report/presentation/core/duplicate.py +16 -0
- data_profiling/report/presentation/core/frequency_table.py +14 -0
- data_profiling/report/presentation/core/frequency_table_small.py +16 -0
- data_profiling/report/presentation/core/html.py +14 -0
- data_profiling/report/presentation/core/image.py +34 -0
- data_profiling/report/presentation/core/item_renderer.py +17 -0
- data_profiling/report/presentation/core/renderable.py +42 -0
- data_profiling/report/presentation/core/root.py +35 -0
- data_profiling/report/presentation/core/sample.py +20 -0
- data_profiling/report/presentation/core/scores.py +32 -0
- data_profiling/report/presentation/core/table.py +26 -0
- data_profiling/report/presentation/core/toggle_button.py +14 -0
- data_profiling/report/presentation/core/variable.py +40 -0
- data_profiling/report/presentation/core/variable_info.py +36 -0
- data_profiling/report/presentation/flavours/__init__.py +9 -0
- data_profiling/report/presentation/flavours/flavour_html.py +64 -0
- data_profiling/report/presentation/flavours/flavour_widget.py +61 -0
- data_profiling/report/presentation/flavours/flavours.py +43 -0
- data_profiling/report/presentation/flavours/html/__init__.py +47 -0
- data_profiling/report/presentation/flavours/html/alerts.py +10 -0
- data_profiling/report/presentation/flavours/html/collapse.py +7 -0
- data_profiling/report/presentation/flavours/html/container.py +58 -0
- data_profiling/report/presentation/flavours/html/correlation_table.py +13 -0
- data_profiling/report/presentation/flavours/html/dropdown.py +7 -0
- data_profiling/report/presentation/flavours/html/duplicate.py +24 -0
- data_profiling/report/presentation/flavours/html/frequency_table.py +20 -0
- data_profiling/report/presentation/flavours/html/frequency_table_small.py +15 -0
- data_profiling/report/presentation/flavours/html/html.py +6 -0
- data_profiling/report/presentation/flavours/html/image.py +7 -0
- data_profiling/report/presentation/flavours/html/root.py +14 -0
- data_profiling/report/presentation/flavours/html/sample.py +12 -0
- data_profiling/report/presentation/flavours/html/scores.py +11 -0
- data_profiling/report/presentation/flavours/html/table.py +7 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant_length.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_dirty_category.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_duplicates.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_empty.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_cardinality.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_correlation.html +4 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_imbalance.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_infinite.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_missing.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_near_duplicates.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_non_stationary.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_seasonal.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_skewed.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_truncated.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_type_date.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_uniform.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_unique.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_unsupported.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_zeros.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts.html +47 -0
- data_profiling/report/presentation/flavours/html/templates/collapse.html +11 -0
- data_profiling/report/presentation/flavours/html/templates/correlation_table.html +5 -0
- data_profiling/report/presentation/flavours/html/templates/diagram.html +11 -0
- data_profiling/report/presentation/flavours/html/templates/dropdown.html +16 -0
- data_profiling/report/presentation/flavours/html/templates/duplicate.html +5 -0
- data_profiling/report/presentation/flavours/html/templates/frequency_table.html +45 -0
- data_profiling/report/presentation/flavours/html/templates/frequency_table_small.html +34 -0
- data_profiling/report/presentation/flavours/html/templates/report.html +26 -0
- data_profiling/report/presentation/flavours/html/templates/sample.html +10 -0
- data_profiling/report/presentation/flavours/html/templates/scores.html +78 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/batch_grid.html +16 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/grid.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/list.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/named_list.html +8 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/overview_tabs.html +30 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/scores.html +3 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/sections.html +13 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/select.html +40 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/tabs.html +30 -0
- data_profiling/report/presentation/flavours/html/templates/table.html +38 -0
- data_profiling/report/presentation/flavours/html/templates/toggle_button.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/variable.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/variable_info.html +49 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.bundle.min.js +7 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.min.css +6 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/cosmo.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/flatly.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/script.js +52 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/simplex.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/style.css +253 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/united.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/footer.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/javascript.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/navigation.html +36 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/style.html +53 -0
- data_profiling/report/presentation/flavours/html/templates.py +76 -0
- data_profiling/report/presentation/flavours/html/toggle_button.py +7 -0
- data_profiling/report/presentation/flavours/html/variable.py +7 -0
- data_profiling/report/presentation/flavours/html/variable_info.py +7 -0
- data_profiling/report/presentation/flavours/widget/__init__.py +49 -0
- data_profiling/report/presentation/flavours/widget/alerts.py +45 -0
- data_profiling/report/presentation/flavours/widget/collapse.py +43 -0
- data_profiling/report/presentation/flavours/widget/container.py +121 -0
- data_profiling/report/presentation/flavours/widget/correlation_table.py +14 -0
- data_profiling/report/presentation/flavours/widget/dropdown.py +31 -0
- data_profiling/report/presentation/flavours/widget/duplicate.py +14 -0
- data_profiling/report/presentation/flavours/widget/frequency_table.py +57 -0
- data_profiling/report/presentation/flavours/widget/frequency_table_small.py +66 -0
- data_profiling/report/presentation/flavours/widget/html.py +11 -0
- data_profiling/report/presentation/flavours/widget/image.py +26 -0
- data_profiling/report/presentation/flavours/widget/notebook.py +81 -0
- data_profiling/report/presentation/flavours/widget/root.py +10 -0
- data_profiling/report/presentation/flavours/widget/sample.py +14 -0
- data_profiling/report/presentation/flavours/widget/table.py +30 -0
- data_profiling/report/presentation/flavours/widget/toggle_button.py +17 -0
- data_profiling/report/presentation/flavours/widget/variable.py +12 -0
- data_profiling/report/presentation/flavours/widget/variable_info.py +11 -0
- data_profiling/report/presentation/frequency_table_utils.py +141 -0
- data_profiling/report/structure/__init__.py +1 -0
- data_profiling/report/structure/correlations.py +123 -0
- data_profiling/report/structure/overview.py +376 -0
- data_profiling/report/structure/report.py +457 -0
- data_profiling/report/structure/variables/__init__.py +35 -0
- data_profiling/report/structure/variables/render_boolean.py +132 -0
- data_profiling/report/structure/variables/render_categorical.py +566 -0
- data_profiling/report/structure/variables/render_common.py +31 -0
- data_profiling/report/structure/variables/render_complex.py +102 -0
- data_profiling/report/structure/variables/render_count.py +172 -0
- data_profiling/report/structure/variables/render_date.py +143 -0
- data_profiling/report/structure/variables/render_file.py +70 -0
- data_profiling/report/structure/variables/render_generic.py +45 -0
- data_profiling/report/structure/variables/render_image.py +204 -0
- data_profiling/report/structure/variables/render_path.py +134 -0
- data_profiling/report/structure/variables/render_real.py +314 -0
- data_profiling/report/structure/variables/render_text.py +189 -0
- data_profiling/report/structure/variables/render_timeseries.py +371 -0
- data_profiling/report/structure/variables/render_url.py +132 -0
- data_profiling/report/utils.py +34 -0
- data_profiling/serialize_report.py +143 -0
- data_profiling/utils/__init__.py +1 -0
- data_profiling/utils/backend.py +9 -0
- data_profiling/utils/cache.py +59 -0
- data_profiling/utils/common.py +142 -0
- data_profiling/utils/compat.py +31 -0
- data_profiling/utils/dataframe.py +238 -0
- data_profiling/utils/logger.py +53 -0
- data_profiling/utils/notebook.py +8 -0
- data_profiling/utils/paths.py +45 -0
- data_profiling/utils/progress_bar.py +15 -0
- data_profiling/utils/styles.py +22 -0
- data_profiling/utils/versions.py +19 -0
- data_profiling/version.py +1 -0
- data_profiling/visualisation/__init__.py +1 -0
- data_profiling/visualisation/context.py +87 -0
- data_profiling/visualisation/missing.py +138 -0
- data_profiling/visualisation/plot.py +1158 -0
- data_profiling/visualisation/utils.py +113 -0
- fg_data_profiling-4.19.0.dist-info/METADATA +362 -0
- fg_data_profiling-4.19.0.dist-info/RECORD +238 -0
- fg_data_profiling-4.19.0.dist-info/WHEEL +6 -0
- fg_data_profiling-4.19.0.dist-info/entry_points.txt +3 -0
- fg_data_profiling-4.19.0.dist-info/licenses/LICENSE +21 -0
- fg_data_profiling-4.19.0.dist-info/top_level.txt +2 -0
- ydata_profiling/__init__.py +43 -0
data_profiling/config.py
ADDED
|
@@ -0,0 +1,496 @@
|
|
|
1
|
+
"""Configuration for the package."""
|
|
2
|
+
from enum import Enum
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Any, Dict, List, Optional, Tuple, Union
|
|
5
|
+
|
|
6
|
+
import yaml
|
|
7
|
+
from pydantic.v1 import BaseModel, BaseSettings, Field, PrivateAttr
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _merge_dictionaries(dict1: dict, dict2: dict) -> dict:
|
|
11
|
+
"""
|
|
12
|
+
Recursive merge dictionaries.
|
|
13
|
+
|
|
14
|
+
:param dict1: Base dictionary to merge.
|
|
15
|
+
:param dict2: Dictionary to merge on top of base dictionary.
|
|
16
|
+
:return: Merged dictionary
|
|
17
|
+
"""
|
|
18
|
+
for key, val in dict1.items():
|
|
19
|
+
if isinstance(val, dict):
|
|
20
|
+
dict2_node = dict2.setdefault(key, {})
|
|
21
|
+
_merge_dictionaries(val, dict2_node)
|
|
22
|
+
else:
|
|
23
|
+
if key not in dict2:
|
|
24
|
+
dict2[key] = val
|
|
25
|
+
|
|
26
|
+
return dict2
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Dataset(BaseModel):
|
|
30
|
+
"""Metadata of the dataset"""
|
|
31
|
+
|
|
32
|
+
description: str = ""
|
|
33
|
+
creator: str = ""
|
|
34
|
+
author: str = ""
|
|
35
|
+
copyright_holder: str = ""
|
|
36
|
+
copyright_year: str = ""
|
|
37
|
+
url: str = ""
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class NumVars(BaseModel):
|
|
41
|
+
quantiles: List[float] = [0.05, 0.25, 0.5, 0.75, 0.95]
|
|
42
|
+
skewness_threshold: int = 20
|
|
43
|
+
low_categorical_threshold: int = 5
|
|
44
|
+
# Set to zero to disable
|
|
45
|
+
chi_squared_threshold: float = 0.999
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class TextVars(BaseModel):
|
|
49
|
+
length: bool = True
|
|
50
|
+
words: bool = True
|
|
51
|
+
characters: bool = True
|
|
52
|
+
redact: bool = False
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class CatVars(BaseModel):
|
|
56
|
+
length: bool = True
|
|
57
|
+
characters: bool = True
|
|
58
|
+
words: bool = True
|
|
59
|
+
# if var has more than threshold categories, it's a text var
|
|
60
|
+
cardinality_threshold: int = 50
|
|
61
|
+
# if var has more than threshold % distinct values, it's a text var
|
|
62
|
+
percentage_cat_threshold: float = 0.5
|
|
63
|
+
imbalance_threshold: float = 0.5
|
|
64
|
+
n_obs: int = 5
|
|
65
|
+
# Set to zero to disable
|
|
66
|
+
chi_squared_threshold: float = 0.999
|
|
67
|
+
coerce_str_to_date: bool = False
|
|
68
|
+
redact: bool = False
|
|
69
|
+
histogram_largest: int = 50
|
|
70
|
+
stop_words: List[str] = []
|
|
71
|
+
dirty_categories: bool = False
|
|
72
|
+
dirty_categories_threshold: float = 0.85
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class BoolVars(BaseModel):
|
|
76
|
+
n_obs: int = 3
|
|
77
|
+
imbalance_threshold: float = 0.5
|
|
78
|
+
|
|
79
|
+
# string to boolean mapping dict
|
|
80
|
+
mappings: Dict[str, bool] = {
|
|
81
|
+
"t": True,
|
|
82
|
+
"f": False,
|
|
83
|
+
"yes": True,
|
|
84
|
+
"no": False,
|
|
85
|
+
"y": True,
|
|
86
|
+
"n": False,
|
|
87
|
+
"true": True,
|
|
88
|
+
"false": False,
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class FileVars(BaseModel):
|
|
93
|
+
active: bool = False
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class PathVars(BaseModel):
|
|
97
|
+
active: bool = False
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class ImageVars(BaseModel):
|
|
101
|
+
active: bool = False
|
|
102
|
+
exif: bool = True
|
|
103
|
+
hash: bool = True
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class UrlVars(BaseModel):
|
|
107
|
+
active: bool = False
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class TimeseriesVars(BaseModel):
|
|
111
|
+
active: bool = False
|
|
112
|
+
sortby: Optional[str] = None
|
|
113
|
+
autocorrelation: float = 0.7
|
|
114
|
+
lags: List[int] = [1, 7, 12, 24, 30]
|
|
115
|
+
significance: float = 0.05
|
|
116
|
+
pacf_acf_lag: int = 100
|
|
117
|
+
autolag: Optional[str] = "AIC"
|
|
118
|
+
maxlag: Optional[int] = None
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class Univariate(BaseModel):
|
|
122
|
+
num: NumVars = NumVars()
|
|
123
|
+
text: TextVars = TextVars()
|
|
124
|
+
cat: CatVars = CatVars()
|
|
125
|
+
image: ImageVars = ImageVars()
|
|
126
|
+
bool: BoolVars = BoolVars()
|
|
127
|
+
path: PathVars = PathVars()
|
|
128
|
+
file: FileVars = FileVars()
|
|
129
|
+
url: UrlVars = UrlVars()
|
|
130
|
+
timeseries: TimeseriesVars = TimeseriesVars()
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
class MissingPlot(BaseModel):
|
|
134
|
+
# Force labels when there are > 50 variables
|
|
135
|
+
force_labels: bool = True
|
|
136
|
+
cmap: str = "RdBu"
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class ImageType(Enum):
|
|
140
|
+
svg = "svg"
|
|
141
|
+
png = "png"
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
class CorrelationPlot(BaseModel):
|
|
145
|
+
cmap: str = "RdBu"
|
|
146
|
+
bad: str = "#000000"
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
class Histogram(BaseModel):
|
|
150
|
+
# Number of bins (set to 0 to automatically detect the bin size)
|
|
151
|
+
bins: int = 50
|
|
152
|
+
# Maximum number of bins (when bins=0)
|
|
153
|
+
max_bins: int = 250
|
|
154
|
+
x_axis_labels: bool = True
|
|
155
|
+
density: bool = False
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
class CatFrequencyPlot(BaseModel):
|
|
159
|
+
show: bool = True # if false, the category frequency plot is turned off
|
|
160
|
+
type: str = "bar" # options: 'bar', 'pie'
|
|
161
|
+
|
|
162
|
+
# The cat frequency plot is only rendered if the number of distinct values is
|
|
163
|
+
# smaller or equal to "max_unique"
|
|
164
|
+
max_unique: int = 10
|
|
165
|
+
|
|
166
|
+
# Colors should be a list of matplotlib recognised strings:
|
|
167
|
+
# --> https://matplotlib.org/stable/tutorials/colors/colors.html
|
|
168
|
+
# --> matplotlib defaults are used by default
|
|
169
|
+
colors: Optional[List[str]] = None
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
class Plot(BaseModel):
|
|
173
|
+
missing: MissingPlot = MissingPlot()
|
|
174
|
+
image_format: ImageType = ImageType.svg
|
|
175
|
+
correlation: CorrelationPlot = CorrelationPlot()
|
|
176
|
+
dpi: int = 800 # PNG dpi
|
|
177
|
+
histogram: Histogram = Histogram()
|
|
178
|
+
scatter_threshold: int = 1000
|
|
179
|
+
cat_freq: CatFrequencyPlot = CatFrequencyPlot()
|
|
180
|
+
font_path: Optional[Union[Path, str]] = None
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
class Theme(Enum):
|
|
184
|
+
united = "united"
|
|
185
|
+
flatly = "flatly"
|
|
186
|
+
cosmo = "cosmo"
|
|
187
|
+
simplex = "simplex"
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
class Style(BaseModel):
|
|
191
|
+
# Primary color used for plotting and text where applicable.
|
|
192
|
+
@property
|
|
193
|
+
def primary_color(self) -> str:
|
|
194
|
+
# This attribute may be deprecated in the future, please use primary_colors[0]
|
|
195
|
+
return self.primary_colors[0]
|
|
196
|
+
|
|
197
|
+
# Primary color used for comparisons (default: blue, red, green)
|
|
198
|
+
primary_colors: List[str] = ["#0d6efd", "#dc3545", "#198754"]
|
|
199
|
+
|
|
200
|
+
# Base64-encoded logo image
|
|
201
|
+
logo: str = ""
|
|
202
|
+
|
|
203
|
+
# HTML Theme (optional, default: None)
|
|
204
|
+
theme: Optional[Theme] = None
|
|
205
|
+
|
|
206
|
+
# Labels used for comparing reports (private attribute)
|
|
207
|
+
_labels: List[str] = PrivateAttr(["_"])
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
class Html(BaseModel):
|
|
211
|
+
# Styling options for the HTML report
|
|
212
|
+
style: Style = Style()
|
|
213
|
+
|
|
214
|
+
# Show navbar
|
|
215
|
+
navbar_show: bool = True
|
|
216
|
+
|
|
217
|
+
# Minify the html
|
|
218
|
+
minify_html: bool = True
|
|
219
|
+
|
|
220
|
+
# Offline support
|
|
221
|
+
use_local_assets: bool = True
|
|
222
|
+
|
|
223
|
+
# If True, single file, else directory with assets
|
|
224
|
+
inline: bool = True
|
|
225
|
+
|
|
226
|
+
# Assets prefix if inline = True
|
|
227
|
+
assets_prefix: Optional[str] = None
|
|
228
|
+
|
|
229
|
+
# Internal usage
|
|
230
|
+
assets_path: Optional[str] = None
|
|
231
|
+
|
|
232
|
+
full_width: bool = False
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
class Duplicates(BaseModel):
|
|
236
|
+
head: int = 10
|
|
237
|
+
key: str = "# duplicates"
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
class Correlation(BaseModel):
|
|
241
|
+
key: str = ""
|
|
242
|
+
calculate: bool = Field(default=True)
|
|
243
|
+
warn_high_correlations: int = Field(default=10)
|
|
244
|
+
threshold: float = Field(default=0.5)
|
|
245
|
+
n_bins: int = Field(default=10)
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
class Correlations(BaseModel):
|
|
249
|
+
pearson: Correlation = Correlation(key="pearson")
|
|
250
|
+
spearman: Correlation = Correlation(key="spearman")
|
|
251
|
+
auto: Correlation = Correlation(key="auto")
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
class Interactions(BaseModel):
|
|
255
|
+
# Set to False to disable scatter plots
|
|
256
|
+
continuous: bool = True
|
|
257
|
+
|
|
258
|
+
targets: List[str] = []
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
class Samples(BaseModel):
|
|
262
|
+
head: int = 10
|
|
263
|
+
tail: int = 10
|
|
264
|
+
random: int = 0
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
class Variables(BaseModel):
|
|
268
|
+
descriptions: dict = {}
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
class IframeAttribute(Enum):
|
|
272
|
+
src = "src"
|
|
273
|
+
srcdoc = "srcdoc"
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
class Iframe(BaseModel):
|
|
277
|
+
height: str = "800px"
|
|
278
|
+
width: str = "100%"
|
|
279
|
+
attribute: IframeAttribute = IframeAttribute.srcdoc
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
class Notebook(BaseModel):
|
|
283
|
+
"""When in a Jupyter notebook"""
|
|
284
|
+
|
|
285
|
+
iframe: Iframe = Iframe()
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
class Report(BaseModel):
|
|
289
|
+
# Numeric precision for displaying statistics
|
|
290
|
+
precision: int = 8
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
class Settings(BaseSettings):
|
|
294
|
+
# Default prefix to avoid collisions with environment variables
|
|
295
|
+
class Config:
|
|
296
|
+
env_prefix = "profile_"
|
|
297
|
+
|
|
298
|
+
# Title of the document
|
|
299
|
+
title: str = "YData Profiling Report"
|
|
300
|
+
|
|
301
|
+
dataset: Dataset = Dataset()
|
|
302
|
+
variables: Variables = Variables()
|
|
303
|
+
infer_dtypes: bool = True
|
|
304
|
+
|
|
305
|
+
# Show the description at each variable (in addition to the overview tab)
|
|
306
|
+
show_variable_description: bool = True
|
|
307
|
+
|
|
308
|
+
# Number of workers (0=multiprocessing.cpu_count())
|
|
309
|
+
pool_size: int = 0
|
|
310
|
+
|
|
311
|
+
# Show the progress bar
|
|
312
|
+
progress_bar: bool = True
|
|
313
|
+
|
|
314
|
+
# Per variable type description settings
|
|
315
|
+
vars: Univariate = Univariate()
|
|
316
|
+
|
|
317
|
+
# Sort the variables. Possible values: ascending, descending or None (leaves original sorting)
|
|
318
|
+
sort: Optional[str] = None
|
|
319
|
+
|
|
320
|
+
missing_diagrams: Dict[str, bool] = {
|
|
321
|
+
"bar": True,
|
|
322
|
+
"matrix": True,
|
|
323
|
+
"heatmap": True,
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
correlation_table: bool = True
|
|
327
|
+
|
|
328
|
+
correlations: Dict[str, Correlation] = {
|
|
329
|
+
"auto": Correlation(key="auto", calculate=True),
|
|
330
|
+
"spearman": Correlation(key="spearman", calculate=False),
|
|
331
|
+
"pearson": Correlation(key="pearson", calculate=False),
|
|
332
|
+
"phi_k": Correlation(key="phi_k", calculate=False),
|
|
333
|
+
"cramers": Correlation(key="cramers", calculate=False),
|
|
334
|
+
"kendall": Correlation(key="kendall", calculate=False),
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
interactions: Interactions = Interactions()
|
|
338
|
+
|
|
339
|
+
categorical_maximum_correlation_distinct: int = 100
|
|
340
|
+
# Use `deep` flag for memory_usage
|
|
341
|
+
memory_deep: bool = False
|
|
342
|
+
plot: Plot = Plot()
|
|
343
|
+
duplicates: Duplicates = Duplicates()
|
|
344
|
+
samples: Samples = Samples()
|
|
345
|
+
|
|
346
|
+
reject_variables: bool = True
|
|
347
|
+
|
|
348
|
+
# The number of observations to show
|
|
349
|
+
n_obs_unique: int = 10
|
|
350
|
+
n_freq_table_max: int = 10
|
|
351
|
+
n_extreme_obs: int = 10
|
|
352
|
+
|
|
353
|
+
# Report rendering
|
|
354
|
+
report: Report = Report()
|
|
355
|
+
html: Html = Html()
|
|
356
|
+
notebook: Notebook = Notebook()
|
|
357
|
+
|
|
358
|
+
def update(self, updates: dict) -> "Settings":
|
|
359
|
+
update = _merge_dictionaries(self.dict(), updates)
|
|
360
|
+
return self.parse_obj(self.copy(update=update))
|
|
361
|
+
|
|
362
|
+
@staticmethod
|
|
363
|
+
def from_file(config_file: Union[Path, str]) -> "Settings":
|
|
364
|
+
"""Create a Settings object from a yaml file.
|
|
365
|
+
|
|
366
|
+
Args:
|
|
367
|
+
config_file: yaml file path
|
|
368
|
+
Returns:
|
|
369
|
+
Settings
|
|
370
|
+
"""
|
|
371
|
+
with open(config_file) as f:
|
|
372
|
+
data = yaml.safe_load(f)
|
|
373
|
+
|
|
374
|
+
return Settings.parse_obj(data)
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
class SparkSettings(Settings):
|
|
378
|
+
"""
|
|
379
|
+
Setting class with the standard report configuration for Spark DataFrames
|
|
380
|
+
All the supported analysis are set to true
|
|
381
|
+
"""
|
|
382
|
+
|
|
383
|
+
vars: Univariate = Univariate()
|
|
384
|
+
|
|
385
|
+
vars.num.low_categorical_threshold = 0
|
|
386
|
+
|
|
387
|
+
infer_dtypes: bool = False
|
|
388
|
+
|
|
389
|
+
correlations: Dict[str, Correlation] = {
|
|
390
|
+
"spearman": Correlation(key="spearman", calculate=True),
|
|
391
|
+
"pearson": Correlation(key="pearson", calculate=True),
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
correlation_table: bool = True
|
|
395
|
+
|
|
396
|
+
interactions: Interactions = Interactions()
|
|
397
|
+
interactions.continuous = False
|
|
398
|
+
|
|
399
|
+
missing_diagrams: Dict[str, bool] = {
|
|
400
|
+
"bar": False,
|
|
401
|
+
"matrix": False,
|
|
402
|
+
"dendrogram": False,
|
|
403
|
+
"heatmap": False,
|
|
404
|
+
}
|
|
405
|
+
samples: Samples = Samples()
|
|
406
|
+
samples.tail = 0
|
|
407
|
+
samples.random = 0
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
class Config:
|
|
411
|
+
arg_groups: Dict[str, Any] = {
|
|
412
|
+
"sensitive": {
|
|
413
|
+
"samples": None,
|
|
414
|
+
"duplicates": None,
|
|
415
|
+
"vars": {"cat": {"redact": True}, "text": {"redact": True}},
|
|
416
|
+
},
|
|
417
|
+
"flatly_theme": {
|
|
418
|
+
"html": {
|
|
419
|
+
"style": {
|
|
420
|
+
"theme": Theme.flatly,
|
|
421
|
+
"primary_color": "#2c3e50",
|
|
422
|
+
}
|
|
423
|
+
}
|
|
424
|
+
},
|
|
425
|
+
"united_theme": {
|
|
426
|
+
"html": {
|
|
427
|
+
"style": {
|
|
428
|
+
"theme": Theme.united,
|
|
429
|
+
"primary_color": "#d34615",
|
|
430
|
+
}
|
|
431
|
+
}
|
|
432
|
+
},
|
|
433
|
+
"explorative": {
|
|
434
|
+
"vars": {
|
|
435
|
+
"cat": {"characters": True, "words": True},
|
|
436
|
+
"url": {"active": True},
|
|
437
|
+
"path": {"active": True},
|
|
438
|
+
"file": {"active": True},
|
|
439
|
+
"image": {"active": True},
|
|
440
|
+
},
|
|
441
|
+
"n_obs_unique": 10,
|
|
442
|
+
"n_extreme_obs": 10,
|
|
443
|
+
"n_freq_table_max": 10,
|
|
444
|
+
"memory_deep": True,
|
|
445
|
+
},
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
_shorthands = {
|
|
449
|
+
"dataset": {
|
|
450
|
+
"creator": "",
|
|
451
|
+
"author": "",
|
|
452
|
+
"description": "",
|
|
453
|
+
"copyright_holder": "",
|
|
454
|
+
"copyright_year": "",
|
|
455
|
+
"url": "",
|
|
456
|
+
},
|
|
457
|
+
"samples": {"head": 0, "tail": 0, "random": 0},
|
|
458
|
+
"duplicates": {"head": 0},
|
|
459
|
+
"interactions": {"targets": [], "continuous": False},
|
|
460
|
+
"missing_diagrams": {
|
|
461
|
+
"bar": False,
|
|
462
|
+
"matrix": False,
|
|
463
|
+
"heatmap": False,
|
|
464
|
+
},
|
|
465
|
+
"correlations": {
|
|
466
|
+
"auto": {"calculate": False},
|
|
467
|
+
"pearson": {"calculate": False},
|
|
468
|
+
"spearman": {"calculate": False},
|
|
469
|
+
"kendall": {"calculate": False},
|
|
470
|
+
"phi_k": {"calculate": False},
|
|
471
|
+
"cramers": {"calculate": False},
|
|
472
|
+
},
|
|
473
|
+
"correlation_table": True,
|
|
474
|
+
}
|
|
475
|
+
|
|
476
|
+
@staticmethod
|
|
477
|
+
def get_arg_groups(key: str) -> dict:
|
|
478
|
+
kwargs = Config.arg_groups[key]
|
|
479
|
+
shorthand_args, _ = Config.shorthands(kwargs, split=False)
|
|
480
|
+
return shorthand_args
|
|
481
|
+
|
|
482
|
+
@staticmethod
|
|
483
|
+
def shorthands(kwargs: dict, split: bool = True) -> Tuple[dict, dict]:
|
|
484
|
+
shorthand_args = {}
|
|
485
|
+
if not split:
|
|
486
|
+
shorthand_args = kwargs
|
|
487
|
+
for key, value in list(kwargs.items()):
|
|
488
|
+
if value is None and key in Config._shorthands:
|
|
489
|
+
shorthand_args[key] = Config._shorthands[key]
|
|
490
|
+
if split:
|
|
491
|
+
del kwargs[key]
|
|
492
|
+
|
|
493
|
+
if split:
|
|
494
|
+
return shorthand_args, kwargs
|
|
495
|
+
else:
|
|
496
|
+
return shorthand_args, {}
|
|
@@ -0,0 +1,223 @@
|
|
|
1
|
+
# Title of the document
|
|
2
|
+
title: YData Profiling Report
|
|
3
|
+
|
|
4
|
+
# Metadata
|
|
5
|
+
dataset:
|
|
6
|
+
description: ""
|
|
7
|
+
creator: ""
|
|
8
|
+
author: ""
|
|
9
|
+
copyright_holder: ""
|
|
10
|
+
copyright_year: ""
|
|
11
|
+
url: ""
|
|
12
|
+
|
|
13
|
+
variables:
|
|
14
|
+
descriptions: {}
|
|
15
|
+
|
|
16
|
+
# infer dtypes
|
|
17
|
+
infer_dtypes: true
|
|
18
|
+
|
|
19
|
+
# Show the description at each variable (in addition to the overview tab)
|
|
20
|
+
show_variable_description: true
|
|
21
|
+
|
|
22
|
+
# Number of workers (0=multiprocessing.cpu_count())
|
|
23
|
+
pool_size: 0
|
|
24
|
+
|
|
25
|
+
# Show the progress bar
|
|
26
|
+
progress_bar: true
|
|
27
|
+
|
|
28
|
+
# Per variable type description settings
|
|
29
|
+
vars:
|
|
30
|
+
num:
|
|
31
|
+
quantiles:
|
|
32
|
+
- 0.05
|
|
33
|
+
- 0.25
|
|
34
|
+
- 0.5
|
|
35
|
+
- 0.75
|
|
36
|
+
- 0.95
|
|
37
|
+
skewness_threshold: 20
|
|
38
|
+
low_categorical_threshold: 5
|
|
39
|
+
# Set to zero to disable
|
|
40
|
+
chi_squared_threshold: 0.999
|
|
41
|
+
cat:
|
|
42
|
+
length: true
|
|
43
|
+
characters: true
|
|
44
|
+
words: true
|
|
45
|
+
cardinality_threshold: 50
|
|
46
|
+
n_obs: 5
|
|
47
|
+
# Set to zero to disable
|
|
48
|
+
chi_squared_threshold: 0.999
|
|
49
|
+
coerce_str_to_date: false
|
|
50
|
+
redact: false
|
|
51
|
+
histogram_largest: 50
|
|
52
|
+
stop_words: []
|
|
53
|
+
bool:
|
|
54
|
+
n_obs: 3
|
|
55
|
+
# string to boolean mapping dict
|
|
56
|
+
mappings:
|
|
57
|
+
t: true
|
|
58
|
+
f: false
|
|
59
|
+
yes: true
|
|
60
|
+
no: false
|
|
61
|
+
y: true
|
|
62
|
+
n: false
|
|
63
|
+
"true": true
|
|
64
|
+
"false": false
|
|
65
|
+
file:
|
|
66
|
+
active: false
|
|
67
|
+
image:
|
|
68
|
+
active: false
|
|
69
|
+
exif: true
|
|
70
|
+
hash: true
|
|
71
|
+
path:
|
|
72
|
+
active: false
|
|
73
|
+
url:
|
|
74
|
+
active: false
|
|
75
|
+
timeseries:
|
|
76
|
+
active: false
|
|
77
|
+
autocorrelation: 0.7
|
|
78
|
+
lags:
|
|
79
|
+
- 1
|
|
80
|
+
- 7
|
|
81
|
+
- 12
|
|
82
|
+
- 24
|
|
83
|
+
- 30
|
|
84
|
+
significance: 0.05
|
|
85
|
+
pacf_acf_lag: 100
|
|
86
|
+
|
|
87
|
+
# Sort the variables. Possible values: "ascending", "descending" or null (leaves original sorting)
|
|
88
|
+
sort: null
|
|
89
|
+
|
|
90
|
+
# which diagrams to show
|
|
91
|
+
missing_diagrams:
|
|
92
|
+
bar: true
|
|
93
|
+
matrix: true
|
|
94
|
+
heatmap: true
|
|
95
|
+
|
|
96
|
+
correlations:
|
|
97
|
+
pearson:
|
|
98
|
+
calculate: false
|
|
99
|
+
warn_high_correlations: true
|
|
100
|
+
threshold: 0.9
|
|
101
|
+
spearman:
|
|
102
|
+
calculate: false
|
|
103
|
+
warn_high_correlations: false
|
|
104
|
+
threshold: 0.9
|
|
105
|
+
kendall:
|
|
106
|
+
calculate: false
|
|
107
|
+
warn_high_correlations: false
|
|
108
|
+
threshold: 0.9
|
|
109
|
+
phi_k:
|
|
110
|
+
calculate: false
|
|
111
|
+
warn_high_correlations: false
|
|
112
|
+
threshold: 0.9
|
|
113
|
+
cramers:
|
|
114
|
+
calculate: false
|
|
115
|
+
warn_high_correlations: true
|
|
116
|
+
threshold: 0.9
|
|
117
|
+
auto:
|
|
118
|
+
calculate: true
|
|
119
|
+
warn_high_correlations: true
|
|
120
|
+
threshold: 0.9
|
|
121
|
+
|
|
122
|
+
# Bivariate / Pairwise relations
|
|
123
|
+
interactions:
|
|
124
|
+
targets: []
|
|
125
|
+
continuous: true
|
|
126
|
+
|
|
127
|
+
# For categorical
|
|
128
|
+
categorical_maximum_correlation_distinct: 100
|
|
129
|
+
|
|
130
|
+
report:
|
|
131
|
+
precision: 10
|
|
132
|
+
|
|
133
|
+
# Plot-specific settings
|
|
134
|
+
plot:
|
|
135
|
+
# Image format (svg or png)
|
|
136
|
+
image_format: svg
|
|
137
|
+
dpi: 800
|
|
138
|
+
|
|
139
|
+
scatter_threshold: 1000
|
|
140
|
+
|
|
141
|
+
correlation:
|
|
142
|
+
cmap: RdBu
|
|
143
|
+
bad: "#000000"
|
|
144
|
+
|
|
145
|
+
missing:
|
|
146
|
+
cmap: RdBu
|
|
147
|
+
# Force labels when there are > 50 variables
|
|
148
|
+
# https://github.com/ResidentMario/missingno/issues/93#issuecomment-513322615
|
|
149
|
+
force_labels: true
|
|
150
|
+
|
|
151
|
+
cat_frequency:
|
|
152
|
+
show: true # if false, the category frequency plot is turned off
|
|
153
|
+
type: bar # options: 'bar', 'pie'
|
|
154
|
+
max_unique: 10
|
|
155
|
+
colors: null # use null for default or give a list of matplotlib recognized strings
|
|
156
|
+
|
|
157
|
+
histogram:
|
|
158
|
+
x_axis_labels: true
|
|
159
|
+
|
|
160
|
+
# Number of bins (set to 0 to automatically detect the bin size)
|
|
161
|
+
bins: 50
|
|
162
|
+
|
|
163
|
+
# Maximum number of bins (when bins=0)
|
|
164
|
+
max_bins: 250
|
|
165
|
+
|
|
166
|
+
font_path: null
|
|
167
|
+
|
|
168
|
+
# The number of observations to show
|
|
169
|
+
n_obs_unique: 5
|
|
170
|
+
n_extreme_obs: 5
|
|
171
|
+
n_freq_table_max: 10
|
|
172
|
+
|
|
173
|
+
# Use `deep` flag for memory_usage
|
|
174
|
+
memory_deep: false
|
|
175
|
+
|
|
176
|
+
# Configuration related to the duplicates
|
|
177
|
+
duplicates:
|
|
178
|
+
head: 10
|
|
179
|
+
key: "# duplicates"
|
|
180
|
+
|
|
181
|
+
# Configuration related to the samples area
|
|
182
|
+
samples:
|
|
183
|
+
head: 10
|
|
184
|
+
tail: 10
|
|
185
|
+
random: 0
|
|
186
|
+
|
|
187
|
+
# Configuration related to the rejection of variables
|
|
188
|
+
reject_variables: true
|
|
189
|
+
|
|
190
|
+
# When in a Jupyter notebook
|
|
191
|
+
notebook:
|
|
192
|
+
iframe:
|
|
193
|
+
height: 800px
|
|
194
|
+
width: 100%
|
|
195
|
+
# or 'src'
|
|
196
|
+
attribute: srcdoc
|
|
197
|
+
|
|
198
|
+
html:
|
|
199
|
+
# Minify the html
|
|
200
|
+
minify_html: true
|
|
201
|
+
|
|
202
|
+
# Offline support
|
|
203
|
+
use_local_assets: true
|
|
204
|
+
|
|
205
|
+
# If true, single file, else directory with assets
|
|
206
|
+
inline: true
|
|
207
|
+
|
|
208
|
+
# Show navbar
|
|
209
|
+
navbar_show: true
|
|
210
|
+
|
|
211
|
+
# Assets prefix if inline = true
|
|
212
|
+
assets_prefix: null
|
|
213
|
+
|
|
214
|
+
# Styling options for the HTML report
|
|
215
|
+
style:
|
|
216
|
+
theme: null
|
|
217
|
+
logo: ""
|
|
218
|
+
primary_colors:
|
|
219
|
+
- "#0d6efd"
|
|
220
|
+
- "#dc3545"
|
|
221
|
+
- "#198754"
|
|
222
|
+
|
|
223
|
+
full_width: false
|