fg-data-profiling 4.19.0__py2.py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- data_profiling/__init__.py +34 -0
- data_profiling/compare_reports.py +359 -0
- data_profiling/config.py +496 -0
- data_profiling/config_default.yaml +223 -0
- data_profiling/config_minimal.yaml +222 -0
- data_profiling/controller/__init__.py +1 -0
- data_profiling/controller/console.py +125 -0
- data_profiling/controller/pandas_decorator.py +21 -0
- data_profiling/expectations_report.py +117 -0
- data_profiling/model/__init__.py +4 -0
- data_profiling/model/alerts.py +780 -0
- data_profiling/model/correlations.py +163 -0
- data_profiling/model/dataframe.py +35 -0
- data_profiling/model/describe.py +210 -0
- data_profiling/model/description.py +108 -0
- data_profiling/model/duplicates.py +14 -0
- data_profiling/model/expectation_algorithms.py +112 -0
- data_profiling/model/handler.py +81 -0
- data_profiling/model/missing.py +146 -0
- data_profiling/model/pairwise.py +33 -0
- data_profiling/model/pandas/__init__.py +55 -0
- data_profiling/model/pandas/correlations_pandas.py +207 -0
- data_profiling/model/pandas/dataframe_pandas.py +26 -0
- data_profiling/model/pandas/describe_boolean_pandas.py +43 -0
- data_profiling/model/pandas/describe_categorical_pandas.py +274 -0
- data_profiling/model/pandas/describe_counts_pandas.py +63 -0
- data_profiling/model/pandas/describe_date_pandas.py +77 -0
- data_profiling/model/pandas/describe_file_pandas.py +56 -0
- data_profiling/model/pandas/describe_generic_pandas.py +36 -0
- data_profiling/model/pandas/describe_image_pandas.py +255 -0
- data_profiling/model/pandas/describe_numeric_pandas.py +175 -0
- data_profiling/model/pandas/describe_path_pandas.py +63 -0
- data_profiling/model/pandas/describe_supported_pandas.py +41 -0
- data_profiling/model/pandas/describe_text_pandas.py +62 -0
- data_profiling/model/pandas/describe_timeseries_pandas.py +222 -0
- data_profiling/model/pandas/describe_url_pandas.py +57 -0
- data_profiling/model/pandas/discretize_pandas.py +81 -0
- data_profiling/model/pandas/duplicates_pandas.py +56 -0
- data_profiling/model/pandas/imbalance_pandas.py +35 -0
- data_profiling/model/pandas/missing_pandas.py +42 -0
- data_profiling/model/pandas/sample_pandas.py +38 -0
- data_profiling/model/pandas/summary_pandas.py +101 -0
- data_profiling/model/pandas/table_pandas.py +56 -0
- data_profiling/model/pandas/timeseries_index_pandas.py +33 -0
- data_profiling/model/pandas/utils_pandas.py +27 -0
- data_profiling/model/sample.py +37 -0
- data_profiling/model/spark/__init__.py +48 -0
- data_profiling/model/spark/correlations_spark.py +152 -0
- data_profiling/model/spark/dataframe_spark.py +34 -0
- data_profiling/model/spark/describe_boolean_spark.py +27 -0
- data_profiling/model/spark/describe_categorical_spark.py +28 -0
- data_profiling/model/spark/describe_counts_spark.py +105 -0
- data_profiling/model/spark/describe_date_spark.py +51 -0
- data_profiling/model/spark/describe_generic_spark.py +30 -0
- data_profiling/model/spark/describe_numeric_spark.py +155 -0
- data_profiling/model/spark/describe_supported_spark.py +33 -0
- data_profiling/model/spark/describe_text_spark.py +25 -0
- data_profiling/model/spark/duplicates_spark.py +54 -0
- data_profiling/model/spark/missing_spark.py +96 -0
- data_profiling/model/spark/sample_spark.py +43 -0
- data_profiling/model/spark/summary_spark.py +95 -0
- data_profiling/model/spark/table_spark.py +58 -0
- data_profiling/model/spark/timeseries_index_spark.py +12 -0
- data_profiling/model/summarizer.py +207 -0
- data_profiling/model/summary.py +66 -0
- data_profiling/model/summary_algorithms.py +276 -0
- data_profiling/model/table.py +10 -0
- data_profiling/model/timeseries_index.py +16 -0
- data_profiling/model/typeset.py +365 -0
- data_profiling/model/typeset_relations.py +143 -0
- data_profiling/profile_report.py +573 -0
- data_profiling/report/__init__.py +4 -0
- data_profiling/report/formatters.py +346 -0
- data_profiling/report/presentation/__init__.py +1 -0
- data_profiling/report/presentation/core/__init__.py +39 -0
- data_profiling/report/presentation/core/alerts.py +18 -0
- data_profiling/report/presentation/core/collapse.py +24 -0
- data_profiling/report/presentation/core/container.py +50 -0
- data_profiling/report/presentation/core/correlation_table.py +21 -0
- data_profiling/report/presentation/core/dropdown.py +44 -0
- data_profiling/report/presentation/core/duplicate.py +16 -0
- data_profiling/report/presentation/core/frequency_table.py +14 -0
- data_profiling/report/presentation/core/frequency_table_small.py +16 -0
- data_profiling/report/presentation/core/html.py +14 -0
- data_profiling/report/presentation/core/image.py +34 -0
- data_profiling/report/presentation/core/item_renderer.py +17 -0
- data_profiling/report/presentation/core/renderable.py +42 -0
- data_profiling/report/presentation/core/root.py +35 -0
- data_profiling/report/presentation/core/sample.py +20 -0
- data_profiling/report/presentation/core/scores.py +32 -0
- data_profiling/report/presentation/core/table.py +26 -0
- data_profiling/report/presentation/core/toggle_button.py +14 -0
- data_profiling/report/presentation/core/variable.py +40 -0
- data_profiling/report/presentation/core/variable_info.py +36 -0
- data_profiling/report/presentation/flavours/__init__.py +9 -0
- data_profiling/report/presentation/flavours/flavour_html.py +64 -0
- data_profiling/report/presentation/flavours/flavour_widget.py +61 -0
- data_profiling/report/presentation/flavours/flavours.py +43 -0
- data_profiling/report/presentation/flavours/html/__init__.py +47 -0
- data_profiling/report/presentation/flavours/html/alerts.py +10 -0
- data_profiling/report/presentation/flavours/html/collapse.py +7 -0
- data_profiling/report/presentation/flavours/html/container.py +58 -0
- data_profiling/report/presentation/flavours/html/correlation_table.py +13 -0
- data_profiling/report/presentation/flavours/html/dropdown.py +7 -0
- data_profiling/report/presentation/flavours/html/duplicate.py +24 -0
- data_profiling/report/presentation/flavours/html/frequency_table.py +20 -0
- data_profiling/report/presentation/flavours/html/frequency_table_small.py +15 -0
- data_profiling/report/presentation/flavours/html/html.py +6 -0
- data_profiling/report/presentation/flavours/html/image.py +7 -0
- data_profiling/report/presentation/flavours/html/root.py +14 -0
- data_profiling/report/presentation/flavours/html/sample.py +12 -0
- data_profiling/report/presentation/flavours/html/scores.py +11 -0
- data_profiling/report/presentation/flavours/html/table.py +7 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant_length.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_dirty_category.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_duplicates.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_empty.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_cardinality.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_correlation.html +4 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_imbalance.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_infinite.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_missing.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_near_duplicates.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_non_stationary.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_seasonal.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_skewed.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_truncated.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_type_date.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_uniform.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_unique.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_unsupported.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_zeros.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts.html +47 -0
- data_profiling/report/presentation/flavours/html/templates/collapse.html +11 -0
- data_profiling/report/presentation/flavours/html/templates/correlation_table.html +5 -0
- data_profiling/report/presentation/flavours/html/templates/diagram.html +11 -0
- data_profiling/report/presentation/flavours/html/templates/dropdown.html +16 -0
- data_profiling/report/presentation/flavours/html/templates/duplicate.html +5 -0
- data_profiling/report/presentation/flavours/html/templates/frequency_table.html +45 -0
- data_profiling/report/presentation/flavours/html/templates/frequency_table_small.html +34 -0
- data_profiling/report/presentation/flavours/html/templates/report.html +26 -0
- data_profiling/report/presentation/flavours/html/templates/sample.html +10 -0
- data_profiling/report/presentation/flavours/html/templates/scores.html +78 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/batch_grid.html +16 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/grid.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/list.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/named_list.html +8 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/overview_tabs.html +30 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/scores.html +3 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/sections.html +13 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/select.html +40 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/tabs.html +30 -0
- data_profiling/report/presentation/flavours/html/templates/table.html +38 -0
- data_profiling/report/presentation/flavours/html/templates/toggle_button.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/variable.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/variable_info.html +49 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.bundle.min.js +7 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.min.css +6 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/cosmo.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/flatly.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/script.js +52 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/simplex.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/style.css +253 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/united.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/footer.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/javascript.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/navigation.html +36 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/style.html +53 -0
- data_profiling/report/presentation/flavours/html/templates.py +76 -0
- data_profiling/report/presentation/flavours/html/toggle_button.py +7 -0
- data_profiling/report/presentation/flavours/html/variable.py +7 -0
- data_profiling/report/presentation/flavours/html/variable_info.py +7 -0
- data_profiling/report/presentation/flavours/widget/__init__.py +49 -0
- data_profiling/report/presentation/flavours/widget/alerts.py +45 -0
- data_profiling/report/presentation/flavours/widget/collapse.py +43 -0
- data_profiling/report/presentation/flavours/widget/container.py +121 -0
- data_profiling/report/presentation/flavours/widget/correlation_table.py +14 -0
- data_profiling/report/presentation/flavours/widget/dropdown.py +31 -0
- data_profiling/report/presentation/flavours/widget/duplicate.py +14 -0
- data_profiling/report/presentation/flavours/widget/frequency_table.py +57 -0
- data_profiling/report/presentation/flavours/widget/frequency_table_small.py +66 -0
- data_profiling/report/presentation/flavours/widget/html.py +11 -0
- data_profiling/report/presentation/flavours/widget/image.py +26 -0
- data_profiling/report/presentation/flavours/widget/notebook.py +81 -0
- data_profiling/report/presentation/flavours/widget/root.py +10 -0
- data_profiling/report/presentation/flavours/widget/sample.py +14 -0
- data_profiling/report/presentation/flavours/widget/table.py +30 -0
- data_profiling/report/presentation/flavours/widget/toggle_button.py +17 -0
- data_profiling/report/presentation/flavours/widget/variable.py +12 -0
- data_profiling/report/presentation/flavours/widget/variable_info.py +11 -0
- data_profiling/report/presentation/frequency_table_utils.py +141 -0
- data_profiling/report/structure/__init__.py +1 -0
- data_profiling/report/structure/correlations.py +123 -0
- data_profiling/report/structure/overview.py +376 -0
- data_profiling/report/structure/report.py +457 -0
- data_profiling/report/structure/variables/__init__.py +35 -0
- data_profiling/report/structure/variables/render_boolean.py +132 -0
- data_profiling/report/structure/variables/render_categorical.py +566 -0
- data_profiling/report/structure/variables/render_common.py +31 -0
- data_profiling/report/structure/variables/render_complex.py +102 -0
- data_profiling/report/structure/variables/render_count.py +172 -0
- data_profiling/report/structure/variables/render_date.py +143 -0
- data_profiling/report/structure/variables/render_file.py +70 -0
- data_profiling/report/structure/variables/render_generic.py +45 -0
- data_profiling/report/structure/variables/render_image.py +204 -0
- data_profiling/report/structure/variables/render_path.py +134 -0
- data_profiling/report/structure/variables/render_real.py +314 -0
- data_profiling/report/structure/variables/render_text.py +189 -0
- data_profiling/report/structure/variables/render_timeseries.py +371 -0
- data_profiling/report/structure/variables/render_url.py +132 -0
- data_profiling/report/utils.py +34 -0
- data_profiling/serialize_report.py +143 -0
- data_profiling/utils/__init__.py +1 -0
- data_profiling/utils/backend.py +9 -0
- data_profiling/utils/cache.py +59 -0
- data_profiling/utils/common.py +142 -0
- data_profiling/utils/compat.py +31 -0
- data_profiling/utils/dataframe.py +238 -0
- data_profiling/utils/logger.py +53 -0
- data_profiling/utils/notebook.py +8 -0
- data_profiling/utils/paths.py +45 -0
- data_profiling/utils/progress_bar.py +15 -0
- data_profiling/utils/styles.py +22 -0
- data_profiling/utils/versions.py +19 -0
- data_profiling/version.py +1 -0
- data_profiling/visualisation/__init__.py +1 -0
- data_profiling/visualisation/context.py +87 -0
- data_profiling/visualisation/missing.py +138 -0
- data_profiling/visualisation/plot.py +1158 -0
- data_profiling/visualisation/utils.py +113 -0
- fg_data_profiling-4.19.0.dist-info/METADATA +362 -0
- fg_data_profiling-4.19.0.dist-info/RECORD +238 -0
- fg_data_profiling-4.19.0.dist-info/WHEEL +6 -0
- fg_data_profiling-4.19.0.dist-info/entry_points.txt +3 -0
- fg_data_profiling-4.19.0.dist-info/licenses/LICENSE +21 -0
- fg_data_profiling-4.19.0.dist-info/top_level.txt +2 -0
- ydata_profiling/__init__.py +43 -0
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
import contextlib
|
|
2
|
+
import string
|
|
3
|
+
from collections import Counter
|
|
4
|
+
from typing import List, Tuple
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
|
|
9
|
+
from data_profiling.config import Settings
|
|
10
|
+
from data_profiling.model.pandas.imbalance_pandas import column_imbalance_score
|
|
11
|
+
from data_profiling.model.pandas.utils_pandas import weighted_median
|
|
12
|
+
from data_profiling.model.summary_algorithms import (
|
|
13
|
+
chi_square,
|
|
14
|
+
describe_categorical_1d,
|
|
15
|
+
histogram_compute,
|
|
16
|
+
series_handle_nulls,
|
|
17
|
+
series_hashable,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def get_character_counts_vc(vc: pd.Series) -> pd.Series:
|
|
22
|
+
series = pd.Series(vc.index, index=vc, dtype=object)
|
|
23
|
+
characters = series[series != ""].apply(list)
|
|
24
|
+
characters = characters.explode()
|
|
25
|
+
|
|
26
|
+
counts = pd.Series(characters.index, index=characters).dropna()
|
|
27
|
+
if len(counts) > 0:
|
|
28
|
+
counts = counts.groupby(level=0, sort=False).sum()
|
|
29
|
+
counts = counts.sort_values(ascending=False)
|
|
30
|
+
# FIXME: correct in split, below should be zero: print(counts.loc[''])
|
|
31
|
+
counts = counts[counts.index.str.len() > 0]
|
|
32
|
+
return counts
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def get_character_counts(series: pd.Series) -> Counter:
|
|
36
|
+
"""Function to return the character counts
|
|
37
|
+
|
|
38
|
+
Args:
|
|
39
|
+
series: the Series to process
|
|
40
|
+
|
|
41
|
+
Returns:
|
|
42
|
+
A dict with character counts
|
|
43
|
+
"""
|
|
44
|
+
return Counter(series.str.cat())
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def counter_to_series(counter: Counter) -> pd.Series:
|
|
48
|
+
if not counter:
|
|
49
|
+
return pd.Series([], dtype=object)
|
|
50
|
+
|
|
51
|
+
counter_as_tuples = counter.most_common()
|
|
52
|
+
items, counts = zip(*counter_as_tuples)
|
|
53
|
+
return pd.Series(counts, index=items)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def unicode_summary_vc(vc: pd.Series) -> dict:
|
|
57
|
+
try:
|
|
58
|
+
from tangled_up_in_unicode import ( # type: ignore
|
|
59
|
+
block,
|
|
60
|
+
block_abbr,
|
|
61
|
+
category,
|
|
62
|
+
category_long,
|
|
63
|
+
script,
|
|
64
|
+
)
|
|
65
|
+
except ImportError:
|
|
66
|
+
from unicodedata import category as _category # pylint: disable=import-error
|
|
67
|
+
|
|
68
|
+
category = _category # type: ignore
|
|
69
|
+
char_handler = lambda char: "(unknown)" # noqa: E731
|
|
70
|
+
block = char_handler
|
|
71
|
+
block_abbr = char_handler
|
|
72
|
+
category_long = char_handler
|
|
73
|
+
script = char_handler
|
|
74
|
+
|
|
75
|
+
# Unicode Character Summaries (category and script name)
|
|
76
|
+
character_counts = get_character_counts_vc(vc)
|
|
77
|
+
|
|
78
|
+
character_counts_series = character_counts
|
|
79
|
+
summary = {
|
|
80
|
+
"n_characters_distinct": len(character_counts_series),
|
|
81
|
+
"n_characters": np.sum(character_counts_series.values),
|
|
82
|
+
"character_counts": character_counts_series,
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
char_to_block = {key: block(key) for key in character_counts.keys()}
|
|
86
|
+
char_to_category_short = {key: category(key) for key in character_counts.keys()}
|
|
87
|
+
char_to_script = {key: script(key) for key in character_counts.keys()}
|
|
88
|
+
|
|
89
|
+
summary.update(
|
|
90
|
+
{
|
|
91
|
+
"category_alias_values": {
|
|
92
|
+
key: category_long(value)
|
|
93
|
+
for key, value in char_to_category_short.items()
|
|
94
|
+
},
|
|
95
|
+
"block_alias_values": {
|
|
96
|
+
key: block_abbr(value) for key, value in char_to_block.items()
|
|
97
|
+
},
|
|
98
|
+
}
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
# Retrieve original distribution
|
|
102
|
+
block_alias_counts: Counter = Counter()
|
|
103
|
+
per_block_char_counts: dict = {
|
|
104
|
+
k: Counter() for k in summary["block_alias_values"].values()
|
|
105
|
+
}
|
|
106
|
+
for char, n_char in character_counts.items():
|
|
107
|
+
block_name = summary["block_alias_values"][char]
|
|
108
|
+
block_alias_counts[block_name] += n_char
|
|
109
|
+
per_block_char_counts[block_name][char] = n_char
|
|
110
|
+
summary["block_alias_counts"] = counter_to_series(block_alias_counts)
|
|
111
|
+
summary["n_block_alias"] = len(summary["block_alias_counts"])
|
|
112
|
+
summary["block_alias_char_counts"] = {
|
|
113
|
+
k: counter_to_series(v) for k, v in per_block_char_counts.items()
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
script_counts: Counter = Counter()
|
|
117
|
+
per_script_char_counts: dict = {k: Counter() for k in char_to_script.values()}
|
|
118
|
+
for char, n_char in character_counts.items():
|
|
119
|
+
script_name = char_to_script[char]
|
|
120
|
+
script_counts[script_name] += n_char
|
|
121
|
+
per_script_char_counts[script_name][char] = n_char
|
|
122
|
+
summary["script_counts"] = counter_to_series(script_counts)
|
|
123
|
+
summary["n_scripts"] = len(summary["script_counts"])
|
|
124
|
+
summary["script_char_counts"] = {
|
|
125
|
+
k: counter_to_series(v) for k, v in per_script_char_counts.items()
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
category_alias_counts: Counter = Counter()
|
|
129
|
+
per_category_alias_char_counts: dict = {
|
|
130
|
+
k: Counter() for k in summary["category_alias_values"].values()
|
|
131
|
+
}
|
|
132
|
+
for char, n_char in character_counts.items():
|
|
133
|
+
category_alias_name = summary["category_alias_values"][char]
|
|
134
|
+
category_alias_counts[category_alias_name] += n_char
|
|
135
|
+
per_category_alias_char_counts[category_alias_name][char] += n_char
|
|
136
|
+
summary["category_alias_counts"] = counter_to_series(category_alias_counts)
|
|
137
|
+
if len(summary["category_alias_counts"]) > 0:
|
|
138
|
+
summary["category_alias_counts"].index = summary[
|
|
139
|
+
"category_alias_counts"
|
|
140
|
+
].index.str.replace("_", " ")
|
|
141
|
+
summary["n_category"] = len(summary["category_alias_counts"])
|
|
142
|
+
summary["category_alias_char_counts"] = {
|
|
143
|
+
k: counter_to_series(v) for k, v in per_category_alias_char_counts.items()
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
with contextlib.suppress(AttributeError):
|
|
147
|
+
summary["category_alias_counts"].index = summary[
|
|
148
|
+
"category_alias_counts"
|
|
149
|
+
].index.str.replace("_", " ")
|
|
150
|
+
|
|
151
|
+
return summary
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def word_summary_vc(vc: pd.Series, stop_words: List[str] = []) -> dict:
|
|
155
|
+
"""Count the number of occurrences of each individual word across
|
|
156
|
+
all lines of the data Series, then sort from the word with the most
|
|
157
|
+
occurrences to the word with the least occurrences. If a list of
|
|
158
|
+
stop words is given, they will be ignored.
|
|
159
|
+
|
|
160
|
+
Args:
|
|
161
|
+
vc: Series containing all unique categories as index and their
|
|
162
|
+
frequency as value. Sorted from the most frequent down.
|
|
163
|
+
stop_words: List of stop words to ignore, empty by default.
|
|
164
|
+
|
|
165
|
+
Returns:
|
|
166
|
+
A dict containing the results as a Series with unique words as
|
|
167
|
+
index and the computed frequency as value
|
|
168
|
+
"""
|
|
169
|
+
# TODO: configurable lowercase/punctuation etc.
|
|
170
|
+
# TODO: remove punctuation in words
|
|
171
|
+
|
|
172
|
+
series = pd.Series(vc.index, index=vc, dtype=object)
|
|
173
|
+
word_lists = series.str.lower().str.split()
|
|
174
|
+
words = word_lists.explode().str.strip(string.punctuation + string.whitespace)
|
|
175
|
+
word_counts = pd.Series(words.index, index=words)
|
|
176
|
+
# fix for pandas 1.0.5
|
|
177
|
+
word_counts = word_counts[word_counts.index.notnull()]
|
|
178
|
+
word_counts = word_counts.groupby(level=0, sort=False).sum()
|
|
179
|
+
word_counts = word_counts.sort_values(ascending=False)
|
|
180
|
+
|
|
181
|
+
# Remove stop words
|
|
182
|
+
if len(stop_words) > 0:
|
|
183
|
+
stop_words = [x.lower() for x in stop_words]
|
|
184
|
+
word_counts = word_counts.loc[~word_counts.index.isin(stop_words)]
|
|
185
|
+
|
|
186
|
+
return {"word_counts": word_counts} if not word_counts.empty else {}
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def length_summary_vc(vc: pd.Series) -> dict:
|
|
190
|
+
series = pd.Series(vc.index, index=vc, dtype=object)
|
|
191
|
+
length = series.str.len()
|
|
192
|
+
length_counts = pd.Series(length.index, index=length)
|
|
193
|
+
length_counts = length_counts.groupby(level=0, sort=False).sum()
|
|
194
|
+
length_counts = length_counts.sort_values(ascending=False)
|
|
195
|
+
|
|
196
|
+
summary = {
|
|
197
|
+
"max_length": np.max(length_counts.index),
|
|
198
|
+
"mean_length": np.average(length_counts.index, weights=length_counts.values)
|
|
199
|
+
if not length_counts.empty
|
|
200
|
+
else np.nan,
|
|
201
|
+
"median_length": weighted_median(
|
|
202
|
+
length_counts.index.values, weights=length_counts.values
|
|
203
|
+
)
|
|
204
|
+
if not length_counts.empty
|
|
205
|
+
else np.nan,
|
|
206
|
+
"min_length": np.min(length_counts.index),
|
|
207
|
+
"length_histogram": length_counts,
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
return summary
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
_displayed_catvar_banner = False
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
@describe_categorical_1d.register
|
|
217
|
+
@series_hashable
|
|
218
|
+
@series_handle_nulls
|
|
219
|
+
def pandas_describe_categorical_1d(
|
|
220
|
+
config: Settings, series: pd.Series, summary: dict
|
|
221
|
+
) -> Tuple[Settings, pd.Series, dict]:
|
|
222
|
+
"""Describe a categorical series.
|
|
223
|
+
|
|
224
|
+
Args:
|
|
225
|
+
config: report Settings object
|
|
226
|
+
series: The Series to describe.
|
|
227
|
+
summary: The dict containing the series description so far.
|
|
228
|
+
|
|
229
|
+
Returns:
|
|
230
|
+
A dict containing calculated series description values.
|
|
231
|
+
"""
|
|
232
|
+
# Global info banner
|
|
233
|
+
global _displayed_catvar_banner
|
|
234
|
+
|
|
235
|
+
# Make sure we deal with strings (Issue #100)
|
|
236
|
+
series = series.astype(str)
|
|
237
|
+
|
|
238
|
+
# Only run if at least 1 non-missing value
|
|
239
|
+
value_counts = summary["value_counts_without_nan"]
|
|
240
|
+
value_counts.index = value_counts.index.astype(str)
|
|
241
|
+
|
|
242
|
+
summary["imbalance"] = column_imbalance_score(value_counts, len(value_counts))
|
|
243
|
+
|
|
244
|
+
redact = config.vars.cat.redact
|
|
245
|
+
if not redact:
|
|
246
|
+
summary.update({"first_rows": series.head(5)})
|
|
247
|
+
|
|
248
|
+
chi_squared_threshold = config.vars.num.chi_squared_threshold
|
|
249
|
+
if chi_squared_threshold > 0.0:
|
|
250
|
+
summary["chi_squared"] = chi_square(histogram=value_counts.values)
|
|
251
|
+
|
|
252
|
+
if config.vars.cat.length:
|
|
253
|
+
summary.update(length_summary_vc(value_counts))
|
|
254
|
+
summary.update(
|
|
255
|
+
histogram_compute(
|
|
256
|
+
config,
|
|
257
|
+
summary["length_histogram"].index.values,
|
|
258
|
+
len(summary["length_histogram"]),
|
|
259
|
+
name="histogram_length",
|
|
260
|
+
weights=summary["length_histogram"].values,
|
|
261
|
+
)
|
|
262
|
+
)
|
|
263
|
+
|
|
264
|
+
if config.vars.cat.characters:
|
|
265
|
+
summary.update(unicode_summary_vc(value_counts))
|
|
266
|
+
|
|
267
|
+
if config.vars.cat.words:
|
|
268
|
+
summary.update(word_summary_vc(value_counts, config.vars.cat.stop_words))
|
|
269
|
+
|
|
270
|
+
if config.vars.cat.dirty_categories: # noqa: SIM102
|
|
271
|
+
if not _displayed_catvar_banner:
|
|
272
|
+
_displayed_catvar_banner = True
|
|
273
|
+
|
|
274
|
+
return config, series, summary
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
from typing import Tuple
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from data_profiling.config import Settings
|
|
6
|
+
from data_profiling.model.summary_algorithms import describe_counts
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@describe_counts.register
|
|
10
|
+
def pandas_describe_counts(
|
|
11
|
+
config: Settings, series: pd.Series, summary: dict
|
|
12
|
+
) -> Tuple[Settings, pd.Series, dict]:
|
|
13
|
+
"""Counts the values in a series (with and without NaN, distinct).
|
|
14
|
+
|
|
15
|
+
Args:
|
|
16
|
+
config: report Settings object
|
|
17
|
+
series: Series for which we want to calculate the values.
|
|
18
|
+
summary: series' summary
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
A dictionary with the count values (with and without NaN, distinct).
|
|
22
|
+
"""
|
|
23
|
+
try:
|
|
24
|
+
value_counts_with_nan = series.value_counts(dropna=False)
|
|
25
|
+
_ = set(value_counts_with_nan.index)
|
|
26
|
+
hashable = True
|
|
27
|
+
except: # noqa: E722
|
|
28
|
+
hashable = False
|
|
29
|
+
|
|
30
|
+
summary["hashable"] = hashable
|
|
31
|
+
|
|
32
|
+
if hashable:
|
|
33
|
+
value_counts_with_nan = value_counts_with_nan[value_counts_with_nan > 0]
|
|
34
|
+
|
|
35
|
+
null_index = value_counts_with_nan.index.isnull()
|
|
36
|
+
if null_index.any():
|
|
37
|
+
n_missing = value_counts_with_nan[null_index].sum()
|
|
38
|
+
value_counts_without_nan = value_counts_with_nan[~null_index]
|
|
39
|
+
else:
|
|
40
|
+
n_missing = 0
|
|
41
|
+
value_counts_without_nan = value_counts_with_nan
|
|
42
|
+
|
|
43
|
+
summary.update(
|
|
44
|
+
{
|
|
45
|
+
"value_counts_without_nan": value_counts_without_nan,
|
|
46
|
+
}
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
try:
|
|
50
|
+
summary["value_counts_index_sorted"] = summary[
|
|
51
|
+
"value_counts_without_nan"
|
|
52
|
+
].sort_index(ascending=True)
|
|
53
|
+
ordering = True
|
|
54
|
+
except TypeError:
|
|
55
|
+
ordering = False
|
|
56
|
+
else:
|
|
57
|
+
n_missing = series.isna().sum()
|
|
58
|
+
ordering = False
|
|
59
|
+
|
|
60
|
+
summary["ordering"] = ordering
|
|
61
|
+
summary["n_missing"] = n_missing
|
|
62
|
+
|
|
63
|
+
return config, series, summary
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
from typing import Tuple
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
from data_profiling.config import Settings
|
|
7
|
+
from data_profiling.model.summary_algorithms import (
|
|
8
|
+
chi_square,
|
|
9
|
+
describe_date_1d,
|
|
10
|
+
histogram_compute,
|
|
11
|
+
series_handle_nulls,
|
|
12
|
+
series_hashable,
|
|
13
|
+
)
|
|
14
|
+
from data_profiling.model.typeset_relations import is_pandas_1
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def to_datetime(series: pd.Series) -> pd.Series:
|
|
18
|
+
if is_pandas_1():
|
|
19
|
+
return pd.to_datetime(series, errors="coerce")
|
|
20
|
+
return pd.to_datetime(series, format="mixed", errors="coerce")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@describe_date_1d.register
|
|
24
|
+
@series_hashable
|
|
25
|
+
@series_handle_nulls
|
|
26
|
+
def pandas_describe_date_1d(
|
|
27
|
+
config: Settings, series: pd.Series, summary: dict
|
|
28
|
+
) -> Tuple[Settings, pd.Series, dict]:
|
|
29
|
+
"""Describe a date series.
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
config: report Settings object
|
|
33
|
+
series: The Series to describe.
|
|
34
|
+
summary: The dict containing the series description so far.
|
|
35
|
+
|
|
36
|
+
Returns:
|
|
37
|
+
A dict containing calculated series description values.
|
|
38
|
+
"""
|
|
39
|
+
og_series = series.dropna()
|
|
40
|
+
series = to_datetime(og_series)
|
|
41
|
+
invalid_values = og_series[series.isna()]
|
|
42
|
+
|
|
43
|
+
series = series.dropna()
|
|
44
|
+
|
|
45
|
+
if summary["value_counts_without_nan"].empty:
|
|
46
|
+
values = series.values
|
|
47
|
+
summary.update(
|
|
48
|
+
{
|
|
49
|
+
"min": pd.NaT,
|
|
50
|
+
"max": pd.NaT,
|
|
51
|
+
"range": 0,
|
|
52
|
+
}
|
|
53
|
+
)
|
|
54
|
+
else:
|
|
55
|
+
summary.update(
|
|
56
|
+
{
|
|
57
|
+
"min": pd.Timestamp.to_pydatetime(series.min()),
|
|
58
|
+
"max": pd.Timestamp.to_pydatetime(series.max()),
|
|
59
|
+
}
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
summary["range"] = summary["max"] - summary["min"]
|
|
63
|
+
|
|
64
|
+
values = series.values.astype(np.int64) // 10**9
|
|
65
|
+
|
|
66
|
+
if config.vars.num.chi_squared_threshold > 0.0:
|
|
67
|
+
summary["chi_squared"] = chi_square(values)
|
|
68
|
+
|
|
69
|
+
summary.update(histogram_compute(config, values, series.nunique()))
|
|
70
|
+
summary.update(
|
|
71
|
+
{
|
|
72
|
+
"invalid_dates": invalid_values.nunique(),
|
|
73
|
+
"n_invalid_dates": len(invalid_values),
|
|
74
|
+
"p_invalid_dates": len(invalid_values) / summary["n"],
|
|
75
|
+
}
|
|
76
|
+
)
|
|
77
|
+
return config, values, summary
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from datetime import datetime
|
|
3
|
+
from typing import Tuple
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
from data_profiling.config import Settings
|
|
8
|
+
from data_profiling.model.summary_algorithms import describe_file_1d, histogram_compute
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def file_summary(series: pd.Series) -> dict:
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
Args:
|
|
15
|
+
series: series to summarize
|
|
16
|
+
|
|
17
|
+
Returns:
|
|
18
|
+
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
# Transform
|
|
22
|
+
stats = series.map(lambda x: os.stat(x))
|
|
23
|
+
|
|
24
|
+
def convert_datetime(x: float) -> str:
|
|
25
|
+
return datetime.fromtimestamp(x).strftime("%Y-%m-%d %H:%M:%S")
|
|
26
|
+
|
|
27
|
+
# Transform some more
|
|
28
|
+
summary = {
|
|
29
|
+
"file_size": stats.map(lambda x: x.st_size),
|
|
30
|
+
"file_created_time": stats.map(lambda x: x.st_ctime).map(convert_datetime),
|
|
31
|
+
"file_accessed_time": stats.map(lambda x: x.st_atime).map(convert_datetime),
|
|
32
|
+
"file_modified_time": stats.map(lambda x: x.st_mtime).map(convert_datetime),
|
|
33
|
+
}
|
|
34
|
+
return summary
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@describe_file_1d.register
|
|
38
|
+
def pandas_describe_file_1d(
|
|
39
|
+
config: Settings, series: pd.Series, summary: dict
|
|
40
|
+
) -> Tuple[Settings, pd.Series, dict]:
|
|
41
|
+
if series.hasnans:
|
|
42
|
+
raise ValueError("May not contain NaNs")
|
|
43
|
+
if not hasattr(series, "str"):
|
|
44
|
+
raise ValueError("series should have .str accessor")
|
|
45
|
+
|
|
46
|
+
summary.update(file_summary(series))
|
|
47
|
+
summary.update(
|
|
48
|
+
histogram_compute(
|
|
49
|
+
config,
|
|
50
|
+
summary["file_size"],
|
|
51
|
+
summary["file_size"].nunique(),
|
|
52
|
+
name="histogram_file_size",
|
|
53
|
+
)
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
return config, series, summary
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
from typing import Tuple
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from data_profiling.config import Settings
|
|
6
|
+
from data_profiling.model.summary_algorithms import describe_generic
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@describe_generic.register
|
|
10
|
+
def pandas_describe_generic(
|
|
11
|
+
config: Settings, series: pd.Series, summary: dict
|
|
12
|
+
) -> Tuple[Settings, pd.Series, dict]:
|
|
13
|
+
"""Describe generic series.
|
|
14
|
+
|
|
15
|
+
Args:
|
|
16
|
+
config: report Settings object
|
|
17
|
+
series: The Series to describe.
|
|
18
|
+
summary: The dict containing the series description so far.
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
A dict containing calculated series description values.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
# number of observations in the Series
|
|
25
|
+
length = len(series)
|
|
26
|
+
|
|
27
|
+
summary.update(
|
|
28
|
+
{
|
|
29
|
+
"n": length,
|
|
30
|
+
"p_missing": summary["n_missing"] / length if length > 0 else 0,
|
|
31
|
+
"count": length - summary["n_missing"],
|
|
32
|
+
"memory_size": series.memory_usage(deep=config.memory_deep),
|
|
33
|
+
}
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
return config, series, summary
|