fg-data-profiling 4.19.0__py2.py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- data_profiling/__init__.py +34 -0
- data_profiling/compare_reports.py +359 -0
- data_profiling/config.py +496 -0
- data_profiling/config_default.yaml +223 -0
- data_profiling/config_minimal.yaml +222 -0
- data_profiling/controller/__init__.py +1 -0
- data_profiling/controller/console.py +125 -0
- data_profiling/controller/pandas_decorator.py +21 -0
- data_profiling/expectations_report.py +117 -0
- data_profiling/model/__init__.py +4 -0
- data_profiling/model/alerts.py +780 -0
- data_profiling/model/correlations.py +163 -0
- data_profiling/model/dataframe.py +35 -0
- data_profiling/model/describe.py +210 -0
- data_profiling/model/description.py +108 -0
- data_profiling/model/duplicates.py +14 -0
- data_profiling/model/expectation_algorithms.py +112 -0
- data_profiling/model/handler.py +81 -0
- data_profiling/model/missing.py +146 -0
- data_profiling/model/pairwise.py +33 -0
- data_profiling/model/pandas/__init__.py +55 -0
- data_profiling/model/pandas/correlations_pandas.py +207 -0
- data_profiling/model/pandas/dataframe_pandas.py +26 -0
- data_profiling/model/pandas/describe_boolean_pandas.py +43 -0
- data_profiling/model/pandas/describe_categorical_pandas.py +274 -0
- data_profiling/model/pandas/describe_counts_pandas.py +63 -0
- data_profiling/model/pandas/describe_date_pandas.py +77 -0
- data_profiling/model/pandas/describe_file_pandas.py +56 -0
- data_profiling/model/pandas/describe_generic_pandas.py +36 -0
- data_profiling/model/pandas/describe_image_pandas.py +255 -0
- data_profiling/model/pandas/describe_numeric_pandas.py +175 -0
- data_profiling/model/pandas/describe_path_pandas.py +63 -0
- data_profiling/model/pandas/describe_supported_pandas.py +41 -0
- data_profiling/model/pandas/describe_text_pandas.py +62 -0
- data_profiling/model/pandas/describe_timeseries_pandas.py +222 -0
- data_profiling/model/pandas/describe_url_pandas.py +57 -0
- data_profiling/model/pandas/discretize_pandas.py +81 -0
- data_profiling/model/pandas/duplicates_pandas.py +56 -0
- data_profiling/model/pandas/imbalance_pandas.py +35 -0
- data_profiling/model/pandas/missing_pandas.py +42 -0
- data_profiling/model/pandas/sample_pandas.py +38 -0
- data_profiling/model/pandas/summary_pandas.py +101 -0
- data_profiling/model/pandas/table_pandas.py +56 -0
- data_profiling/model/pandas/timeseries_index_pandas.py +33 -0
- data_profiling/model/pandas/utils_pandas.py +27 -0
- data_profiling/model/sample.py +37 -0
- data_profiling/model/spark/__init__.py +48 -0
- data_profiling/model/spark/correlations_spark.py +152 -0
- data_profiling/model/spark/dataframe_spark.py +34 -0
- data_profiling/model/spark/describe_boolean_spark.py +27 -0
- data_profiling/model/spark/describe_categorical_spark.py +28 -0
- data_profiling/model/spark/describe_counts_spark.py +105 -0
- data_profiling/model/spark/describe_date_spark.py +51 -0
- data_profiling/model/spark/describe_generic_spark.py +30 -0
- data_profiling/model/spark/describe_numeric_spark.py +155 -0
- data_profiling/model/spark/describe_supported_spark.py +33 -0
- data_profiling/model/spark/describe_text_spark.py +25 -0
- data_profiling/model/spark/duplicates_spark.py +54 -0
- data_profiling/model/spark/missing_spark.py +96 -0
- data_profiling/model/spark/sample_spark.py +43 -0
- data_profiling/model/spark/summary_spark.py +95 -0
- data_profiling/model/spark/table_spark.py +58 -0
- data_profiling/model/spark/timeseries_index_spark.py +12 -0
- data_profiling/model/summarizer.py +207 -0
- data_profiling/model/summary.py +66 -0
- data_profiling/model/summary_algorithms.py +276 -0
- data_profiling/model/table.py +10 -0
- data_profiling/model/timeseries_index.py +16 -0
- data_profiling/model/typeset.py +365 -0
- data_profiling/model/typeset_relations.py +143 -0
- data_profiling/profile_report.py +573 -0
- data_profiling/report/__init__.py +4 -0
- data_profiling/report/formatters.py +346 -0
- data_profiling/report/presentation/__init__.py +1 -0
- data_profiling/report/presentation/core/__init__.py +39 -0
- data_profiling/report/presentation/core/alerts.py +18 -0
- data_profiling/report/presentation/core/collapse.py +24 -0
- data_profiling/report/presentation/core/container.py +50 -0
- data_profiling/report/presentation/core/correlation_table.py +21 -0
- data_profiling/report/presentation/core/dropdown.py +44 -0
- data_profiling/report/presentation/core/duplicate.py +16 -0
- data_profiling/report/presentation/core/frequency_table.py +14 -0
- data_profiling/report/presentation/core/frequency_table_small.py +16 -0
- data_profiling/report/presentation/core/html.py +14 -0
- data_profiling/report/presentation/core/image.py +34 -0
- data_profiling/report/presentation/core/item_renderer.py +17 -0
- data_profiling/report/presentation/core/renderable.py +42 -0
- data_profiling/report/presentation/core/root.py +35 -0
- data_profiling/report/presentation/core/sample.py +20 -0
- data_profiling/report/presentation/core/scores.py +32 -0
- data_profiling/report/presentation/core/table.py +26 -0
- data_profiling/report/presentation/core/toggle_button.py +14 -0
- data_profiling/report/presentation/core/variable.py +40 -0
- data_profiling/report/presentation/core/variable_info.py +36 -0
- data_profiling/report/presentation/flavours/__init__.py +9 -0
- data_profiling/report/presentation/flavours/flavour_html.py +64 -0
- data_profiling/report/presentation/flavours/flavour_widget.py +61 -0
- data_profiling/report/presentation/flavours/flavours.py +43 -0
- data_profiling/report/presentation/flavours/html/__init__.py +47 -0
- data_profiling/report/presentation/flavours/html/alerts.py +10 -0
- data_profiling/report/presentation/flavours/html/collapse.py +7 -0
- data_profiling/report/presentation/flavours/html/container.py +58 -0
- data_profiling/report/presentation/flavours/html/correlation_table.py +13 -0
- data_profiling/report/presentation/flavours/html/dropdown.py +7 -0
- data_profiling/report/presentation/flavours/html/duplicate.py +24 -0
- data_profiling/report/presentation/flavours/html/frequency_table.py +20 -0
- data_profiling/report/presentation/flavours/html/frequency_table_small.py +15 -0
- data_profiling/report/presentation/flavours/html/html.py +6 -0
- data_profiling/report/presentation/flavours/html/image.py +7 -0
- data_profiling/report/presentation/flavours/html/root.py +14 -0
- data_profiling/report/presentation/flavours/html/sample.py +12 -0
- data_profiling/report/presentation/flavours/html/scores.py +11 -0
- data_profiling/report/presentation/flavours/html/table.py +7 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant_length.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_dirty_category.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_duplicates.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_empty.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_cardinality.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_correlation.html +4 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_imbalance.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_infinite.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_missing.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_near_duplicates.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_non_stationary.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_seasonal.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_skewed.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_truncated.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_type_date.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_uniform.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_unique.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_unsupported.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_zeros.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts.html +47 -0
- data_profiling/report/presentation/flavours/html/templates/collapse.html +11 -0
- data_profiling/report/presentation/flavours/html/templates/correlation_table.html +5 -0
- data_profiling/report/presentation/flavours/html/templates/diagram.html +11 -0
- data_profiling/report/presentation/flavours/html/templates/dropdown.html +16 -0
- data_profiling/report/presentation/flavours/html/templates/duplicate.html +5 -0
- data_profiling/report/presentation/flavours/html/templates/frequency_table.html +45 -0
- data_profiling/report/presentation/flavours/html/templates/frequency_table_small.html +34 -0
- data_profiling/report/presentation/flavours/html/templates/report.html +26 -0
- data_profiling/report/presentation/flavours/html/templates/sample.html +10 -0
- data_profiling/report/presentation/flavours/html/templates/scores.html +78 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/batch_grid.html +16 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/grid.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/list.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/named_list.html +8 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/overview_tabs.html +30 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/scores.html +3 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/sections.html +13 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/select.html +40 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/tabs.html +30 -0
- data_profiling/report/presentation/flavours/html/templates/table.html +38 -0
- data_profiling/report/presentation/flavours/html/templates/toggle_button.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/variable.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/variable_info.html +49 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.bundle.min.js +7 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.min.css +6 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/cosmo.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/flatly.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/script.js +52 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/simplex.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/style.css +253 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/united.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/footer.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/javascript.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/navigation.html +36 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/style.html +53 -0
- data_profiling/report/presentation/flavours/html/templates.py +76 -0
- data_profiling/report/presentation/flavours/html/toggle_button.py +7 -0
- data_profiling/report/presentation/flavours/html/variable.py +7 -0
- data_profiling/report/presentation/flavours/html/variable_info.py +7 -0
- data_profiling/report/presentation/flavours/widget/__init__.py +49 -0
- data_profiling/report/presentation/flavours/widget/alerts.py +45 -0
- data_profiling/report/presentation/flavours/widget/collapse.py +43 -0
- data_profiling/report/presentation/flavours/widget/container.py +121 -0
- data_profiling/report/presentation/flavours/widget/correlation_table.py +14 -0
- data_profiling/report/presentation/flavours/widget/dropdown.py +31 -0
- data_profiling/report/presentation/flavours/widget/duplicate.py +14 -0
- data_profiling/report/presentation/flavours/widget/frequency_table.py +57 -0
- data_profiling/report/presentation/flavours/widget/frequency_table_small.py +66 -0
- data_profiling/report/presentation/flavours/widget/html.py +11 -0
- data_profiling/report/presentation/flavours/widget/image.py +26 -0
- data_profiling/report/presentation/flavours/widget/notebook.py +81 -0
- data_profiling/report/presentation/flavours/widget/root.py +10 -0
- data_profiling/report/presentation/flavours/widget/sample.py +14 -0
- data_profiling/report/presentation/flavours/widget/table.py +30 -0
- data_profiling/report/presentation/flavours/widget/toggle_button.py +17 -0
- data_profiling/report/presentation/flavours/widget/variable.py +12 -0
- data_profiling/report/presentation/flavours/widget/variable_info.py +11 -0
- data_profiling/report/presentation/frequency_table_utils.py +141 -0
- data_profiling/report/structure/__init__.py +1 -0
- data_profiling/report/structure/correlations.py +123 -0
- data_profiling/report/structure/overview.py +376 -0
- data_profiling/report/structure/report.py +457 -0
- data_profiling/report/structure/variables/__init__.py +35 -0
- data_profiling/report/structure/variables/render_boolean.py +132 -0
- data_profiling/report/structure/variables/render_categorical.py +566 -0
- data_profiling/report/structure/variables/render_common.py +31 -0
- data_profiling/report/structure/variables/render_complex.py +102 -0
- data_profiling/report/structure/variables/render_count.py +172 -0
- data_profiling/report/structure/variables/render_date.py +143 -0
- data_profiling/report/structure/variables/render_file.py +70 -0
- data_profiling/report/structure/variables/render_generic.py +45 -0
- data_profiling/report/structure/variables/render_image.py +204 -0
- data_profiling/report/structure/variables/render_path.py +134 -0
- data_profiling/report/structure/variables/render_real.py +314 -0
- data_profiling/report/structure/variables/render_text.py +189 -0
- data_profiling/report/structure/variables/render_timeseries.py +371 -0
- data_profiling/report/structure/variables/render_url.py +132 -0
- data_profiling/report/utils.py +34 -0
- data_profiling/serialize_report.py +143 -0
- data_profiling/utils/__init__.py +1 -0
- data_profiling/utils/backend.py +9 -0
- data_profiling/utils/cache.py +59 -0
- data_profiling/utils/common.py +142 -0
- data_profiling/utils/compat.py +31 -0
- data_profiling/utils/dataframe.py +238 -0
- data_profiling/utils/logger.py +53 -0
- data_profiling/utils/notebook.py +8 -0
- data_profiling/utils/paths.py +45 -0
- data_profiling/utils/progress_bar.py +15 -0
- data_profiling/utils/styles.py +22 -0
- data_profiling/utils/versions.py +19 -0
- data_profiling/version.py +1 -0
- data_profiling/visualisation/__init__.py +1 -0
- data_profiling/visualisation/context.py +87 -0
- data_profiling/visualisation/missing.py +138 -0
- data_profiling/visualisation/plot.py +1158 -0
- data_profiling/visualisation/utils.py +113 -0
- fg_data_profiling-4.19.0.dist-info/METADATA +362 -0
- fg_data_profiling-4.19.0.dist-info/RECORD +238 -0
- fg_data_profiling-4.19.0.dist-info/WHEEL +6 -0
- fg_data_profiling-4.19.0.dist-info/entry_points.txt +3 -0
- fg_data_profiling-4.19.0.dist-info/licenses/LICENSE +21 -0
- fg_data_profiling-4.19.0.dist-info/top_level.txt +2 -0
- ydata_profiling/__init__.py +43 -0
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
from functools import partial
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Optional, Tuple, Union
|
|
4
|
+
|
|
5
|
+
import filetype
|
|
6
|
+
import imagehash
|
|
7
|
+
import pandas as pd
|
|
8
|
+
from PIL import ExifTags, Image
|
|
9
|
+
|
|
10
|
+
from data_profiling.config import Settings
|
|
11
|
+
from data_profiling.model.summary_algorithms import (
|
|
12
|
+
describe_image_1d,
|
|
13
|
+
named_aggregate_summary,
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def open_image(path: Path) -> Optional[Image.Image]:
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
Args:
|
|
21
|
+
path:
|
|
22
|
+
|
|
23
|
+
Returns:
|
|
24
|
+
|
|
25
|
+
"""
|
|
26
|
+
try:
|
|
27
|
+
return Image.open(path)
|
|
28
|
+
except (OSError, AttributeError):
|
|
29
|
+
return None
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def is_image_truncated(image: Image) -> bool:
|
|
33
|
+
"""Returns True if the path refers to a truncated image
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
image:
|
|
37
|
+
|
|
38
|
+
Returns:
|
|
39
|
+
True if the image is truncated
|
|
40
|
+
"""
|
|
41
|
+
try:
|
|
42
|
+
image.load()
|
|
43
|
+
except (OSError, AttributeError):
|
|
44
|
+
return True
|
|
45
|
+
else:
|
|
46
|
+
return False
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def get_image_shape(image: Image) -> Optional[Tuple[int, int]]:
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
Args:
|
|
53
|
+
image:
|
|
54
|
+
|
|
55
|
+
Returns:
|
|
56
|
+
|
|
57
|
+
"""
|
|
58
|
+
try:
|
|
59
|
+
return image.size
|
|
60
|
+
except (OSError, AttributeError):
|
|
61
|
+
return None
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def hash_image(image: Image) -> Optional[str]:
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
Args:
|
|
68
|
+
image:
|
|
69
|
+
|
|
70
|
+
Returns:
|
|
71
|
+
|
|
72
|
+
"""
|
|
73
|
+
try:
|
|
74
|
+
return str(imagehash.phash(image))
|
|
75
|
+
except (OSError, AttributeError):
|
|
76
|
+
return None
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def decode_byte_exif(exif_val: Union[str, bytes]) -> str:
|
|
80
|
+
"""Decode byte encodings
|
|
81
|
+
|
|
82
|
+
Args:
|
|
83
|
+
exif_val:
|
|
84
|
+
|
|
85
|
+
Returns:
|
|
86
|
+
|
|
87
|
+
"""
|
|
88
|
+
if isinstance(exif_val, str):
|
|
89
|
+
return exif_val
|
|
90
|
+
else:
|
|
91
|
+
return exif_val.decode()
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def extract_exif(image: Image) -> dict:
|
|
95
|
+
"""
|
|
96
|
+
|
|
97
|
+
Args:
|
|
98
|
+
image:
|
|
99
|
+
|
|
100
|
+
Returns:
|
|
101
|
+
|
|
102
|
+
"""
|
|
103
|
+
try:
|
|
104
|
+
exif_data = image._getexif()
|
|
105
|
+
if exif_data is not None:
|
|
106
|
+
exif = {
|
|
107
|
+
ExifTags.TAGS[k]: decode_byte_exif(v)
|
|
108
|
+
for k, v in exif_data.items()
|
|
109
|
+
if k in ExifTags.TAGS
|
|
110
|
+
}
|
|
111
|
+
else:
|
|
112
|
+
exif = {}
|
|
113
|
+
except (AttributeError, OSError):
|
|
114
|
+
# Not all file types (e.g. .gif) have exif information.
|
|
115
|
+
exif = {}
|
|
116
|
+
|
|
117
|
+
return exif
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def path_is_image(p: Path) -> bool:
|
|
121
|
+
guess = filetype.guess(str(p))
|
|
122
|
+
return guess is not None and guess.mime.startswith("image/")
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def count_duplicate_hashes(image_descriptions: dict) -> int:
|
|
126
|
+
"""
|
|
127
|
+
|
|
128
|
+
Args:
|
|
129
|
+
image_descriptions:
|
|
130
|
+
|
|
131
|
+
Returns:
|
|
132
|
+
|
|
133
|
+
"""
|
|
134
|
+
counts = pd.Series(
|
|
135
|
+
[x["hash"] for x in image_descriptions if "hash" in x]
|
|
136
|
+
).value_counts()
|
|
137
|
+
return counts.sum() - len(counts)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def extract_exif_series(image_exifs: list) -> dict:
|
|
141
|
+
"""
|
|
142
|
+
|
|
143
|
+
Args:
|
|
144
|
+
image_exifs:
|
|
145
|
+
|
|
146
|
+
Returns:
|
|
147
|
+
|
|
148
|
+
"""
|
|
149
|
+
exif_keys = []
|
|
150
|
+
exif_values: dict = {}
|
|
151
|
+
|
|
152
|
+
for image_exif in image_exifs:
|
|
153
|
+
# Extract key
|
|
154
|
+
exif_keys.extend(list(image_exif.keys()))
|
|
155
|
+
|
|
156
|
+
# Extract values per key
|
|
157
|
+
for exif_key, exif_val in image_exif.items():
|
|
158
|
+
if exif_key not in exif_values:
|
|
159
|
+
exif_values[exif_key] = []
|
|
160
|
+
|
|
161
|
+
exif_values[exif_key].append(exif_val)
|
|
162
|
+
|
|
163
|
+
series = {"exif_keys": pd.Series(exif_keys, dtype=object).value_counts().to_dict()}
|
|
164
|
+
|
|
165
|
+
for k, v in exif_values.items():
|
|
166
|
+
series[k] = pd.Series(v).value_counts()
|
|
167
|
+
|
|
168
|
+
return series
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def extract_image_information(
|
|
172
|
+
path: Path, exif: bool = False, hash: bool = False
|
|
173
|
+
) -> dict:
|
|
174
|
+
"""Extracts all image information per file, as opening files is slow
|
|
175
|
+
|
|
176
|
+
Args:
|
|
177
|
+
path: Path to the image
|
|
178
|
+
exif: extract exif information
|
|
179
|
+
hash: calculate hash (for duplicate detection)
|
|
180
|
+
|
|
181
|
+
Returns:
|
|
182
|
+
A dict containing image information
|
|
183
|
+
"""
|
|
184
|
+
information: dict = {}
|
|
185
|
+
image = open_image(path)
|
|
186
|
+
information["opened"] = image is not None
|
|
187
|
+
if image is not None:
|
|
188
|
+
information["truncated"] = is_image_truncated(image)
|
|
189
|
+
if not information["truncated"]:
|
|
190
|
+
information["size"] = image.size
|
|
191
|
+
if exif:
|
|
192
|
+
information["exif"] = extract_exif(image)
|
|
193
|
+
if hash:
|
|
194
|
+
information["hash"] = hash_image(image)
|
|
195
|
+
|
|
196
|
+
return information
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def image_summary(series: pd.Series, exif: bool = False, hash: bool = False) -> dict:
|
|
200
|
+
"""
|
|
201
|
+
|
|
202
|
+
Args:
|
|
203
|
+
series: series to summarize
|
|
204
|
+
exif: extract exif information
|
|
205
|
+
hash: calculate hash (for duplicate detection)
|
|
206
|
+
|
|
207
|
+
Returns:
|
|
208
|
+
|
|
209
|
+
"""
|
|
210
|
+
|
|
211
|
+
image_information = series.apply(
|
|
212
|
+
partial(extract_image_information, exif=exif, hash=hash)
|
|
213
|
+
)
|
|
214
|
+
summary = {
|
|
215
|
+
"n_truncated": sum(
|
|
216
|
+
1 for x in image_information if "truncated" in x and x["truncated"]
|
|
217
|
+
),
|
|
218
|
+
"image_dimensions": pd.Series(
|
|
219
|
+
[x["size"] for x in image_information if "size" in x],
|
|
220
|
+
name="image_dimensions",
|
|
221
|
+
),
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
image_widths = summary["image_dimensions"].map(lambda x: x[0])
|
|
225
|
+
summary.update(named_aggregate_summary(image_widths, "width"))
|
|
226
|
+
image_heights = summary["image_dimensions"].map(lambda x: x[1])
|
|
227
|
+
summary.update(named_aggregate_summary(image_heights, "height"))
|
|
228
|
+
image_areas = image_widths * image_heights
|
|
229
|
+
summary.update(named_aggregate_summary(image_areas, "area"))
|
|
230
|
+
|
|
231
|
+
if hash:
|
|
232
|
+
summary["n_duplicate_hash"] = count_duplicate_hashes(image_information)
|
|
233
|
+
|
|
234
|
+
if exif:
|
|
235
|
+
exif_series = extract_exif_series(
|
|
236
|
+
[x["exif"] for x in image_information if "exif" in x]
|
|
237
|
+
)
|
|
238
|
+
summary["exif_keys_counts"] = exif_series["exif_keys"]
|
|
239
|
+
summary["exif_data"] = exif_series
|
|
240
|
+
|
|
241
|
+
return summary
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
@describe_image_1d.register
|
|
245
|
+
def pandas_describe_image_1d(
|
|
246
|
+
config: Settings, series: pd.Series, summary: dict
|
|
247
|
+
) -> Tuple[Settings, pd.Series, dict]:
|
|
248
|
+
if series.hasnans:
|
|
249
|
+
raise ValueError("May not contain NaNs")
|
|
250
|
+
if not hasattr(series, "str"):
|
|
251
|
+
raise ValueError("series should have .str accessor")
|
|
252
|
+
|
|
253
|
+
summary.update(image_summary(series, config.vars.image.exif))
|
|
254
|
+
|
|
255
|
+
return config, series, summary
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
from typing import Any, Dict, Tuple
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
from data_profiling.utils.compat import pandas_version_info
|
|
7
|
+
|
|
8
|
+
if pandas_version_info() >= (1, 5):
|
|
9
|
+
from pandas.core.arrays.integer import IntegerDtype
|
|
10
|
+
else:
|
|
11
|
+
from pandas.core.arrays.integer import _IntegerDtype as IntegerDtype
|
|
12
|
+
|
|
13
|
+
from data_profiling.config import Settings
|
|
14
|
+
from data_profiling.model.summary_algorithms import (
|
|
15
|
+
chi_square,
|
|
16
|
+
describe_numeric_1d,
|
|
17
|
+
histogram_compute,
|
|
18
|
+
series_handle_nulls,
|
|
19
|
+
series_hashable,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def mad(arr: np.ndarray) -> np.ndarray:
|
|
24
|
+
"""Median Absolute Deviation: a "Robust" version of standard deviation.
|
|
25
|
+
Indices variability of the sample.
|
|
26
|
+
https://en.wikipedia.org/wiki/Median_absolute_deviation
|
|
27
|
+
"""
|
|
28
|
+
return np.median(np.abs(arr - np.median(arr)))
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def numeric_stats_pandas(series: pd.Series) -> Dict[str, Any]:
|
|
32
|
+
return {
|
|
33
|
+
"mean": series.mean(),
|
|
34
|
+
"std": series.std(),
|
|
35
|
+
"variance": series.var(),
|
|
36
|
+
"min": series.min(),
|
|
37
|
+
"max": series.max(),
|
|
38
|
+
# Unbiased kurtosis obtained using Fisher's definition (kurtosis of normal == 0.0). Normalized by N-1.
|
|
39
|
+
"kurtosis": series.kurt(),
|
|
40
|
+
# Unbiased skew normalized by N-1
|
|
41
|
+
"skewness": series.skew(),
|
|
42
|
+
"sum": series.sum(),
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def numeric_stats_numpy(
|
|
47
|
+
present_values: np.ndarray, series: pd.Series, series_description: Dict[str, Any]
|
|
48
|
+
) -> Dict[str, Any]:
|
|
49
|
+
vc = series_description["value_counts_without_nan"]
|
|
50
|
+
index_values = vc.index.values
|
|
51
|
+
|
|
52
|
+
# FIXME: can be performance optimized by using weights in std, var, kurt and skew...
|
|
53
|
+
if len(index_values):
|
|
54
|
+
return {
|
|
55
|
+
"mean": np.average(index_values, weights=vc.values),
|
|
56
|
+
"std": np.std(present_values, ddof=1),
|
|
57
|
+
"variance": np.var(present_values, ddof=1),
|
|
58
|
+
"min": np.min(index_values),
|
|
59
|
+
"max": np.max(index_values),
|
|
60
|
+
# Unbiased kurtosis obtained using Fisher's definition (kurtosis of normal == 0.0). Normalized by N-1.
|
|
61
|
+
"kurtosis": series.kurt(),
|
|
62
|
+
# Unbiased skew normalized by N-1
|
|
63
|
+
"skewness": series.skew(),
|
|
64
|
+
"sum": np.dot(index_values, vc.values),
|
|
65
|
+
}
|
|
66
|
+
else: # Empty numerical series
|
|
67
|
+
return {
|
|
68
|
+
"mean": np.nan,
|
|
69
|
+
"std": 0.0,
|
|
70
|
+
"variance": 0.0,
|
|
71
|
+
"min": np.nan,
|
|
72
|
+
"max": np.nan,
|
|
73
|
+
"kurtosis": 0.0,
|
|
74
|
+
"skewness": 0.0,
|
|
75
|
+
"sum": 0,
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
@describe_numeric_1d.register
|
|
80
|
+
@series_hashable
|
|
81
|
+
@series_handle_nulls
|
|
82
|
+
def pandas_describe_numeric_1d(
|
|
83
|
+
config: Settings, series: pd.Series, summary: dict
|
|
84
|
+
) -> Tuple[Settings, pd.Series, dict]:
|
|
85
|
+
"""Describe a numeric series.
|
|
86
|
+
|
|
87
|
+
Args:
|
|
88
|
+
config: report Settings object
|
|
89
|
+
series: The Series to describe.
|
|
90
|
+
summary: The dict containing the series description so far.
|
|
91
|
+
|
|
92
|
+
Returns:
|
|
93
|
+
A dict containing calculated series description values.
|
|
94
|
+
"""
|
|
95
|
+
|
|
96
|
+
chi_squared_threshold = config.vars.num.chi_squared_threshold
|
|
97
|
+
quantiles = config.vars.num.quantiles
|
|
98
|
+
|
|
99
|
+
value_counts = summary["value_counts_without_nan"]
|
|
100
|
+
|
|
101
|
+
negative_index = value_counts.index < 0
|
|
102
|
+
summary["n_negative"] = value_counts.loc[negative_index].sum()
|
|
103
|
+
summary["p_negative"] = summary["n_negative"] / summary["n"]
|
|
104
|
+
|
|
105
|
+
infinity_values = [np.inf, -np.inf]
|
|
106
|
+
infinity_index = value_counts.index.isin(infinity_values)
|
|
107
|
+
summary["n_infinite"] = value_counts.loc[infinity_index].sum()
|
|
108
|
+
|
|
109
|
+
summary["n_zeros"] = 0
|
|
110
|
+
if 0 in value_counts.index:
|
|
111
|
+
summary["n_zeros"] = value_counts.loc[0]
|
|
112
|
+
|
|
113
|
+
stats = summary
|
|
114
|
+
|
|
115
|
+
if isinstance(series.dtype, IntegerDtype):
|
|
116
|
+
stats.update(numeric_stats_pandas(series))
|
|
117
|
+
present_values = series.astype(str(series.dtype).lower())
|
|
118
|
+
finite_values = present_values
|
|
119
|
+
else:
|
|
120
|
+
present_values = series.values
|
|
121
|
+
finite_values = present_values[np.isfinite(present_values)]
|
|
122
|
+
stats.update(numeric_stats_numpy(present_values, series, summary))
|
|
123
|
+
|
|
124
|
+
stats.update(
|
|
125
|
+
{
|
|
126
|
+
"mad": mad(present_values),
|
|
127
|
+
}
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
if chi_squared_threshold > 0.0:
|
|
131
|
+
stats["chi_squared"] = chi_square(finite_values)
|
|
132
|
+
|
|
133
|
+
stats["range"] = stats["max"] - stats["min"]
|
|
134
|
+
stats.update(
|
|
135
|
+
{
|
|
136
|
+
f"{percentile:.0%}": value
|
|
137
|
+
for percentile, value in series.quantile(quantiles).to_dict().items()
|
|
138
|
+
}
|
|
139
|
+
)
|
|
140
|
+
stats["iqr"] = stats["75%"] - stats["25%"]
|
|
141
|
+
stats["cv"] = stats["std"] / stats["mean"] if stats["mean"] else np.nan
|
|
142
|
+
stats["p_zeros"] = stats["n_zeros"] / summary["n"]
|
|
143
|
+
stats["p_infinite"] = summary["n_infinite"] / summary["n"]
|
|
144
|
+
|
|
145
|
+
stats["monotonic_increase"] = series.is_monotonic_increasing
|
|
146
|
+
stats["monotonic_decrease"] = series.is_monotonic_decreasing
|
|
147
|
+
|
|
148
|
+
stats["monotonic_increase_strict"] = (
|
|
149
|
+
stats["monotonic_increase"] and series.is_unique
|
|
150
|
+
)
|
|
151
|
+
stats["monotonic_decrease_strict"] = (
|
|
152
|
+
stats["monotonic_decrease"] and series.is_unique
|
|
153
|
+
)
|
|
154
|
+
if summary["monotonic_increase_strict"]:
|
|
155
|
+
stats["monotonic"] = 2
|
|
156
|
+
elif summary["monotonic_decrease_strict"]:
|
|
157
|
+
stats["monotonic"] = -2
|
|
158
|
+
elif summary["monotonic_increase"]:
|
|
159
|
+
stats["monotonic"] = 1
|
|
160
|
+
elif summary["monotonic_decrease"]:
|
|
161
|
+
stats["monotonic"] = -1
|
|
162
|
+
else:
|
|
163
|
+
stats["monotonic"] = 0
|
|
164
|
+
|
|
165
|
+
if len(value_counts[~infinity_index].index.values) > 0:
|
|
166
|
+
stats.update(
|
|
167
|
+
histogram_compute(
|
|
168
|
+
config,
|
|
169
|
+
value_counts[~infinity_index].index.values,
|
|
170
|
+
summary["n_distinct"],
|
|
171
|
+
weights=value_counts[~infinity_index].values,
|
|
172
|
+
)
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
return config, series, stats
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from typing import Tuple
|
|
3
|
+
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
from data_profiling.config import Settings
|
|
7
|
+
from data_profiling.model.summary_algorithms import describe_path_1d
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def path_summary(series: pd.Series) -> dict:
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
Args:
|
|
14
|
+
series: series to summarize
|
|
15
|
+
|
|
16
|
+
Returns:
|
|
17
|
+
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
# TODO: optimize using value counts
|
|
21
|
+
summary = {
|
|
22
|
+
"common_prefix": os.path.commonprefix(series.values.tolist())
|
|
23
|
+
or "No common prefix",
|
|
24
|
+
"stem_counts": series.map(lambda x: os.path.splitext(x)[0]).value_counts(),
|
|
25
|
+
"suffix_counts": series.map(lambda x: os.path.splitext(x)[1]).value_counts(),
|
|
26
|
+
"name_counts": series.map(lambda x: os.path.basename(x)).value_counts(),
|
|
27
|
+
"parent_counts": series.map(lambda x: os.path.dirname(x)).value_counts(),
|
|
28
|
+
"anchor_counts": series.map(lambda x: os.path.splitdrive(x)[0]).value_counts(),
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
summary["n_stem_unique"] = len(summary["stem_counts"])
|
|
32
|
+
summary["n_suffix_unique"] = len(summary["suffix_counts"])
|
|
33
|
+
summary["n_name_unique"] = len(summary["name_counts"])
|
|
34
|
+
summary["n_parent_unique"] = len(summary["parent_counts"])
|
|
35
|
+
summary["n_anchor_unique"] = len(summary["anchor_counts"])
|
|
36
|
+
|
|
37
|
+
return summary
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@describe_path_1d.register
|
|
41
|
+
def pandas_describe_path_1d(
|
|
42
|
+
config: Settings, series: pd.Series, summary: dict
|
|
43
|
+
) -> Tuple[Settings, pd.Series, dict]:
|
|
44
|
+
"""Describe a path series.
|
|
45
|
+
|
|
46
|
+
Args:
|
|
47
|
+
config: report Settings object
|
|
48
|
+
series: The Series to describe.
|
|
49
|
+
summary: The dict containing the series description so far.
|
|
50
|
+
|
|
51
|
+
Returns:
|
|
52
|
+
A dict containing calculated series description values.
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
# Make sure we deal with strings (Issue #100)
|
|
56
|
+
if series.hasnans:
|
|
57
|
+
raise ValueError("May not contain NaNs")
|
|
58
|
+
if not hasattr(series, "str"):
|
|
59
|
+
raise ValueError("series should have .str accessor")
|
|
60
|
+
|
|
61
|
+
summary.update(path_summary(series))
|
|
62
|
+
|
|
63
|
+
return config, series, summary
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
from typing import Tuple
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from data_profiling.config import Settings
|
|
6
|
+
from data_profiling.model.summary_algorithms import describe_supported, series_hashable
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@describe_supported.register
|
|
10
|
+
@series_hashable
|
|
11
|
+
def pandas_describe_supported(
|
|
12
|
+
config: Settings, series: pd.Series, series_description: dict
|
|
13
|
+
) -> Tuple[Settings, pd.Series, dict]:
|
|
14
|
+
"""Describe a supported series.
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
config: report Settings object
|
|
18
|
+
series: The Series to describe.
|
|
19
|
+
series_description: The dict containing the series description so far.
|
|
20
|
+
|
|
21
|
+
Returns:
|
|
22
|
+
A dict containing calculated series description values.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
# number of non-NaN observations in the Series
|
|
26
|
+
count = series_description["count"]
|
|
27
|
+
|
|
28
|
+
value_counts = series_description["value_counts_without_nan"]
|
|
29
|
+
distinct_count = len(value_counts)
|
|
30
|
+
unique_count = value_counts.where(value_counts == 1).count()
|
|
31
|
+
|
|
32
|
+
stats = {
|
|
33
|
+
"n_distinct": distinct_count,
|
|
34
|
+
"p_distinct": distinct_count / count if count > 0 else 0,
|
|
35
|
+
"is_unique": unique_count == count and count > 0,
|
|
36
|
+
"n_unique": unique_count,
|
|
37
|
+
"p_unique": unique_count / count if count > 0 else 0,
|
|
38
|
+
}
|
|
39
|
+
stats.update(series_description)
|
|
40
|
+
|
|
41
|
+
return config, series, stats
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
from typing import Tuple
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from data_profiling.config import Settings
|
|
6
|
+
from data_profiling.model.pandas.describe_categorical_pandas import (
|
|
7
|
+
length_summary_vc,
|
|
8
|
+
unicode_summary_vc,
|
|
9
|
+
word_summary_vc,
|
|
10
|
+
)
|
|
11
|
+
from data_profiling.model.summary_algorithms import (
|
|
12
|
+
histogram_compute,
|
|
13
|
+
series_handle_nulls,
|
|
14
|
+
series_hashable,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@series_hashable
|
|
19
|
+
@series_handle_nulls
|
|
20
|
+
def pandas_describe_text_1d(
|
|
21
|
+
config: Settings,
|
|
22
|
+
series: pd.Series,
|
|
23
|
+
summary: dict,
|
|
24
|
+
) -> Tuple[Settings, pd.Series, dict]:
|
|
25
|
+
"""Describe string series.
|
|
26
|
+
|
|
27
|
+
Args:
|
|
28
|
+
config: report Settings object
|
|
29
|
+
series: The Series to describe.
|
|
30
|
+
summary: The dict containing the series description so far.
|
|
31
|
+
|
|
32
|
+
Returns:
|
|
33
|
+
A dict containing calculated series description values.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
series = series.astype(str)
|
|
37
|
+
|
|
38
|
+
# Only run if at least 1 non-missing value
|
|
39
|
+
value_counts = summary["value_counts_without_nan"]
|
|
40
|
+
value_counts.index = value_counts.index.astype(str)
|
|
41
|
+
|
|
42
|
+
summary.update({"first_rows": series.head(5)})
|
|
43
|
+
|
|
44
|
+
if config.vars.text.length:
|
|
45
|
+
summary.update(length_summary_vc(value_counts))
|
|
46
|
+
summary.update(
|
|
47
|
+
histogram_compute(
|
|
48
|
+
config,
|
|
49
|
+
summary["length_histogram"].index.values,
|
|
50
|
+
len(summary["length_histogram"]),
|
|
51
|
+
name="histogram_length",
|
|
52
|
+
weights=summary["length_histogram"].values,
|
|
53
|
+
)
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
if config.vars.text.characters:
|
|
57
|
+
summary.update(unicode_summary_vc(value_counts))
|
|
58
|
+
|
|
59
|
+
if config.vars.text.words:
|
|
60
|
+
summary.update(word_summary_vc(value_counts, config.vars.cat.stop_words))
|
|
61
|
+
|
|
62
|
+
return config, series, summary
|