fg-data-profiling 4.19.0__py2.py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (238) hide show
  1. data_profiling/__init__.py +34 -0
  2. data_profiling/compare_reports.py +359 -0
  3. data_profiling/config.py +496 -0
  4. data_profiling/config_default.yaml +223 -0
  5. data_profiling/config_minimal.yaml +222 -0
  6. data_profiling/controller/__init__.py +1 -0
  7. data_profiling/controller/console.py +125 -0
  8. data_profiling/controller/pandas_decorator.py +21 -0
  9. data_profiling/expectations_report.py +117 -0
  10. data_profiling/model/__init__.py +4 -0
  11. data_profiling/model/alerts.py +780 -0
  12. data_profiling/model/correlations.py +163 -0
  13. data_profiling/model/dataframe.py +35 -0
  14. data_profiling/model/describe.py +210 -0
  15. data_profiling/model/description.py +108 -0
  16. data_profiling/model/duplicates.py +14 -0
  17. data_profiling/model/expectation_algorithms.py +112 -0
  18. data_profiling/model/handler.py +81 -0
  19. data_profiling/model/missing.py +146 -0
  20. data_profiling/model/pairwise.py +33 -0
  21. data_profiling/model/pandas/__init__.py +55 -0
  22. data_profiling/model/pandas/correlations_pandas.py +207 -0
  23. data_profiling/model/pandas/dataframe_pandas.py +26 -0
  24. data_profiling/model/pandas/describe_boolean_pandas.py +43 -0
  25. data_profiling/model/pandas/describe_categorical_pandas.py +274 -0
  26. data_profiling/model/pandas/describe_counts_pandas.py +63 -0
  27. data_profiling/model/pandas/describe_date_pandas.py +77 -0
  28. data_profiling/model/pandas/describe_file_pandas.py +56 -0
  29. data_profiling/model/pandas/describe_generic_pandas.py +36 -0
  30. data_profiling/model/pandas/describe_image_pandas.py +255 -0
  31. data_profiling/model/pandas/describe_numeric_pandas.py +175 -0
  32. data_profiling/model/pandas/describe_path_pandas.py +63 -0
  33. data_profiling/model/pandas/describe_supported_pandas.py +41 -0
  34. data_profiling/model/pandas/describe_text_pandas.py +62 -0
  35. data_profiling/model/pandas/describe_timeseries_pandas.py +222 -0
  36. data_profiling/model/pandas/describe_url_pandas.py +57 -0
  37. data_profiling/model/pandas/discretize_pandas.py +81 -0
  38. data_profiling/model/pandas/duplicates_pandas.py +56 -0
  39. data_profiling/model/pandas/imbalance_pandas.py +35 -0
  40. data_profiling/model/pandas/missing_pandas.py +42 -0
  41. data_profiling/model/pandas/sample_pandas.py +38 -0
  42. data_profiling/model/pandas/summary_pandas.py +101 -0
  43. data_profiling/model/pandas/table_pandas.py +56 -0
  44. data_profiling/model/pandas/timeseries_index_pandas.py +33 -0
  45. data_profiling/model/pandas/utils_pandas.py +27 -0
  46. data_profiling/model/sample.py +37 -0
  47. data_profiling/model/spark/__init__.py +48 -0
  48. data_profiling/model/spark/correlations_spark.py +152 -0
  49. data_profiling/model/spark/dataframe_spark.py +34 -0
  50. data_profiling/model/spark/describe_boolean_spark.py +27 -0
  51. data_profiling/model/spark/describe_categorical_spark.py +28 -0
  52. data_profiling/model/spark/describe_counts_spark.py +105 -0
  53. data_profiling/model/spark/describe_date_spark.py +51 -0
  54. data_profiling/model/spark/describe_generic_spark.py +30 -0
  55. data_profiling/model/spark/describe_numeric_spark.py +155 -0
  56. data_profiling/model/spark/describe_supported_spark.py +33 -0
  57. data_profiling/model/spark/describe_text_spark.py +25 -0
  58. data_profiling/model/spark/duplicates_spark.py +54 -0
  59. data_profiling/model/spark/missing_spark.py +96 -0
  60. data_profiling/model/spark/sample_spark.py +43 -0
  61. data_profiling/model/spark/summary_spark.py +95 -0
  62. data_profiling/model/spark/table_spark.py +58 -0
  63. data_profiling/model/spark/timeseries_index_spark.py +12 -0
  64. data_profiling/model/summarizer.py +207 -0
  65. data_profiling/model/summary.py +66 -0
  66. data_profiling/model/summary_algorithms.py +276 -0
  67. data_profiling/model/table.py +10 -0
  68. data_profiling/model/timeseries_index.py +16 -0
  69. data_profiling/model/typeset.py +365 -0
  70. data_profiling/model/typeset_relations.py +143 -0
  71. data_profiling/profile_report.py +573 -0
  72. data_profiling/report/__init__.py +4 -0
  73. data_profiling/report/formatters.py +346 -0
  74. data_profiling/report/presentation/__init__.py +1 -0
  75. data_profiling/report/presentation/core/__init__.py +39 -0
  76. data_profiling/report/presentation/core/alerts.py +18 -0
  77. data_profiling/report/presentation/core/collapse.py +24 -0
  78. data_profiling/report/presentation/core/container.py +50 -0
  79. data_profiling/report/presentation/core/correlation_table.py +21 -0
  80. data_profiling/report/presentation/core/dropdown.py +44 -0
  81. data_profiling/report/presentation/core/duplicate.py +16 -0
  82. data_profiling/report/presentation/core/frequency_table.py +14 -0
  83. data_profiling/report/presentation/core/frequency_table_small.py +16 -0
  84. data_profiling/report/presentation/core/html.py +14 -0
  85. data_profiling/report/presentation/core/image.py +34 -0
  86. data_profiling/report/presentation/core/item_renderer.py +17 -0
  87. data_profiling/report/presentation/core/renderable.py +42 -0
  88. data_profiling/report/presentation/core/root.py +35 -0
  89. data_profiling/report/presentation/core/sample.py +20 -0
  90. data_profiling/report/presentation/core/scores.py +32 -0
  91. data_profiling/report/presentation/core/table.py +26 -0
  92. data_profiling/report/presentation/core/toggle_button.py +14 -0
  93. data_profiling/report/presentation/core/variable.py +40 -0
  94. data_profiling/report/presentation/core/variable_info.py +36 -0
  95. data_profiling/report/presentation/flavours/__init__.py +9 -0
  96. data_profiling/report/presentation/flavours/flavour_html.py +64 -0
  97. data_profiling/report/presentation/flavours/flavour_widget.py +61 -0
  98. data_profiling/report/presentation/flavours/flavours.py +43 -0
  99. data_profiling/report/presentation/flavours/html/__init__.py +47 -0
  100. data_profiling/report/presentation/flavours/html/alerts.py +10 -0
  101. data_profiling/report/presentation/flavours/html/collapse.py +7 -0
  102. data_profiling/report/presentation/flavours/html/container.py +58 -0
  103. data_profiling/report/presentation/flavours/html/correlation_table.py +13 -0
  104. data_profiling/report/presentation/flavours/html/dropdown.py +7 -0
  105. data_profiling/report/presentation/flavours/html/duplicate.py +24 -0
  106. data_profiling/report/presentation/flavours/html/frequency_table.py +20 -0
  107. data_profiling/report/presentation/flavours/html/frequency_table_small.py +15 -0
  108. data_profiling/report/presentation/flavours/html/html.py +6 -0
  109. data_profiling/report/presentation/flavours/html/image.py +7 -0
  110. data_profiling/report/presentation/flavours/html/root.py +14 -0
  111. data_profiling/report/presentation/flavours/html/sample.py +12 -0
  112. data_profiling/report/presentation/flavours/html/scores.py +11 -0
  113. data_profiling/report/presentation/flavours/html/table.py +7 -0
  114. data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant.html +1 -0
  115. data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant_length.html +1 -0
  116. data_profiling/report/presentation/flavours/html/templates/alerts/alert_dirty_category.html +1 -0
  117. data_profiling/report/presentation/flavours/html/templates/alerts/alert_duplicates.html +1 -0
  118. data_profiling/report/presentation/flavours/html/templates/alerts/alert_empty.html +1 -0
  119. data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_cardinality.html +1 -0
  120. data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_correlation.html +4 -0
  121. data_profiling/report/presentation/flavours/html/templates/alerts/alert_imbalance.html +1 -0
  122. data_profiling/report/presentation/flavours/html/templates/alerts/alert_infinite.html +1 -0
  123. data_profiling/report/presentation/flavours/html/templates/alerts/alert_missing.html +1 -0
  124. data_profiling/report/presentation/flavours/html/templates/alerts/alert_near_duplicates.html +1 -0
  125. data_profiling/report/presentation/flavours/html/templates/alerts/alert_non_stationary.html +1 -0
  126. data_profiling/report/presentation/flavours/html/templates/alerts/alert_seasonal.html +1 -0
  127. data_profiling/report/presentation/flavours/html/templates/alerts/alert_skewed.html +1 -0
  128. data_profiling/report/presentation/flavours/html/templates/alerts/alert_truncated.html +1 -0
  129. data_profiling/report/presentation/flavours/html/templates/alerts/alert_type_date.html +1 -0
  130. data_profiling/report/presentation/flavours/html/templates/alerts/alert_uniform.html +1 -0
  131. data_profiling/report/presentation/flavours/html/templates/alerts/alert_unique.html +1 -0
  132. data_profiling/report/presentation/flavours/html/templates/alerts/alert_unsupported.html +1 -0
  133. data_profiling/report/presentation/flavours/html/templates/alerts/alert_zeros.html +1 -0
  134. data_profiling/report/presentation/flavours/html/templates/alerts.html +47 -0
  135. data_profiling/report/presentation/flavours/html/templates/collapse.html +11 -0
  136. data_profiling/report/presentation/flavours/html/templates/correlation_table.html +5 -0
  137. data_profiling/report/presentation/flavours/html/templates/diagram.html +11 -0
  138. data_profiling/report/presentation/flavours/html/templates/dropdown.html +16 -0
  139. data_profiling/report/presentation/flavours/html/templates/duplicate.html +5 -0
  140. data_profiling/report/presentation/flavours/html/templates/frequency_table.html +45 -0
  141. data_profiling/report/presentation/flavours/html/templates/frequency_table_small.html +34 -0
  142. data_profiling/report/presentation/flavours/html/templates/report.html +26 -0
  143. data_profiling/report/presentation/flavours/html/templates/sample.html +10 -0
  144. data_profiling/report/presentation/flavours/html/templates/scores.html +78 -0
  145. data_profiling/report/presentation/flavours/html/templates/sequence/batch_grid.html +16 -0
  146. data_profiling/report/presentation/flavours/html/templates/sequence/grid.html +18 -0
  147. data_profiling/report/presentation/flavours/html/templates/sequence/list.html +7 -0
  148. data_profiling/report/presentation/flavours/html/templates/sequence/named_list.html +8 -0
  149. data_profiling/report/presentation/flavours/html/templates/sequence/overview_tabs.html +30 -0
  150. data_profiling/report/presentation/flavours/html/templates/sequence/scores.html +3 -0
  151. data_profiling/report/presentation/flavours/html/templates/sequence/sections.html +13 -0
  152. data_profiling/report/presentation/flavours/html/templates/sequence/select.html +40 -0
  153. data_profiling/report/presentation/flavours/html/templates/sequence/tabs.html +30 -0
  154. data_profiling/report/presentation/flavours/html/templates/table.html +38 -0
  155. data_profiling/report/presentation/flavours/html/templates/toggle_button.html +18 -0
  156. data_profiling/report/presentation/flavours/html/templates/variable.html +7 -0
  157. data_profiling/report/presentation/flavours/html/templates/variable_info.html +49 -0
  158. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.bundle.min.js +7 -0
  159. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.min.css +6 -0
  160. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/cosmo.bootstrap.min.css +12 -0
  161. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/flatly.bootstrap.min.css +12 -0
  162. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/script.js +52 -0
  163. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/simplex.bootstrap.min.css +12 -0
  164. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/style.css +253 -0
  165. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/united.bootstrap.min.css +12 -0
  166. data_profiling/report/presentation/flavours/html/templates/wrapper/footer.html +7 -0
  167. data_profiling/report/presentation/flavours/html/templates/wrapper/javascript.html +18 -0
  168. data_profiling/report/presentation/flavours/html/templates/wrapper/navigation.html +36 -0
  169. data_profiling/report/presentation/flavours/html/templates/wrapper/style.html +53 -0
  170. data_profiling/report/presentation/flavours/html/templates.py +76 -0
  171. data_profiling/report/presentation/flavours/html/toggle_button.py +7 -0
  172. data_profiling/report/presentation/flavours/html/variable.py +7 -0
  173. data_profiling/report/presentation/flavours/html/variable_info.py +7 -0
  174. data_profiling/report/presentation/flavours/widget/__init__.py +49 -0
  175. data_profiling/report/presentation/flavours/widget/alerts.py +45 -0
  176. data_profiling/report/presentation/flavours/widget/collapse.py +43 -0
  177. data_profiling/report/presentation/flavours/widget/container.py +121 -0
  178. data_profiling/report/presentation/flavours/widget/correlation_table.py +14 -0
  179. data_profiling/report/presentation/flavours/widget/dropdown.py +31 -0
  180. data_profiling/report/presentation/flavours/widget/duplicate.py +14 -0
  181. data_profiling/report/presentation/flavours/widget/frequency_table.py +57 -0
  182. data_profiling/report/presentation/flavours/widget/frequency_table_small.py +66 -0
  183. data_profiling/report/presentation/flavours/widget/html.py +11 -0
  184. data_profiling/report/presentation/flavours/widget/image.py +26 -0
  185. data_profiling/report/presentation/flavours/widget/notebook.py +81 -0
  186. data_profiling/report/presentation/flavours/widget/root.py +10 -0
  187. data_profiling/report/presentation/flavours/widget/sample.py +14 -0
  188. data_profiling/report/presentation/flavours/widget/table.py +30 -0
  189. data_profiling/report/presentation/flavours/widget/toggle_button.py +17 -0
  190. data_profiling/report/presentation/flavours/widget/variable.py +12 -0
  191. data_profiling/report/presentation/flavours/widget/variable_info.py +11 -0
  192. data_profiling/report/presentation/frequency_table_utils.py +141 -0
  193. data_profiling/report/structure/__init__.py +1 -0
  194. data_profiling/report/structure/correlations.py +123 -0
  195. data_profiling/report/structure/overview.py +376 -0
  196. data_profiling/report/structure/report.py +457 -0
  197. data_profiling/report/structure/variables/__init__.py +35 -0
  198. data_profiling/report/structure/variables/render_boolean.py +132 -0
  199. data_profiling/report/structure/variables/render_categorical.py +566 -0
  200. data_profiling/report/structure/variables/render_common.py +31 -0
  201. data_profiling/report/structure/variables/render_complex.py +102 -0
  202. data_profiling/report/structure/variables/render_count.py +172 -0
  203. data_profiling/report/structure/variables/render_date.py +143 -0
  204. data_profiling/report/structure/variables/render_file.py +70 -0
  205. data_profiling/report/structure/variables/render_generic.py +45 -0
  206. data_profiling/report/structure/variables/render_image.py +204 -0
  207. data_profiling/report/structure/variables/render_path.py +134 -0
  208. data_profiling/report/structure/variables/render_real.py +314 -0
  209. data_profiling/report/structure/variables/render_text.py +189 -0
  210. data_profiling/report/structure/variables/render_timeseries.py +371 -0
  211. data_profiling/report/structure/variables/render_url.py +132 -0
  212. data_profiling/report/utils.py +34 -0
  213. data_profiling/serialize_report.py +143 -0
  214. data_profiling/utils/__init__.py +1 -0
  215. data_profiling/utils/backend.py +9 -0
  216. data_profiling/utils/cache.py +59 -0
  217. data_profiling/utils/common.py +142 -0
  218. data_profiling/utils/compat.py +31 -0
  219. data_profiling/utils/dataframe.py +238 -0
  220. data_profiling/utils/logger.py +53 -0
  221. data_profiling/utils/notebook.py +8 -0
  222. data_profiling/utils/paths.py +45 -0
  223. data_profiling/utils/progress_bar.py +15 -0
  224. data_profiling/utils/styles.py +22 -0
  225. data_profiling/utils/versions.py +19 -0
  226. data_profiling/version.py +1 -0
  227. data_profiling/visualisation/__init__.py +1 -0
  228. data_profiling/visualisation/context.py +87 -0
  229. data_profiling/visualisation/missing.py +138 -0
  230. data_profiling/visualisation/plot.py +1158 -0
  231. data_profiling/visualisation/utils.py +113 -0
  232. fg_data_profiling-4.19.0.dist-info/METADATA +362 -0
  233. fg_data_profiling-4.19.0.dist-info/RECORD +238 -0
  234. fg_data_profiling-4.19.0.dist-info/WHEEL +6 -0
  235. fg_data_profiling-4.19.0.dist-info/entry_points.txt +3 -0
  236. fg_data_profiling-4.19.0.dist-info/licenses/LICENSE +21 -0
  237. fg_data_profiling-4.19.0.dist-info/top_level.txt +2 -0
  238. ydata_profiling/__init__.py +43 -0
@@ -0,0 +1,255 @@
1
+ from functools import partial
2
+ from pathlib import Path
3
+ from typing import Optional, Tuple, Union
4
+
5
+ import filetype
6
+ import imagehash
7
+ import pandas as pd
8
+ from PIL import ExifTags, Image
9
+
10
+ from data_profiling.config import Settings
11
+ from data_profiling.model.summary_algorithms import (
12
+ describe_image_1d,
13
+ named_aggregate_summary,
14
+ )
15
+
16
+
17
+ def open_image(path: Path) -> Optional[Image.Image]:
18
+ """
19
+
20
+ Args:
21
+ path:
22
+
23
+ Returns:
24
+
25
+ """
26
+ try:
27
+ return Image.open(path)
28
+ except (OSError, AttributeError):
29
+ return None
30
+
31
+
32
+ def is_image_truncated(image: Image) -> bool:
33
+ """Returns True if the path refers to a truncated image
34
+
35
+ Args:
36
+ image:
37
+
38
+ Returns:
39
+ True if the image is truncated
40
+ """
41
+ try:
42
+ image.load()
43
+ except (OSError, AttributeError):
44
+ return True
45
+ else:
46
+ return False
47
+
48
+
49
+ def get_image_shape(image: Image) -> Optional[Tuple[int, int]]:
50
+ """
51
+
52
+ Args:
53
+ image:
54
+
55
+ Returns:
56
+
57
+ """
58
+ try:
59
+ return image.size
60
+ except (OSError, AttributeError):
61
+ return None
62
+
63
+
64
+ def hash_image(image: Image) -> Optional[str]:
65
+ """
66
+
67
+ Args:
68
+ image:
69
+
70
+ Returns:
71
+
72
+ """
73
+ try:
74
+ return str(imagehash.phash(image))
75
+ except (OSError, AttributeError):
76
+ return None
77
+
78
+
79
+ def decode_byte_exif(exif_val: Union[str, bytes]) -> str:
80
+ """Decode byte encodings
81
+
82
+ Args:
83
+ exif_val:
84
+
85
+ Returns:
86
+
87
+ """
88
+ if isinstance(exif_val, str):
89
+ return exif_val
90
+ else:
91
+ return exif_val.decode()
92
+
93
+
94
+ def extract_exif(image: Image) -> dict:
95
+ """
96
+
97
+ Args:
98
+ image:
99
+
100
+ Returns:
101
+
102
+ """
103
+ try:
104
+ exif_data = image._getexif()
105
+ if exif_data is not None:
106
+ exif = {
107
+ ExifTags.TAGS[k]: decode_byte_exif(v)
108
+ for k, v in exif_data.items()
109
+ if k in ExifTags.TAGS
110
+ }
111
+ else:
112
+ exif = {}
113
+ except (AttributeError, OSError):
114
+ # Not all file types (e.g. .gif) have exif information.
115
+ exif = {}
116
+
117
+ return exif
118
+
119
+
120
+ def path_is_image(p: Path) -> bool:
121
+ guess = filetype.guess(str(p))
122
+ return guess is not None and guess.mime.startswith("image/")
123
+
124
+
125
+ def count_duplicate_hashes(image_descriptions: dict) -> int:
126
+ """
127
+
128
+ Args:
129
+ image_descriptions:
130
+
131
+ Returns:
132
+
133
+ """
134
+ counts = pd.Series(
135
+ [x["hash"] for x in image_descriptions if "hash" in x]
136
+ ).value_counts()
137
+ return counts.sum() - len(counts)
138
+
139
+
140
+ def extract_exif_series(image_exifs: list) -> dict:
141
+ """
142
+
143
+ Args:
144
+ image_exifs:
145
+
146
+ Returns:
147
+
148
+ """
149
+ exif_keys = []
150
+ exif_values: dict = {}
151
+
152
+ for image_exif in image_exifs:
153
+ # Extract key
154
+ exif_keys.extend(list(image_exif.keys()))
155
+
156
+ # Extract values per key
157
+ for exif_key, exif_val in image_exif.items():
158
+ if exif_key not in exif_values:
159
+ exif_values[exif_key] = []
160
+
161
+ exif_values[exif_key].append(exif_val)
162
+
163
+ series = {"exif_keys": pd.Series(exif_keys, dtype=object).value_counts().to_dict()}
164
+
165
+ for k, v in exif_values.items():
166
+ series[k] = pd.Series(v).value_counts()
167
+
168
+ return series
169
+
170
+
171
+ def extract_image_information(
172
+ path: Path, exif: bool = False, hash: bool = False
173
+ ) -> dict:
174
+ """Extracts all image information per file, as opening files is slow
175
+
176
+ Args:
177
+ path: Path to the image
178
+ exif: extract exif information
179
+ hash: calculate hash (for duplicate detection)
180
+
181
+ Returns:
182
+ A dict containing image information
183
+ """
184
+ information: dict = {}
185
+ image = open_image(path)
186
+ information["opened"] = image is not None
187
+ if image is not None:
188
+ information["truncated"] = is_image_truncated(image)
189
+ if not information["truncated"]:
190
+ information["size"] = image.size
191
+ if exif:
192
+ information["exif"] = extract_exif(image)
193
+ if hash:
194
+ information["hash"] = hash_image(image)
195
+
196
+ return information
197
+
198
+
199
+ def image_summary(series: pd.Series, exif: bool = False, hash: bool = False) -> dict:
200
+ """
201
+
202
+ Args:
203
+ series: series to summarize
204
+ exif: extract exif information
205
+ hash: calculate hash (for duplicate detection)
206
+
207
+ Returns:
208
+
209
+ """
210
+
211
+ image_information = series.apply(
212
+ partial(extract_image_information, exif=exif, hash=hash)
213
+ )
214
+ summary = {
215
+ "n_truncated": sum(
216
+ 1 for x in image_information if "truncated" in x and x["truncated"]
217
+ ),
218
+ "image_dimensions": pd.Series(
219
+ [x["size"] for x in image_information if "size" in x],
220
+ name="image_dimensions",
221
+ ),
222
+ }
223
+
224
+ image_widths = summary["image_dimensions"].map(lambda x: x[0])
225
+ summary.update(named_aggregate_summary(image_widths, "width"))
226
+ image_heights = summary["image_dimensions"].map(lambda x: x[1])
227
+ summary.update(named_aggregate_summary(image_heights, "height"))
228
+ image_areas = image_widths * image_heights
229
+ summary.update(named_aggregate_summary(image_areas, "area"))
230
+
231
+ if hash:
232
+ summary["n_duplicate_hash"] = count_duplicate_hashes(image_information)
233
+
234
+ if exif:
235
+ exif_series = extract_exif_series(
236
+ [x["exif"] for x in image_information if "exif" in x]
237
+ )
238
+ summary["exif_keys_counts"] = exif_series["exif_keys"]
239
+ summary["exif_data"] = exif_series
240
+
241
+ return summary
242
+
243
+
244
+ @describe_image_1d.register
245
+ def pandas_describe_image_1d(
246
+ config: Settings, series: pd.Series, summary: dict
247
+ ) -> Tuple[Settings, pd.Series, dict]:
248
+ if series.hasnans:
249
+ raise ValueError("May not contain NaNs")
250
+ if not hasattr(series, "str"):
251
+ raise ValueError("series should have .str accessor")
252
+
253
+ summary.update(image_summary(series, config.vars.image.exif))
254
+
255
+ return config, series, summary
@@ -0,0 +1,175 @@
1
+ from typing import Any, Dict, Tuple
2
+
3
+ import numpy as np
4
+ import pandas as pd
5
+
6
+ from data_profiling.utils.compat import pandas_version_info
7
+
8
+ if pandas_version_info() >= (1, 5):
9
+ from pandas.core.arrays.integer import IntegerDtype
10
+ else:
11
+ from pandas.core.arrays.integer import _IntegerDtype as IntegerDtype
12
+
13
+ from data_profiling.config import Settings
14
+ from data_profiling.model.summary_algorithms import (
15
+ chi_square,
16
+ describe_numeric_1d,
17
+ histogram_compute,
18
+ series_handle_nulls,
19
+ series_hashable,
20
+ )
21
+
22
+
23
+ def mad(arr: np.ndarray) -> np.ndarray:
24
+ """Median Absolute Deviation: a "Robust" version of standard deviation.
25
+ Indices variability of the sample.
26
+ https://en.wikipedia.org/wiki/Median_absolute_deviation
27
+ """
28
+ return np.median(np.abs(arr - np.median(arr)))
29
+
30
+
31
+ def numeric_stats_pandas(series: pd.Series) -> Dict[str, Any]:
32
+ return {
33
+ "mean": series.mean(),
34
+ "std": series.std(),
35
+ "variance": series.var(),
36
+ "min": series.min(),
37
+ "max": series.max(),
38
+ # Unbiased kurtosis obtained using Fisher's definition (kurtosis of normal == 0.0). Normalized by N-1.
39
+ "kurtosis": series.kurt(),
40
+ # Unbiased skew normalized by N-1
41
+ "skewness": series.skew(),
42
+ "sum": series.sum(),
43
+ }
44
+
45
+
46
+ def numeric_stats_numpy(
47
+ present_values: np.ndarray, series: pd.Series, series_description: Dict[str, Any]
48
+ ) -> Dict[str, Any]:
49
+ vc = series_description["value_counts_without_nan"]
50
+ index_values = vc.index.values
51
+
52
+ # FIXME: can be performance optimized by using weights in std, var, kurt and skew...
53
+ if len(index_values):
54
+ return {
55
+ "mean": np.average(index_values, weights=vc.values),
56
+ "std": np.std(present_values, ddof=1),
57
+ "variance": np.var(present_values, ddof=1),
58
+ "min": np.min(index_values),
59
+ "max": np.max(index_values),
60
+ # Unbiased kurtosis obtained using Fisher's definition (kurtosis of normal == 0.0). Normalized by N-1.
61
+ "kurtosis": series.kurt(),
62
+ # Unbiased skew normalized by N-1
63
+ "skewness": series.skew(),
64
+ "sum": np.dot(index_values, vc.values),
65
+ }
66
+ else: # Empty numerical series
67
+ return {
68
+ "mean": np.nan,
69
+ "std": 0.0,
70
+ "variance": 0.0,
71
+ "min": np.nan,
72
+ "max": np.nan,
73
+ "kurtosis": 0.0,
74
+ "skewness": 0.0,
75
+ "sum": 0,
76
+ }
77
+
78
+
79
+ @describe_numeric_1d.register
80
+ @series_hashable
81
+ @series_handle_nulls
82
+ def pandas_describe_numeric_1d(
83
+ config: Settings, series: pd.Series, summary: dict
84
+ ) -> Tuple[Settings, pd.Series, dict]:
85
+ """Describe a numeric series.
86
+
87
+ Args:
88
+ config: report Settings object
89
+ series: The Series to describe.
90
+ summary: The dict containing the series description so far.
91
+
92
+ Returns:
93
+ A dict containing calculated series description values.
94
+ """
95
+
96
+ chi_squared_threshold = config.vars.num.chi_squared_threshold
97
+ quantiles = config.vars.num.quantiles
98
+
99
+ value_counts = summary["value_counts_without_nan"]
100
+
101
+ negative_index = value_counts.index < 0
102
+ summary["n_negative"] = value_counts.loc[negative_index].sum()
103
+ summary["p_negative"] = summary["n_negative"] / summary["n"]
104
+
105
+ infinity_values = [np.inf, -np.inf]
106
+ infinity_index = value_counts.index.isin(infinity_values)
107
+ summary["n_infinite"] = value_counts.loc[infinity_index].sum()
108
+
109
+ summary["n_zeros"] = 0
110
+ if 0 in value_counts.index:
111
+ summary["n_zeros"] = value_counts.loc[0]
112
+
113
+ stats = summary
114
+
115
+ if isinstance(series.dtype, IntegerDtype):
116
+ stats.update(numeric_stats_pandas(series))
117
+ present_values = series.astype(str(series.dtype).lower())
118
+ finite_values = present_values
119
+ else:
120
+ present_values = series.values
121
+ finite_values = present_values[np.isfinite(present_values)]
122
+ stats.update(numeric_stats_numpy(present_values, series, summary))
123
+
124
+ stats.update(
125
+ {
126
+ "mad": mad(present_values),
127
+ }
128
+ )
129
+
130
+ if chi_squared_threshold > 0.0:
131
+ stats["chi_squared"] = chi_square(finite_values)
132
+
133
+ stats["range"] = stats["max"] - stats["min"]
134
+ stats.update(
135
+ {
136
+ f"{percentile:.0%}": value
137
+ for percentile, value in series.quantile(quantiles).to_dict().items()
138
+ }
139
+ )
140
+ stats["iqr"] = stats["75%"] - stats["25%"]
141
+ stats["cv"] = stats["std"] / stats["mean"] if stats["mean"] else np.nan
142
+ stats["p_zeros"] = stats["n_zeros"] / summary["n"]
143
+ stats["p_infinite"] = summary["n_infinite"] / summary["n"]
144
+
145
+ stats["monotonic_increase"] = series.is_monotonic_increasing
146
+ stats["monotonic_decrease"] = series.is_monotonic_decreasing
147
+
148
+ stats["monotonic_increase_strict"] = (
149
+ stats["monotonic_increase"] and series.is_unique
150
+ )
151
+ stats["monotonic_decrease_strict"] = (
152
+ stats["monotonic_decrease"] and series.is_unique
153
+ )
154
+ if summary["monotonic_increase_strict"]:
155
+ stats["monotonic"] = 2
156
+ elif summary["monotonic_decrease_strict"]:
157
+ stats["monotonic"] = -2
158
+ elif summary["monotonic_increase"]:
159
+ stats["monotonic"] = 1
160
+ elif summary["monotonic_decrease"]:
161
+ stats["monotonic"] = -1
162
+ else:
163
+ stats["monotonic"] = 0
164
+
165
+ if len(value_counts[~infinity_index].index.values) > 0:
166
+ stats.update(
167
+ histogram_compute(
168
+ config,
169
+ value_counts[~infinity_index].index.values,
170
+ summary["n_distinct"],
171
+ weights=value_counts[~infinity_index].values,
172
+ )
173
+ )
174
+
175
+ return config, series, stats
@@ -0,0 +1,63 @@
1
+ import os
2
+ from typing import Tuple
3
+
4
+ import pandas as pd
5
+
6
+ from data_profiling.config import Settings
7
+ from data_profiling.model.summary_algorithms import describe_path_1d
8
+
9
+
10
+ def path_summary(series: pd.Series) -> dict:
11
+ """
12
+
13
+ Args:
14
+ series: series to summarize
15
+
16
+ Returns:
17
+
18
+ """
19
+
20
+ # TODO: optimize using value counts
21
+ summary = {
22
+ "common_prefix": os.path.commonprefix(series.values.tolist())
23
+ or "No common prefix",
24
+ "stem_counts": series.map(lambda x: os.path.splitext(x)[0]).value_counts(),
25
+ "suffix_counts": series.map(lambda x: os.path.splitext(x)[1]).value_counts(),
26
+ "name_counts": series.map(lambda x: os.path.basename(x)).value_counts(),
27
+ "parent_counts": series.map(lambda x: os.path.dirname(x)).value_counts(),
28
+ "anchor_counts": series.map(lambda x: os.path.splitdrive(x)[0]).value_counts(),
29
+ }
30
+
31
+ summary["n_stem_unique"] = len(summary["stem_counts"])
32
+ summary["n_suffix_unique"] = len(summary["suffix_counts"])
33
+ summary["n_name_unique"] = len(summary["name_counts"])
34
+ summary["n_parent_unique"] = len(summary["parent_counts"])
35
+ summary["n_anchor_unique"] = len(summary["anchor_counts"])
36
+
37
+ return summary
38
+
39
+
40
+ @describe_path_1d.register
41
+ def pandas_describe_path_1d(
42
+ config: Settings, series: pd.Series, summary: dict
43
+ ) -> Tuple[Settings, pd.Series, dict]:
44
+ """Describe a path series.
45
+
46
+ Args:
47
+ config: report Settings object
48
+ series: The Series to describe.
49
+ summary: The dict containing the series description so far.
50
+
51
+ Returns:
52
+ A dict containing calculated series description values.
53
+ """
54
+
55
+ # Make sure we deal with strings (Issue #100)
56
+ if series.hasnans:
57
+ raise ValueError("May not contain NaNs")
58
+ if not hasattr(series, "str"):
59
+ raise ValueError("series should have .str accessor")
60
+
61
+ summary.update(path_summary(series))
62
+
63
+ return config, series, summary
@@ -0,0 +1,41 @@
1
+ from typing import Tuple
2
+
3
+ import pandas as pd
4
+
5
+ from data_profiling.config import Settings
6
+ from data_profiling.model.summary_algorithms import describe_supported, series_hashable
7
+
8
+
9
+ @describe_supported.register
10
+ @series_hashable
11
+ def pandas_describe_supported(
12
+ config: Settings, series: pd.Series, series_description: dict
13
+ ) -> Tuple[Settings, pd.Series, dict]:
14
+ """Describe a supported series.
15
+
16
+ Args:
17
+ config: report Settings object
18
+ series: The Series to describe.
19
+ series_description: The dict containing the series description so far.
20
+
21
+ Returns:
22
+ A dict containing calculated series description values.
23
+ """
24
+
25
+ # number of non-NaN observations in the Series
26
+ count = series_description["count"]
27
+
28
+ value_counts = series_description["value_counts_without_nan"]
29
+ distinct_count = len(value_counts)
30
+ unique_count = value_counts.where(value_counts == 1).count()
31
+
32
+ stats = {
33
+ "n_distinct": distinct_count,
34
+ "p_distinct": distinct_count / count if count > 0 else 0,
35
+ "is_unique": unique_count == count and count > 0,
36
+ "n_unique": unique_count,
37
+ "p_unique": unique_count / count if count > 0 else 0,
38
+ }
39
+ stats.update(series_description)
40
+
41
+ return config, series, stats
@@ -0,0 +1,62 @@
1
+ from typing import Tuple
2
+
3
+ import pandas as pd
4
+
5
+ from data_profiling.config import Settings
6
+ from data_profiling.model.pandas.describe_categorical_pandas import (
7
+ length_summary_vc,
8
+ unicode_summary_vc,
9
+ word_summary_vc,
10
+ )
11
+ from data_profiling.model.summary_algorithms import (
12
+ histogram_compute,
13
+ series_handle_nulls,
14
+ series_hashable,
15
+ )
16
+
17
+
18
+ @series_hashable
19
+ @series_handle_nulls
20
+ def pandas_describe_text_1d(
21
+ config: Settings,
22
+ series: pd.Series,
23
+ summary: dict,
24
+ ) -> Tuple[Settings, pd.Series, dict]:
25
+ """Describe string series.
26
+
27
+ Args:
28
+ config: report Settings object
29
+ series: The Series to describe.
30
+ summary: The dict containing the series description so far.
31
+
32
+ Returns:
33
+ A dict containing calculated series description values.
34
+ """
35
+
36
+ series = series.astype(str)
37
+
38
+ # Only run if at least 1 non-missing value
39
+ value_counts = summary["value_counts_without_nan"]
40
+ value_counts.index = value_counts.index.astype(str)
41
+
42
+ summary.update({"first_rows": series.head(5)})
43
+
44
+ if config.vars.text.length:
45
+ summary.update(length_summary_vc(value_counts))
46
+ summary.update(
47
+ histogram_compute(
48
+ config,
49
+ summary["length_histogram"].index.values,
50
+ len(summary["length_histogram"]),
51
+ name="histogram_length",
52
+ weights=summary["length_histogram"].values,
53
+ )
54
+ )
55
+
56
+ if config.vars.text.characters:
57
+ summary.update(unicode_summary_vc(value_counts))
58
+
59
+ if config.vars.text.words:
60
+ summary.update(word_summary_vc(value_counts, config.vars.cat.stop_words))
61
+
62
+ return config, series, summary