fg-data-profiling 4.19.0__py2.py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (238) hide show
  1. data_profiling/__init__.py +34 -0
  2. data_profiling/compare_reports.py +359 -0
  3. data_profiling/config.py +496 -0
  4. data_profiling/config_default.yaml +223 -0
  5. data_profiling/config_minimal.yaml +222 -0
  6. data_profiling/controller/__init__.py +1 -0
  7. data_profiling/controller/console.py +125 -0
  8. data_profiling/controller/pandas_decorator.py +21 -0
  9. data_profiling/expectations_report.py +117 -0
  10. data_profiling/model/__init__.py +4 -0
  11. data_profiling/model/alerts.py +780 -0
  12. data_profiling/model/correlations.py +163 -0
  13. data_profiling/model/dataframe.py +35 -0
  14. data_profiling/model/describe.py +210 -0
  15. data_profiling/model/description.py +108 -0
  16. data_profiling/model/duplicates.py +14 -0
  17. data_profiling/model/expectation_algorithms.py +112 -0
  18. data_profiling/model/handler.py +81 -0
  19. data_profiling/model/missing.py +146 -0
  20. data_profiling/model/pairwise.py +33 -0
  21. data_profiling/model/pandas/__init__.py +55 -0
  22. data_profiling/model/pandas/correlations_pandas.py +207 -0
  23. data_profiling/model/pandas/dataframe_pandas.py +26 -0
  24. data_profiling/model/pandas/describe_boolean_pandas.py +43 -0
  25. data_profiling/model/pandas/describe_categorical_pandas.py +274 -0
  26. data_profiling/model/pandas/describe_counts_pandas.py +63 -0
  27. data_profiling/model/pandas/describe_date_pandas.py +77 -0
  28. data_profiling/model/pandas/describe_file_pandas.py +56 -0
  29. data_profiling/model/pandas/describe_generic_pandas.py +36 -0
  30. data_profiling/model/pandas/describe_image_pandas.py +255 -0
  31. data_profiling/model/pandas/describe_numeric_pandas.py +175 -0
  32. data_profiling/model/pandas/describe_path_pandas.py +63 -0
  33. data_profiling/model/pandas/describe_supported_pandas.py +41 -0
  34. data_profiling/model/pandas/describe_text_pandas.py +62 -0
  35. data_profiling/model/pandas/describe_timeseries_pandas.py +222 -0
  36. data_profiling/model/pandas/describe_url_pandas.py +57 -0
  37. data_profiling/model/pandas/discretize_pandas.py +81 -0
  38. data_profiling/model/pandas/duplicates_pandas.py +56 -0
  39. data_profiling/model/pandas/imbalance_pandas.py +35 -0
  40. data_profiling/model/pandas/missing_pandas.py +42 -0
  41. data_profiling/model/pandas/sample_pandas.py +38 -0
  42. data_profiling/model/pandas/summary_pandas.py +101 -0
  43. data_profiling/model/pandas/table_pandas.py +56 -0
  44. data_profiling/model/pandas/timeseries_index_pandas.py +33 -0
  45. data_profiling/model/pandas/utils_pandas.py +27 -0
  46. data_profiling/model/sample.py +37 -0
  47. data_profiling/model/spark/__init__.py +48 -0
  48. data_profiling/model/spark/correlations_spark.py +152 -0
  49. data_profiling/model/spark/dataframe_spark.py +34 -0
  50. data_profiling/model/spark/describe_boolean_spark.py +27 -0
  51. data_profiling/model/spark/describe_categorical_spark.py +28 -0
  52. data_profiling/model/spark/describe_counts_spark.py +105 -0
  53. data_profiling/model/spark/describe_date_spark.py +51 -0
  54. data_profiling/model/spark/describe_generic_spark.py +30 -0
  55. data_profiling/model/spark/describe_numeric_spark.py +155 -0
  56. data_profiling/model/spark/describe_supported_spark.py +33 -0
  57. data_profiling/model/spark/describe_text_spark.py +25 -0
  58. data_profiling/model/spark/duplicates_spark.py +54 -0
  59. data_profiling/model/spark/missing_spark.py +96 -0
  60. data_profiling/model/spark/sample_spark.py +43 -0
  61. data_profiling/model/spark/summary_spark.py +95 -0
  62. data_profiling/model/spark/table_spark.py +58 -0
  63. data_profiling/model/spark/timeseries_index_spark.py +12 -0
  64. data_profiling/model/summarizer.py +207 -0
  65. data_profiling/model/summary.py +66 -0
  66. data_profiling/model/summary_algorithms.py +276 -0
  67. data_profiling/model/table.py +10 -0
  68. data_profiling/model/timeseries_index.py +16 -0
  69. data_profiling/model/typeset.py +365 -0
  70. data_profiling/model/typeset_relations.py +143 -0
  71. data_profiling/profile_report.py +573 -0
  72. data_profiling/report/__init__.py +4 -0
  73. data_profiling/report/formatters.py +346 -0
  74. data_profiling/report/presentation/__init__.py +1 -0
  75. data_profiling/report/presentation/core/__init__.py +39 -0
  76. data_profiling/report/presentation/core/alerts.py +18 -0
  77. data_profiling/report/presentation/core/collapse.py +24 -0
  78. data_profiling/report/presentation/core/container.py +50 -0
  79. data_profiling/report/presentation/core/correlation_table.py +21 -0
  80. data_profiling/report/presentation/core/dropdown.py +44 -0
  81. data_profiling/report/presentation/core/duplicate.py +16 -0
  82. data_profiling/report/presentation/core/frequency_table.py +14 -0
  83. data_profiling/report/presentation/core/frequency_table_small.py +16 -0
  84. data_profiling/report/presentation/core/html.py +14 -0
  85. data_profiling/report/presentation/core/image.py +34 -0
  86. data_profiling/report/presentation/core/item_renderer.py +17 -0
  87. data_profiling/report/presentation/core/renderable.py +42 -0
  88. data_profiling/report/presentation/core/root.py +35 -0
  89. data_profiling/report/presentation/core/sample.py +20 -0
  90. data_profiling/report/presentation/core/scores.py +32 -0
  91. data_profiling/report/presentation/core/table.py +26 -0
  92. data_profiling/report/presentation/core/toggle_button.py +14 -0
  93. data_profiling/report/presentation/core/variable.py +40 -0
  94. data_profiling/report/presentation/core/variable_info.py +36 -0
  95. data_profiling/report/presentation/flavours/__init__.py +9 -0
  96. data_profiling/report/presentation/flavours/flavour_html.py +64 -0
  97. data_profiling/report/presentation/flavours/flavour_widget.py +61 -0
  98. data_profiling/report/presentation/flavours/flavours.py +43 -0
  99. data_profiling/report/presentation/flavours/html/__init__.py +47 -0
  100. data_profiling/report/presentation/flavours/html/alerts.py +10 -0
  101. data_profiling/report/presentation/flavours/html/collapse.py +7 -0
  102. data_profiling/report/presentation/flavours/html/container.py +58 -0
  103. data_profiling/report/presentation/flavours/html/correlation_table.py +13 -0
  104. data_profiling/report/presentation/flavours/html/dropdown.py +7 -0
  105. data_profiling/report/presentation/flavours/html/duplicate.py +24 -0
  106. data_profiling/report/presentation/flavours/html/frequency_table.py +20 -0
  107. data_profiling/report/presentation/flavours/html/frequency_table_small.py +15 -0
  108. data_profiling/report/presentation/flavours/html/html.py +6 -0
  109. data_profiling/report/presentation/flavours/html/image.py +7 -0
  110. data_profiling/report/presentation/flavours/html/root.py +14 -0
  111. data_profiling/report/presentation/flavours/html/sample.py +12 -0
  112. data_profiling/report/presentation/flavours/html/scores.py +11 -0
  113. data_profiling/report/presentation/flavours/html/table.py +7 -0
  114. data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant.html +1 -0
  115. data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant_length.html +1 -0
  116. data_profiling/report/presentation/flavours/html/templates/alerts/alert_dirty_category.html +1 -0
  117. data_profiling/report/presentation/flavours/html/templates/alerts/alert_duplicates.html +1 -0
  118. data_profiling/report/presentation/flavours/html/templates/alerts/alert_empty.html +1 -0
  119. data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_cardinality.html +1 -0
  120. data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_correlation.html +4 -0
  121. data_profiling/report/presentation/flavours/html/templates/alerts/alert_imbalance.html +1 -0
  122. data_profiling/report/presentation/flavours/html/templates/alerts/alert_infinite.html +1 -0
  123. data_profiling/report/presentation/flavours/html/templates/alerts/alert_missing.html +1 -0
  124. data_profiling/report/presentation/flavours/html/templates/alerts/alert_near_duplicates.html +1 -0
  125. data_profiling/report/presentation/flavours/html/templates/alerts/alert_non_stationary.html +1 -0
  126. data_profiling/report/presentation/flavours/html/templates/alerts/alert_seasonal.html +1 -0
  127. data_profiling/report/presentation/flavours/html/templates/alerts/alert_skewed.html +1 -0
  128. data_profiling/report/presentation/flavours/html/templates/alerts/alert_truncated.html +1 -0
  129. data_profiling/report/presentation/flavours/html/templates/alerts/alert_type_date.html +1 -0
  130. data_profiling/report/presentation/flavours/html/templates/alerts/alert_uniform.html +1 -0
  131. data_profiling/report/presentation/flavours/html/templates/alerts/alert_unique.html +1 -0
  132. data_profiling/report/presentation/flavours/html/templates/alerts/alert_unsupported.html +1 -0
  133. data_profiling/report/presentation/flavours/html/templates/alerts/alert_zeros.html +1 -0
  134. data_profiling/report/presentation/flavours/html/templates/alerts.html +47 -0
  135. data_profiling/report/presentation/flavours/html/templates/collapse.html +11 -0
  136. data_profiling/report/presentation/flavours/html/templates/correlation_table.html +5 -0
  137. data_profiling/report/presentation/flavours/html/templates/diagram.html +11 -0
  138. data_profiling/report/presentation/flavours/html/templates/dropdown.html +16 -0
  139. data_profiling/report/presentation/flavours/html/templates/duplicate.html +5 -0
  140. data_profiling/report/presentation/flavours/html/templates/frequency_table.html +45 -0
  141. data_profiling/report/presentation/flavours/html/templates/frequency_table_small.html +34 -0
  142. data_profiling/report/presentation/flavours/html/templates/report.html +26 -0
  143. data_profiling/report/presentation/flavours/html/templates/sample.html +10 -0
  144. data_profiling/report/presentation/flavours/html/templates/scores.html +78 -0
  145. data_profiling/report/presentation/flavours/html/templates/sequence/batch_grid.html +16 -0
  146. data_profiling/report/presentation/flavours/html/templates/sequence/grid.html +18 -0
  147. data_profiling/report/presentation/flavours/html/templates/sequence/list.html +7 -0
  148. data_profiling/report/presentation/flavours/html/templates/sequence/named_list.html +8 -0
  149. data_profiling/report/presentation/flavours/html/templates/sequence/overview_tabs.html +30 -0
  150. data_profiling/report/presentation/flavours/html/templates/sequence/scores.html +3 -0
  151. data_profiling/report/presentation/flavours/html/templates/sequence/sections.html +13 -0
  152. data_profiling/report/presentation/flavours/html/templates/sequence/select.html +40 -0
  153. data_profiling/report/presentation/flavours/html/templates/sequence/tabs.html +30 -0
  154. data_profiling/report/presentation/flavours/html/templates/table.html +38 -0
  155. data_profiling/report/presentation/flavours/html/templates/toggle_button.html +18 -0
  156. data_profiling/report/presentation/flavours/html/templates/variable.html +7 -0
  157. data_profiling/report/presentation/flavours/html/templates/variable_info.html +49 -0
  158. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.bundle.min.js +7 -0
  159. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.min.css +6 -0
  160. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/cosmo.bootstrap.min.css +12 -0
  161. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/flatly.bootstrap.min.css +12 -0
  162. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/script.js +52 -0
  163. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/simplex.bootstrap.min.css +12 -0
  164. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/style.css +253 -0
  165. data_profiling/report/presentation/flavours/html/templates/wrapper/assets/united.bootstrap.min.css +12 -0
  166. data_profiling/report/presentation/flavours/html/templates/wrapper/footer.html +7 -0
  167. data_profiling/report/presentation/flavours/html/templates/wrapper/javascript.html +18 -0
  168. data_profiling/report/presentation/flavours/html/templates/wrapper/navigation.html +36 -0
  169. data_profiling/report/presentation/flavours/html/templates/wrapper/style.html +53 -0
  170. data_profiling/report/presentation/flavours/html/templates.py +76 -0
  171. data_profiling/report/presentation/flavours/html/toggle_button.py +7 -0
  172. data_profiling/report/presentation/flavours/html/variable.py +7 -0
  173. data_profiling/report/presentation/flavours/html/variable_info.py +7 -0
  174. data_profiling/report/presentation/flavours/widget/__init__.py +49 -0
  175. data_profiling/report/presentation/flavours/widget/alerts.py +45 -0
  176. data_profiling/report/presentation/flavours/widget/collapse.py +43 -0
  177. data_profiling/report/presentation/flavours/widget/container.py +121 -0
  178. data_profiling/report/presentation/flavours/widget/correlation_table.py +14 -0
  179. data_profiling/report/presentation/flavours/widget/dropdown.py +31 -0
  180. data_profiling/report/presentation/flavours/widget/duplicate.py +14 -0
  181. data_profiling/report/presentation/flavours/widget/frequency_table.py +57 -0
  182. data_profiling/report/presentation/flavours/widget/frequency_table_small.py +66 -0
  183. data_profiling/report/presentation/flavours/widget/html.py +11 -0
  184. data_profiling/report/presentation/flavours/widget/image.py +26 -0
  185. data_profiling/report/presentation/flavours/widget/notebook.py +81 -0
  186. data_profiling/report/presentation/flavours/widget/root.py +10 -0
  187. data_profiling/report/presentation/flavours/widget/sample.py +14 -0
  188. data_profiling/report/presentation/flavours/widget/table.py +30 -0
  189. data_profiling/report/presentation/flavours/widget/toggle_button.py +17 -0
  190. data_profiling/report/presentation/flavours/widget/variable.py +12 -0
  191. data_profiling/report/presentation/flavours/widget/variable_info.py +11 -0
  192. data_profiling/report/presentation/frequency_table_utils.py +141 -0
  193. data_profiling/report/structure/__init__.py +1 -0
  194. data_profiling/report/structure/correlations.py +123 -0
  195. data_profiling/report/structure/overview.py +376 -0
  196. data_profiling/report/structure/report.py +457 -0
  197. data_profiling/report/structure/variables/__init__.py +35 -0
  198. data_profiling/report/structure/variables/render_boolean.py +132 -0
  199. data_profiling/report/structure/variables/render_categorical.py +566 -0
  200. data_profiling/report/structure/variables/render_common.py +31 -0
  201. data_profiling/report/structure/variables/render_complex.py +102 -0
  202. data_profiling/report/structure/variables/render_count.py +172 -0
  203. data_profiling/report/structure/variables/render_date.py +143 -0
  204. data_profiling/report/structure/variables/render_file.py +70 -0
  205. data_profiling/report/structure/variables/render_generic.py +45 -0
  206. data_profiling/report/structure/variables/render_image.py +204 -0
  207. data_profiling/report/structure/variables/render_path.py +134 -0
  208. data_profiling/report/structure/variables/render_real.py +314 -0
  209. data_profiling/report/structure/variables/render_text.py +189 -0
  210. data_profiling/report/structure/variables/render_timeseries.py +371 -0
  211. data_profiling/report/structure/variables/render_url.py +132 -0
  212. data_profiling/report/utils.py +34 -0
  213. data_profiling/serialize_report.py +143 -0
  214. data_profiling/utils/__init__.py +1 -0
  215. data_profiling/utils/backend.py +9 -0
  216. data_profiling/utils/cache.py +59 -0
  217. data_profiling/utils/common.py +142 -0
  218. data_profiling/utils/compat.py +31 -0
  219. data_profiling/utils/dataframe.py +238 -0
  220. data_profiling/utils/logger.py +53 -0
  221. data_profiling/utils/notebook.py +8 -0
  222. data_profiling/utils/paths.py +45 -0
  223. data_profiling/utils/progress_bar.py +15 -0
  224. data_profiling/utils/styles.py +22 -0
  225. data_profiling/utils/versions.py +19 -0
  226. data_profiling/version.py +1 -0
  227. data_profiling/visualisation/__init__.py +1 -0
  228. data_profiling/visualisation/context.py +87 -0
  229. data_profiling/visualisation/missing.py +138 -0
  230. data_profiling/visualisation/plot.py +1158 -0
  231. data_profiling/visualisation/utils.py +113 -0
  232. fg_data_profiling-4.19.0.dist-info/METADATA +362 -0
  233. fg_data_profiling-4.19.0.dist-info/RECORD +238 -0
  234. fg_data_profiling-4.19.0.dist-info/WHEEL +6 -0
  235. fg_data_profiling-4.19.0.dist-info/entry_points.txt +3 -0
  236. fg_data_profiling-4.19.0.dist-info/licenses/LICENSE +21 -0
  237. fg_data_profiling-4.19.0.dist-info/top_level.txt +2 -0
  238. ydata_profiling/__init__.py +43 -0
@@ -0,0 +1,274 @@
1
+ import contextlib
2
+ import string
3
+ from collections import Counter
4
+ from typing import List, Tuple
5
+
6
+ import numpy as np
7
+ import pandas as pd
8
+
9
+ from data_profiling.config import Settings
10
+ from data_profiling.model.pandas.imbalance_pandas import column_imbalance_score
11
+ from data_profiling.model.pandas.utils_pandas import weighted_median
12
+ from data_profiling.model.summary_algorithms import (
13
+ chi_square,
14
+ describe_categorical_1d,
15
+ histogram_compute,
16
+ series_handle_nulls,
17
+ series_hashable,
18
+ )
19
+
20
+
21
+ def get_character_counts_vc(vc: pd.Series) -> pd.Series:
22
+ series = pd.Series(vc.index, index=vc, dtype=object)
23
+ characters = series[series != ""].apply(list)
24
+ characters = characters.explode()
25
+
26
+ counts = pd.Series(characters.index, index=characters).dropna()
27
+ if len(counts) > 0:
28
+ counts = counts.groupby(level=0, sort=False).sum()
29
+ counts = counts.sort_values(ascending=False)
30
+ # FIXME: correct in split, below should be zero: print(counts.loc[''])
31
+ counts = counts[counts.index.str.len() > 0]
32
+ return counts
33
+
34
+
35
+ def get_character_counts(series: pd.Series) -> Counter:
36
+ """Function to return the character counts
37
+
38
+ Args:
39
+ series: the Series to process
40
+
41
+ Returns:
42
+ A dict with character counts
43
+ """
44
+ return Counter(series.str.cat())
45
+
46
+
47
+ def counter_to_series(counter: Counter) -> pd.Series:
48
+ if not counter:
49
+ return pd.Series([], dtype=object)
50
+
51
+ counter_as_tuples = counter.most_common()
52
+ items, counts = zip(*counter_as_tuples)
53
+ return pd.Series(counts, index=items)
54
+
55
+
56
+ def unicode_summary_vc(vc: pd.Series) -> dict:
57
+ try:
58
+ from tangled_up_in_unicode import ( # type: ignore
59
+ block,
60
+ block_abbr,
61
+ category,
62
+ category_long,
63
+ script,
64
+ )
65
+ except ImportError:
66
+ from unicodedata import category as _category # pylint: disable=import-error
67
+
68
+ category = _category # type: ignore
69
+ char_handler = lambda char: "(unknown)" # noqa: E731
70
+ block = char_handler
71
+ block_abbr = char_handler
72
+ category_long = char_handler
73
+ script = char_handler
74
+
75
+ # Unicode Character Summaries (category and script name)
76
+ character_counts = get_character_counts_vc(vc)
77
+
78
+ character_counts_series = character_counts
79
+ summary = {
80
+ "n_characters_distinct": len(character_counts_series),
81
+ "n_characters": np.sum(character_counts_series.values),
82
+ "character_counts": character_counts_series,
83
+ }
84
+
85
+ char_to_block = {key: block(key) for key in character_counts.keys()}
86
+ char_to_category_short = {key: category(key) for key in character_counts.keys()}
87
+ char_to_script = {key: script(key) for key in character_counts.keys()}
88
+
89
+ summary.update(
90
+ {
91
+ "category_alias_values": {
92
+ key: category_long(value)
93
+ for key, value in char_to_category_short.items()
94
+ },
95
+ "block_alias_values": {
96
+ key: block_abbr(value) for key, value in char_to_block.items()
97
+ },
98
+ }
99
+ )
100
+
101
+ # Retrieve original distribution
102
+ block_alias_counts: Counter = Counter()
103
+ per_block_char_counts: dict = {
104
+ k: Counter() for k in summary["block_alias_values"].values()
105
+ }
106
+ for char, n_char in character_counts.items():
107
+ block_name = summary["block_alias_values"][char]
108
+ block_alias_counts[block_name] += n_char
109
+ per_block_char_counts[block_name][char] = n_char
110
+ summary["block_alias_counts"] = counter_to_series(block_alias_counts)
111
+ summary["n_block_alias"] = len(summary["block_alias_counts"])
112
+ summary["block_alias_char_counts"] = {
113
+ k: counter_to_series(v) for k, v in per_block_char_counts.items()
114
+ }
115
+
116
+ script_counts: Counter = Counter()
117
+ per_script_char_counts: dict = {k: Counter() for k in char_to_script.values()}
118
+ for char, n_char in character_counts.items():
119
+ script_name = char_to_script[char]
120
+ script_counts[script_name] += n_char
121
+ per_script_char_counts[script_name][char] = n_char
122
+ summary["script_counts"] = counter_to_series(script_counts)
123
+ summary["n_scripts"] = len(summary["script_counts"])
124
+ summary["script_char_counts"] = {
125
+ k: counter_to_series(v) for k, v in per_script_char_counts.items()
126
+ }
127
+
128
+ category_alias_counts: Counter = Counter()
129
+ per_category_alias_char_counts: dict = {
130
+ k: Counter() for k in summary["category_alias_values"].values()
131
+ }
132
+ for char, n_char in character_counts.items():
133
+ category_alias_name = summary["category_alias_values"][char]
134
+ category_alias_counts[category_alias_name] += n_char
135
+ per_category_alias_char_counts[category_alias_name][char] += n_char
136
+ summary["category_alias_counts"] = counter_to_series(category_alias_counts)
137
+ if len(summary["category_alias_counts"]) > 0:
138
+ summary["category_alias_counts"].index = summary[
139
+ "category_alias_counts"
140
+ ].index.str.replace("_", " ")
141
+ summary["n_category"] = len(summary["category_alias_counts"])
142
+ summary["category_alias_char_counts"] = {
143
+ k: counter_to_series(v) for k, v in per_category_alias_char_counts.items()
144
+ }
145
+
146
+ with contextlib.suppress(AttributeError):
147
+ summary["category_alias_counts"].index = summary[
148
+ "category_alias_counts"
149
+ ].index.str.replace("_", " ")
150
+
151
+ return summary
152
+
153
+
154
+ def word_summary_vc(vc: pd.Series, stop_words: List[str] = []) -> dict:
155
+ """Count the number of occurrences of each individual word across
156
+ all lines of the data Series, then sort from the word with the most
157
+ occurrences to the word with the least occurrences. If a list of
158
+ stop words is given, they will be ignored.
159
+
160
+ Args:
161
+ vc: Series containing all unique categories as index and their
162
+ frequency as value. Sorted from the most frequent down.
163
+ stop_words: List of stop words to ignore, empty by default.
164
+
165
+ Returns:
166
+ A dict containing the results as a Series with unique words as
167
+ index and the computed frequency as value
168
+ """
169
+ # TODO: configurable lowercase/punctuation etc.
170
+ # TODO: remove punctuation in words
171
+
172
+ series = pd.Series(vc.index, index=vc, dtype=object)
173
+ word_lists = series.str.lower().str.split()
174
+ words = word_lists.explode().str.strip(string.punctuation + string.whitespace)
175
+ word_counts = pd.Series(words.index, index=words)
176
+ # fix for pandas 1.0.5
177
+ word_counts = word_counts[word_counts.index.notnull()]
178
+ word_counts = word_counts.groupby(level=0, sort=False).sum()
179
+ word_counts = word_counts.sort_values(ascending=False)
180
+
181
+ # Remove stop words
182
+ if len(stop_words) > 0:
183
+ stop_words = [x.lower() for x in stop_words]
184
+ word_counts = word_counts.loc[~word_counts.index.isin(stop_words)]
185
+
186
+ return {"word_counts": word_counts} if not word_counts.empty else {}
187
+
188
+
189
+ def length_summary_vc(vc: pd.Series) -> dict:
190
+ series = pd.Series(vc.index, index=vc, dtype=object)
191
+ length = series.str.len()
192
+ length_counts = pd.Series(length.index, index=length)
193
+ length_counts = length_counts.groupby(level=0, sort=False).sum()
194
+ length_counts = length_counts.sort_values(ascending=False)
195
+
196
+ summary = {
197
+ "max_length": np.max(length_counts.index),
198
+ "mean_length": np.average(length_counts.index, weights=length_counts.values)
199
+ if not length_counts.empty
200
+ else np.nan,
201
+ "median_length": weighted_median(
202
+ length_counts.index.values, weights=length_counts.values
203
+ )
204
+ if not length_counts.empty
205
+ else np.nan,
206
+ "min_length": np.min(length_counts.index),
207
+ "length_histogram": length_counts,
208
+ }
209
+
210
+ return summary
211
+
212
+
213
+ _displayed_catvar_banner = False
214
+
215
+
216
+ @describe_categorical_1d.register
217
+ @series_hashable
218
+ @series_handle_nulls
219
+ def pandas_describe_categorical_1d(
220
+ config: Settings, series: pd.Series, summary: dict
221
+ ) -> Tuple[Settings, pd.Series, dict]:
222
+ """Describe a categorical series.
223
+
224
+ Args:
225
+ config: report Settings object
226
+ series: The Series to describe.
227
+ summary: The dict containing the series description so far.
228
+
229
+ Returns:
230
+ A dict containing calculated series description values.
231
+ """
232
+ # Global info banner
233
+ global _displayed_catvar_banner
234
+
235
+ # Make sure we deal with strings (Issue #100)
236
+ series = series.astype(str)
237
+
238
+ # Only run if at least 1 non-missing value
239
+ value_counts = summary["value_counts_without_nan"]
240
+ value_counts.index = value_counts.index.astype(str)
241
+
242
+ summary["imbalance"] = column_imbalance_score(value_counts, len(value_counts))
243
+
244
+ redact = config.vars.cat.redact
245
+ if not redact:
246
+ summary.update({"first_rows": series.head(5)})
247
+
248
+ chi_squared_threshold = config.vars.num.chi_squared_threshold
249
+ if chi_squared_threshold > 0.0:
250
+ summary["chi_squared"] = chi_square(histogram=value_counts.values)
251
+
252
+ if config.vars.cat.length:
253
+ summary.update(length_summary_vc(value_counts))
254
+ summary.update(
255
+ histogram_compute(
256
+ config,
257
+ summary["length_histogram"].index.values,
258
+ len(summary["length_histogram"]),
259
+ name="histogram_length",
260
+ weights=summary["length_histogram"].values,
261
+ )
262
+ )
263
+
264
+ if config.vars.cat.characters:
265
+ summary.update(unicode_summary_vc(value_counts))
266
+
267
+ if config.vars.cat.words:
268
+ summary.update(word_summary_vc(value_counts, config.vars.cat.stop_words))
269
+
270
+ if config.vars.cat.dirty_categories: # noqa: SIM102
271
+ if not _displayed_catvar_banner:
272
+ _displayed_catvar_banner = True
273
+
274
+ return config, series, summary
@@ -0,0 +1,63 @@
1
+ from typing import Tuple
2
+
3
+ import pandas as pd
4
+
5
+ from data_profiling.config import Settings
6
+ from data_profiling.model.summary_algorithms import describe_counts
7
+
8
+
9
+ @describe_counts.register
10
+ def pandas_describe_counts(
11
+ config: Settings, series: pd.Series, summary: dict
12
+ ) -> Tuple[Settings, pd.Series, dict]:
13
+ """Counts the values in a series (with and without NaN, distinct).
14
+
15
+ Args:
16
+ config: report Settings object
17
+ series: Series for which we want to calculate the values.
18
+ summary: series' summary
19
+
20
+ Returns:
21
+ A dictionary with the count values (with and without NaN, distinct).
22
+ """
23
+ try:
24
+ value_counts_with_nan = series.value_counts(dropna=False)
25
+ _ = set(value_counts_with_nan.index)
26
+ hashable = True
27
+ except: # noqa: E722
28
+ hashable = False
29
+
30
+ summary["hashable"] = hashable
31
+
32
+ if hashable:
33
+ value_counts_with_nan = value_counts_with_nan[value_counts_with_nan > 0]
34
+
35
+ null_index = value_counts_with_nan.index.isnull()
36
+ if null_index.any():
37
+ n_missing = value_counts_with_nan[null_index].sum()
38
+ value_counts_without_nan = value_counts_with_nan[~null_index]
39
+ else:
40
+ n_missing = 0
41
+ value_counts_without_nan = value_counts_with_nan
42
+
43
+ summary.update(
44
+ {
45
+ "value_counts_without_nan": value_counts_without_nan,
46
+ }
47
+ )
48
+
49
+ try:
50
+ summary["value_counts_index_sorted"] = summary[
51
+ "value_counts_without_nan"
52
+ ].sort_index(ascending=True)
53
+ ordering = True
54
+ except TypeError:
55
+ ordering = False
56
+ else:
57
+ n_missing = series.isna().sum()
58
+ ordering = False
59
+
60
+ summary["ordering"] = ordering
61
+ summary["n_missing"] = n_missing
62
+
63
+ return config, series, summary
@@ -0,0 +1,77 @@
1
+ from typing import Tuple
2
+
3
+ import numpy as np
4
+ import pandas as pd
5
+
6
+ from data_profiling.config import Settings
7
+ from data_profiling.model.summary_algorithms import (
8
+ chi_square,
9
+ describe_date_1d,
10
+ histogram_compute,
11
+ series_handle_nulls,
12
+ series_hashable,
13
+ )
14
+ from data_profiling.model.typeset_relations import is_pandas_1
15
+
16
+
17
+ def to_datetime(series: pd.Series) -> pd.Series:
18
+ if is_pandas_1():
19
+ return pd.to_datetime(series, errors="coerce")
20
+ return pd.to_datetime(series, format="mixed", errors="coerce")
21
+
22
+
23
+ @describe_date_1d.register
24
+ @series_hashable
25
+ @series_handle_nulls
26
+ def pandas_describe_date_1d(
27
+ config: Settings, series: pd.Series, summary: dict
28
+ ) -> Tuple[Settings, pd.Series, dict]:
29
+ """Describe a date series.
30
+
31
+ Args:
32
+ config: report Settings object
33
+ series: The Series to describe.
34
+ summary: The dict containing the series description so far.
35
+
36
+ Returns:
37
+ A dict containing calculated series description values.
38
+ """
39
+ og_series = series.dropna()
40
+ series = to_datetime(og_series)
41
+ invalid_values = og_series[series.isna()]
42
+
43
+ series = series.dropna()
44
+
45
+ if summary["value_counts_without_nan"].empty:
46
+ values = series.values
47
+ summary.update(
48
+ {
49
+ "min": pd.NaT,
50
+ "max": pd.NaT,
51
+ "range": 0,
52
+ }
53
+ )
54
+ else:
55
+ summary.update(
56
+ {
57
+ "min": pd.Timestamp.to_pydatetime(series.min()),
58
+ "max": pd.Timestamp.to_pydatetime(series.max()),
59
+ }
60
+ )
61
+
62
+ summary["range"] = summary["max"] - summary["min"]
63
+
64
+ values = series.values.astype(np.int64) // 10**9
65
+
66
+ if config.vars.num.chi_squared_threshold > 0.0:
67
+ summary["chi_squared"] = chi_square(values)
68
+
69
+ summary.update(histogram_compute(config, values, series.nunique()))
70
+ summary.update(
71
+ {
72
+ "invalid_dates": invalid_values.nunique(),
73
+ "n_invalid_dates": len(invalid_values),
74
+ "p_invalid_dates": len(invalid_values) / summary["n"],
75
+ }
76
+ )
77
+ return config, values, summary
@@ -0,0 +1,56 @@
1
+ import os
2
+ from datetime import datetime
3
+ from typing import Tuple
4
+
5
+ import pandas as pd
6
+
7
+ from data_profiling.config import Settings
8
+ from data_profiling.model.summary_algorithms import describe_file_1d, histogram_compute
9
+
10
+
11
+ def file_summary(series: pd.Series) -> dict:
12
+ """
13
+
14
+ Args:
15
+ series: series to summarize
16
+
17
+ Returns:
18
+
19
+ """
20
+
21
+ # Transform
22
+ stats = series.map(lambda x: os.stat(x))
23
+
24
+ def convert_datetime(x: float) -> str:
25
+ return datetime.fromtimestamp(x).strftime("%Y-%m-%d %H:%M:%S")
26
+
27
+ # Transform some more
28
+ summary = {
29
+ "file_size": stats.map(lambda x: x.st_size),
30
+ "file_created_time": stats.map(lambda x: x.st_ctime).map(convert_datetime),
31
+ "file_accessed_time": stats.map(lambda x: x.st_atime).map(convert_datetime),
32
+ "file_modified_time": stats.map(lambda x: x.st_mtime).map(convert_datetime),
33
+ }
34
+ return summary
35
+
36
+
37
+ @describe_file_1d.register
38
+ def pandas_describe_file_1d(
39
+ config: Settings, series: pd.Series, summary: dict
40
+ ) -> Tuple[Settings, pd.Series, dict]:
41
+ if series.hasnans:
42
+ raise ValueError("May not contain NaNs")
43
+ if not hasattr(series, "str"):
44
+ raise ValueError("series should have .str accessor")
45
+
46
+ summary.update(file_summary(series))
47
+ summary.update(
48
+ histogram_compute(
49
+ config,
50
+ summary["file_size"],
51
+ summary["file_size"].nunique(),
52
+ name="histogram_file_size",
53
+ )
54
+ )
55
+
56
+ return config, series, summary
@@ -0,0 +1,36 @@
1
+ from typing import Tuple
2
+
3
+ import pandas as pd
4
+
5
+ from data_profiling.config import Settings
6
+ from data_profiling.model.summary_algorithms import describe_generic
7
+
8
+
9
+ @describe_generic.register
10
+ def pandas_describe_generic(
11
+ config: Settings, series: pd.Series, summary: dict
12
+ ) -> Tuple[Settings, pd.Series, dict]:
13
+ """Describe generic series.
14
+
15
+ Args:
16
+ config: report Settings object
17
+ series: The Series to describe.
18
+ summary: The dict containing the series description so far.
19
+
20
+ Returns:
21
+ A dict containing calculated series description values.
22
+ """
23
+
24
+ # number of observations in the Series
25
+ length = len(series)
26
+
27
+ summary.update(
28
+ {
29
+ "n": length,
30
+ "p_missing": summary["n_missing"] / length if length > 0 else 0,
31
+ "count": length - summary["n_missing"],
32
+ "memory_size": series.memory_usage(deep=config.memory_deep),
33
+ }
34
+ )
35
+
36
+ return config, series, summary