fg-data-profiling 4.19.0__py2.py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- data_profiling/__init__.py +34 -0
- data_profiling/compare_reports.py +359 -0
- data_profiling/config.py +496 -0
- data_profiling/config_default.yaml +223 -0
- data_profiling/config_minimal.yaml +222 -0
- data_profiling/controller/__init__.py +1 -0
- data_profiling/controller/console.py +125 -0
- data_profiling/controller/pandas_decorator.py +21 -0
- data_profiling/expectations_report.py +117 -0
- data_profiling/model/__init__.py +4 -0
- data_profiling/model/alerts.py +780 -0
- data_profiling/model/correlations.py +163 -0
- data_profiling/model/dataframe.py +35 -0
- data_profiling/model/describe.py +210 -0
- data_profiling/model/description.py +108 -0
- data_profiling/model/duplicates.py +14 -0
- data_profiling/model/expectation_algorithms.py +112 -0
- data_profiling/model/handler.py +81 -0
- data_profiling/model/missing.py +146 -0
- data_profiling/model/pairwise.py +33 -0
- data_profiling/model/pandas/__init__.py +55 -0
- data_profiling/model/pandas/correlations_pandas.py +207 -0
- data_profiling/model/pandas/dataframe_pandas.py +26 -0
- data_profiling/model/pandas/describe_boolean_pandas.py +43 -0
- data_profiling/model/pandas/describe_categorical_pandas.py +274 -0
- data_profiling/model/pandas/describe_counts_pandas.py +63 -0
- data_profiling/model/pandas/describe_date_pandas.py +77 -0
- data_profiling/model/pandas/describe_file_pandas.py +56 -0
- data_profiling/model/pandas/describe_generic_pandas.py +36 -0
- data_profiling/model/pandas/describe_image_pandas.py +255 -0
- data_profiling/model/pandas/describe_numeric_pandas.py +175 -0
- data_profiling/model/pandas/describe_path_pandas.py +63 -0
- data_profiling/model/pandas/describe_supported_pandas.py +41 -0
- data_profiling/model/pandas/describe_text_pandas.py +62 -0
- data_profiling/model/pandas/describe_timeseries_pandas.py +222 -0
- data_profiling/model/pandas/describe_url_pandas.py +57 -0
- data_profiling/model/pandas/discretize_pandas.py +81 -0
- data_profiling/model/pandas/duplicates_pandas.py +56 -0
- data_profiling/model/pandas/imbalance_pandas.py +35 -0
- data_profiling/model/pandas/missing_pandas.py +42 -0
- data_profiling/model/pandas/sample_pandas.py +38 -0
- data_profiling/model/pandas/summary_pandas.py +101 -0
- data_profiling/model/pandas/table_pandas.py +56 -0
- data_profiling/model/pandas/timeseries_index_pandas.py +33 -0
- data_profiling/model/pandas/utils_pandas.py +27 -0
- data_profiling/model/sample.py +37 -0
- data_profiling/model/spark/__init__.py +48 -0
- data_profiling/model/spark/correlations_spark.py +152 -0
- data_profiling/model/spark/dataframe_spark.py +34 -0
- data_profiling/model/spark/describe_boolean_spark.py +27 -0
- data_profiling/model/spark/describe_categorical_spark.py +28 -0
- data_profiling/model/spark/describe_counts_spark.py +105 -0
- data_profiling/model/spark/describe_date_spark.py +51 -0
- data_profiling/model/spark/describe_generic_spark.py +30 -0
- data_profiling/model/spark/describe_numeric_spark.py +155 -0
- data_profiling/model/spark/describe_supported_spark.py +33 -0
- data_profiling/model/spark/describe_text_spark.py +25 -0
- data_profiling/model/spark/duplicates_spark.py +54 -0
- data_profiling/model/spark/missing_spark.py +96 -0
- data_profiling/model/spark/sample_spark.py +43 -0
- data_profiling/model/spark/summary_spark.py +95 -0
- data_profiling/model/spark/table_spark.py +58 -0
- data_profiling/model/spark/timeseries_index_spark.py +12 -0
- data_profiling/model/summarizer.py +207 -0
- data_profiling/model/summary.py +66 -0
- data_profiling/model/summary_algorithms.py +276 -0
- data_profiling/model/table.py +10 -0
- data_profiling/model/timeseries_index.py +16 -0
- data_profiling/model/typeset.py +365 -0
- data_profiling/model/typeset_relations.py +143 -0
- data_profiling/profile_report.py +573 -0
- data_profiling/report/__init__.py +4 -0
- data_profiling/report/formatters.py +346 -0
- data_profiling/report/presentation/__init__.py +1 -0
- data_profiling/report/presentation/core/__init__.py +39 -0
- data_profiling/report/presentation/core/alerts.py +18 -0
- data_profiling/report/presentation/core/collapse.py +24 -0
- data_profiling/report/presentation/core/container.py +50 -0
- data_profiling/report/presentation/core/correlation_table.py +21 -0
- data_profiling/report/presentation/core/dropdown.py +44 -0
- data_profiling/report/presentation/core/duplicate.py +16 -0
- data_profiling/report/presentation/core/frequency_table.py +14 -0
- data_profiling/report/presentation/core/frequency_table_small.py +16 -0
- data_profiling/report/presentation/core/html.py +14 -0
- data_profiling/report/presentation/core/image.py +34 -0
- data_profiling/report/presentation/core/item_renderer.py +17 -0
- data_profiling/report/presentation/core/renderable.py +42 -0
- data_profiling/report/presentation/core/root.py +35 -0
- data_profiling/report/presentation/core/sample.py +20 -0
- data_profiling/report/presentation/core/scores.py +32 -0
- data_profiling/report/presentation/core/table.py +26 -0
- data_profiling/report/presentation/core/toggle_button.py +14 -0
- data_profiling/report/presentation/core/variable.py +40 -0
- data_profiling/report/presentation/core/variable_info.py +36 -0
- data_profiling/report/presentation/flavours/__init__.py +9 -0
- data_profiling/report/presentation/flavours/flavour_html.py +64 -0
- data_profiling/report/presentation/flavours/flavour_widget.py +61 -0
- data_profiling/report/presentation/flavours/flavours.py +43 -0
- data_profiling/report/presentation/flavours/html/__init__.py +47 -0
- data_profiling/report/presentation/flavours/html/alerts.py +10 -0
- data_profiling/report/presentation/flavours/html/collapse.py +7 -0
- data_profiling/report/presentation/flavours/html/container.py +58 -0
- data_profiling/report/presentation/flavours/html/correlation_table.py +13 -0
- data_profiling/report/presentation/flavours/html/dropdown.py +7 -0
- data_profiling/report/presentation/flavours/html/duplicate.py +24 -0
- data_profiling/report/presentation/flavours/html/frequency_table.py +20 -0
- data_profiling/report/presentation/flavours/html/frequency_table_small.py +15 -0
- data_profiling/report/presentation/flavours/html/html.py +6 -0
- data_profiling/report/presentation/flavours/html/image.py +7 -0
- data_profiling/report/presentation/flavours/html/root.py +14 -0
- data_profiling/report/presentation/flavours/html/sample.py +12 -0
- data_profiling/report/presentation/flavours/html/scores.py +11 -0
- data_profiling/report/presentation/flavours/html/table.py +7 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_constant_length.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_dirty_category.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_duplicates.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_empty.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_cardinality.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_high_correlation.html +4 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_imbalance.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_infinite.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_missing.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_near_duplicates.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_non_stationary.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_seasonal.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_skewed.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_truncated.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_type_date.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_uniform.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_unique.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_unsupported.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts/alert_zeros.html +1 -0
- data_profiling/report/presentation/flavours/html/templates/alerts.html +47 -0
- data_profiling/report/presentation/flavours/html/templates/collapse.html +11 -0
- data_profiling/report/presentation/flavours/html/templates/correlation_table.html +5 -0
- data_profiling/report/presentation/flavours/html/templates/diagram.html +11 -0
- data_profiling/report/presentation/flavours/html/templates/dropdown.html +16 -0
- data_profiling/report/presentation/flavours/html/templates/duplicate.html +5 -0
- data_profiling/report/presentation/flavours/html/templates/frequency_table.html +45 -0
- data_profiling/report/presentation/flavours/html/templates/frequency_table_small.html +34 -0
- data_profiling/report/presentation/flavours/html/templates/report.html +26 -0
- data_profiling/report/presentation/flavours/html/templates/sample.html +10 -0
- data_profiling/report/presentation/flavours/html/templates/scores.html +78 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/batch_grid.html +16 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/grid.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/list.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/named_list.html +8 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/overview_tabs.html +30 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/scores.html +3 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/sections.html +13 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/select.html +40 -0
- data_profiling/report/presentation/flavours/html/templates/sequence/tabs.html +30 -0
- data_profiling/report/presentation/flavours/html/templates/table.html +38 -0
- data_profiling/report/presentation/flavours/html/templates/toggle_button.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/variable.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/variable_info.html +49 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.bundle.min.js +7 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/bootstrap.min.css +6 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/cosmo.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/flatly.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/script.js +52 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/simplex.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/style.css +253 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/assets/united.bootstrap.min.css +12 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/footer.html +7 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/javascript.html +18 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/navigation.html +36 -0
- data_profiling/report/presentation/flavours/html/templates/wrapper/style.html +53 -0
- data_profiling/report/presentation/flavours/html/templates.py +76 -0
- data_profiling/report/presentation/flavours/html/toggle_button.py +7 -0
- data_profiling/report/presentation/flavours/html/variable.py +7 -0
- data_profiling/report/presentation/flavours/html/variable_info.py +7 -0
- data_profiling/report/presentation/flavours/widget/__init__.py +49 -0
- data_profiling/report/presentation/flavours/widget/alerts.py +45 -0
- data_profiling/report/presentation/flavours/widget/collapse.py +43 -0
- data_profiling/report/presentation/flavours/widget/container.py +121 -0
- data_profiling/report/presentation/flavours/widget/correlation_table.py +14 -0
- data_profiling/report/presentation/flavours/widget/dropdown.py +31 -0
- data_profiling/report/presentation/flavours/widget/duplicate.py +14 -0
- data_profiling/report/presentation/flavours/widget/frequency_table.py +57 -0
- data_profiling/report/presentation/flavours/widget/frequency_table_small.py +66 -0
- data_profiling/report/presentation/flavours/widget/html.py +11 -0
- data_profiling/report/presentation/flavours/widget/image.py +26 -0
- data_profiling/report/presentation/flavours/widget/notebook.py +81 -0
- data_profiling/report/presentation/flavours/widget/root.py +10 -0
- data_profiling/report/presentation/flavours/widget/sample.py +14 -0
- data_profiling/report/presentation/flavours/widget/table.py +30 -0
- data_profiling/report/presentation/flavours/widget/toggle_button.py +17 -0
- data_profiling/report/presentation/flavours/widget/variable.py +12 -0
- data_profiling/report/presentation/flavours/widget/variable_info.py +11 -0
- data_profiling/report/presentation/frequency_table_utils.py +141 -0
- data_profiling/report/structure/__init__.py +1 -0
- data_profiling/report/structure/correlations.py +123 -0
- data_profiling/report/structure/overview.py +376 -0
- data_profiling/report/structure/report.py +457 -0
- data_profiling/report/structure/variables/__init__.py +35 -0
- data_profiling/report/structure/variables/render_boolean.py +132 -0
- data_profiling/report/structure/variables/render_categorical.py +566 -0
- data_profiling/report/structure/variables/render_common.py +31 -0
- data_profiling/report/structure/variables/render_complex.py +102 -0
- data_profiling/report/structure/variables/render_count.py +172 -0
- data_profiling/report/structure/variables/render_date.py +143 -0
- data_profiling/report/structure/variables/render_file.py +70 -0
- data_profiling/report/structure/variables/render_generic.py +45 -0
- data_profiling/report/structure/variables/render_image.py +204 -0
- data_profiling/report/structure/variables/render_path.py +134 -0
- data_profiling/report/structure/variables/render_real.py +314 -0
- data_profiling/report/structure/variables/render_text.py +189 -0
- data_profiling/report/structure/variables/render_timeseries.py +371 -0
- data_profiling/report/structure/variables/render_url.py +132 -0
- data_profiling/report/utils.py +34 -0
- data_profiling/serialize_report.py +143 -0
- data_profiling/utils/__init__.py +1 -0
- data_profiling/utils/backend.py +9 -0
- data_profiling/utils/cache.py +59 -0
- data_profiling/utils/common.py +142 -0
- data_profiling/utils/compat.py +31 -0
- data_profiling/utils/dataframe.py +238 -0
- data_profiling/utils/logger.py +53 -0
- data_profiling/utils/notebook.py +8 -0
- data_profiling/utils/paths.py +45 -0
- data_profiling/utils/progress_bar.py +15 -0
- data_profiling/utils/styles.py +22 -0
- data_profiling/utils/versions.py +19 -0
- data_profiling/version.py +1 -0
- data_profiling/visualisation/__init__.py +1 -0
- data_profiling/visualisation/context.py +87 -0
- data_profiling/visualisation/missing.py +138 -0
- data_profiling/visualisation/plot.py +1158 -0
- data_profiling/visualisation/utils.py +113 -0
- fg_data_profiling-4.19.0.dist-info/METADATA +362 -0
- fg_data_profiling-4.19.0.dist-info/RECORD +238 -0
- fg_data_profiling-4.19.0.dist-info/WHEEL +6 -0
- fg_data_profiling-4.19.0.dist-info/entry_points.txt +3 -0
- fg_data_profiling-4.19.0.dist-info/licenses/LICENSE +21 -0
- fg_data_profiling-4.19.0.dist-info/top_level.txt +2 -0
- ydata_profiling/__init__.py +43 -0
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
# mypy: ignore-errors
|
|
2
|
+
|
|
3
|
+
"""Correlations between variables."""
|
|
4
|
+
|
|
5
|
+
import warnings
|
|
6
|
+
from typing import Dict, List, Optional, Sized, no_type_check
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
import pandas as pd
|
|
10
|
+
|
|
11
|
+
from data_profiling.config import Settings
|
|
12
|
+
|
|
13
|
+
try:
|
|
14
|
+
from pandas.core.base import DataError
|
|
15
|
+
except ImportError:
|
|
16
|
+
from pandas.errors import DataError
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class CorrelationBackend:
|
|
20
|
+
"""Helper class to select and cache the appropriate correlation backend (Pandas or Spark)."""
|
|
21
|
+
|
|
22
|
+
@no_type_check
|
|
23
|
+
def __init__(self, df: Sized):
|
|
24
|
+
"""Determine backend once and store it for all correlation computations."""
|
|
25
|
+
if isinstance(df, pd.DataFrame):
|
|
26
|
+
from data_profiling.model.pandas import (
|
|
27
|
+
correlations_pandas as correlation_backend, # type: ignore
|
|
28
|
+
)
|
|
29
|
+
else:
|
|
30
|
+
from data_profiling.model.spark import (
|
|
31
|
+
correlations_spark as correlation_backend, # type: ignore
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
self.backend = correlation_backend
|
|
35
|
+
|
|
36
|
+
def get_method(self, method_name: str): # noqa: ANN201
|
|
37
|
+
"""Retrieve the appropriate correlation method class from the backend."""
|
|
38
|
+
if hasattr(self.backend, method_name):
|
|
39
|
+
return getattr(self.backend, method_name)
|
|
40
|
+
raise AttributeError(
|
|
41
|
+
f"Correlation method '{method_name}' is not available in the backend."
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class Correlation:
|
|
46
|
+
_method_name: str = ""
|
|
47
|
+
|
|
48
|
+
def compute(
|
|
49
|
+
self, config: Settings, df: Sized, summary: dict, backend: CorrelationBackend
|
|
50
|
+
) -> Optional[Sized]:
|
|
51
|
+
"""Computes correlation using the correct backend (Pandas or Spark)."""
|
|
52
|
+
try:
|
|
53
|
+
method = backend.get_method(self._method_name)
|
|
54
|
+
except AttributeError as ex:
|
|
55
|
+
raise NotImplementedError() from ex
|
|
56
|
+
else:
|
|
57
|
+
return method(config, df, summary)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class Auto(Correlation):
|
|
61
|
+
"""Automatically selects the appropriate correlation method based on the DataFrame type."""
|
|
62
|
+
|
|
63
|
+
_method_name = "auto_compute"
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class Spearman(Correlation):
|
|
67
|
+
_method_name = "spearman_compute"
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class Pearson(Correlation):
|
|
71
|
+
_method_name = "pearson_compute"
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class Kendall(Correlation):
|
|
75
|
+
_method_name = "kendall_compute"
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class Cramers(Correlation):
|
|
79
|
+
_method_name = "cramers_compute"
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class PhiK(Correlation):
|
|
83
|
+
_method_name = "phik_compute"
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def warn_correlation(correlation_name: str, error: str) -> None:
|
|
87
|
+
warnings.warn(
|
|
88
|
+
f"""There was an attempt to calculate the {correlation_name} correlation, but this failed.
|
|
89
|
+
To hide this warning, disable the calculation
|
|
90
|
+
(using `df.profile_report(correlations={{\"{correlation_name}\": {{\"calculate\": False}}}})`
|
|
91
|
+
If this is problematic for your use case, please report this as an issue:
|
|
92
|
+
https://github.com/Data-Centric-AI-Community/data-profiling/issues
|
|
93
|
+
(include the error message: '{error}')"""
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def calculate_correlation(
|
|
98
|
+
config: Settings, df: Sized, correlation_name: str, summary: dict
|
|
99
|
+
) -> Optional[Sized]:
|
|
100
|
+
"""Calculate the correlation coefficients between variables for the correlation types selected in the config
|
|
101
|
+
(auto, pearson, spearman, kendall, phi_k, cramers).
|
|
102
|
+
|
|
103
|
+
Args:
|
|
104
|
+
config: report Settings object
|
|
105
|
+
df: The DataFrame with variables.
|
|
106
|
+
correlation_name:
|
|
107
|
+
summary: summary dictionary
|
|
108
|
+
|
|
109
|
+
Returns:
|
|
110
|
+
The correlation matrices for the given correlation measures. Return None if correlation is empty.
|
|
111
|
+
"""
|
|
112
|
+
backend = CorrelationBackend(df)
|
|
113
|
+
|
|
114
|
+
correlation_measures = {
|
|
115
|
+
"auto": Auto,
|
|
116
|
+
"pearson": Pearson,
|
|
117
|
+
"spearman": Spearman,
|
|
118
|
+
"kendall": Kendall,
|
|
119
|
+
"cramers": Cramers,
|
|
120
|
+
"phi_k": PhiK,
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
correlation = None
|
|
124
|
+
try:
|
|
125
|
+
correlation = correlation_measures[correlation_name]().compute(
|
|
126
|
+
config, df, summary, backend
|
|
127
|
+
)
|
|
128
|
+
except (ValueError, AssertionError, TypeError, DataError, IndexError) as e:
|
|
129
|
+
warn_correlation(correlation_name, str(e))
|
|
130
|
+
|
|
131
|
+
return correlation if correlation is not None and len(correlation) > 0 else None
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def perform_check_correlation(
|
|
135
|
+
correlation_matrix: pd.DataFrame, threshold: float
|
|
136
|
+
) -> Dict[str, List[str]]:
|
|
137
|
+
"""Check whether selected variables are highly correlated values in the correlation matrix.
|
|
138
|
+
|
|
139
|
+
Args:
|
|
140
|
+
correlation_matrix: The correlation matrix for the DataFrame.
|
|
141
|
+
threshold:.
|
|
142
|
+
|
|
143
|
+
Returns:
|
|
144
|
+
The variables that are highly correlated.
|
|
145
|
+
"""
|
|
146
|
+
|
|
147
|
+
cols = correlation_matrix.columns
|
|
148
|
+
bool_index = abs(correlation_matrix.values) >= threshold
|
|
149
|
+
np.fill_diagonal(bool_index, False)
|
|
150
|
+
return {
|
|
151
|
+
col: cols[bool_index[i]].values.tolist()
|
|
152
|
+
for i, col in enumerate(cols)
|
|
153
|
+
if any(bool_index[i])
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def get_active_correlations(config: Settings) -> List[str]:
|
|
158
|
+
correlation_names = [
|
|
159
|
+
correlation_name
|
|
160
|
+
for correlation_name in config.correlations.keys()
|
|
161
|
+
if config.correlations[correlation_name].calculate
|
|
162
|
+
]
|
|
163
|
+
return correlation_names
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import importlib
|
|
2
|
+
from typing import Any
|
|
3
|
+
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
from data_profiling.config import Settings
|
|
7
|
+
from data_profiling.model.pandas.dataframe_pandas import pandas_preprocess
|
|
8
|
+
|
|
9
|
+
spec = importlib.util.find_spec("pyspark")
|
|
10
|
+
if spec is None:
|
|
11
|
+
from typing import TypeVar
|
|
12
|
+
|
|
13
|
+
sparkDataFrame = TypeVar("sparkDataFrame")
|
|
14
|
+
else:
|
|
15
|
+
from pyspark.sql import DataFrame as sparkDataFrame # type: ignore
|
|
16
|
+
|
|
17
|
+
from data_profiling.model.spark.dataframe_spark import spark_preprocess
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def preprocess(config: Settings, df: Any) -> Any:
|
|
21
|
+
"""
|
|
22
|
+
Search for invalid columns datatypes as well as ensures column names follow the expected rules
|
|
23
|
+
Args:
|
|
24
|
+
config: ydataprofiling Settings class
|
|
25
|
+
df: a pandas or spark dataframe
|
|
26
|
+
|
|
27
|
+
Returns: a pandas or spark dataframe
|
|
28
|
+
"""
|
|
29
|
+
if isinstance(df, pd.DataFrame):
|
|
30
|
+
df = pandas_preprocess(config=config, df=df)
|
|
31
|
+
elif isinstance(df, sparkDataFrame): # type: ignore
|
|
32
|
+
df = spark_preprocess(config=config, df=df)
|
|
33
|
+
else:
|
|
34
|
+
return NotImplementedError()
|
|
35
|
+
return df
|
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""Organize the calculation of statistics for each series in this DataFrame."""
|
|
2
|
+
from datetime import datetime
|
|
3
|
+
from typing import Any, Dict, Optional, Union
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
from tqdm.auto import tqdm
|
|
7
|
+
from visions import VisionsTypeset
|
|
8
|
+
|
|
9
|
+
from data_profiling.config import Settings
|
|
10
|
+
from data_profiling.model import BaseAnalysis, BaseDescription
|
|
11
|
+
from data_profiling.model.alerts import get_alerts
|
|
12
|
+
from data_profiling.model.correlations import (
|
|
13
|
+
calculate_correlation,
|
|
14
|
+
get_active_correlations,
|
|
15
|
+
)
|
|
16
|
+
from data_profiling.model.dataframe import preprocess
|
|
17
|
+
from data_profiling.model.description import TimeIndexAnalysis
|
|
18
|
+
from data_profiling.model.duplicates import get_duplicates
|
|
19
|
+
from data_profiling.model.missing import get_missing_active, get_missing_diagram
|
|
20
|
+
from data_profiling.model.pairwise import get_scatter_plot, get_scatter_tasks
|
|
21
|
+
from data_profiling.model.sample import get_custom_sample, get_sample
|
|
22
|
+
from data_profiling.model.summarizer import BaseSummarizer
|
|
23
|
+
from data_profiling.model.summary import get_series_descriptions
|
|
24
|
+
from data_profiling.model.table import get_table_stats
|
|
25
|
+
from data_profiling.model.timeseries_index import get_time_index_description
|
|
26
|
+
from data_profiling.utils.progress_bar import progress
|
|
27
|
+
from data_profiling.version import __version__
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def describe(
|
|
31
|
+
config: Settings,
|
|
32
|
+
df: Union[pd.DataFrame, "pyspark.sql.DataFrame"], # type: ignore[name-defined] # noqa: F821
|
|
33
|
+
summarizer: BaseSummarizer,
|
|
34
|
+
typeset: VisionsTypeset,
|
|
35
|
+
sample: Optional[dict] = None,
|
|
36
|
+
) -> BaseDescription: # noqa: TC301
|
|
37
|
+
"""Calculate the statistics for each series in this DataFrame.
|
|
38
|
+
|
|
39
|
+
Args:
|
|
40
|
+
config: report Settings object
|
|
41
|
+
df: DataFrame.
|
|
42
|
+
summarizer: summarizer object
|
|
43
|
+
typeset: visions typeset
|
|
44
|
+
sample: optional, dict with custom sample
|
|
45
|
+
|
|
46
|
+
Returns:
|
|
47
|
+
This function returns a dictionary containing:
|
|
48
|
+
- table: overall statistics.
|
|
49
|
+
- variables: descriptions per series.
|
|
50
|
+
- correlations: correlation matrices.
|
|
51
|
+
- missing: missing value diagrams.
|
|
52
|
+
- alerts: direct special attention to these patterns in your data.
|
|
53
|
+
- package: package details.
|
|
54
|
+
"""
|
|
55
|
+
# ** Validate Input types **
|
|
56
|
+
if not isinstance(config, Settings):
|
|
57
|
+
raise TypeError(f"`config` must be of type `Settings`, got {type(config)}")
|
|
58
|
+
|
|
59
|
+
# Validate df input type
|
|
60
|
+
|
|
61
|
+
if not isinstance(df, pd.DataFrame):
|
|
62
|
+
try:
|
|
63
|
+
from pyspark.sql import DataFrame as SparkDataFrame # type: ignore
|
|
64
|
+
|
|
65
|
+
if not isinstance(df, SparkDataFrame): # noqa: TC301
|
|
66
|
+
raise TypeError( # noqa: TC301
|
|
67
|
+
f"`df` must be either a `pandas.DataFrame` or a `pyspark.sql.DataFrame`, but got {type(df)}."
|
|
68
|
+
)
|
|
69
|
+
except ImportError as ex:
|
|
70
|
+
raise TypeError(
|
|
71
|
+
f"`df must be either a `pandas.DataFrame` or a `pyspark.sql.DataFrame`, but got {type(df)}."
|
|
72
|
+
f"If using Spark, make sure PySpark is installed."
|
|
73
|
+
) from ex
|
|
74
|
+
|
|
75
|
+
df = preprocess(config, df)
|
|
76
|
+
|
|
77
|
+
number_of_tasks = 5
|
|
78
|
+
|
|
79
|
+
with tqdm(
|
|
80
|
+
total=number_of_tasks,
|
|
81
|
+
desc="Summarize dataset",
|
|
82
|
+
disable=not config.progress_bar,
|
|
83
|
+
position=0,
|
|
84
|
+
) as pbar:
|
|
85
|
+
date_start = datetime.utcnow()
|
|
86
|
+
|
|
87
|
+
# Variable-specific
|
|
88
|
+
pbar.total += len(df.columns)
|
|
89
|
+
series_description = get_series_descriptions(
|
|
90
|
+
config, df, summarizer, typeset, pbar
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
pbar.set_postfix_str("Get variable types")
|
|
94
|
+
pbar.total += 1
|
|
95
|
+
variables = {
|
|
96
|
+
column: description["type"]
|
|
97
|
+
for column, description in series_description.items()
|
|
98
|
+
}
|
|
99
|
+
supported_columns = [
|
|
100
|
+
column
|
|
101
|
+
for column, type_name in variables.items()
|
|
102
|
+
if type_name != "Unsupported"
|
|
103
|
+
]
|
|
104
|
+
interval_columns = [
|
|
105
|
+
column
|
|
106
|
+
for column, type_name in variables.items()
|
|
107
|
+
if type_name in {"Numeric", "TimeSeries"}
|
|
108
|
+
]
|
|
109
|
+
pbar.update()
|
|
110
|
+
|
|
111
|
+
# Table statistics
|
|
112
|
+
table_stats = progress(get_table_stats, pbar, "Get dataframe statistics")(
|
|
113
|
+
config, df, series_description
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
# Get correlations
|
|
117
|
+
if table_stats["n"] != 0:
|
|
118
|
+
correlation_names = get_active_correlations(config)
|
|
119
|
+
pbar.total += len(correlation_names)
|
|
120
|
+
|
|
121
|
+
correlations = {
|
|
122
|
+
correlation_name: progress(
|
|
123
|
+
calculate_correlation,
|
|
124
|
+
pbar,
|
|
125
|
+
f"Calculate {correlation_name} correlation",
|
|
126
|
+
)(config, df, correlation_name, series_description)
|
|
127
|
+
for correlation_name in correlation_names
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
# make sure correlations is not None
|
|
131
|
+
correlations = {
|
|
132
|
+
key: value for key, value in correlations.items() if value is not None
|
|
133
|
+
}
|
|
134
|
+
else:
|
|
135
|
+
correlations = {}
|
|
136
|
+
|
|
137
|
+
# Scatter matrix
|
|
138
|
+
pbar.set_postfix_str("Get scatter matrix")
|
|
139
|
+
scatter_tasks = get_scatter_tasks(config, interval_columns)
|
|
140
|
+
pbar.total += len(scatter_tasks)
|
|
141
|
+
scatter_matrix: Dict[Any, Dict[Any, Any]] = {
|
|
142
|
+
x: {y: None} for x, y in scatter_tasks
|
|
143
|
+
}
|
|
144
|
+
for x, y in scatter_tasks:
|
|
145
|
+
scatter_matrix[x][y] = progress(
|
|
146
|
+
get_scatter_plot, pbar, f"scatter {x}, {y}"
|
|
147
|
+
)(config, df, x, y, interval_columns)
|
|
148
|
+
|
|
149
|
+
# missing diagrams
|
|
150
|
+
missing_map = get_missing_active(config, table_stats)
|
|
151
|
+
pbar.total += len(missing_map)
|
|
152
|
+
missing = {
|
|
153
|
+
name: progress(get_missing_diagram, pbar, f"Missing diagram {name}")(
|
|
154
|
+
config, df, settings
|
|
155
|
+
)
|
|
156
|
+
for name, settings in missing_map.items()
|
|
157
|
+
}
|
|
158
|
+
missing = {name: value for name, value in missing.items() if value is not None}
|
|
159
|
+
|
|
160
|
+
# Sample
|
|
161
|
+
pbar.set_postfix_str("Take sample")
|
|
162
|
+
if sample is None:
|
|
163
|
+
samples = get_sample(config, df)
|
|
164
|
+
else:
|
|
165
|
+
samples = get_custom_sample(sample)
|
|
166
|
+
pbar.update()
|
|
167
|
+
|
|
168
|
+
# Duplicates
|
|
169
|
+
metrics, duplicates = progress(get_duplicates, pbar, "Detecting duplicates")(
|
|
170
|
+
config, df, supported_columns
|
|
171
|
+
)
|
|
172
|
+
table_stats.update(metrics)
|
|
173
|
+
|
|
174
|
+
alerts = progress(get_alerts, pbar, "Get alerts")(
|
|
175
|
+
config, table_stats, series_description, correlations
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
if config.vars.timeseries.active:
|
|
179
|
+
tsindex_description = get_time_index_description(config, df, table_stats)
|
|
180
|
+
|
|
181
|
+
pbar.set_postfix_str("Get reproduction details")
|
|
182
|
+
package = {
|
|
183
|
+
"data_profiling_version": __version__,
|
|
184
|
+
"data_profiling_config": config.json(),
|
|
185
|
+
}
|
|
186
|
+
pbar.update()
|
|
187
|
+
|
|
188
|
+
pbar.set_postfix_str("Completed")
|
|
189
|
+
|
|
190
|
+
date_end = datetime.utcnow()
|
|
191
|
+
|
|
192
|
+
analysis = BaseAnalysis(config.title, date_start, date_end)
|
|
193
|
+
time_index_analysis = None
|
|
194
|
+
if config.vars.timeseries.active and tsindex_description:
|
|
195
|
+
time_index_analysis = TimeIndexAnalysis(**tsindex_description)
|
|
196
|
+
|
|
197
|
+
description = BaseDescription(
|
|
198
|
+
analysis=analysis,
|
|
199
|
+
time_index_analysis=time_index_analysis,
|
|
200
|
+
table=table_stats,
|
|
201
|
+
variables=series_description,
|
|
202
|
+
scatter=scatter_matrix,
|
|
203
|
+
correlations=correlations,
|
|
204
|
+
missing=missing,
|
|
205
|
+
alerts=alerts,
|
|
206
|
+
package=package,
|
|
207
|
+
sample=samples,
|
|
208
|
+
duplicates=duplicates,
|
|
209
|
+
)
|
|
210
|
+
return description
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
from datetime import datetime, timedelta
|
|
3
|
+
from typing import Any, Dict, List, Optional, Union
|
|
4
|
+
|
|
5
|
+
from pandas import Timedelta
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass
|
|
9
|
+
class BaseAnalysis:
|
|
10
|
+
"""Description of base analysis module of report.
|
|
11
|
+
Overall info about report.
|
|
12
|
+
|
|
13
|
+
Attributes
|
|
14
|
+
title (str): Title of report.
|
|
15
|
+
date_start (Union[datetime, List[datetime]]): Start of generating description.
|
|
16
|
+
date_end (Union[datetime, List[datetime]]): End of generating description.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
title: str
|
|
20
|
+
date_start: Union[datetime, List[datetime]]
|
|
21
|
+
date_end: Union[datetime, List[datetime]]
|
|
22
|
+
|
|
23
|
+
def __init__(self, title: str, date_start: datetime, date_end: datetime) -> None:
|
|
24
|
+
self.title = title
|
|
25
|
+
self.date_start = date_start
|
|
26
|
+
self.date_end = date_end
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
def duration(self) -> Union[timedelta, List[timedelta]]:
|
|
30
|
+
if isinstance(self.date_start, datetime) and isinstance(
|
|
31
|
+
self.date_end, datetime
|
|
32
|
+
):
|
|
33
|
+
return self.date_end - self.date_start
|
|
34
|
+
if isinstance(self.date_start, list) and isinstance(self.date_end, list):
|
|
35
|
+
return [
|
|
36
|
+
self.date_end[i] - self.date_start[i]
|
|
37
|
+
for i in range(len(self.date_start))
|
|
38
|
+
]
|
|
39
|
+
else:
|
|
40
|
+
raise TypeError()
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass
|
|
44
|
+
class TimeIndexAnalysis:
|
|
45
|
+
"""Description of timeseries index analysis module of report.
|
|
46
|
+
|
|
47
|
+
Attributes:
|
|
48
|
+
n_series (Union[int, List[int]): Number of time series identified in the dataset.
|
|
49
|
+
length (Union[int, List[int]): Number of data points in the time series.
|
|
50
|
+
start (Any): Starting point of the time series.
|
|
51
|
+
end (Any): Ending point of the time series.
|
|
52
|
+
period (Union[float, List[float]): Average interval between data points in the time series.
|
|
53
|
+
frequency (Union[Optional[str], List[Optional[str]]): A string alias given to useful common time series frequencies, e.g. H - hours.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
n_series: Union[int, List[int]]
|
|
57
|
+
length: Union[int, List[int]]
|
|
58
|
+
start: Any
|
|
59
|
+
end: Any
|
|
60
|
+
period: Union[float, List[float], Timedelta, List[Timedelta]]
|
|
61
|
+
frequency: Union[Optional[str], List[Optional[str]]]
|
|
62
|
+
|
|
63
|
+
def __init__(
|
|
64
|
+
self,
|
|
65
|
+
n_series: int,
|
|
66
|
+
length: int,
|
|
67
|
+
start: Any,
|
|
68
|
+
end: Any,
|
|
69
|
+
period: float,
|
|
70
|
+
frequency: Optional[str] = None,
|
|
71
|
+
) -> None:
|
|
72
|
+
self.n_series = n_series
|
|
73
|
+
self.length = length
|
|
74
|
+
self.start = start
|
|
75
|
+
self.end = end
|
|
76
|
+
self.period = period
|
|
77
|
+
self.frequency = frequency
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass
|
|
81
|
+
class BaseDescription:
|
|
82
|
+
"""Description of DataFrame.
|
|
83
|
+
|
|
84
|
+
Attributes:
|
|
85
|
+
analysis (BaseAnalysis): Base info about report. Title, start time and end time of description generating.
|
|
86
|
+
time_index_analysis (Optional[TimeIndexAnalysis]): Description of timeseries index analysis module of report.
|
|
87
|
+
table (Any): DataFrame statistic. Base information about DataFrame.
|
|
88
|
+
variables (Dict[str, Any]): Description of variables (columns) of DataFrame. Key is column name, value is description dictionary.
|
|
89
|
+
scatter (Any): Pairwise scatter for all variables. Plot interactions between variables.
|
|
90
|
+
correlations (Dict[str, Any]): Prepare correlation matrix for DataFrame
|
|
91
|
+
missing (Dict[str, Any]): Describe missing values.
|
|
92
|
+
alerts (Any): Take alerts from all modules (variables, scatter, correlations), and group them.
|
|
93
|
+
package (Dict[str, Any]): Contains version of data-profiling and config.
|
|
94
|
+
sample (Any): Sample of data.
|
|
95
|
+
duplicates (Any): Description of duplicates.
|
|
96
|
+
"""
|
|
97
|
+
|
|
98
|
+
analysis: BaseAnalysis
|
|
99
|
+
time_index_analysis: Optional[TimeIndexAnalysis]
|
|
100
|
+
table: Any
|
|
101
|
+
variables: Dict[str, Any]
|
|
102
|
+
scatter: Any
|
|
103
|
+
correlations: Dict[str, Any]
|
|
104
|
+
missing: Dict[str, Any]
|
|
105
|
+
alerts: Any
|
|
106
|
+
package: Dict[str, Any]
|
|
107
|
+
sample: Any
|
|
108
|
+
duplicates: Any
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
from typing import Any, Dict, Optional, Sequence, Tuple, TypeVar
|
|
2
|
+
|
|
3
|
+
from multimethod import multimethod
|
|
4
|
+
|
|
5
|
+
from data_profiling.config import Settings
|
|
6
|
+
|
|
7
|
+
T = TypeVar("T")
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@multimethod
|
|
11
|
+
def get_duplicates(
|
|
12
|
+
config: Settings, df: T, supported_columns: Sequence
|
|
13
|
+
) -> Tuple[Dict[str, Any], Optional[T]]:
|
|
14
|
+
raise NotImplementedError()
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
from typing import Any, Tuple
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def generic_expectations(
|
|
5
|
+
name: str, summary: dict, batch: Any, *args
|
|
6
|
+
) -> Tuple[str, dict, Any]:
|
|
7
|
+
batch.expect_column_to_exist(name)
|
|
8
|
+
|
|
9
|
+
if summary["n_missing"] == 0:
|
|
10
|
+
batch.expect_column_values_to_not_be_null(name)
|
|
11
|
+
|
|
12
|
+
if summary["p_unique"] == 1.0:
|
|
13
|
+
batch.expect_column_values_to_be_unique(name)
|
|
14
|
+
|
|
15
|
+
return name, summary, batch
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def numeric_expectations(
|
|
19
|
+
name: str, summary: dict, batch: Any, *args
|
|
20
|
+
) -> Tuple[str, dict, Any]:
|
|
21
|
+
from great_expectations.profile.base import ProfilerTypeMapping
|
|
22
|
+
|
|
23
|
+
numeric_type_names = (
|
|
24
|
+
ProfilerTypeMapping.INT_TYPE_NAMES + ProfilerTypeMapping.FLOAT_TYPE_NAMES
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
batch.expect_column_values_to_be_in_type_list(
|
|
28
|
+
name,
|
|
29
|
+
numeric_type_names,
|
|
30
|
+
meta={
|
|
31
|
+
"notes": {
|
|
32
|
+
"format": "markdown",
|
|
33
|
+
"content": [
|
|
34
|
+
"The column values should be stored in one of these types."
|
|
35
|
+
],
|
|
36
|
+
}
|
|
37
|
+
},
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
if summary["monotonic_increase"]:
|
|
41
|
+
batch.expect_column_values_to_be_increasing(
|
|
42
|
+
name, strictly=summary["monotonic_increase_strict"]
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
if summary["monotonic_decrease"]:
|
|
46
|
+
batch.expect_column_values_to_be_decreasing(
|
|
47
|
+
name, strictly=summary["monotonic_decrease_strict"]
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
if any(k in summary for k in ["min", "max"]):
|
|
51
|
+
batch.expect_column_values_to_be_between(
|
|
52
|
+
name, min_value=summary.get("min"), max_value=summary.get("max")
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
return name, summary, batch
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def categorical_expectations(
|
|
59
|
+
name: str, summary: dict, batch: Any, *args
|
|
60
|
+
) -> Tuple[str, dict, Any]:
|
|
61
|
+
# Use for both categorical and special case (boolean)
|
|
62
|
+
absolute_threshold = 10
|
|
63
|
+
relative_threshold = 0.2
|
|
64
|
+
if (
|
|
65
|
+
summary["n_distinct"] < absolute_threshold
|
|
66
|
+
or summary["p_distinct"] < relative_threshold
|
|
67
|
+
):
|
|
68
|
+
batch.expect_column_values_to_be_in_set(
|
|
69
|
+
name, set(summary["value_counts_without_nan"].keys())
|
|
70
|
+
)
|
|
71
|
+
return name, summary, batch
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def path_expectations(
|
|
75
|
+
name: str, summary: dict, batch: Any, *args
|
|
76
|
+
) -> Tuple[str, dict, Any]:
|
|
77
|
+
return name, summary, batch
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def datetime_expectations(
|
|
81
|
+
name: str, summary: dict, batch: Any, *args
|
|
82
|
+
) -> Tuple[str, dict, Any]:
|
|
83
|
+
if any(k in summary for k in ["min", "max"]):
|
|
84
|
+
batch.expect_column_values_to_be_between(
|
|
85
|
+
name,
|
|
86
|
+
min_value=summary.get("min"),
|
|
87
|
+
max_value=summary.get("max"),
|
|
88
|
+
parse_strings_as_datetimes=True,
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
return name, summary, batch
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def image_expectations(
|
|
95
|
+
name: str, summary: dict, batch: Any, *args
|
|
96
|
+
) -> Tuple[str, dict, Any]:
|
|
97
|
+
return name, summary, batch
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def url_expectations(
|
|
101
|
+
name: str, summary: dict, batch: Any, *args
|
|
102
|
+
) -> Tuple[str, dict, Any]:
|
|
103
|
+
return name, summary, batch
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def file_expectations(
|
|
107
|
+
name: str, summary: dict, batch: Any, *args
|
|
108
|
+
) -> Tuple[str, dict, Any]:
|
|
109
|
+
# By definition within our type logic, a file exists (as it's a path that also exists)
|
|
110
|
+
batch.expect_file_to_exist(name)
|
|
111
|
+
|
|
112
|
+
return name, summary, batch
|