datapilot-kit 0.3.0rc1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. datapilot_kit-0.3.0rc1/LICENSE +21 -0
  2. datapilot_kit-0.3.0rc1/MANIFEST.in +17 -0
  3. datapilot_kit-0.3.0rc1/PKG-INFO +103 -0
  4. datapilot_kit-0.3.0rc1/README.md +64 -0
  5. datapilot_kit-0.3.0rc1/datapilot/__init__.py +11 -0
  6. datapilot_kit-0.3.0rc1/datapilot/analysis/__init__.py +3 -0
  7. datapilot_kit-0.3.0rc1/datapilot/analysis/correlation.py +58 -0
  8. datapilot_kit-0.3.0rc1/datapilot/analysis/datatype.py +48 -0
  9. datapilot_kit-0.3.0rc1/datapilot/analysis/duplicate.py +40 -0
  10. datapilot_kit-0.3.0rc1/datapilot/analysis/health.py +163 -0
  11. datapilot_kit-0.3.0rc1/datapilot/analysis/insights.py +101 -0
  12. datapilot_kit-0.3.0rc1/datapilot/analysis/missing.py +56 -0
  13. datapilot_kit-0.3.0rc1/datapilot/analysis/models.py +108 -0
  14. datapilot_kit-0.3.0rc1/datapilot/analysis/outliers.py +85 -0
  15. datapilot_kit-0.3.0rc1/datapilot/analysis/statistics.py +46 -0
  16. datapilot_kit-0.3.0rc1/datapilot/analysis/summary.py +36 -0
  17. datapilot_kit-0.3.0rc1/datapilot/api/__init__.py +7 -0
  18. datapilot_kit-0.3.0rc1/datapilot/api/analyze.py +16 -0
  19. datapilot_kit-0.3.0rc1/datapilot/assets/style.css +1081 -0
  20. datapilot_kit-0.3.0rc1/datapilot/cli/__init__.py +3 -0
  21. datapilot_kit-0.3.0rc1/datapilot/core/__init__.py +3 -0
  22. datapilot_kit-0.3.0rc1/datapilot/core/loader.py +72 -0
  23. datapilot_kit-0.3.0rc1/datapilot/core/report.py +104 -0
  24. datapilot_kit-0.3.0rc1/datapilot/interpretation/__init__.py +3 -0
  25. datapilot_kit-0.3.0rc1/datapilot/llm/__init__.py +0 -0
  26. datapilot_kit-0.3.0rc1/datapilot/recommendation/__init__.py +0 -0
  27. datapilot_kit-0.3.0rc1/datapilot/reporting/__init__.py +3 -0
  28. datapilot_kit-0.3.0rc1/datapilot/reporting/fragments.py +329 -0
  29. datapilot_kit-0.3.0rc1/datapilot/reporting/html.py +51 -0
  30. datapilot_kit-0.3.0rc1/datapilot/reporting/placeholders.py +133 -0
  31. datapilot_kit-0.3.0rc1/datapilot/reporting/renderer.py +57 -0
  32. datapilot_kit-0.3.0rc1/datapilot/templates/report.html +528 -0
  33. datapilot_kit-0.3.0rc1/datapilot/ui/__init__.py +0 -0
  34. datapilot_kit-0.3.0rc1/datapilot/ui/cards.py +70 -0
  35. datapilot_kit-0.3.0rc1/datapilot/ui/console.py +12 -0
  36. datapilot_kit-0.3.0rc1/datapilot/ui/dashboard.py +90 -0
  37. datapilot_kit-0.3.0rc1/datapilot/ui/panels.py +16 -0
  38. datapilot_kit-0.3.0rc1/datapilot/ui/renderer.py +21 -0
  39. datapilot_kit-0.3.0rc1/datapilot/ui/sections.py +89 -0
  40. datapilot_kit-0.3.0rc1/datapilot/ui/tables.py +21 -0
  41. datapilot_kit-0.3.0rc1/datapilot/ui/theme.py +20 -0
  42. datapilot_kit-0.3.0rc1/datapilot/utils/__init__.py +3 -0
  43. datapilot_kit-0.3.0rc1/datapilot/visualization/__init__.py +0 -0
  44. datapilot_kit-0.3.0rc1/datapilot_kit.egg-info/PKG-INFO +103 -0
  45. datapilot_kit-0.3.0rc1/datapilot_kit.egg-info/SOURCES.txt +48 -0
  46. datapilot_kit-0.3.0rc1/datapilot_kit.egg-info/dependency_links.txt +1 -0
  47. datapilot_kit-0.3.0rc1/datapilot_kit.egg-info/requires.txt +14 -0
  48. datapilot_kit-0.3.0rc1/datapilot_kit.egg-info/top_level.txt +2 -0
  49. datapilot_kit-0.3.0rc1/pyproject.toml +91 -0
  50. datapilot_kit-0.3.0rc1/setup.cfg +4 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Avishkar Gopale
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,17 @@
1
+ include LICENSE
2
+ include README.md
3
+
4
+ graft datapilot
5
+
6
+ prune .venv
7
+ prune datapilot-test
8
+ prune build
9
+ prune dist
10
+ prune reports
11
+ prune datasets
12
+ prune docs
13
+ prune scripts
14
+ prune tests
15
+
16
+ global-exclude *.py[cod]
17
+ global-exclude __pycache__
@@ -0,0 +1,103 @@
1
+ Metadata-Version: 2.4
2
+ Name: datapilot-kit
3
+ Version: 0.3.0rc1
4
+ Summary: An open-source Python library for deterministic exploratory data analysis and dataset understanding.
5
+ Author: Avishkar D. Gopale
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/Aviii-3085/datapilot
8
+ Project-URL: Repository, https://github.com/Aviii-3085/datapilot
9
+ Project-URL: Issues, https://github.com/Aviii-3085/datapilot/issues
10
+ Project-URL: Documentation, https://github.com/Aviii-3085/datapilot/tree/main/docs
11
+ Keywords: eda,exploratory-data-analysis,data-analysis,data-science,dataset-profiling,machine-learning,analytics,python
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Intended Audience :: Education
16
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Operating System :: OS Independent
22
+ Requires-Python: >=3.10
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE
25
+ Requires-Dist: pandas>=2.2
26
+ Requires-Dist: numpy>=2.0
27
+ Requires-Dist: scipy>=1.14
28
+ Requires-Dist: matplotlib>=3.9
29
+ Requires-Dist: plotly>=6.0
30
+ Requires-Dist: jinja2>=3.1
31
+ Requires-Dist: rich>=14.0
32
+ Requires-Dist: typer>=0.16
33
+ Provides-Extra: dev
34
+ Requires-Dist: black; extra == "dev"
35
+ Requires-Dist: ruff; extra == "dev"
36
+ Requires-Dist: mypy; extra == "dev"
37
+ Requires-Dist: pytest; extra == "dev"
38
+ Dynamic: license-file
39
+
40
+ # Datapilot
41
+
42
+ > **The first step after loading your dataset.**
43
+
44
+ Datapilot is an open-source Python library that automates exploratory data analysis by generating deterministic dataset insights, health assessments, statistical summaries, and professional HTML reports.
45
+
46
+ Every data project begins with understanding the data. Datapilot helps you understand your dataset before building machine learning models, dashboards, or AI-powered applications.
47
+
48
+ ---
49
+
50
+ ## Installation
51
+
52
+ ```bash
53
+ pip install datapilot-kit
54
+ ```
55
+
56
+ ---
57
+
58
+ ## Quick Start
59
+
60
+ ```python
61
+ import pandas as pd
62
+
63
+ from datapilot import analyze
64
+
65
+ df = pd.read_csv("dataset.csv")
66
+
67
+ report = analyze(df)
68
+ ```
69
+
70
+ ---
71
+
72
+ ## Features
73
+
74
+ - Dataset Summary
75
+ - Dataset Health Score
76
+ - Missing Value Analysis
77
+ - Duplicate Detection
78
+ - Data Type Analysis
79
+ - Statistical Summaries
80
+ - Outlier Detection
81
+ - Correlation Analysis
82
+ - Actionable Insights
83
+ - Professional HTML Reports
84
+
85
+ ---
86
+
87
+ ## Documentation
88
+
89
+ Project documentation is available in the `docs/` directory.
90
+
91
+ - Vision
92
+ - User Journey
93
+ - API Philosophy
94
+ - API Review
95
+ - Roadmap
96
+
97
+ ---
98
+
99
+ ## Status
100
+
101
+ 🚧 Datapilot v0.3 — Public Preview (In Development)
102
+
103
+ The project is currently being prepared for its first public release.
@@ -0,0 +1,64 @@
1
+ # Datapilot
2
+
3
+ > **The first step after loading your dataset.**
4
+
5
+ Datapilot is an open-source Python library that automates exploratory data analysis by generating deterministic dataset insights, health assessments, statistical summaries, and professional HTML reports.
6
+
7
+ Every data project begins with understanding the data. Datapilot helps you understand your dataset before building machine learning models, dashboards, or AI-powered applications.
8
+
9
+ ---
10
+
11
+ ## Installation
12
+
13
+ ```bash
14
+ pip install datapilot-kit
15
+ ```
16
+
17
+ ---
18
+
19
+ ## Quick Start
20
+
21
+ ```python
22
+ import pandas as pd
23
+
24
+ from datapilot import analyze
25
+
26
+ df = pd.read_csv("dataset.csv")
27
+
28
+ report = analyze(df)
29
+ ```
30
+
31
+ ---
32
+
33
+ ## Features
34
+
35
+ - Dataset Summary
36
+ - Dataset Health Score
37
+ - Missing Value Analysis
38
+ - Duplicate Detection
39
+ - Data Type Analysis
40
+ - Statistical Summaries
41
+ - Outlier Detection
42
+ - Correlation Analysis
43
+ - Actionable Insights
44
+ - Professional HTML Reports
45
+
46
+ ---
47
+
48
+ ## Documentation
49
+
50
+ Project documentation is available in the `docs/` directory.
51
+
52
+ - Vision
53
+ - User Journey
54
+ - API Philosophy
55
+ - API Review
56
+ - Roadmap
57
+
58
+ ---
59
+
60
+ ## Status
61
+
62
+ 🚧 Datapilot v0.3 — Public Preview (In Development)
63
+
64
+ The project is currently being prepared for its first public release.
@@ -0,0 +1,11 @@
1
+ """
2
+ Datapilot
3
+
4
+ An intelligent Python library for exploratory data analysis.
5
+ """
6
+
7
+ __version__ = "0.3.0rc1"
8
+
9
+ from .api import analyze
10
+
11
+ __all__ = ["analyze"]
@@ -0,0 +1,3 @@
1
+ """
2
+ Analysis modules for Datapilot.
3
+ """
@@ -0,0 +1,58 @@
1
+ """
2
+ Correlation analysis for Datapilot.
3
+ """
4
+
5
+ from typing import SupportsFloat, cast
6
+
7
+ import pandas as pd
8
+
9
+ from .models import CorrelationSummary
10
+
11
+
12
+ def generate_correlation_summary(
13
+ dataframe: pd.DataFrame,
14
+ ) -> CorrelationSummary:
15
+ """
16
+ Generate Pearson correlation analysis for numeric columns.
17
+ """
18
+
19
+ numeric_dataframe = dataframe.select_dtypes(
20
+ include="number",
21
+ )
22
+
23
+ correlation_matrix = numeric_dataframe.corr()
24
+
25
+ strong_positive_pairs: dict[str, float] = {}
26
+ strong_negative_pairs: dict[str, float] = {}
27
+
28
+ columns = list(correlation_matrix.columns)
29
+
30
+ for i, left in enumerate(columns):
31
+ for right in columns[i + 1:]:
32
+
33
+ correlation = cast(
34
+ SupportsFloat,
35
+ correlation_matrix.loc[left, right],
36
+ )
37
+
38
+ value = float(correlation)
39
+
40
+ pair = f"{left} ↔ {right}"
41
+
42
+ if value >= 0.70:
43
+ strong_positive_pairs[pair] = round(
44
+ value,
45
+ 3,
46
+ )
47
+
48
+ elif value <= -0.70:
49
+ strong_negative_pairs[pair] = round(
50
+ value,
51
+ 3,
52
+ )
53
+
54
+ return CorrelationSummary(
55
+ correlation_matrix=correlation_matrix,
56
+ strong_positive_pairs=strong_positive_pairs,
57
+ strong_negative_pairs=strong_negative_pairs,
58
+ )
@@ -0,0 +1,48 @@
1
+ """
2
+ Data type analysis for Datapilot.
3
+ """
4
+
5
+ import pandas as pd
6
+
7
+ from .models import DataTypeSummary
8
+
9
+
10
+ def generate_data_type_summary(
11
+ dataframe: pd.DataFrame,
12
+ ) -> DataTypeSummary:
13
+ """
14
+ Generate data type statistics for a dataset.
15
+
16
+ Parameters
17
+ ----------
18
+ dataframe : pandas.DataFrame
19
+ Dataset to analyze.
20
+
21
+ Returns
22
+ -------
23
+ DataTypeSummary
24
+ Summary of dataset column types.
25
+ """
26
+
27
+ numeric_columns = dataframe.select_dtypes(
28
+ include=["number"]
29
+ ).columns.tolist()
30
+
31
+ categorical_columns = dataframe.select_dtypes(
32
+ include=["object", "string", "category"]
33
+ ).columns.tolist()
34
+
35
+ boolean_columns = dataframe.select_dtypes(
36
+ include=["bool"]
37
+ ).columns.tolist()
38
+
39
+ datetime_columns = dataframe.select_dtypes(
40
+ include=["datetime", "datetimetz"]
41
+ ).columns.tolist()
42
+
43
+ return DataTypeSummary(
44
+ numeric_columns=numeric_columns,
45
+ categorical_columns=categorical_columns,
46
+ boolean_columns=boolean_columns,
47
+ datetime_columns=datetime_columns,
48
+ )
@@ -0,0 +1,40 @@
1
+ """
2
+ Duplicate row analysis for Datapilot.
3
+ """
4
+
5
+ import pandas as pd
6
+
7
+ from .models import DuplicateSummary
8
+
9
+
10
+ def generate_duplicate_summary(
11
+ dataframe: pd.DataFrame,
12
+ ) -> DuplicateSummary:
13
+ """
14
+ Generate duplicate row statistics for a dataset.
15
+
16
+ Parameters
17
+ ----------
18
+ dataframe : pandas.DataFrame
19
+ Dataset to analyze.
20
+
21
+ Returns
22
+ -------
23
+ DuplicateSummary
24
+ Summary of duplicate rows in the dataset.
25
+ """
26
+
27
+ total_duplicates = int(dataframe.duplicated().sum())
28
+
29
+ total_rows = len(dataframe)
30
+
31
+ duplicate_percentage = (
32
+ round((total_duplicates / total_rows) * 100, 2)
33
+ if total_rows > 0
34
+ else 0.0
35
+ )
36
+
37
+ return DuplicateSummary(
38
+ total_duplicates=total_duplicates,
39
+ duplicate_percentage=duplicate_percentage,
40
+ )
@@ -0,0 +1,163 @@
1
+ """
2
+ Dataset health assessment for Datapilot.
3
+ """
4
+
5
+ import pandas as pd
6
+
7
+ from .datatype import generate_data_type_summary
8
+ from .duplicate import generate_duplicate_summary
9
+ from .missing import generate_missing_value_summary
10
+ from .models import DatasetHealth
11
+
12
+
13
+ def generate_dataset_health(
14
+ dataframe: pd.DataFrame,
15
+ ) -> DatasetHealth:
16
+ """
17
+ Generate an overall health assessment for a dataset.
18
+ """
19
+
20
+ missing = generate_missing_value_summary(dataframe)
21
+ duplicates = generate_duplicate_summary(dataframe)
22
+ data_types = generate_data_type_summary(dataframe)
23
+
24
+ score = 100.0
25
+
26
+ # -----------------------------
27
+ # Missing Value Penalty (40 pts)
28
+ # -----------------------------
29
+ score -= (missing.missing_percentage / 100) * 40
30
+
31
+ # -----------------------------
32
+ # Duplicate Penalty (35 pts)
33
+ # -----------------------------
34
+ score -= (duplicates.duplicate_percentage / 100) * 35
35
+
36
+ # -----------------------------
37
+ # Structure Penalty (25 pts)
38
+ # -----------------------------
39
+ total_columns = len(dataframe.columns)
40
+
41
+ recognized_columns = (
42
+ len(data_types.numeric_columns)
43
+ + len(data_types.categorical_columns)
44
+ + len(data_types.boolean_columns)
45
+ + len(data_types.datetime_columns)
46
+ )
47
+
48
+ unrecognized_columns = max(
49
+ total_columns - recognized_columns,
50
+ 0,
51
+ )
52
+
53
+ if total_columns > 0:
54
+ score -= (
55
+ unrecognized_columns / total_columns
56
+ ) * 25
57
+
58
+ score = max(0, min(100, round(score)))
59
+
60
+ # -----------------------------
61
+ # Grade
62
+ # -----------------------------
63
+ if score >= 95:
64
+ grade = "A+"
65
+ status = "Excellent"
66
+
67
+ elif score >= 90:
68
+ grade = "A"
69
+ status = "Healthy"
70
+
71
+ elif score >= 80:
72
+ grade = "B"
73
+ status = "Good"
74
+
75
+ elif score >= 70:
76
+ grade = "C"
77
+ status = "Fair"
78
+
79
+ elif score >= 60:
80
+ grade = "D"
81
+ status = "Poor"
82
+
83
+ else:
84
+ grade = "F"
85
+ status = "Critical"
86
+
87
+ # -----------------------------
88
+ # ML Readiness
89
+ # -----------------------------
90
+ ml_ready = (
91
+ score >= 85
92
+ and missing.missing_percentage <= 20
93
+ and duplicates.duplicate_percentage <= 5
94
+ )
95
+
96
+ # -----------------------------
97
+ # Strengths
98
+ # -----------------------------
99
+ strengths = []
100
+
101
+ if missing.total_missing == 0:
102
+ strengths.append("No missing values detected.")
103
+
104
+ if duplicates.total_duplicates == 0:
105
+ strengths.append("No duplicate rows detected.")
106
+
107
+ if unrecognized_columns == 0:
108
+ strengths.append("All columns have recognized data types.")
109
+
110
+ # -----------------------------
111
+ # Weaknesses
112
+ # -----------------------------
113
+ weaknesses = []
114
+
115
+ if missing.total_missing > 0:
116
+ weaknesses.append(
117
+ f"{missing.total_missing} missing values detected."
118
+ )
119
+
120
+ if duplicates.total_duplicates > 0:
121
+ weaknesses.append(
122
+ f"{duplicates.total_duplicates} duplicate rows detected."
123
+ )
124
+
125
+ if unrecognized_columns > 0:
126
+ weaknesses.append(
127
+ f"{unrecognized_columns} columns have unsupported data types."
128
+ )
129
+
130
+ # -----------------------------
131
+ # Recommendations
132
+ # -----------------------------
133
+ recommendations = []
134
+
135
+ if missing.missing_percentage > 0:
136
+ recommendations.append(
137
+ "Handle missing values before further analysis."
138
+ )
139
+
140
+ if duplicates.total_duplicates > 0:
141
+ recommendations.append(
142
+ "Remove duplicate rows."
143
+ )
144
+
145
+ if unrecognized_columns > 0:
146
+ recommendations.append(
147
+ "Review unsupported column data types."
148
+ )
149
+
150
+ if not recommendations:
151
+ recommendations.append(
152
+ "Dataset is ready for analysis."
153
+ )
154
+
155
+ return DatasetHealth(
156
+ score=score,
157
+ grade=grade,
158
+ status=status,
159
+ ml_ready=ml_ready,
160
+ strengths=strengths,
161
+ weaknesses=weaknesses,
162
+ recommendations=recommendations,
163
+ )
@@ -0,0 +1,101 @@
1
+ """
2
+ Insight generation for Datapilot.
3
+ """
4
+
5
+ import pandas as pd
6
+
7
+ from .correlation import generate_correlation_summary
8
+ from .duplicate import generate_duplicate_summary
9
+ from .missing import generate_missing_value_summary
10
+ from .models import InsightSummary
11
+ from .outliers import generate_outlier_summary
12
+
13
+
14
+ def generate_insight_summary(
15
+ dataframe: pd.DataFrame,
16
+ ) -> InsightSummary:
17
+ """
18
+ Generate actionable dataset insights.
19
+ """
20
+
21
+ insights: list[str] = []
22
+
23
+ recommendations: list[str] = []
24
+
25
+ missing = generate_missing_value_summary(
26
+ dataframe
27
+ )
28
+
29
+ duplicates = generate_duplicate_summary(
30
+ dataframe
31
+ )
32
+
33
+ outliers = generate_outlier_summary(
34
+ dataframe
35
+ )
36
+
37
+ correlation = generate_correlation_summary(
38
+ dataframe
39
+ )
40
+
41
+ if missing.total_missing > 0:
42
+
43
+ insights.append(
44
+ f"Dataset contains "
45
+ f"{missing.total_missing} missing values."
46
+ )
47
+
48
+ recommendations.append(
49
+ "Handle missing values before "
50
+ "training machine learning models."
51
+ )
52
+
53
+ if duplicates.total_duplicates > 0:
54
+
55
+ insights.append(
56
+ f"Dataset contains "
57
+ f"{duplicates.total_duplicates} duplicate rows."
58
+ )
59
+
60
+ recommendations.append(
61
+ "Remove duplicate rows."
62
+ )
63
+
64
+ if outliers.total_outliers > 0:
65
+
66
+ insights.append(
67
+ f"Detected "
68
+ f"{outliers.total_outliers} statistical outliers."
69
+ )
70
+
71
+ recommendations.append(
72
+ "Review outliers before modelling."
73
+ )
74
+
75
+ if correlation.strong_positive_pairs:
76
+
77
+ insights.append(
78
+ "Highly correlated numeric features detected."
79
+ )
80
+
81
+ recommendations.append(
82
+ "Review correlated features to "
83
+ "reduce multicollinearity."
84
+ )
85
+
86
+ if correlation.strong_negative_pairs:
87
+
88
+ insights.append(
89
+ "Strong negative correlations detected."
90
+ )
91
+
92
+ if not insights:
93
+
94
+ insights.append(
95
+ "No significant data quality issues detected."
96
+ )
97
+
98
+ return InsightSummary(
99
+ insights=insights,
100
+ recommendations=recommendations,
101
+ )
@@ -0,0 +1,56 @@
1
+ """
2
+ Missing value analysis for Datapilot.
3
+ """
4
+
5
+ import pandas as pd
6
+
7
+ from .models import MissingValueSummary
8
+
9
+
10
+ def generate_missing_value_summary(
11
+ dataframe: pd.DataFrame,
12
+ ) -> MissingValueSummary:
13
+ """
14
+ Generate missing value statistics for a dataset.
15
+
16
+ Parameters
17
+ ----------
18
+ dataframe : pandas.DataFrame
19
+ Dataset to analyze.
20
+
21
+ Returns
22
+ -------
23
+ MissingValueSummary
24
+ Summary of missing values in the dataset.
25
+ """
26
+
27
+ total_missing = int(dataframe.isna().sum().sum())
28
+
29
+ total_cells = dataframe.shape[0] * dataframe.shape[1]
30
+
31
+ missing_percentage = (
32
+ round((total_missing / total_cells) * 100, 2)
33
+ if total_cells > 0
34
+ else 0.0
35
+ )
36
+
37
+ missing_per_column = dataframe.isna().sum()
38
+
39
+ columns_with_missing = {
40
+ str(column): int(count)
41
+ for column, count in missing_per_column.items()
42
+ if count > 0
43
+ }
44
+
45
+ columns_without_missing = [
46
+ str(column)
47
+ for column, count in missing_per_column.items()
48
+ if count == 0
49
+ ]
50
+
51
+ return MissingValueSummary(
52
+ total_missing=total_missing,
53
+ missing_percentage=missing_percentage,
54
+ columns_with_missing=columns_with_missing,
55
+ columns_without_missing=columns_without_missing,
56
+ )