datapilot-kit 0.3.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. datapilot/__init__.py +11 -0
  2. datapilot/analysis/__init__.py +3 -0
  3. datapilot/analysis/correlation.py +58 -0
  4. datapilot/analysis/datatype.py +48 -0
  5. datapilot/analysis/duplicate.py +40 -0
  6. datapilot/analysis/health.py +163 -0
  7. datapilot/analysis/insights.py +101 -0
  8. datapilot/analysis/missing.py +56 -0
  9. datapilot/analysis/models.py +108 -0
  10. datapilot/analysis/outliers.py +85 -0
  11. datapilot/analysis/statistics.py +46 -0
  12. datapilot/analysis/summary.py +36 -0
  13. datapilot/api/__init__.py +7 -0
  14. datapilot/api/analyze.py +16 -0
  15. datapilot/assets/style.css +1081 -0
  16. datapilot/cli/__init__.py +3 -0
  17. datapilot/core/__init__.py +3 -0
  18. datapilot/core/loader.py +72 -0
  19. datapilot/core/report.py +104 -0
  20. datapilot/interpretation/__init__.py +3 -0
  21. datapilot/llm/__init__.py +0 -0
  22. datapilot/recommendation/__init__.py +0 -0
  23. datapilot/reporting/__init__.py +3 -0
  24. datapilot/reporting/fragments.py +329 -0
  25. datapilot/reporting/html.py +51 -0
  26. datapilot/reporting/placeholders.py +133 -0
  27. datapilot/reporting/renderer.py +57 -0
  28. datapilot/templates/report.html +528 -0
  29. datapilot/ui/__init__.py +0 -0
  30. datapilot/ui/cards.py +70 -0
  31. datapilot/ui/console.py +12 -0
  32. datapilot/ui/dashboard.py +90 -0
  33. datapilot/ui/panels.py +16 -0
  34. datapilot/ui/renderer.py +21 -0
  35. datapilot/ui/sections.py +89 -0
  36. datapilot/ui/tables.py +21 -0
  37. datapilot/ui/theme.py +20 -0
  38. datapilot/utils/__init__.py +3 -0
  39. datapilot/visualization/__init__.py +0 -0
  40. datapilot_kit-0.3.0rc1.dist-info/METADATA +103 -0
  41. datapilot_kit-0.3.0rc1.dist-info/RECORD +44 -0
  42. datapilot_kit-0.3.0rc1.dist-info/WHEEL +5 -0
  43. datapilot_kit-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
  44. datapilot_kit-0.3.0rc1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,3 @@
1
+ """
2
+ Command-line interface for Datapilot.
3
+ """
@@ -0,0 +1,3 @@
1
+ """
2
+ Core modules for Datapilot.
3
+ """
@@ -0,0 +1,72 @@
1
+ """
2
+ Dataset loading utilities for Datapilot.
3
+
4
+ This module is responsible for loading supported dataset formats
5
+ and converting them into a pandas DataFrame.
6
+ """
7
+
8
+ from pathlib import Path
9
+ from typing import Any
10
+
11
+ import pandas as pd
12
+
13
+ CSV_EXTENSIONS = {".csv"}
14
+ EXCEL_EXTENSIONS = {".xlsx", ".xls"}
15
+
16
+
17
+ def load_dataset(data: Any) -> pd.DataFrame:
18
+ """
19
+ Load a supported dataset into a pandas DataFrame.
20
+ """
21
+
22
+ if isinstance(data, pd.DataFrame):
23
+ return _load_dataframe(data)
24
+
25
+ if isinstance(data, (str, Path)):
26
+ return _load_file(Path(data))
27
+
28
+ raise TypeError(
29
+ f"Unsupported input type: {type(data).__name__}"
30
+ )
31
+
32
+
33
+ def _load_dataframe(dataframe: pd.DataFrame) -> pd.DataFrame:
34
+ """
35
+ Return a copy of the provided DataFrame.
36
+ """
37
+
38
+ return dataframe.copy(deep=True)
39
+
40
+
41
+ def _load_file(path: Path) -> pd.DataFrame:
42
+ """
43
+ Load a dataset from a supported file.
44
+ """
45
+
46
+ suffix = path.suffix.lower()
47
+
48
+ if suffix in CSV_EXTENSIONS:
49
+ return _load_csv(path)
50
+
51
+ if suffix in EXCEL_EXTENSIONS:
52
+ return _load_excel(path)
53
+
54
+ raise ValueError(
55
+ f"Unsupported file format: '{suffix}'"
56
+ )
57
+
58
+
59
+ def _load_csv(path: Path) -> pd.DataFrame:
60
+ """
61
+ Load a CSV file.
62
+ """
63
+
64
+ return pd.read_csv(path)
65
+
66
+
67
+ def _load_excel(path: Path) -> pd.DataFrame:
68
+ """
69
+ Load an Excel file.
70
+ """
71
+
72
+ return pd.read_excel(path)
@@ -0,0 +1,104 @@
1
+ """
2
+ Report object for Datapilot.
3
+
4
+ The Report class is the primary interface returned by the
5
+ public `analyze()` function.
6
+ """
7
+
8
+ import pandas as pd
9
+
10
+ from ..analysis.correlation import (
11
+ generate_correlation_summary,
12
+ )
13
+ from ..analysis.datatype import generate_data_type_summary
14
+ from ..analysis.duplicate import generate_duplicate_summary
15
+ from ..analysis.health import generate_dataset_health
16
+ from ..analysis.insights import (
17
+ generate_insight_summary,
18
+ )
19
+ from ..analysis.missing import generate_missing_value_summary
20
+ from ..analysis.models import (
21
+ CorrelationSummary,
22
+ DatasetHealth,
23
+ DatasetSummary,
24
+ DataTypeSummary,
25
+ DuplicateSummary,
26
+ InsightSummary,
27
+ MissingValueSummary,
28
+ OutlierSummary,
29
+ StatisticsSummary,
30
+ )
31
+ from ..analysis.outliers import generate_outlier_summary
32
+ from ..analysis.statistics import generate_statistics_summary
33
+ from ..analysis.summary import generate_summary
34
+
35
+
36
+ class Report:
37
+ """
38
+ Represents the analysis results of a dataset.
39
+ """
40
+
41
+ def __init__(self, dataframe: pd.DataFrame) -> None:
42
+ """
43
+ Initialize a Report object.
44
+
45
+ Parameters
46
+ ----------
47
+ dataframe : pandas.DataFrame
48
+ Dataset to analyze.
49
+ """
50
+ self._df = dataframe
51
+
52
+ def summary(self) -> DatasetSummary:
53
+ """
54
+ Return a summary of the dataset.
55
+ """
56
+ return generate_summary(self._df)
57
+
58
+ def missing_values(self) -> MissingValueSummary:
59
+ """
60
+ Return missing value statistics for the dataset.
61
+ """
62
+ return generate_missing_value_summary(self._df)
63
+
64
+ def duplicates(self) -> DuplicateSummary:
65
+ """
66
+ Return duplicate row statistics for the dataset.
67
+ """
68
+ return generate_duplicate_summary(self._df)
69
+
70
+ def data_types(self) -> DataTypeSummary:
71
+ """
72
+ Return data type statistics for the dataset.
73
+ """
74
+ return generate_data_type_summary(self._df)
75
+
76
+ def dataset_health(self) -> DatasetHealth:
77
+ """
78
+ Return the overall health assessment of the dataset.
79
+ """
80
+ return generate_dataset_health(self._df)
81
+
82
+ def statistics(self) -> StatisticsSummary:
83
+ """
84
+ Return statistical summary.
85
+ """
86
+ return generate_statistics_summary(self._df)
87
+
88
+ def outliers(self) -> OutlierSummary:
89
+ """
90
+ Return outlier statistics.
91
+ """
92
+ return generate_outlier_summary(self._df)
93
+
94
+ def correlation(self) -> CorrelationSummary:
95
+ """
96
+ Return correlation analysis.
97
+ """
98
+ return generate_correlation_summary(self._df)
99
+
100
+ def insights(self) -> InsightSummary:
101
+ """
102
+ Return generated dataset insights.
103
+ """
104
+ return generate_insight_summary(self._df)
@@ -0,0 +1,3 @@
1
+ """
2
+ Interpretation modules for Datapilot.
3
+ """
File without changes
File without changes
@@ -0,0 +1,3 @@
1
+ """
2
+ Reporting utilities for Datapilot.
3
+ """
@@ -0,0 +1,329 @@
1
+ """
2
+ HTML fragment builders for Datapilot reports.
3
+ """
4
+
5
+ from datapilot.core.report import Report
6
+
7
+
8
+ def build_column_name_chips(
9
+ report: Report,
10
+ ) -> str:
11
+ """
12
+ Build HTML chips for dataset column names.
13
+ """
14
+
15
+ summary = report.summary()
16
+
17
+ chips: list[str] = []
18
+
19
+ for column in summary.column_names:
20
+ chips.append(
21
+ (
22
+ '<span class="column-chip">'
23
+ f"{column}"
24
+ "</span>"
25
+ )
26
+ )
27
+
28
+ return "\n".join(chips)
29
+
30
+
31
+ def build_health_headline(
32
+ report: Report,
33
+ ) -> str:
34
+ """
35
+ Build health headline.
36
+ """
37
+
38
+ health = report.dataset_health()
39
+
40
+ return (
41
+ f"Overall dataset quality is "
42
+ f"{health.status.lower()}."
43
+ )
44
+
45
+
46
+ def build_health_summary(
47
+ report: Report,
48
+ ) -> str:
49
+ """
50
+ Build health summary.
51
+ """
52
+
53
+ health = report.dataset_health()
54
+
55
+ return (
56
+ f"The dataset received a score of "
57
+ f"{health.score}/100 with grade "
58
+ f"{health.grade}."
59
+ )
60
+
61
+
62
+ def build_html_list(
63
+ items: list[str],
64
+ ) -> str:
65
+ """
66
+ Convert a list of strings into HTML list items.
67
+ """
68
+
69
+ if not items:
70
+ return "<li>None</li>"
71
+
72
+ return "\n".join(
73
+ f"<li>{item}</li>"
74
+ for item in items
75
+ )
76
+
77
+
78
+ def build_missing_table(
79
+ report: Report,
80
+ ) -> str:
81
+ """
82
+ Build missing values table.
83
+ """
84
+
85
+ missing = report.missing_values()
86
+
87
+ rows: list[str] = []
88
+
89
+ if not missing.columns_with_missing:
90
+ return (
91
+ "<p>No missing values were detected.</p>"
92
+ )
93
+
94
+ rows.append(
95
+ """
96
+ <table class="report-table">
97
+ <thead>
98
+ <tr>
99
+ <th>Column</th>
100
+ <th>Missing Values</th>
101
+ </tr>
102
+ </thead>
103
+ <tbody>
104
+ """
105
+ )
106
+
107
+ for column, count in (
108
+ missing.columns_with_missing.items()
109
+ ):
110
+ rows.append(
111
+ f"""
112
+ <tr>
113
+ <td>{column}</td>
114
+ <td>{count}</td>
115
+ </tr>
116
+ """
117
+ )
118
+
119
+ rows.append(
120
+ """
121
+ </tbody>
122
+ </table>
123
+ """
124
+ )
125
+
126
+ return "\n".join(rows)
127
+
128
+ def build_statistics_table(
129
+ report: Report,
130
+ ) -> str:
131
+ """
132
+ Build statistics table.
133
+ """
134
+
135
+ statistics = report.statistics()
136
+
137
+ rows: list[str] = []
138
+
139
+ if not statistics.column_statistics:
140
+ return (
141
+ "<p>No numeric columns were detected.</p>"
142
+ )
143
+
144
+ rows.append(
145
+ """
146
+ <table class="report-table">
147
+ <thead>
148
+ <tr>
149
+ <th>Column</th>
150
+ <th>Mean</th>
151
+ <th>Median</th>
152
+ <th>Std</th>
153
+ <th>Min</th>
154
+ <th>Max</th>
155
+ </tr>
156
+ </thead>
157
+ <tbody>
158
+ """
159
+ )
160
+
161
+ for (
162
+ column,
163
+ values,
164
+ ) in statistics.column_statistics.items():
165
+ rows.append(
166
+ f"""
167
+ <tr>
168
+ <td>{column}</td>
169
+ <td>{values["mean"]:.2f}</td>
170
+ <td>{values["median"]:.2f}</td>
171
+ <td>{values["std"]:.2f}</td>
172
+ <td>{values["min"]:.2f}</td>
173
+ <td>{values["max"]:.2f}</td>
174
+ </tr>
175
+ """
176
+ )
177
+
178
+ rows.append(
179
+ """
180
+ </tbody>
181
+ </table>
182
+ """
183
+ )
184
+
185
+ return "\n".join(rows)
186
+
187
+ def build_outlier_cards(
188
+ report: Report,
189
+ ) -> str:
190
+ """
191
+ Build HTML cards for outlier analysis.
192
+ """
193
+
194
+ outliers = report.outliers()
195
+
196
+ if not outliers.columns_with_outliers:
197
+ return (
198
+ "<p>No outliers were detected.</p>"
199
+ )
200
+
201
+ cards: list[str] = []
202
+
203
+ for column, count in (
204
+ outliers.columns_with_outliers.items()
205
+ ):
206
+ cards.append(
207
+ f"""
208
+ <div class="outlier-card">
209
+ <h4>{column}</h4>
210
+ <p>{count} outliers detected</p>
211
+ </div>
212
+ """
213
+ )
214
+
215
+ return "\n".join(cards)
216
+
217
+ def build_correlation_table(
218
+ report: Report,
219
+ ) -> str:
220
+ """
221
+ Build correlation table.
222
+ """
223
+
224
+ correlation = report.correlation()
225
+
226
+ rows: list[str] = []
227
+
228
+ if (
229
+ not correlation.strong_positive_pairs
230
+ and not correlation.strong_negative_pairs
231
+ ):
232
+ return (
233
+ "<p>No strong correlations were detected.</p>"
234
+ )
235
+
236
+ rows.append(
237
+ """
238
+ <table class="report-table">
239
+ <thead>
240
+ <tr>
241
+ <th>Relationship</th>
242
+ <th>Type</th>
243
+ <th>Correlation</th>
244
+ </tr>
245
+ </thead>
246
+ <tbody>
247
+ """
248
+ )
249
+
250
+ for pair, value in (
251
+ correlation.strong_positive_pairs.items()
252
+ ):
253
+ rows.append(
254
+ f"""
255
+ <tr>
256
+ <td>{pair}</td>
257
+ <td>Positive</td>
258
+ <td>{value:.2f}</td>
259
+ </tr>
260
+ """
261
+ )
262
+
263
+ for pair, value in (
264
+ correlation.strong_negative_pairs.items()
265
+ ):
266
+ rows.append(
267
+ f"""
268
+ <tr>
269
+ <td>{pair}</td>
270
+ <td>Negative</td>
271
+ <td>{value:.2f}</td>
272
+ </tr>
273
+ """
274
+ )
275
+
276
+ rows.append(
277
+ """
278
+ </tbody>
279
+ </table>
280
+ """
281
+ )
282
+
283
+ return "\n".join(rows)
284
+
285
+ def build_insights(
286
+ report: Report,
287
+ ) -> str:
288
+ """
289
+ Build insights HTML.
290
+ """
291
+
292
+ insight_summary = report.insights()
293
+
294
+ rows: list[str] = []
295
+
296
+ rows.append("<ul>")
297
+
298
+ for insight in insight_summary.insights:
299
+ rows.append(
300
+ f"<li>{insight}</li>"
301
+ )
302
+
303
+ rows.append("</ul>")
304
+
305
+ return "\n".join(rows)
306
+
307
+ def build_insight_recommendations(
308
+ report: Report,
309
+ ) -> str:
310
+ """
311
+ Build recommendation HTML.
312
+ """
313
+
314
+ insight_summary = report.insights()
315
+
316
+ rows: list[str] = []
317
+
318
+ rows.append("<ul>")
319
+
320
+ for recommendation in (
321
+ insight_summary.recommendations
322
+ ):
323
+ rows.append(
324
+ f"<li>{recommendation}</li>"
325
+ )
326
+
327
+ rows.append("</ul>")
328
+
329
+ return "\n".join(rows)
@@ -0,0 +1,51 @@
1
+ """
2
+ HTML report generation for Datapilot.
3
+ """
4
+
5
+ import shutil
6
+ from pathlib import Path
7
+
8
+ from datapilot.core.report import Report
9
+
10
+ from .renderer import HTMLRenderer
11
+
12
+
13
+ def generate_html_report(
14
+ report: Report,
15
+ output_path: str | Path,
16
+ ) -> None:
17
+ """
18
+ Generate a Datapilot HTML report.
19
+ """
20
+
21
+ output_path = Path(output_path)
22
+
23
+ project_root = (
24
+ Path(__file__).parent.parent
25
+ )
26
+
27
+ template = (
28
+ project_root
29
+ / "templates"
30
+ / "report.html"
31
+ )
32
+
33
+ css_source = (
34
+ project_root
35
+ / "assets"
36
+ / "style.css"
37
+ )
38
+
39
+ renderer = HTMLRenderer(
40
+ template_path=template,
41
+ )
42
+
43
+ renderer.render(
44
+ report=report,
45
+ output_path=output_path,
46
+ )
47
+
48
+ shutil.copy2(
49
+ css_source,
50
+ output_path.parent / "style.css",
51
+ )
@@ -0,0 +1,133 @@
1
+ """
2
+ Placeholder generation for Datapilot HTML reports.
3
+ """
4
+
5
+ from datetime import datetime
6
+
7
+ from datapilot.core.report import Report
8
+
9
+ from .fragments import (
10
+ build_column_name_chips,
11
+ build_correlation_table,
12
+ build_health_headline,
13
+ build_health_summary,
14
+ build_html_list,
15
+ build_insight_recommendations,
16
+ build_insights,
17
+ build_missing_table,
18
+ build_outlier_cards,
19
+ build_statistics_table,
20
+ )
21
+
22
+ DATAPILOT_VERSION = "0.2.0"
23
+
24
+
25
+ def build_placeholders(
26
+ report: Report,
27
+ ) -> dict[str, str]:
28
+ """
29
+ Build placeholder dictionary for the HTML template.
30
+ """
31
+
32
+ summary = report.summary()
33
+ health = report.dataset_health()
34
+ missing = report.missing_values()
35
+ duplicates = report.duplicates()
36
+
37
+ health_class = (
38
+ "kpi-card--good"
39
+ if health.score >= 90
40
+ else "kpi-card--warn"
41
+ if health.score >= 75
42
+ else "kpi-card--bad"
43
+ )
44
+
45
+ return {
46
+ "{{DATASET_NAME}}": "Dataset",
47
+ "{{REPORT_GENERATED_AT}}": datetime.now().strftime(
48
+ "%d %b %Y %H:%M",
49
+ ),
50
+ "{{DATAPILOT_VERSION}}": DATAPILOT_VERSION,
51
+ "{{REPORT_SUMMARY}}": (
52
+ "Automated Data Quality Assessment Report"
53
+ ),
54
+
55
+ "{{ROWS}}": str(summary.rows),
56
+ "{{COLUMNS}}": str(summary.columns),
57
+ "{{MEMORY_USAGE}}": (
58
+ f"{summary.memory_usage_mb:.2f} MB"
59
+ ),
60
+ "{{FILE_FORMAT}}": "CSV",
61
+
62
+ "{{COLUMN_NAME_CHIPS}}": (
63
+ build_column_name_chips(report)
64
+ ),
65
+
66
+ "{{HEALTH_SCORE}}": str(health.score),
67
+ "{{GRADE}}": health.grade,
68
+ "{{HEALTH_STATUS}}": health.status,
69
+
70
+ "{{HEALTH_HEADLINE}}": (
71
+ build_health_headline(report)
72
+ ),
73
+ "{{HEALTH_SUMMARY}}": (
74
+ build_health_summary(report)
75
+ ),
76
+
77
+ "{{STRENGTHS_LIST}}": (
78
+ build_html_list(
79
+ health.strengths,
80
+ )
81
+ ),
82
+ "{{WEAKNESSES_LIST}}": (
83
+ build_html_list(
84
+ health.weaknesses,
85
+ )
86
+ ),
87
+ "{{RECOMMENDATIONS_LIST}}": (
88
+ build_html_list(
89
+ health.recommendations,
90
+ )
91
+ ),
92
+
93
+ "{{HEALTH_SCORE_SEVERITY_CLASS}}": health_class,
94
+ "{{HEALTH_SCORE_SEVERITY_CLASS_GAUGE}}": health_class,
95
+
96
+ "{{MISSING_VALUES_COUNT}}": str(
97
+ missing.total_missing,
98
+ ),
99
+ "{{MISSING_VALUES_PERCENT}}": (
100
+ f"{missing.missing_percentage:.1f}"
101
+ ),
102
+
103
+ "{{MISSING_TABLE}}": (
104
+ build_missing_table(report)
105
+ ),
106
+
107
+ "{{STATISTICS_TABLE}}": (
108
+ build_statistics_table(report)
109
+ ),
110
+
111
+ "{{OUTLIER_CARDS}}": (
112
+ build_outlier_cards(report)
113
+ ),
114
+
115
+ "{{CORRELATION_TABLE}}": (
116
+ build_correlation_table(report)
117
+ ),
118
+
119
+ "{{INSIGHTS}}": (
120
+ build_insights(report)
121
+ ),
122
+
123
+ "{{RECOMMENDATIONS}}": (
124
+ build_insight_recommendations(report)
125
+ ),
126
+
127
+ "{{DUPLICATE_ROWS_COUNT}}": str(
128
+ duplicates.total_duplicates,
129
+ ),
130
+ "{{DUPLICATE_ROWS_PERCENT}}": (
131
+ f"{duplicates.duplicate_percentage:.1f}"
132
+ ),
133
+ }