datapilot-kit 0.3.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- datapilot/__init__.py +11 -0
- datapilot/analysis/__init__.py +3 -0
- datapilot/analysis/correlation.py +58 -0
- datapilot/analysis/datatype.py +48 -0
- datapilot/analysis/duplicate.py +40 -0
- datapilot/analysis/health.py +163 -0
- datapilot/analysis/insights.py +101 -0
- datapilot/analysis/missing.py +56 -0
- datapilot/analysis/models.py +108 -0
- datapilot/analysis/outliers.py +85 -0
- datapilot/analysis/statistics.py +46 -0
- datapilot/analysis/summary.py +36 -0
- datapilot/api/__init__.py +7 -0
- datapilot/api/analyze.py +16 -0
- datapilot/assets/style.css +1081 -0
- datapilot/cli/__init__.py +3 -0
- datapilot/core/__init__.py +3 -0
- datapilot/core/loader.py +72 -0
- datapilot/core/report.py +104 -0
- datapilot/interpretation/__init__.py +3 -0
- datapilot/llm/__init__.py +0 -0
- datapilot/recommendation/__init__.py +0 -0
- datapilot/reporting/__init__.py +3 -0
- datapilot/reporting/fragments.py +329 -0
- datapilot/reporting/html.py +51 -0
- datapilot/reporting/placeholders.py +133 -0
- datapilot/reporting/renderer.py +57 -0
- datapilot/templates/report.html +528 -0
- datapilot/ui/__init__.py +0 -0
- datapilot/ui/cards.py +70 -0
- datapilot/ui/console.py +12 -0
- datapilot/ui/dashboard.py +90 -0
- datapilot/ui/panels.py +16 -0
- datapilot/ui/renderer.py +21 -0
- datapilot/ui/sections.py +89 -0
- datapilot/ui/tables.py +21 -0
- datapilot/ui/theme.py +20 -0
- datapilot/utils/__init__.py +3 -0
- datapilot/visualization/__init__.py +0 -0
- datapilot_kit-0.3.0rc1.dist-info/METADATA +103 -0
- datapilot_kit-0.3.0rc1.dist-info/RECORD +44 -0
- datapilot_kit-0.3.0rc1.dist-info/WHEEL +5 -0
- datapilot_kit-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
- datapilot_kit-0.3.0rc1.dist-info/top_level.txt +1 -0
datapilot/core/loader.py
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Dataset loading utilities for Datapilot.
|
|
3
|
+
|
|
4
|
+
This module is responsible for loading supported dataset formats
|
|
5
|
+
and converting them into a pandas DataFrame.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
import pandas as pd
|
|
12
|
+
|
|
13
|
+
CSV_EXTENSIONS = {".csv"}
|
|
14
|
+
EXCEL_EXTENSIONS = {".xlsx", ".xls"}
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def load_dataset(data: Any) -> pd.DataFrame:
|
|
18
|
+
"""
|
|
19
|
+
Load a supported dataset into a pandas DataFrame.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
if isinstance(data, pd.DataFrame):
|
|
23
|
+
return _load_dataframe(data)
|
|
24
|
+
|
|
25
|
+
if isinstance(data, (str, Path)):
|
|
26
|
+
return _load_file(Path(data))
|
|
27
|
+
|
|
28
|
+
raise TypeError(
|
|
29
|
+
f"Unsupported input type: {type(data).__name__}"
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _load_dataframe(dataframe: pd.DataFrame) -> pd.DataFrame:
|
|
34
|
+
"""
|
|
35
|
+
Return a copy of the provided DataFrame.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
return dataframe.copy(deep=True)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _load_file(path: Path) -> pd.DataFrame:
|
|
42
|
+
"""
|
|
43
|
+
Load a dataset from a supported file.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
suffix = path.suffix.lower()
|
|
47
|
+
|
|
48
|
+
if suffix in CSV_EXTENSIONS:
|
|
49
|
+
return _load_csv(path)
|
|
50
|
+
|
|
51
|
+
if suffix in EXCEL_EXTENSIONS:
|
|
52
|
+
return _load_excel(path)
|
|
53
|
+
|
|
54
|
+
raise ValueError(
|
|
55
|
+
f"Unsupported file format: '{suffix}'"
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _load_csv(path: Path) -> pd.DataFrame:
|
|
60
|
+
"""
|
|
61
|
+
Load a CSV file.
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
return pd.read_csv(path)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _load_excel(path: Path) -> pd.DataFrame:
|
|
68
|
+
"""
|
|
69
|
+
Load an Excel file.
|
|
70
|
+
"""
|
|
71
|
+
|
|
72
|
+
return pd.read_excel(path)
|
datapilot/core/report.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Report object for Datapilot.
|
|
3
|
+
|
|
4
|
+
The Report class is the primary interface returned by the
|
|
5
|
+
public `analyze()` function.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import pandas as pd
|
|
9
|
+
|
|
10
|
+
from ..analysis.correlation import (
|
|
11
|
+
generate_correlation_summary,
|
|
12
|
+
)
|
|
13
|
+
from ..analysis.datatype import generate_data_type_summary
|
|
14
|
+
from ..analysis.duplicate import generate_duplicate_summary
|
|
15
|
+
from ..analysis.health import generate_dataset_health
|
|
16
|
+
from ..analysis.insights import (
|
|
17
|
+
generate_insight_summary,
|
|
18
|
+
)
|
|
19
|
+
from ..analysis.missing import generate_missing_value_summary
|
|
20
|
+
from ..analysis.models import (
|
|
21
|
+
CorrelationSummary,
|
|
22
|
+
DatasetHealth,
|
|
23
|
+
DatasetSummary,
|
|
24
|
+
DataTypeSummary,
|
|
25
|
+
DuplicateSummary,
|
|
26
|
+
InsightSummary,
|
|
27
|
+
MissingValueSummary,
|
|
28
|
+
OutlierSummary,
|
|
29
|
+
StatisticsSummary,
|
|
30
|
+
)
|
|
31
|
+
from ..analysis.outliers import generate_outlier_summary
|
|
32
|
+
from ..analysis.statistics import generate_statistics_summary
|
|
33
|
+
from ..analysis.summary import generate_summary
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class Report:
|
|
37
|
+
"""
|
|
38
|
+
Represents the analysis results of a dataset.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
def __init__(self, dataframe: pd.DataFrame) -> None:
|
|
42
|
+
"""
|
|
43
|
+
Initialize a Report object.
|
|
44
|
+
|
|
45
|
+
Parameters
|
|
46
|
+
----------
|
|
47
|
+
dataframe : pandas.DataFrame
|
|
48
|
+
Dataset to analyze.
|
|
49
|
+
"""
|
|
50
|
+
self._df = dataframe
|
|
51
|
+
|
|
52
|
+
def summary(self) -> DatasetSummary:
|
|
53
|
+
"""
|
|
54
|
+
Return a summary of the dataset.
|
|
55
|
+
"""
|
|
56
|
+
return generate_summary(self._df)
|
|
57
|
+
|
|
58
|
+
def missing_values(self) -> MissingValueSummary:
|
|
59
|
+
"""
|
|
60
|
+
Return missing value statistics for the dataset.
|
|
61
|
+
"""
|
|
62
|
+
return generate_missing_value_summary(self._df)
|
|
63
|
+
|
|
64
|
+
def duplicates(self) -> DuplicateSummary:
|
|
65
|
+
"""
|
|
66
|
+
Return duplicate row statistics for the dataset.
|
|
67
|
+
"""
|
|
68
|
+
return generate_duplicate_summary(self._df)
|
|
69
|
+
|
|
70
|
+
def data_types(self) -> DataTypeSummary:
|
|
71
|
+
"""
|
|
72
|
+
Return data type statistics for the dataset.
|
|
73
|
+
"""
|
|
74
|
+
return generate_data_type_summary(self._df)
|
|
75
|
+
|
|
76
|
+
def dataset_health(self) -> DatasetHealth:
|
|
77
|
+
"""
|
|
78
|
+
Return the overall health assessment of the dataset.
|
|
79
|
+
"""
|
|
80
|
+
return generate_dataset_health(self._df)
|
|
81
|
+
|
|
82
|
+
def statistics(self) -> StatisticsSummary:
|
|
83
|
+
"""
|
|
84
|
+
Return statistical summary.
|
|
85
|
+
"""
|
|
86
|
+
return generate_statistics_summary(self._df)
|
|
87
|
+
|
|
88
|
+
def outliers(self) -> OutlierSummary:
|
|
89
|
+
"""
|
|
90
|
+
Return outlier statistics.
|
|
91
|
+
"""
|
|
92
|
+
return generate_outlier_summary(self._df)
|
|
93
|
+
|
|
94
|
+
def correlation(self) -> CorrelationSummary:
|
|
95
|
+
"""
|
|
96
|
+
Return correlation analysis.
|
|
97
|
+
"""
|
|
98
|
+
return generate_correlation_summary(self._df)
|
|
99
|
+
|
|
100
|
+
def insights(self) -> InsightSummary:
|
|
101
|
+
"""
|
|
102
|
+
Return generated dataset insights.
|
|
103
|
+
"""
|
|
104
|
+
return generate_insight_summary(self._df)
|
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
"""
|
|
2
|
+
HTML fragment builders for Datapilot reports.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from datapilot.core.report import Report
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def build_column_name_chips(
|
|
9
|
+
report: Report,
|
|
10
|
+
) -> str:
|
|
11
|
+
"""
|
|
12
|
+
Build HTML chips for dataset column names.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
summary = report.summary()
|
|
16
|
+
|
|
17
|
+
chips: list[str] = []
|
|
18
|
+
|
|
19
|
+
for column in summary.column_names:
|
|
20
|
+
chips.append(
|
|
21
|
+
(
|
|
22
|
+
'<span class="column-chip">'
|
|
23
|
+
f"{column}"
|
|
24
|
+
"</span>"
|
|
25
|
+
)
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
return "\n".join(chips)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def build_health_headline(
|
|
32
|
+
report: Report,
|
|
33
|
+
) -> str:
|
|
34
|
+
"""
|
|
35
|
+
Build health headline.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
health = report.dataset_health()
|
|
39
|
+
|
|
40
|
+
return (
|
|
41
|
+
f"Overall dataset quality is "
|
|
42
|
+
f"{health.status.lower()}."
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def build_health_summary(
|
|
47
|
+
report: Report,
|
|
48
|
+
) -> str:
|
|
49
|
+
"""
|
|
50
|
+
Build health summary.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
health = report.dataset_health()
|
|
54
|
+
|
|
55
|
+
return (
|
|
56
|
+
f"The dataset received a score of "
|
|
57
|
+
f"{health.score}/100 with grade "
|
|
58
|
+
f"{health.grade}."
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def build_html_list(
|
|
63
|
+
items: list[str],
|
|
64
|
+
) -> str:
|
|
65
|
+
"""
|
|
66
|
+
Convert a list of strings into HTML list items.
|
|
67
|
+
"""
|
|
68
|
+
|
|
69
|
+
if not items:
|
|
70
|
+
return "<li>None</li>"
|
|
71
|
+
|
|
72
|
+
return "\n".join(
|
|
73
|
+
f"<li>{item}</li>"
|
|
74
|
+
for item in items
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def build_missing_table(
|
|
79
|
+
report: Report,
|
|
80
|
+
) -> str:
|
|
81
|
+
"""
|
|
82
|
+
Build missing values table.
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
missing = report.missing_values()
|
|
86
|
+
|
|
87
|
+
rows: list[str] = []
|
|
88
|
+
|
|
89
|
+
if not missing.columns_with_missing:
|
|
90
|
+
return (
|
|
91
|
+
"<p>No missing values were detected.</p>"
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
rows.append(
|
|
95
|
+
"""
|
|
96
|
+
<table class="report-table">
|
|
97
|
+
<thead>
|
|
98
|
+
<tr>
|
|
99
|
+
<th>Column</th>
|
|
100
|
+
<th>Missing Values</th>
|
|
101
|
+
</tr>
|
|
102
|
+
</thead>
|
|
103
|
+
<tbody>
|
|
104
|
+
"""
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
for column, count in (
|
|
108
|
+
missing.columns_with_missing.items()
|
|
109
|
+
):
|
|
110
|
+
rows.append(
|
|
111
|
+
f"""
|
|
112
|
+
<tr>
|
|
113
|
+
<td>{column}</td>
|
|
114
|
+
<td>{count}</td>
|
|
115
|
+
</tr>
|
|
116
|
+
"""
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
rows.append(
|
|
120
|
+
"""
|
|
121
|
+
</tbody>
|
|
122
|
+
</table>
|
|
123
|
+
"""
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
return "\n".join(rows)
|
|
127
|
+
|
|
128
|
+
def build_statistics_table(
|
|
129
|
+
report: Report,
|
|
130
|
+
) -> str:
|
|
131
|
+
"""
|
|
132
|
+
Build statistics table.
|
|
133
|
+
"""
|
|
134
|
+
|
|
135
|
+
statistics = report.statistics()
|
|
136
|
+
|
|
137
|
+
rows: list[str] = []
|
|
138
|
+
|
|
139
|
+
if not statistics.column_statistics:
|
|
140
|
+
return (
|
|
141
|
+
"<p>No numeric columns were detected.</p>"
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
rows.append(
|
|
145
|
+
"""
|
|
146
|
+
<table class="report-table">
|
|
147
|
+
<thead>
|
|
148
|
+
<tr>
|
|
149
|
+
<th>Column</th>
|
|
150
|
+
<th>Mean</th>
|
|
151
|
+
<th>Median</th>
|
|
152
|
+
<th>Std</th>
|
|
153
|
+
<th>Min</th>
|
|
154
|
+
<th>Max</th>
|
|
155
|
+
</tr>
|
|
156
|
+
</thead>
|
|
157
|
+
<tbody>
|
|
158
|
+
"""
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
for (
|
|
162
|
+
column,
|
|
163
|
+
values,
|
|
164
|
+
) in statistics.column_statistics.items():
|
|
165
|
+
rows.append(
|
|
166
|
+
f"""
|
|
167
|
+
<tr>
|
|
168
|
+
<td>{column}</td>
|
|
169
|
+
<td>{values["mean"]:.2f}</td>
|
|
170
|
+
<td>{values["median"]:.2f}</td>
|
|
171
|
+
<td>{values["std"]:.2f}</td>
|
|
172
|
+
<td>{values["min"]:.2f}</td>
|
|
173
|
+
<td>{values["max"]:.2f}</td>
|
|
174
|
+
</tr>
|
|
175
|
+
"""
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
rows.append(
|
|
179
|
+
"""
|
|
180
|
+
</tbody>
|
|
181
|
+
</table>
|
|
182
|
+
"""
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
return "\n".join(rows)
|
|
186
|
+
|
|
187
|
+
def build_outlier_cards(
|
|
188
|
+
report: Report,
|
|
189
|
+
) -> str:
|
|
190
|
+
"""
|
|
191
|
+
Build HTML cards for outlier analysis.
|
|
192
|
+
"""
|
|
193
|
+
|
|
194
|
+
outliers = report.outliers()
|
|
195
|
+
|
|
196
|
+
if not outliers.columns_with_outliers:
|
|
197
|
+
return (
|
|
198
|
+
"<p>No outliers were detected.</p>"
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
cards: list[str] = []
|
|
202
|
+
|
|
203
|
+
for column, count in (
|
|
204
|
+
outliers.columns_with_outliers.items()
|
|
205
|
+
):
|
|
206
|
+
cards.append(
|
|
207
|
+
f"""
|
|
208
|
+
<div class="outlier-card">
|
|
209
|
+
<h4>{column}</h4>
|
|
210
|
+
<p>{count} outliers detected</p>
|
|
211
|
+
</div>
|
|
212
|
+
"""
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
return "\n".join(cards)
|
|
216
|
+
|
|
217
|
+
def build_correlation_table(
|
|
218
|
+
report: Report,
|
|
219
|
+
) -> str:
|
|
220
|
+
"""
|
|
221
|
+
Build correlation table.
|
|
222
|
+
"""
|
|
223
|
+
|
|
224
|
+
correlation = report.correlation()
|
|
225
|
+
|
|
226
|
+
rows: list[str] = []
|
|
227
|
+
|
|
228
|
+
if (
|
|
229
|
+
not correlation.strong_positive_pairs
|
|
230
|
+
and not correlation.strong_negative_pairs
|
|
231
|
+
):
|
|
232
|
+
return (
|
|
233
|
+
"<p>No strong correlations were detected.</p>"
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
rows.append(
|
|
237
|
+
"""
|
|
238
|
+
<table class="report-table">
|
|
239
|
+
<thead>
|
|
240
|
+
<tr>
|
|
241
|
+
<th>Relationship</th>
|
|
242
|
+
<th>Type</th>
|
|
243
|
+
<th>Correlation</th>
|
|
244
|
+
</tr>
|
|
245
|
+
</thead>
|
|
246
|
+
<tbody>
|
|
247
|
+
"""
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
for pair, value in (
|
|
251
|
+
correlation.strong_positive_pairs.items()
|
|
252
|
+
):
|
|
253
|
+
rows.append(
|
|
254
|
+
f"""
|
|
255
|
+
<tr>
|
|
256
|
+
<td>{pair}</td>
|
|
257
|
+
<td>Positive</td>
|
|
258
|
+
<td>{value:.2f}</td>
|
|
259
|
+
</tr>
|
|
260
|
+
"""
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
for pair, value in (
|
|
264
|
+
correlation.strong_negative_pairs.items()
|
|
265
|
+
):
|
|
266
|
+
rows.append(
|
|
267
|
+
f"""
|
|
268
|
+
<tr>
|
|
269
|
+
<td>{pair}</td>
|
|
270
|
+
<td>Negative</td>
|
|
271
|
+
<td>{value:.2f}</td>
|
|
272
|
+
</tr>
|
|
273
|
+
"""
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
rows.append(
|
|
277
|
+
"""
|
|
278
|
+
</tbody>
|
|
279
|
+
</table>
|
|
280
|
+
"""
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
return "\n".join(rows)
|
|
284
|
+
|
|
285
|
+
def build_insights(
|
|
286
|
+
report: Report,
|
|
287
|
+
) -> str:
|
|
288
|
+
"""
|
|
289
|
+
Build insights HTML.
|
|
290
|
+
"""
|
|
291
|
+
|
|
292
|
+
insight_summary = report.insights()
|
|
293
|
+
|
|
294
|
+
rows: list[str] = []
|
|
295
|
+
|
|
296
|
+
rows.append("<ul>")
|
|
297
|
+
|
|
298
|
+
for insight in insight_summary.insights:
|
|
299
|
+
rows.append(
|
|
300
|
+
f"<li>{insight}</li>"
|
|
301
|
+
)
|
|
302
|
+
|
|
303
|
+
rows.append("</ul>")
|
|
304
|
+
|
|
305
|
+
return "\n".join(rows)
|
|
306
|
+
|
|
307
|
+
def build_insight_recommendations(
|
|
308
|
+
report: Report,
|
|
309
|
+
) -> str:
|
|
310
|
+
"""
|
|
311
|
+
Build recommendation HTML.
|
|
312
|
+
"""
|
|
313
|
+
|
|
314
|
+
insight_summary = report.insights()
|
|
315
|
+
|
|
316
|
+
rows: list[str] = []
|
|
317
|
+
|
|
318
|
+
rows.append("<ul>")
|
|
319
|
+
|
|
320
|
+
for recommendation in (
|
|
321
|
+
insight_summary.recommendations
|
|
322
|
+
):
|
|
323
|
+
rows.append(
|
|
324
|
+
f"<li>{recommendation}</li>"
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
rows.append("</ul>")
|
|
328
|
+
|
|
329
|
+
return "\n".join(rows)
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""
|
|
2
|
+
HTML report generation for Datapilot.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import shutil
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from datapilot.core.report import Report
|
|
9
|
+
|
|
10
|
+
from .renderer import HTMLRenderer
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def generate_html_report(
|
|
14
|
+
report: Report,
|
|
15
|
+
output_path: str | Path,
|
|
16
|
+
) -> None:
|
|
17
|
+
"""
|
|
18
|
+
Generate a Datapilot HTML report.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
output_path = Path(output_path)
|
|
22
|
+
|
|
23
|
+
project_root = (
|
|
24
|
+
Path(__file__).parent.parent
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
template = (
|
|
28
|
+
project_root
|
|
29
|
+
/ "templates"
|
|
30
|
+
/ "report.html"
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
css_source = (
|
|
34
|
+
project_root
|
|
35
|
+
/ "assets"
|
|
36
|
+
/ "style.css"
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
renderer = HTMLRenderer(
|
|
40
|
+
template_path=template,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
renderer.render(
|
|
44
|
+
report=report,
|
|
45
|
+
output_path=output_path,
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
shutil.copy2(
|
|
49
|
+
css_source,
|
|
50
|
+
output_path.parent / "style.css",
|
|
51
|
+
)
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Placeholder generation for Datapilot HTML reports.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from datetime import datetime
|
|
6
|
+
|
|
7
|
+
from datapilot.core.report import Report
|
|
8
|
+
|
|
9
|
+
from .fragments import (
|
|
10
|
+
build_column_name_chips,
|
|
11
|
+
build_correlation_table,
|
|
12
|
+
build_health_headline,
|
|
13
|
+
build_health_summary,
|
|
14
|
+
build_html_list,
|
|
15
|
+
build_insight_recommendations,
|
|
16
|
+
build_insights,
|
|
17
|
+
build_missing_table,
|
|
18
|
+
build_outlier_cards,
|
|
19
|
+
build_statistics_table,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
DATAPILOT_VERSION = "0.2.0"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def build_placeholders(
|
|
26
|
+
report: Report,
|
|
27
|
+
) -> dict[str, str]:
|
|
28
|
+
"""
|
|
29
|
+
Build placeholder dictionary for the HTML template.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
summary = report.summary()
|
|
33
|
+
health = report.dataset_health()
|
|
34
|
+
missing = report.missing_values()
|
|
35
|
+
duplicates = report.duplicates()
|
|
36
|
+
|
|
37
|
+
health_class = (
|
|
38
|
+
"kpi-card--good"
|
|
39
|
+
if health.score >= 90
|
|
40
|
+
else "kpi-card--warn"
|
|
41
|
+
if health.score >= 75
|
|
42
|
+
else "kpi-card--bad"
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
return {
|
|
46
|
+
"{{DATASET_NAME}}": "Dataset",
|
|
47
|
+
"{{REPORT_GENERATED_AT}}": datetime.now().strftime(
|
|
48
|
+
"%d %b %Y %H:%M",
|
|
49
|
+
),
|
|
50
|
+
"{{DATAPILOT_VERSION}}": DATAPILOT_VERSION,
|
|
51
|
+
"{{REPORT_SUMMARY}}": (
|
|
52
|
+
"Automated Data Quality Assessment Report"
|
|
53
|
+
),
|
|
54
|
+
|
|
55
|
+
"{{ROWS}}": str(summary.rows),
|
|
56
|
+
"{{COLUMNS}}": str(summary.columns),
|
|
57
|
+
"{{MEMORY_USAGE}}": (
|
|
58
|
+
f"{summary.memory_usage_mb:.2f} MB"
|
|
59
|
+
),
|
|
60
|
+
"{{FILE_FORMAT}}": "CSV",
|
|
61
|
+
|
|
62
|
+
"{{COLUMN_NAME_CHIPS}}": (
|
|
63
|
+
build_column_name_chips(report)
|
|
64
|
+
),
|
|
65
|
+
|
|
66
|
+
"{{HEALTH_SCORE}}": str(health.score),
|
|
67
|
+
"{{GRADE}}": health.grade,
|
|
68
|
+
"{{HEALTH_STATUS}}": health.status,
|
|
69
|
+
|
|
70
|
+
"{{HEALTH_HEADLINE}}": (
|
|
71
|
+
build_health_headline(report)
|
|
72
|
+
),
|
|
73
|
+
"{{HEALTH_SUMMARY}}": (
|
|
74
|
+
build_health_summary(report)
|
|
75
|
+
),
|
|
76
|
+
|
|
77
|
+
"{{STRENGTHS_LIST}}": (
|
|
78
|
+
build_html_list(
|
|
79
|
+
health.strengths,
|
|
80
|
+
)
|
|
81
|
+
),
|
|
82
|
+
"{{WEAKNESSES_LIST}}": (
|
|
83
|
+
build_html_list(
|
|
84
|
+
health.weaknesses,
|
|
85
|
+
)
|
|
86
|
+
),
|
|
87
|
+
"{{RECOMMENDATIONS_LIST}}": (
|
|
88
|
+
build_html_list(
|
|
89
|
+
health.recommendations,
|
|
90
|
+
)
|
|
91
|
+
),
|
|
92
|
+
|
|
93
|
+
"{{HEALTH_SCORE_SEVERITY_CLASS}}": health_class,
|
|
94
|
+
"{{HEALTH_SCORE_SEVERITY_CLASS_GAUGE}}": health_class,
|
|
95
|
+
|
|
96
|
+
"{{MISSING_VALUES_COUNT}}": str(
|
|
97
|
+
missing.total_missing,
|
|
98
|
+
),
|
|
99
|
+
"{{MISSING_VALUES_PERCENT}}": (
|
|
100
|
+
f"{missing.missing_percentage:.1f}"
|
|
101
|
+
),
|
|
102
|
+
|
|
103
|
+
"{{MISSING_TABLE}}": (
|
|
104
|
+
build_missing_table(report)
|
|
105
|
+
),
|
|
106
|
+
|
|
107
|
+
"{{STATISTICS_TABLE}}": (
|
|
108
|
+
build_statistics_table(report)
|
|
109
|
+
),
|
|
110
|
+
|
|
111
|
+
"{{OUTLIER_CARDS}}": (
|
|
112
|
+
build_outlier_cards(report)
|
|
113
|
+
),
|
|
114
|
+
|
|
115
|
+
"{{CORRELATION_TABLE}}": (
|
|
116
|
+
build_correlation_table(report)
|
|
117
|
+
),
|
|
118
|
+
|
|
119
|
+
"{{INSIGHTS}}": (
|
|
120
|
+
build_insights(report)
|
|
121
|
+
),
|
|
122
|
+
|
|
123
|
+
"{{RECOMMENDATIONS}}": (
|
|
124
|
+
build_insight_recommendations(report)
|
|
125
|
+
),
|
|
126
|
+
|
|
127
|
+
"{{DUPLICATE_ROWS_COUNT}}": str(
|
|
128
|
+
duplicates.total_duplicates,
|
|
129
|
+
),
|
|
130
|
+
"{{DUPLICATE_ROWS_PERCENT}}": (
|
|
131
|
+
f"{duplicates.duplicate_percentage:.1f}"
|
|
132
|
+
),
|
|
133
|
+
}
|