datapilot-kit 0.4.0__tar.gz → 0.4.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {datapilot_kit-0.4.0/datapilot_kit.egg-info → datapilot_kit-0.4.1}/PKG-INFO +1 -1
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/__init__.py +2 -1
- datapilot_kit-0.4.1/datapilot/analysis/consistency.py +68 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/health.py +27 -23
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/models.py +10 -1
- datapilot_kit-0.4.1/datapilot/analysis/scoring.py +51 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1/datapilot_kit.egg-info}/PKG-INFO +1 -1
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot_kit.egg-info/SOURCES.txt +2 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/pyproject.toml +1 -1
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/LICENSE +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/MANIFEST.in +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/README.md +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/SPECIFICATION.md +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/__init__.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/correlation.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/data_integrity.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/datatype.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/duplicate.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/insights.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/missing.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/ml_readiness.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/notebook_readiness.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/outliers.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/statistical_profile.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/statistics.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/summary.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/api/__init__.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/api/analyze.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/assets/style.css +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/cli/__init__.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/core/__init__.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/core/loader.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/core/report.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/interpretation/__init__.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/llm/__init__.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/recommendation/__init__.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/reporting/__init__.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/reporting/fragments.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/reporting/html.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/reporting/placeholders.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/reporting/renderer.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/templates/report.html +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/__init__.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/cards.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/console.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/dashboard.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/panels.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/renderer.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/sections.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/tables.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/theme.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/utils/__init__.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/visualization/__init__.py +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot_kit.egg-info/dependency_links.txt +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot_kit.egg-info/requires.txt +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot_kit.egg-info/top_level.txt +0 -0
- {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/setup.cfg +0 -0
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Deterministic consistency analysis for Datapilot v0.4.1.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from collections import Counter
|
|
6
|
+
|
|
7
|
+
import pandas as pd
|
|
8
|
+
|
|
9
|
+
from .models import ConsistencySummary
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def generate_consistency_summary(
|
|
13
|
+
dataframe: pd.DataFrame,
|
|
14
|
+
) -> ConsistencySummary:
|
|
15
|
+
"""
|
|
16
|
+
Detect objectively observable type inconsistencies.
|
|
17
|
+
|
|
18
|
+
Missing values are excluded from type analysis because they are
|
|
19
|
+
assessed separately by Dataset Completeness.
|
|
20
|
+
|
|
21
|
+
Consistency degradation is based on the proportion of non-missing
|
|
22
|
+
values whose underlying type differs from the dominant type within
|
|
23
|
+
their column.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
inconsistent_columns: list[str] = []
|
|
27
|
+
inconsistent_value_count = 0
|
|
28
|
+
total_non_missing = 0
|
|
29
|
+
|
|
30
|
+
for column in dataframe.columns:
|
|
31
|
+
series = dataframe[column].dropna()
|
|
32
|
+
|
|
33
|
+
if series.empty:
|
|
34
|
+
continue
|
|
35
|
+
|
|
36
|
+
type_counts = Counter(
|
|
37
|
+
type(value)
|
|
38
|
+
for value in series
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
column_total = len(series)
|
|
42
|
+
dominant_count = max(type_counts.values())
|
|
43
|
+
column_inconsistent_count = (
|
|
44
|
+
column_total - dominant_count
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
total_non_missing += column_total
|
|
48
|
+
|
|
49
|
+
if column_inconsistent_count > 0:
|
|
50
|
+
inconsistent_columns.append(str(column))
|
|
51
|
+
inconsistent_value_count += column_inconsistent_count
|
|
52
|
+
|
|
53
|
+
if total_non_missing == 0:
|
|
54
|
+
consistency_percentage = 0.0
|
|
55
|
+
else:
|
|
56
|
+
consistency_percentage = (
|
|
57
|
+
inconsistent_value_count
|
|
58
|
+
/ total_non_missing
|
|
59
|
+
) * 100.0
|
|
60
|
+
|
|
61
|
+
return ConsistencySummary(
|
|
62
|
+
inconsistent_columns=inconsistent_columns,
|
|
63
|
+
inconsistent_value_count=inconsistent_value_count,
|
|
64
|
+
consistency_percentage=round(
|
|
65
|
+
consistency_percentage,
|
|
66
|
+
2,
|
|
67
|
+
),
|
|
68
|
+
)
|
|
@@ -5,9 +5,11 @@ Dataset health assessment for Datapilot v0.4.
|
|
|
5
5
|
import pandas as pd
|
|
6
6
|
|
|
7
7
|
from .datatype import generate_data_type_summary
|
|
8
|
+
from .consistency import generate_consistency_summary
|
|
8
9
|
from .duplicate import generate_duplicate_summary
|
|
9
10
|
from .missing import generate_missing_value_summary
|
|
10
11
|
from .models import DatasetHealth
|
|
12
|
+
from .scoring import degradation_score
|
|
11
13
|
|
|
12
14
|
|
|
13
15
|
def _grade_and_status(score: float) -> tuple[str, str]:
|
|
@@ -46,30 +48,23 @@ def generate_dataset_health(
|
|
|
46
48
|
missing = generate_missing_value_summary(dataframe)
|
|
47
49
|
duplicates = generate_duplicate_summary(dataframe)
|
|
48
50
|
data_types = generate_data_type_summary(dataframe)
|
|
51
|
+
consistency = generate_consistency_summary(dataframe)
|
|
49
52
|
|
|
50
53
|
# ------------------------------------------------------------------
|
|
51
54
|
# Completeness
|
|
52
55
|
# ------------------------------------------------------------------
|
|
53
56
|
|
|
54
|
-
completeness_score =
|
|
55
|
-
|
|
56
|
-
min(
|
|
57
|
-
100.0,
|
|
58
|
-
100.0 - missing.missing_percentage,
|
|
59
|
-
),
|
|
57
|
+
completeness_score = degradation_score(
|
|
58
|
+
missing.missing_percentage,
|
|
60
59
|
)
|
|
61
60
|
|
|
62
61
|
# ------------------------------------------------------------------
|
|
63
62
|
# Duplicates
|
|
64
63
|
# ------------------------------------------------------------------
|
|
65
64
|
|
|
66
|
-
duplicate_score =
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
100.0,
|
|
70
|
-
100.0 - duplicates.duplicate_percentage,
|
|
71
|
-
),
|
|
72
|
-
)
|
|
65
|
+
duplicate_score = degradation_score(
|
|
66
|
+
duplicates.duplicate_percentage,
|
|
67
|
+
)
|
|
73
68
|
|
|
74
69
|
# ------------------------------------------------------------------
|
|
75
70
|
# Structure
|
|
@@ -140,16 +135,10 @@ def generate_dataset_health(
|
|
|
140
135
|
# ------------------------------------------------------------------
|
|
141
136
|
# Consistency
|
|
142
137
|
# ------------------------------------------------------------------
|
|
143
|
-
#
|
|
144
|
-
# v0.4 currently has no objectively defined consistency metric
|
|
145
|
-
# implemented in the analysis layer.
|
|
146
|
-
#
|
|
147
|
-
# Therefore consistency is neutral until its metric specification
|
|
148
|
-
# is finalized. This prevents an unsupported penalty from being
|
|
149
|
-
# introduced into Dataset Health.
|
|
150
|
-
#
|
|
151
138
|
|
|
152
|
-
consistency_score =
|
|
139
|
+
consistency_score = degradation_score(
|
|
140
|
+
consistency.consistency_percentage,
|
|
141
|
+
)
|
|
153
142
|
|
|
154
143
|
# ------------------------------------------------------------------
|
|
155
144
|
# Overall score
|
|
@@ -187,6 +176,11 @@ def generate_dataset_health(
|
|
|
187
176
|
"No duplicate rows detected."
|
|
188
177
|
)
|
|
189
178
|
|
|
179
|
+
if not consistency.inconsistent_columns:
|
|
180
|
+
strengths.append(
|
|
181
|
+
"No mixed underlying value types detected."
|
|
182
|
+
)
|
|
183
|
+
|
|
190
184
|
if recognized_columns == total_columns:
|
|
191
185
|
strengths.append(
|
|
192
186
|
"All columns have recognized data types."
|
|
@@ -223,6 +217,11 @@ def generate_dataset_health(
|
|
|
223
217
|
0,
|
|
224
218
|
)
|
|
225
219
|
|
|
220
|
+
if consistency.inconsistent_columns:
|
|
221
|
+
weaknesses.append(
|
|
222
|
+
f"{len(consistency.inconsistent_columns)} columns contain mixed underlying value types."
|
|
223
|
+
)
|
|
224
|
+
|
|
226
225
|
if unrecognized_columns > 0:
|
|
227
226
|
weaknesses.append(
|
|
228
227
|
f"{unrecognized_columns} columns have unsupported "
|
|
@@ -260,6 +259,11 @@ def generate_dataset_health(
|
|
|
260
259
|
"Review unsupported column data types."
|
|
261
260
|
)
|
|
262
261
|
|
|
262
|
+
if consistency.inconsistent_columns:
|
|
263
|
+
recommendations.append(
|
|
264
|
+
"Review columns containing mixed underlying value types."
|
|
265
|
+
)
|
|
266
|
+
|
|
263
267
|
if not recommendations:
|
|
264
268
|
recommendations.append(
|
|
265
269
|
"No major structural data-quality issues detected."
|
|
@@ -288,4 +292,4 @@ def generate_dataset_health(
|
|
|
288
292
|
strengths=strengths,
|
|
289
293
|
weaknesses=weaknesses,
|
|
290
294
|
recommendations=recommendations,
|
|
291
|
-
)
|
|
295
|
+
)
|
|
@@ -162,4 +162,13 @@ class InsightSummary:
|
|
|
162
162
|
"""
|
|
163
163
|
|
|
164
164
|
insights: list[str]
|
|
165
|
-
recommendations: list[str]
|
|
165
|
+
recommendations: list[str]
|
|
166
|
+
@dataclass(slots=True)
|
|
167
|
+
class ConsistencySummary:
|
|
168
|
+
"""
|
|
169
|
+
Summary of objectively detectable consistency signals.
|
|
170
|
+
"""
|
|
171
|
+
|
|
172
|
+
inconsistent_columns: list[str]
|
|
173
|
+
inconsistent_value_count: int
|
|
174
|
+
consistency_percentage: float
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Deterministic scoring utilities for Datapilot.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
DEFAULT_DEGRADATION_EXPONENT = 1.2
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def degradation_score(
|
|
12
|
+
degradation_percentage: float,
|
|
13
|
+
exponent: float = DEFAULT_DEGRADATION_EXPONENT,
|
|
14
|
+
) -> float:
|
|
15
|
+
"""
|
|
16
|
+
Convert a degradation percentage into a continuous 0-100 score.
|
|
17
|
+
|
|
18
|
+
The score:
|
|
19
|
+
- is 100 when degradation is 0%
|
|
20
|
+
- reaches 0 when degradation is 100%
|
|
21
|
+
- decreases monotonically as degradation increases
|
|
22
|
+
- remains bounded between 0 and 100
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
if not math.isfinite(degradation_percentage):
|
|
26
|
+
raise ValueError(
|
|
27
|
+
"degradation_percentage must be finite."
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
if not math.isfinite(exponent) or exponent <= 0:
|
|
31
|
+
raise ValueError(
|
|
32
|
+
"exponent must be greater than 0."
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
degradation = max(
|
|
36
|
+
0.0,
|
|
37
|
+
min(
|
|
38
|
+
100.0,
|
|
39
|
+
degradation_percentage,
|
|
40
|
+
),
|
|
41
|
+
) / 100.0
|
|
42
|
+
|
|
43
|
+
score = (
|
|
44
|
+
100.0
|
|
45
|
+
* ((1.0 - degradation) ** exponent)
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
return round(
|
|
49
|
+
max(0.0, min(100.0, score)),
|
|
50
|
+
2,
|
|
51
|
+
)
|
|
@@ -5,6 +5,7 @@ SPECIFICATION.md
|
|
|
5
5
|
pyproject.toml
|
|
6
6
|
datapilot/__init__.py
|
|
7
7
|
datapilot/analysis/__init__.py
|
|
8
|
+
datapilot/analysis/consistency.py
|
|
8
9
|
datapilot/analysis/correlation.py
|
|
9
10
|
datapilot/analysis/data_integrity.py
|
|
10
11
|
datapilot/analysis/datatype.py
|
|
@@ -16,6 +17,7 @@ datapilot/analysis/ml_readiness.py
|
|
|
16
17
|
datapilot/analysis/models.py
|
|
17
18
|
datapilot/analysis/notebook_readiness.py
|
|
18
19
|
datapilot/analysis/outliers.py
|
|
20
|
+
datapilot/analysis/scoring.py
|
|
19
21
|
datapilot/analysis/statistical_profile.py
|
|
20
22
|
datapilot/analysis/statistics.py
|
|
21
23
|
datapilot/analysis/summary.py
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "datapilot-kit"
|
|
7
|
-
version = "0.4.
|
|
7
|
+
version = "0.4.1"
|
|
8
8
|
description = "An open-source Python library for deterministic exploratory data analysis and dataset understanding."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|