datapilot-kit 0.4.0__tar.gz → 0.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. {datapilot_kit-0.4.0/datapilot_kit.egg-info → datapilot_kit-0.4.1}/PKG-INFO +1 -1
  2. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/__init__.py +2 -1
  3. datapilot_kit-0.4.1/datapilot/analysis/consistency.py +68 -0
  4. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/health.py +27 -23
  5. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/models.py +10 -1
  6. datapilot_kit-0.4.1/datapilot/analysis/scoring.py +51 -0
  7. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1/datapilot_kit.egg-info}/PKG-INFO +1 -1
  8. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot_kit.egg-info/SOURCES.txt +2 -0
  9. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/pyproject.toml +1 -1
  10. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/LICENSE +0 -0
  11. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/MANIFEST.in +0 -0
  12. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/README.md +0 -0
  13. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/SPECIFICATION.md +0 -0
  14. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/__init__.py +0 -0
  15. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/correlation.py +0 -0
  16. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/data_integrity.py +0 -0
  17. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/datatype.py +0 -0
  18. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/duplicate.py +0 -0
  19. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/insights.py +0 -0
  20. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/missing.py +0 -0
  21. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/ml_readiness.py +0 -0
  22. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/notebook_readiness.py +0 -0
  23. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/outliers.py +0 -0
  24. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/statistical_profile.py +0 -0
  25. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/statistics.py +0 -0
  26. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/analysis/summary.py +0 -0
  27. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/api/__init__.py +0 -0
  28. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/api/analyze.py +0 -0
  29. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/assets/style.css +0 -0
  30. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/cli/__init__.py +0 -0
  31. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/core/__init__.py +0 -0
  32. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/core/loader.py +0 -0
  33. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/core/report.py +0 -0
  34. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/interpretation/__init__.py +0 -0
  35. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/llm/__init__.py +0 -0
  36. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/recommendation/__init__.py +0 -0
  37. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/reporting/__init__.py +0 -0
  38. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/reporting/fragments.py +0 -0
  39. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/reporting/html.py +0 -0
  40. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/reporting/placeholders.py +0 -0
  41. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/reporting/renderer.py +0 -0
  42. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/templates/report.html +0 -0
  43. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/__init__.py +0 -0
  44. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/cards.py +0 -0
  45. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/console.py +0 -0
  46. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/dashboard.py +0 -0
  47. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/panels.py +0 -0
  48. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/renderer.py +0 -0
  49. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/sections.py +0 -0
  50. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/tables.py +0 -0
  51. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/ui/theme.py +0 -0
  52. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/utils/__init__.py +0 -0
  53. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot/visualization/__init__.py +0 -0
  54. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot_kit.egg-info/dependency_links.txt +0 -0
  55. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot_kit.egg-info/requires.txt +0 -0
  56. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/datapilot_kit.egg-info/top_level.txt +0 -0
  57. {datapilot_kit-0.4.0 → datapilot_kit-0.4.1}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: datapilot-kit
3
- Version: 0.4.0
3
+ Version: 0.4.1
4
4
  Summary: An open-source Python library for deterministic exploratory data analysis and dataset understanding.
5
5
  Author: Avishkar D. Gopale
6
6
  License-Expression: MIT
@@ -4,8 +4,9 @@ Datapilot
4
4
  An intelligent Python library for exploratory data analysis.
5
5
  """
6
6
 
7
- __version__ = "0.4.0"
7
+ __version__ = "0.4.1"
8
8
 
9
9
  from .api import analyze
10
10
 
11
11
  __all__ = ["analyze"]
12
+
@@ -0,0 +1,68 @@
1
+ """
2
+ Deterministic consistency analysis for Datapilot v0.4.1.
3
+ """
4
+
5
+ from collections import Counter
6
+
7
+ import pandas as pd
8
+
9
+ from .models import ConsistencySummary
10
+
11
+
12
+ def generate_consistency_summary(
13
+ dataframe: pd.DataFrame,
14
+ ) -> ConsistencySummary:
15
+ """
16
+ Detect objectively observable type inconsistencies.
17
+
18
+ Missing values are excluded from type analysis because they are
19
+ assessed separately by Dataset Completeness.
20
+
21
+ Consistency degradation is based on the proportion of non-missing
22
+ values whose underlying type differs from the dominant type within
23
+ their column.
24
+ """
25
+
26
+ inconsistent_columns: list[str] = []
27
+ inconsistent_value_count = 0
28
+ total_non_missing = 0
29
+
30
+ for column in dataframe.columns:
31
+ series = dataframe[column].dropna()
32
+
33
+ if series.empty:
34
+ continue
35
+
36
+ type_counts = Counter(
37
+ type(value)
38
+ for value in series
39
+ )
40
+
41
+ column_total = len(series)
42
+ dominant_count = max(type_counts.values())
43
+ column_inconsistent_count = (
44
+ column_total - dominant_count
45
+ )
46
+
47
+ total_non_missing += column_total
48
+
49
+ if column_inconsistent_count > 0:
50
+ inconsistent_columns.append(str(column))
51
+ inconsistent_value_count += column_inconsistent_count
52
+
53
+ if total_non_missing == 0:
54
+ consistency_percentage = 0.0
55
+ else:
56
+ consistency_percentage = (
57
+ inconsistent_value_count
58
+ / total_non_missing
59
+ ) * 100.0
60
+
61
+ return ConsistencySummary(
62
+ inconsistent_columns=inconsistent_columns,
63
+ inconsistent_value_count=inconsistent_value_count,
64
+ consistency_percentage=round(
65
+ consistency_percentage,
66
+ 2,
67
+ ),
68
+ )
@@ -5,9 +5,11 @@ Dataset health assessment for Datapilot v0.4.
5
5
  import pandas as pd
6
6
 
7
7
  from .datatype import generate_data_type_summary
8
+ from .consistency import generate_consistency_summary
8
9
  from .duplicate import generate_duplicate_summary
9
10
  from .missing import generate_missing_value_summary
10
11
  from .models import DatasetHealth
12
+ from .scoring import degradation_score
11
13
 
12
14
 
13
15
  def _grade_and_status(score: float) -> tuple[str, str]:
@@ -46,30 +48,23 @@ def generate_dataset_health(
46
48
  missing = generate_missing_value_summary(dataframe)
47
49
  duplicates = generate_duplicate_summary(dataframe)
48
50
  data_types = generate_data_type_summary(dataframe)
51
+ consistency = generate_consistency_summary(dataframe)
49
52
 
50
53
  # ------------------------------------------------------------------
51
54
  # Completeness
52
55
  # ------------------------------------------------------------------
53
56
 
54
- completeness_score = max(
55
- 0.0,
56
- min(
57
- 100.0,
58
- 100.0 - missing.missing_percentage,
59
- ),
57
+ completeness_score = degradation_score(
58
+ missing.missing_percentage,
60
59
  )
61
60
 
62
61
  # ------------------------------------------------------------------
63
62
  # Duplicates
64
63
  # ------------------------------------------------------------------
65
64
 
66
- duplicate_score = max(
67
- 0.0,
68
- min(
69
- 100.0,
70
- 100.0 - duplicates.duplicate_percentage,
71
- ),
72
- )
65
+ duplicate_score = degradation_score(
66
+ duplicates.duplicate_percentage,
67
+ )
73
68
 
74
69
  # ------------------------------------------------------------------
75
70
  # Structure
@@ -140,16 +135,10 @@ def generate_dataset_health(
140
135
  # ------------------------------------------------------------------
141
136
  # Consistency
142
137
  # ------------------------------------------------------------------
143
- #
144
- # v0.4 currently has no objectively defined consistency metric
145
- # implemented in the analysis layer.
146
- #
147
- # Therefore consistency is neutral until its metric specification
148
- # is finalized. This prevents an unsupported penalty from being
149
- # introduced into Dataset Health.
150
- #
151
138
 
152
- consistency_score = 100.0
139
+ consistency_score = degradation_score(
140
+ consistency.consistency_percentage,
141
+ )
153
142
 
154
143
  # ------------------------------------------------------------------
155
144
  # Overall score
@@ -187,6 +176,11 @@ def generate_dataset_health(
187
176
  "No duplicate rows detected."
188
177
  )
189
178
 
179
+ if not consistency.inconsistent_columns:
180
+ strengths.append(
181
+ "No mixed underlying value types detected."
182
+ )
183
+
190
184
  if recognized_columns == total_columns:
191
185
  strengths.append(
192
186
  "All columns have recognized data types."
@@ -223,6 +217,11 @@ def generate_dataset_health(
223
217
  0,
224
218
  )
225
219
 
220
+ if consistency.inconsistent_columns:
221
+ weaknesses.append(
222
+ f"{len(consistency.inconsistent_columns)} columns contain mixed underlying value types."
223
+ )
224
+
226
225
  if unrecognized_columns > 0:
227
226
  weaknesses.append(
228
227
  f"{unrecognized_columns} columns have unsupported "
@@ -260,6 +259,11 @@ def generate_dataset_health(
260
259
  "Review unsupported column data types."
261
260
  )
262
261
 
262
+ if consistency.inconsistent_columns:
263
+ recommendations.append(
264
+ "Review columns containing mixed underlying value types."
265
+ )
266
+
263
267
  if not recommendations:
264
268
  recommendations.append(
265
269
  "No major structural data-quality issues detected."
@@ -288,4 +292,4 @@ def generate_dataset_health(
288
292
  strengths=strengths,
289
293
  weaknesses=weaknesses,
290
294
  recommendations=recommendations,
291
- )
295
+ )
@@ -162,4 +162,13 @@ class InsightSummary:
162
162
  """
163
163
 
164
164
  insights: list[str]
165
- recommendations: list[str]
165
+ recommendations: list[str]
166
+ @dataclass(slots=True)
167
+ class ConsistencySummary:
168
+ """
169
+ Summary of objectively detectable consistency signals.
170
+ """
171
+
172
+ inconsistent_columns: list[str]
173
+ inconsistent_value_count: int
174
+ consistency_percentage: float
@@ -0,0 +1,51 @@
1
+ """
2
+ Deterministic scoring utilities for Datapilot.
3
+ """
4
+
5
+ import math
6
+
7
+
8
+ DEFAULT_DEGRADATION_EXPONENT = 1.2
9
+
10
+
11
+ def degradation_score(
12
+ degradation_percentage: float,
13
+ exponent: float = DEFAULT_DEGRADATION_EXPONENT,
14
+ ) -> float:
15
+ """
16
+ Convert a degradation percentage into a continuous 0-100 score.
17
+
18
+ The score:
19
+ - is 100 when degradation is 0%
20
+ - reaches 0 when degradation is 100%
21
+ - decreases monotonically as degradation increases
22
+ - remains bounded between 0 and 100
23
+ """
24
+
25
+ if not math.isfinite(degradation_percentage):
26
+ raise ValueError(
27
+ "degradation_percentage must be finite."
28
+ )
29
+
30
+ if not math.isfinite(exponent) or exponent <= 0:
31
+ raise ValueError(
32
+ "exponent must be greater than 0."
33
+ )
34
+
35
+ degradation = max(
36
+ 0.0,
37
+ min(
38
+ 100.0,
39
+ degradation_percentage,
40
+ ),
41
+ ) / 100.0
42
+
43
+ score = (
44
+ 100.0
45
+ * ((1.0 - degradation) ** exponent)
46
+ )
47
+
48
+ return round(
49
+ max(0.0, min(100.0, score)),
50
+ 2,
51
+ )
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: datapilot-kit
3
- Version: 0.4.0
3
+ Version: 0.4.1
4
4
  Summary: An open-source Python library for deterministic exploratory data analysis and dataset understanding.
5
5
  Author: Avishkar D. Gopale
6
6
  License-Expression: MIT
@@ -5,6 +5,7 @@ SPECIFICATION.md
5
5
  pyproject.toml
6
6
  datapilot/__init__.py
7
7
  datapilot/analysis/__init__.py
8
+ datapilot/analysis/consistency.py
8
9
  datapilot/analysis/correlation.py
9
10
  datapilot/analysis/data_integrity.py
10
11
  datapilot/analysis/datatype.py
@@ -16,6 +17,7 @@ datapilot/analysis/ml_readiness.py
16
17
  datapilot/analysis/models.py
17
18
  datapilot/analysis/notebook_readiness.py
18
19
  datapilot/analysis/outliers.py
20
+ datapilot/analysis/scoring.py
19
21
  datapilot/analysis/statistical_profile.py
20
22
  datapilot/analysis/statistics.py
21
23
  datapilot/analysis/summary.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "datapilot-kit"
7
- version = "0.4.0"
7
+ version = "0.4.1"
8
8
  description = "An open-source Python library for deterministic exploratory data analysis and dataset understanding."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
File without changes
File without changes
File without changes
File without changes