datasetdna 0.1.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- datasetdna/__init__.py +1 -0
- datasetdna/cli.py +307 -0
- datasetdna/profiler/__init__.py +0 -0
- datasetdna/profiler/cardinality.py +63 -0
- datasetdna/profiler/categorical.py +117 -0
- datasetdna/profiler/correlations.py +313 -0
- datasetdna/profiler/duplicates.py +188 -0
- datasetdna/profiler/missing.py +25 -0
- datasetdna/profiler/numerical.py +88 -0
- datasetdna/profiler/outliers.py +200 -0
- datasetdna/profiler/overview.py +16 -0
- datasetdna/profiler/schema.py +12 -0
- datasetdna/profiler/target.py +273 -0
- datasetdna/profiler/types.py +204 -0
- datasetdna/recommendations/recommendations.py +838 -0
- datasetdna/reporting/__init__.py +0 -0
- datasetdna/reporting/console.py +1040 -0
- datasetdna/reporting/html.py +2068 -0
- datasetdna/scoring/health_score.py +785 -0
- datasetdna/utils/__init__.py +0 -0
- datasetdna/utils/helpers.py +387 -0
- datasetdna-0.1.2.dist-info/METADATA +176 -0
- datasetdna-0.1.2.dist-info/RECORD +27 -0
- datasetdna-0.1.2.dist-info/WHEEL +5 -0
- datasetdna-0.1.2.dist-info/entry_points.txt +2 -0
- datasetdna-0.1.2.dist-info/licenses/LICENSE +21 -0
- datasetdna-0.1.2.dist-info/top_level.txt +1 -0
datasetdna/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
datasetdna/cli.py
ADDED
|
@@ -0,0 +1,307 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
|
|
5
|
+
import typer
|
|
6
|
+
|
|
7
|
+
from datasetdna.profiler.overview import check_overview
|
|
8
|
+
from datasetdna.profiler.schema import check_schema
|
|
9
|
+
from datasetdna.profiler.missing import check_missing
|
|
10
|
+
from datasetdna.profiler.duplicates import check_duplicates
|
|
11
|
+
from datasetdna.profiler.cardinality import check_cardinality
|
|
12
|
+
from datasetdna.profiler.numerical import check_numerical
|
|
13
|
+
from datasetdna.profiler.categorical import check_categorical
|
|
14
|
+
from datasetdna.profiler.outliers import check_outliers
|
|
15
|
+
from datasetdna.profiler.correlations import check_correlations
|
|
16
|
+
from datasetdna.profiler.target import check_target
|
|
17
|
+
|
|
18
|
+
from datasetdna.scoring.health_score import (
|
|
19
|
+
calculate_health_score,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
from datasetdna.recommendations.recommendations import (
|
|
23
|
+
generate_recommendations,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
from datasetdna.reporting.console import (
|
|
27
|
+
console,
|
|
28
|
+
render_report,
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
from datasetdna.reporting.html import (
|
|
32
|
+
render_html_report,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
from datasetdna.utils.helpers import (
|
|
36
|
+
LARGE_FILE_SIZE_BYTES,
|
|
37
|
+
LARGE_FILE_SAMPLE_SIZE,
|
|
38
|
+
load_dataset,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
app = typer.Typer(
|
|
43
|
+
help="DatasetDNA - Automated Dataset Health Profiler",
|
|
44
|
+
add_completion=False,
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
# =============================================================
|
|
49
|
+
# TARGET INFERENCE
|
|
50
|
+
# =============================================================
|
|
51
|
+
|
|
52
|
+
TARGET_COLUMN_CANDIDATES = (
|
|
53
|
+
"target",
|
|
54
|
+
"label",
|
|
55
|
+
"churn",
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def infer_target(
|
|
60
|
+
df,
|
|
61
|
+
) -> tuple[str | None, bool]:
|
|
62
|
+
"""
|
|
63
|
+
Infer a target column when the user does not provide one.
|
|
64
|
+
|
|
65
|
+
Priority:
|
|
66
|
+
target -> label -> churn
|
|
67
|
+
|
|
68
|
+
Returns:
|
|
69
|
+
(column_name, inferred)
|
|
70
|
+
"""
|
|
71
|
+
|
|
72
|
+
normalized_columns = {
|
|
73
|
+
column.strip().lower(): column
|
|
74
|
+
for column in df.columns
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
for candidate in TARGET_COLUMN_CANDIDATES:
|
|
78
|
+
|
|
79
|
+
if candidate in normalized_columns:
|
|
80
|
+
return (
|
|
81
|
+
normalized_columns[candidate],
|
|
82
|
+
True,
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
return None, False
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
# =============================================================
|
|
89
|
+
# LARGE FILE WARNING
|
|
90
|
+
# =============================================================
|
|
91
|
+
|
|
92
|
+
def warn_if_large_file(
|
|
93
|
+
path: str,
|
|
94
|
+
) -> None:
|
|
95
|
+
"""
|
|
96
|
+
Warn the user when DatasetDNA will analyze a sample
|
|
97
|
+
instead of the complete dataset.
|
|
98
|
+
"""
|
|
99
|
+
|
|
100
|
+
if not os.path.exists(path):
|
|
101
|
+
return
|
|
102
|
+
|
|
103
|
+
if os.path.getsize(path) > LARGE_FILE_SIZE_BYTES:
|
|
104
|
+
typer.echo(
|
|
105
|
+
f"Large file detected — analyzing a "
|
|
106
|
+
f"{LARGE_FILE_SAMPLE_SIZE:,}-row sample."
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
# =============================================================
|
|
111
|
+
# CLI COMMAND
|
|
112
|
+
# =============================================================
|
|
113
|
+
|
|
114
|
+
@app.command()
|
|
115
|
+
def profile(
|
|
116
|
+
file: str = typer.Argument(
|
|
117
|
+
...,
|
|
118
|
+
help="Path to the CSV file.",
|
|
119
|
+
),
|
|
120
|
+
target: str | None = typer.Option(
|
|
121
|
+
None,
|
|
122
|
+
"--target",
|
|
123
|
+
"-t",
|
|
124
|
+
help="Optional target column.",
|
|
125
|
+
),
|
|
126
|
+
html: bool = typer.Option(
|
|
127
|
+
False,
|
|
128
|
+
"--html",
|
|
129
|
+
help="Generate an HTML report.",
|
|
130
|
+
),
|
|
131
|
+
output: str = typer.Option(
|
|
132
|
+
"datasetdna_report.html",
|
|
133
|
+
"--output",
|
|
134
|
+
"-o",
|
|
135
|
+
help="HTML output file path.",
|
|
136
|
+
),
|
|
137
|
+
):
|
|
138
|
+
"""
|
|
139
|
+
Profile a CSV dataset and generate a health report.
|
|
140
|
+
"""
|
|
141
|
+
|
|
142
|
+
try:
|
|
143
|
+
|
|
144
|
+
# ====================================================
|
|
145
|
+
# LARGE FILE WARNING
|
|
146
|
+
# ====================================================
|
|
147
|
+
|
|
148
|
+
warn_if_large_file(file)
|
|
149
|
+
|
|
150
|
+
# ====================================================
|
|
151
|
+
# LOAD DATASET
|
|
152
|
+
# ====================================================
|
|
153
|
+
|
|
154
|
+
df = load_dataset(file)
|
|
155
|
+
|
|
156
|
+
# ====================================================
|
|
157
|
+
# TARGET INFERENCE
|
|
158
|
+
# ====================================================
|
|
159
|
+
|
|
160
|
+
target_was_inferred = False
|
|
161
|
+
|
|
162
|
+
if target is None:
|
|
163
|
+
|
|
164
|
+
target, target_was_inferred = infer_target(
|
|
165
|
+
df
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
# ====================================================
|
|
169
|
+
# RUN PROFILERS
|
|
170
|
+
# ====================================================
|
|
171
|
+
|
|
172
|
+
overview = check_overview(df)
|
|
173
|
+
|
|
174
|
+
schema = check_schema(df)
|
|
175
|
+
|
|
176
|
+
missing = check_missing(df)
|
|
177
|
+
|
|
178
|
+
duplicates = check_duplicates(df)
|
|
179
|
+
|
|
180
|
+
cardinality = check_cardinality(df)
|
|
181
|
+
|
|
182
|
+
numerical = check_numerical(df)
|
|
183
|
+
|
|
184
|
+
categorical = check_categorical(df)
|
|
185
|
+
|
|
186
|
+
outliers = check_outliers(df)
|
|
187
|
+
|
|
188
|
+
correlations = check_correlations(df)
|
|
189
|
+
|
|
190
|
+
target_result = check_target(
|
|
191
|
+
df,
|
|
192
|
+
target,
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
# ====================================================
|
|
196
|
+
# MARK INFERRED TARGET
|
|
197
|
+
# ====================================================
|
|
198
|
+
|
|
199
|
+
if target_was_inferred:
|
|
200
|
+
target_result["inferred"] = True
|
|
201
|
+
|
|
202
|
+
console.print(
|
|
203
|
+
f"[yellow]ℹ Auto-inferred '{target}' as the primary target variable.[/yellow]\n"
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
# ====================================================
|
|
207
|
+
# COLLECT RESULTS
|
|
208
|
+
# ====================================================
|
|
209
|
+
|
|
210
|
+
results = {
|
|
211
|
+
"overview": overview,
|
|
212
|
+
"schema": schema,
|
|
213
|
+
"missing": missing,
|
|
214
|
+
"duplicates": duplicates,
|
|
215
|
+
"cardinality": cardinality,
|
|
216
|
+
"numerical": numerical,
|
|
217
|
+
"categorical": categorical,
|
|
218
|
+
"outliers": outliers,
|
|
219
|
+
"correlations": correlations,
|
|
220
|
+
"target": target_result,
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
# ====================================================
|
|
224
|
+
# HEALTH SCORE
|
|
225
|
+
# ====================================================
|
|
226
|
+
|
|
227
|
+
health = calculate_health_score(
|
|
228
|
+
results
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
results["health"] = health
|
|
232
|
+
|
|
233
|
+
# ====================================================
|
|
234
|
+
# RECOMMENDATIONS
|
|
235
|
+
# ====================================================
|
|
236
|
+
|
|
237
|
+
recommendations = generate_recommendations(
|
|
238
|
+
results
|
|
239
|
+
)
|
|
240
|
+
|
|
241
|
+
results["recommendations"] = recommendations
|
|
242
|
+
|
|
243
|
+
# ====================================================
|
|
244
|
+
# CONSOLE REPORT
|
|
245
|
+
# ====================================================
|
|
246
|
+
|
|
247
|
+
render_report(
|
|
248
|
+
results,
|
|
249
|
+
health,
|
|
250
|
+
recommendations,
|
|
251
|
+
)
|
|
252
|
+
|
|
253
|
+
# ====================================================
|
|
254
|
+
# HTML REPORT
|
|
255
|
+
# ====================================================
|
|
256
|
+
|
|
257
|
+
if html:
|
|
258
|
+
|
|
259
|
+
output_path = render_html_report(
|
|
260
|
+
results,
|
|
261
|
+
output_path=output,
|
|
262
|
+
)
|
|
263
|
+
|
|
264
|
+
typer.echo(
|
|
265
|
+
f"\nHTML report generated: {output_path}"
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
except FileNotFoundError as error:
|
|
269
|
+
|
|
270
|
+
typer.echo(
|
|
271
|
+
f"Error: {error}",
|
|
272
|
+
err=True,
|
|
273
|
+
)
|
|
274
|
+
|
|
275
|
+
raise typer.Exit(
|
|
276
|
+
code=1
|
|
277
|
+
)
|
|
278
|
+
|
|
279
|
+
except ValueError as error:
|
|
280
|
+
|
|
281
|
+
typer.echo(
|
|
282
|
+
f"Error: {error}",
|
|
283
|
+
err=True,
|
|
284
|
+
)
|
|
285
|
+
|
|
286
|
+
raise typer.Exit(
|
|
287
|
+
code=1
|
|
288
|
+
)
|
|
289
|
+
|
|
290
|
+
except Exception as error:
|
|
291
|
+
|
|
292
|
+
typer.echo(
|
|
293
|
+
f"Unexpected error: {error}",
|
|
294
|
+
err=True,
|
|
295
|
+
)
|
|
296
|
+
|
|
297
|
+
raise typer.Exit(
|
|
298
|
+
code=1
|
|
299
|
+
)
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
# =============================================================
|
|
303
|
+
# ENTRY POINT
|
|
304
|
+
# =============================================================
|
|
305
|
+
|
|
306
|
+
if __name__ == "__main__":
|
|
307
|
+
app()
|
|
File without changes
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from datasetdna.profiler.types import check_types
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def check_cardinality(
|
|
9
|
+
df: pd.DataFrame,
|
|
10
|
+
) -> dict:
|
|
11
|
+
"""
|
|
12
|
+
Analyze column cardinality.
|
|
13
|
+
|
|
14
|
+
Cardinality is calculated for every column, but identifier
|
|
15
|
+
columns are marked as ID-like so downstream scoring can
|
|
16
|
+
distinguish expected high cardinality from suspicious
|
|
17
|
+
high cardinality.
|
|
18
|
+
|
|
19
|
+
Identifier columns are NOT removed from the raw cardinality
|
|
20
|
+
report because their uniqueness is still useful information.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
result = {}
|
|
24
|
+
|
|
25
|
+
types = check_types(df)
|
|
26
|
+
|
|
27
|
+
total_rows = len(df)
|
|
28
|
+
|
|
29
|
+
for column in df.columns:
|
|
30
|
+
|
|
31
|
+
unique_count = int(
|
|
32
|
+
df[column].nunique(
|
|
33
|
+
dropna=True
|
|
34
|
+
)
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
unique_percentage = (
|
|
38
|
+
(
|
|
39
|
+
unique_count
|
|
40
|
+
/ total_rows
|
|
41
|
+
)
|
|
42
|
+
* 100
|
|
43
|
+
if total_rows > 0
|
|
44
|
+
else 0.0
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
column_type = types[column]
|
|
48
|
+
|
|
49
|
+
result[column] = {
|
|
50
|
+
"unique_count": unique_count,
|
|
51
|
+
"unique_percentage": round(
|
|
52
|
+
float(unique_percentage),
|
|
53
|
+
2,
|
|
54
|
+
),
|
|
55
|
+
"detected_type": column_type[
|
|
56
|
+
"detected_type"
|
|
57
|
+
],
|
|
58
|
+
"is_id_like": column_type[
|
|
59
|
+
"is_id_like"
|
|
60
|
+
],
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
return result
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from datasetdna.profiler.types import check_types
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def check_categorical(
|
|
9
|
+
df: pd.DataFrame,
|
|
10
|
+
) -> dict:
|
|
11
|
+
"""
|
|
12
|
+
Analyze categorical columns and binary numeric columns.
|
|
13
|
+
|
|
14
|
+
Columns detected as:
|
|
15
|
+
- categorical
|
|
16
|
+
- binary numeric (0/1)
|
|
17
|
+
|
|
18
|
+
are analyzed.
|
|
19
|
+
|
|
20
|
+
Columns detected as:
|
|
21
|
+
- id
|
|
22
|
+
- numeric (non-binary)
|
|
23
|
+
- date
|
|
24
|
+
- boolean
|
|
25
|
+
- empty
|
|
26
|
+
|
|
27
|
+
are excluded automatically.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
result = {}
|
|
31
|
+
|
|
32
|
+
types = check_types(df)
|
|
33
|
+
|
|
34
|
+
for column in df.columns:
|
|
35
|
+
|
|
36
|
+
detected_type = types[column]["detected_type"]
|
|
37
|
+
|
|
38
|
+
# -----------------------------------------------------
|
|
39
|
+
# Determine whether column should be analyzed
|
|
40
|
+
# -----------------------------------------------------
|
|
41
|
+
|
|
42
|
+
if detected_type == "categorical":
|
|
43
|
+
should_analyze = True
|
|
44
|
+
|
|
45
|
+
elif detected_type == "numeric":
|
|
46
|
+
# Binary numeric columns such as 0/1 behave like
|
|
47
|
+
# categorical flags and should have frequency analysis.
|
|
48
|
+
unique_values = df[column].dropna().unique()
|
|
49
|
+
|
|
50
|
+
should_analyze = (
|
|
51
|
+
len(unique_values) <= 2
|
|
52
|
+
and set(unique_values).issubset({0, 1})
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
else:
|
|
56
|
+
should_analyze = False
|
|
57
|
+
|
|
58
|
+
if not should_analyze:
|
|
59
|
+
continue
|
|
60
|
+
|
|
61
|
+
series = df[column].dropna()
|
|
62
|
+
|
|
63
|
+
# -----------------------------------------------------
|
|
64
|
+
# Empty column
|
|
65
|
+
# -----------------------------------------------------
|
|
66
|
+
|
|
67
|
+
if series.empty:
|
|
68
|
+
|
|
69
|
+
result[column] = {
|
|
70
|
+
"unique_count": 0,
|
|
71
|
+
"total_values": 0,
|
|
72
|
+
"categories": {},
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
continue
|
|
76
|
+
|
|
77
|
+
# -----------------------------------------------------
|
|
78
|
+
# Category frequencies
|
|
79
|
+
# -----------------------------------------------------
|
|
80
|
+
|
|
81
|
+
value_counts = series.value_counts()
|
|
82
|
+
|
|
83
|
+
total_values = len(series)
|
|
84
|
+
|
|
85
|
+
categories = {}
|
|
86
|
+
|
|
87
|
+
for category, count in value_counts.items():
|
|
88
|
+
|
|
89
|
+
percentage = (
|
|
90
|
+
count
|
|
91
|
+
/ total_values
|
|
92
|
+
* 100
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
categories[str(category)] = {
|
|
96
|
+
"count": int(count),
|
|
97
|
+
"percentage": round(
|
|
98
|
+
float(percentage),
|
|
99
|
+
2,
|
|
100
|
+
),
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
# -----------------------------------------------------
|
|
104
|
+
# Result
|
|
105
|
+
# -----------------------------------------------------
|
|
106
|
+
|
|
107
|
+
result[column] = {
|
|
108
|
+
"unique_count": int(
|
|
109
|
+
series.nunique()
|
|
110
|
+
),
|
|
111
|
+
"total_values": int(
|
|
112
|
+
total_values
|
|
113
|
+
),
|
|
114
|
+
"categories": categories,
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
return result
|