datasetdna 0.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
datasetdna/__init__.py ADDED
@@ -0,0 +1 @@
1
+ __version__ = "0.1.0"
datasetdna/cli.py ADDED
@@ -0,0 +1,307 @@
1
+ from __future__ import annotations
2
+
3
+ import os
4
+
5
+ import typer
6
+
7
+ from datasetdna.profiler.overview import check_overview
8
+ from datasetdna.profiler.schema import check_schema
9
+ from datasetdna.profiler.missing import check_missing
10
+ from datasetdna.profiler.duplicates import check_duplicates
11
+ from datasetdna.profiler.cardinality import check_cardinality
12
+ from datasetdna.profiler.numerical import check_numerical
13
+ from datasetdna.profiler.categorical import check_categorical
14
+ from datasetdna.profiler.outliers import check_outliers
15
+ from datasetdna.profiler.correlations import check_correlations
16
+ from datasetdna.profiler.target import check_target
17
+
18
+ from datasetdna.scoring.health_score import (
19
+ calculate_health_score,
20
+ )
21
+
22
+ from datasetdna.recommendations.recommendations import (
23
+ generate_recommendations,
24
+ )
25
+
26
+ from datasetdna.reporting.console import (
27
+ console,
28
+ render_report,
29
+ )
30
+
31
+ from datasetdna.reporting.html import (
32
+ render_html_report,
33
+ )
34
+
35
+ from datasetdna.utils.helpers import (
36
+ LARGE_FILE_SIZE_BYTES,
37
+ LARGE_FILE_SAMPLE_SIZE,
38
+ load_dataset,
39
+ )
40
+
41
+
42
+ app = typer.Typer(
43
+ help="DatasetDNA - Automated Dataset Health Profiler",
44
+ add_completion=False,
45
+ )
46
+
47
+
48
+ # =============================================================
49
+ # TARGET INFERENCE
50
+ # =============================================================
51
+
52
+ TARGET_COLUMN_CANDIDATES = (
53
+ "target",
54
+ "label",
55
+ "churn",
56
+ )
57
+
58
+
59
+ def infer_target(
60
+ df,
61
+ ) -> tuple[str | None, bool]:
62
+ """
63
+ Infer a target column when the user does not provide one.
64
+
65
+ Priority:
66
+ target -> label -> churn
67
+
68
+ Returns:
69
+ (column_name, inferred)
70
+ """
71
+
72
+ normalized_columns = {
73
+ column.strip().lower(): column
74
+ for column in df.columns
75
+ }
76
+
77
+ for candidate in TARGET_COLUMN_CANDIDATES:
78
+
79
+ if candidate in normalized_columns:
80
+ return (
81
+ normalized_columns[candidate],
82
+ True,
83
+ )
84
+
85
+ return None, False
86
+
87
+
88
+ # =============================================================
89
+ # LARGE FILE WARNING
90
+ # =============================================================
91
+
92
+ def warn_if_large_file(
93
+ path: str,
94
+ ) -> None:
95
+ """
96
+ Warn the user when DatasetDNA will analyze a sample
97
+ instead of the complete dataset.
98
+ """
99
+
100
+ if not os.path.exists(path):
101
+ return
102
+
103
+ if os.path.getsize(path) > LARGE_FILE_SIZE_BYTES:
104
+ typer.echo(
105
+ f"Large file detected — analyzing a "
106
+ f"{LARGE_FILE_SAMPLE_SIZE:,}-row sample."
107
+ )
108
+
109
+
110
+ # =============================================================
111
+ # CLI COMMAND
112
+ # =============================================================
113
+
114
+ @app.command()
115
+ def profile(
116
+ file: str = typer.Argument(
117
+ ...,
118
+ help="Path to the CSV file.",
119
+ ),
120
+ target: str | None = typer.Option(
121
+ None,
122
+ "--target",
123
+ "-t",
124
+ help="Optional target column.",
125
+ ),
126
+ html: bool = typer.Option(
127
+ False,
128
+ "--html",
129
+ help="Generate an HTML report.",
130
+ ),
131
+ output: str = typer.Option(
132
+ "datasetdna_report.html",
133
+ "--output",
134
+ "-o",
135
+ help="HTML output file path.",
136
+ ),
137
+ ):
138
+ """
139
+ Profile a CSV dataset and generate a health report.
140
+ """
141
+
142
+ try:
143
+
144
+ # ====================================================
145
+ # LARGE FILE WARNING
146
+ # ====================================================
147
+
148
+ warn_if_large_file(file)
149
+
150
+ # ====================================================
151
+ # LOAD DATASET
152
+ # ====================================================
153
+
154
+ df = load_dataset(file)
155
+
156
+ # ====================================================
157
+ # TARGET INFERENCE
158
+ # ====================================================
159
+
160
+ target_was_inferred = False
161
+
162
+ if target is None:
163
+
164
+ target, target_was_inferred = infer_target(
165
+ df
166
+ )
167
+
168
+ # ====================================================
169
+ # RUN PROFILERS
170
+ # ====================================================
171
+
172
+ overview = check_overview(df)
173
+
174
+ schema = check_schema(df)
175
+
176
+ missing = check_missing(df)
177
+
178
+ duplicates = check_duplicates(df)
179
+
180
+ cardinality = check_cardinality(df)
181
+
182
+ numerical = check_numerical(df)
183
+
184
+ categorical = check_categorical(df)
185
+
186
+ outliers = check_outliers(df)
187
+
188
+ correlations = check_correlations(df)
189
+
190
+ target_result = check_target(
191
+ df,
192
+ target,
193
+ )
194
+
195
+ # ====================================================
196
+ # MARK INFERRED TARGET
197
+ # ====================================================
198
+
199
+ if target_was_inferred:
200
+ target_result["inferred"] = True
201
+
202
+ console.print(
203
+ f"[yellow]ℹ Auto-inferred '{target}' as the primary target variable.[/yellow]\n"
204
+ )
205
+
206
+ # ====================================================
207
+ # COLLECT RESULTS
208
+ # ====================================================
209
+
210
+ results = {
211
+ "overview": overview,
212
+ "schema": schema,
213
+ "missing": missing,
214
+ "duplicates": duplicates,
215
+ "cardinality": cardinality,
216
+ "numerical": numerical,
217
+ "categorical": categorical,
218
+ "outliers": outliers,
219
+ "correlations": correlations,
220
+ "target": target_result,
221
+ }
222
+
223
+ # ====================================================
224
+ # HEALTH SCORE
225
+ # ====================================================
226
+
227
+ health = calculate_health_score(
228
+ results
229
+ )
230
+
231
+ results["health"] = health
232
+
233
+ # ====================================================
234
+ # RECOMMENDATIONS
235
+ # ====================================================
236
+
237
+ recommendations = generate_recommendations(
238
+ results
239
+ )
240
+
241
+ results["recommendations"] = recommendations
242
+
243
+ # ====================================================
244
+ # CONSOLE REPORT
245
+ # ====================================================
246
+
247
+ render_report(
248
+ results,
249
+ health,
250
+ recommendations,
251
+ )
252
+
253
+ # ====================================================
254
+ # HTML REPORT
255
+ # ====================================================
256
+
257
+ if html:
258
+
259
+ output_path = render_html_report(
260
+ results,
261
+ output_path=output,
262
+ )
263
+
264
+ typer.echo(
265
+ f"\nHTML report generated: {output_path}"
266
+ )
267
+
268
+ except FileNotFoundError as error:
269
+
270
+ typer.echo(
271
+ f"Error: {error}",
272
+ err=True,
273
+ )
274
+
275
+ raise typer.Exit(
276
+ code=1
277
+ )
278
+
279
+ except ValueError as error:
280
+
281
+ typer.echo(
282
+ f"Error: {error}",
283
+ err=True,
284
+ )
285
+
286
+ raise typer.Exit(
287
+ code=1
288
+ )
289
+
290
+ except Exception as error:
291
+
292
+ typer.echo(
293
+ f"Unexpected error: {error}",
294
+ err=True,
295
+ )
296
+
297
+ raise typer.Exit(
298
+ code=1
299
+ )
300
+
301
+
302
+ # =============================================================
303
+ # ENTRY POINT
304
+ # =============================================================
305
+
306
+ if __name__ == "__main__":
307
+ app()
File without changes
@@ -0,0 +1,63 @@
1
+ from __future__ import annotations
2
+
3
+ import pandas as pd
4
+
5
+ from datasetdna.profiler.types import check_types
6
+
7
+
8
+ def check_cardinality(
9
+ df: pd.DataFrame,
10
+ ) -> dict:
11
+ """
12
+ Analyze column cardinality.
13
+
14
+ Cardinality is calculated for every column, but identifier
15
+ columns are marked as ID-like so downstream scoring can
16
+ distinguish expected high cardinality from suspicious
17
+ high cardinality.
18
+
19
+ Identifier columns are NOT removed from the raw cardinality
20
+ report because their uniqueness is still useful information.
21
+ """
22
+
23
+ result = {}
24
+
25
+ types = check_types(df)
26
+
27
+ total_rows = len(df)
28
+
29
+ for column in df.columns:
30
+
31
+ unique_count = int(
32
+ df[column].nunique(
33
+ dropna=True
34
+ )
35
+ )
36
+
37
+ unique_percentage = (
38
+ (
39
+ unique_count
40
+ / total_rows
41
+ )
42
+ * 100
43
+ if total_rows > 0
44
+ else 0.0
45
+ )
46
+
47
+ column_type = types[column]
48
+
49
+ result[column] = {
50
+ "unique_count": unique_count,
51
+ "unique_percentage": round(
52
+ float(unique_percentage),
53
+ 2,
54
+ ),
55
+ "detected_type": column_type[
56
+ "detected_type"
57
+ ],
58
+ "is_id_like": column_type[
59
+ "is_id_like"
60
+ ],
61
+ }
62
+
63
+ return result
@@ -0,0 +1,117 @@
1
+ from __future__ import annotations
2
+
3
+ import pandas as pd
4
+
5
+ from datasetdna.profiler.types import check_types
6
+
7
+
8
+ def check_categorical(
9
+ df: pd.DataFrame,
10
+ ) -> dict:
11
+ """
12
+ Analyze categorical columns and binary numeric columns.
13
+
14
+ Columns detected as:
15
+ - categorical
16
+ - binary numeric (0/1)
17
+
18
+ are analyzed.
19
+
20
+ Columns detected as:
21
+ - id
22
+ - numeric (non-binary)
23
+ - date
24
+ - boolean
25
+ - empty
26
+
27
+ are excluded automatically.
28
+ """
29
+
30
+ result = {}
31
+
32
+ types = check_types(df)
33
+
34
+ for column in df.columns:
35
+
36
+ detected_type = types[column]["detected_type"]
37
+
38
+ # -----------------------------------------------------
39
+ # Determine whether column should be analyzed
40
+ # -----------------------------------------------------
41
+
42
+ if detected_type == "categorical":
43
+ should_analyze = True
44
+
45
+ elif detected_type == "numeric":
46
+ # Binary numeric columns such as 0/1 behave like
47
+ # categorical flags and should have frequency analysis.
48
+ unique_values = df[column].dropna().unique()
49
+
50
+ should_analyze = (
51
+ len(unique_values) <= 2
52
+ and set(unique_values).issubset({0, 1})
53
+ )
54
+
55
+ else:
56
+ should_analyze = False
57
+
58
+ if not should_analyze:
59
+ continue
60
+
61
+ series = df[column].dropna()
62
+
63
+ # -----------------------------------------------------
64
+ # Empty column
65
+ # -----------------------------------------------------
66
+
67
+ if series.empty:
68
+
69
+ result[column] = {
70
+ "unique_count": 0,
71
+ "total_values": 0,
72
+ "categories": {},
73
+ }
74
+
75
+ continue
76
+
77
+ # -----------------------------------------------------
78
+ # Category frequencies
79
+ # -----------------------------------------------------
80
+
81
+ value_counts = series.value_counts()
82
+
83
+ total_values = len(series)
84
+
85
+ categories = {}
86
+
87
+ for category, count in value_counts.items():
88
+
89
+ percentage = (
90
+ count
91
+ / total_values
92
+ * 100
93
+ )
94
+
95
+ categories[str(category)] = {
96
+ "count": int(count),
97
+ "percentage": round(
98
+ float(percentage),
99
+ 2,
100
+ ),
101
+ }
102
+
103
+ # -----------------------------------------------------
104
+ # Result
105
+ # -----------------------------------------------------
106
+
107
+ result[column] = {
108
+ "unique_count": int(
109
+ series.nunique()
110
+ ),
111
+ "total_values": int(
112
+ total_values
113
+ ),
114
+ "categories": categories,
115
+ }
116
+
117
+ return result