autoforge-engine 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autoforge_engine-0.1.0.dist-info/METADATA +105 -0
- autoforge_engine-0.1.0.dist-info/RECORD +32 -0
- autoforge_engine-0.1.0.dist-info/WHEEL +5 -0
- autoforge_engine-0.1.0.dist-info/entry_points.txt +2 -0
- autoforge_engine-0.1.0.dist-info/licenses/LICENSE +0 -0
- autoforge_engine-0.1.0.dist-info/top_level.txt +1 -0
- modelforge/artifact_manager.py +485 -0
- modelforge/automl.py +1472 -0
- modelforge/cli.py +1258 -0
- modelforge/column_intelligence.py +404 -0
- modelforge/config.py +580 -0
- modelforge/cross_validation.py +749 -0
- modelforge/data_audit.py +392 -0
- modelforge/data_loader.py +76 -0
- modelforge/evaluation.py +397 -0
- modelforge/experiment_tracker.py +490 -0
- modelforge/explainability.py +346 -0
- modelforge/feature_engineering.py +393 -0
- modelforge/feature_selection.py +528 -0
- modelforge/hyperparameter_optimization.py +593 -0
- modelforge/model_registry.py +684 -0
- modelforge/model_screening.py +531 -0
- modelforge/persistence.py +456 -0
- modelforge/pipeline_generator.py +278 -0
- modelforge/prediction_validator.py +316 -0
- modelforge/preprocessing.py +179 -0
- modelforge/profiler.py +85 -0
- modelforge/ranking.py +351 -0
- modelforge/reproducibility.py +295 -0
- modelforge/reproducibility_integration.py +192 -0
- modelforge/run_manager.py +200 -0
- modelforge/target_selector.py +108 -0
|
@@ -0,0 +1,316 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class PredictionSchemaError(ValueError):
|
|
9
|
+
"""Raised when prediction data does not match the expected schema."""
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class PredictionSchemaValidator:
|
|
13
|
+
"""
|
|
14
|
+
Validate prediction data against a training feature schema.
|
|
15
|
+
|
|
16
|
+
The validator checks:
|
|
17
|
+
- required columns
|
|
18
|
+
- unexpected columns
|
|
19
|
+
- column order
|
|
20
|
+
- empty datasets
|
|
21
|
+
- basic data types
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
def __init__(
|
|
25
|
+
self,
|
|
26
|
+
allow_extra_columns: bool = True,
|
|
27
|
+
enforce_column_order: bool = False,
|
|
28
|
+
enforce_dtypes: bool = False,
|
|
29
|
+
):
|
|
30
|
+
if not isinstance(
|
|
31
|
+
allow_extra_columns,
|
|
32
|
+
bool,
|
|
33
|
+
):
|
|
34
|
+
raise TypeError(
|
|
35
|
+
"allow_extra_columns must be a boolean."
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
if not isinstance(
|
|
39
|
+
enforce_column_order,
|
|
40
|
+
bool,
|
|
41
|
+
):
|
|
42
|
+
raise TypeError(
|
|
43
|
+
"enforce_column_order must be a boolean."
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
if not isinstance(
|
|
47
|
+
enforce_dtypes,
|
|
48
|
+
bool,
|
|
49
|
+
):
|
|
50
|
+
raise TypeError(
|
|
51
|
+
"enforce_dtypes must be a boolean."
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
self.allow_extra_columns = (
|
|
55
|
+
allow_extra_columns
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
self.enforce_column_order = (
|
|
59
|
+
enforce_column_order
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
self.enforce_dtypes = (
|
|
63
|
+
enforce_dtypes
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
def validate(
|
|
67
|
+
self,
|
|
68
|
+
data: pd.DataFrame,
|
|
69
|
+
expected_columns: list[str],
|
|
70
|
+
expected_dtypes: dict[str, str] | None = None,
|
|
71
|
+
) -> dict[str, Any]:
|
|
72
|
+
"""
|
|
73
|
+
Validate prediction data.
|
|
74
|
+
|
|
75
|
+
Returns a validation report.
|
|
76
|
+
|
|
77
|
+
Raises:
|
|
78
|
+
TypeError:
|
|
79
|
+
If inputs have invalid types.
|
|
80
|
+
|
|
81
|
+
PredictionSchemaError:
|
|
82
|
+
If the prediction schema is invalid.
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
if not isinstance(
|
|
86
|
+
data,
|
|
87
|
+
pd.DataFrame,
|
|
88
|
+
):
|
|
89
|
+
raise TypeError(
|
|
90
|
+
"data must be a pandas DataFrame."
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
if not isinstance(
|
|
94
|
+
expected_columns,
|
|
95
|
+
list,
|
|
96
|
+
):
|
|
97
|
+
raise TypeError(
|
|
98
|
+
"expected_columns must be a list."
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
if not all(
|
|
102
|
+
isinstance(column, str)
|
|
103
|
+
for column in expected_columns
|
|
104
|
+
):
|
|
105
|
+
raise TypeError(
|
|
106
|
+
"expected_columns must contain strings."
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
if expected_dtypes is not None:
|
|
110
|
+
if not isinstance(
|
|
111
|
+
expected_dtypes,
|
|
112
|
+
dict,
|
|
113
|
+
):
|
|
114
|
+
raise TypeError(
|
|
115
|
+
"expected_dtypes must be a dictionary."
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
if data.empty:
|
|
119
|
+
raise PredictionSchemaError(
|
|
120
|
+
"Prediction data is empty."
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
actual_columns = list(
|
|
124
|
+
data.columns
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
missing_columns = [
|
|
128
|
+
column
|
|
129
|
+
for column in expected_columns
|
|
130
|
+
if column not in actual_columns
|
|
131
|
+
]
|
|
132
|
+
|
|
133
|
+
unexpected_columns = [
|
|
134
|
+
column
|
|
135
|
+
for column in actual_columns
|
|
136
|
+
if column not in expected_columns
|
|
137
|
+
]
|
|
138
|
+
|
|
139
|
+
order_matches = (
|
|
140
|
+
actual_columns
|
|
141
|
+
== expected_columns
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
dtype_mismatches = {}
|
|
145
|
+
|
|
146
|
+
if (
|
|
147
|
+
self.enforce_dtypes
|
|
148
|
+
and expected_dtypes is not None
|
|
149
|
+
):
|
|
150
|
+
for column in expected_columns:
|
|
151
|
+
if column not in data.columns:
|
|
152
|
+
continue
|
|
153
|
+
|
|
154
|
+
expected_dtype = str(
|
|
155
|
+
expected_dtypes.get(
|
|
156
|
+
column,
|
|
157
|
+
"",
|
|
158
|
+
)
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
actual_dtype = str(
|
|
162
|
+
data[column].dtype
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
if (
|
|
166
|
+
expected_dtype
|
|
167
|
+
and actual_dtype
|
|
168
|
+
!= expected_dtype
|
|
169
|
+
):
|
|
170
|
+
dtype_mismatches[column] = {
|
|
171
|
+
"expected": expected_dtype,
|
|
172
|
+
"actual": actual_dtype,
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
errors = []
|
|
176
|
+
|
|
177
|
+
if missing_columns:
|
|
178
|
+
errors.append(
|
|
179
|
+
"Missing required columns: "
|
|
180
|
+
+ ", ".join(
|
|
181
|
+
missing_columns
|
|
182
|
+
)
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
if (
|
|
186
|
+
unexpected_columns
|
|
187
|
+
and not self.allow_extra_columns
|
|
188
|
+
):
|
|
189
|
+
errors.append(
|
|
190
|
+
"Unexpected columns: "
|
|
191
|
+
+ ", ".join(
|
|
192
|
+
unexpected_columns
|
|
193
|
+
)
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
if (
|
|
197
|
+
self.enforce_column_order
|
|
198
|
+
and not order_matches
|
|
199
|
+
):
|
|
200
|
+
errors.append(
|
|
201
|
+
"Column order does not match "
|
|
202
|
+
"the expected feature order."
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
if dtype_mismatches:
|
|
206
|
+
errors.append(
|
|
207
|
+
"Data type mismatch for columns: "
|
|
208
|
+
+ ", ".join(
|
|
209
|
+
dtype_mismatches.keys()
|
|
210
|
+
)
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
valid = not errors
|
|
214
|
+
|
|
215
|
+
report = {
|
|
216
|
+
"valid": valid,
|
|
217
|
+
"expected_columns": expected_columns,
|
|
218
|
+
"actual_columns": actual_columns,
|
|
219
|
+
"missing_columns": missing_columns,
|
|
220
|
+
"unexpected_columns": unexpected_columns,
|
|
221
|
+
"order_matches": order_matches,
|
|
222
|
+
"dtype_mismatches": dtype_mismatches,
|
|
223
|
+
"row_count": len(data),
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
if not valid:
|
|
227
|
+
raise PredictionSchemaError(
|
|
228
|
+
"Prediction data failed schema "
|
|
229
|
+
"validation. "
|
|
230
|
+
+ " ".join(errors)
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
return report
|
|
234
|
+
|
|
235
|
+
def validate_and_align(
|
|
236
|
+
self,
|
|
237
|
+
data: pd.DataFrame,
|
|
238
|
+
expected_columns: list[str],
|
|
239
|
+
expected_dtypes: dict[str, str] | None = None,
|
|
240
|
+
) -> pd.DataFrame:
|
|
241
|
+
"""
|
|
242
|
+
Validate prediction data and return it aligned
|
|
243
|
+
to the expected feature order.
|
|
244
|
+
|
|
245
|
+
Extra columns are removed when they are allowed.
|
|
246
|
+
"""
|
|
247
|
+
|
|
248
|
+
self.validate(
|
|
249
|
+
data=data,
|
|
250
|
+
expected_columns=expected_columns,
|
|
251
|
+
expected_dtypes=expected_dtypes,
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
aligned = data.copy()
|
|
255
|
+
|
|
256
|
+
aligned = aligned[
|
|
257
|
+
expected_columns
|
|
258
|
+
]
|
|
259
|
+
|
|
260
|
+
return aligned
|
|
261
|
+
|
|
262
|
+
@staticmethod
|
|
263
|
+
def infer_schema(
|
|
264
|
+
data: pd.DataFrame,
|
|
265
|
+
) -> dict[str, Any]:
|
|
266
|
+
"""
|
|
267
|
+
Infer a prediction schema from a DataFrame.
|
|
268
|
+
"""
|
|
269
|
+
|
|
270
|
+
if not isinstance(
|
|
271
|
+
data,
|
|
272
|
+
pd.DataFrame,
|
|
273
|
+
):
|
|
274
|
+
raise TypeError(
|
|
275
|
+
"data must be a pandas DataFrame."
|
|
276
|
+
)
|
|
277
|
+
|
|
278
|
+
if data.empty:
|
|
279
|
+
raise PredictionSchemaError(
|
|
280
|
+
"Cannot infer schema from "
|
|
281
|
+
"an empty DataFrame."
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
return {
|
|
285
|
+
"columns": list(
|
|
286
|
+
data.columns
|
|
287
|
+
),
|
|
288
|
+
"dtypes": {
|
|
289
|
+
column: str(
|
|
290
|
+
data[column].dtype
|
|
291
|
+
)
|
|
292
|
+
for column in data.columns
|
|
293
|
+
},
|
|
294
|
+
"n_features": len(
|
|
295
|
+
data.columns
|
|
296
|
+
),
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
@staticmethod
|
|
300
|
+
def format_error(
|
|
301
|
+
error: Exception,
|
|
302
|
+
) -> str:
|
|
303
|
+
"""
|
|
304
|
+
Return a clean user-facing error message.
|
|
305
|
+
"""
|
|
306
|
+
|
|
307
|
+
if isinstance(
|
|
308
|
+
error,
|
|
309
|
+
PredictionSchemaError,
|
|
310
|
+
):
|
|
311
|
+
return str(error)
|
|
312
|
+
|
|
313
|
+
return (
|
|
314
|
+
"Prediction validation failed: "
|
|
315
|
+
f"{error}"
|
|
316
|
+
)
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
from typing import Iterable
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.compose import ColumnTransformer
|
|
5
|
+
from sklearn.impute import SimpleImputer
|
|
6
|
+
from sklearn.pipeline import Pipeline
|
|
7
|
+
from sklearn.preprocessing import (
|
|
8
|
+
OneHotEncoder,
|
|
9
|
+
StandardScaler,
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class PreprocessingEngine:
|
|
14
|
+
"""
|
|
15
|
+
Build leakage-safe preprocessing pipelines for ModelForge.
|
|
16
|
+
|
|
17
|
+
All preprocessing operations are placed inside a scikit-learn
|
|
18
|
+
Pipeline / ColumnTransformer so that transformations are fitted
|
|
19
|
+
only on training data during cross-validation.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
SCALE_MODELS = {
|
|
23
|
+
"knn",
|
|
24
|
+
"svm",
|
|
25
|
+
"svr",
|
|
26
|
+
"linear_regression",
|
|
27
|
+
"ridge",
|
|
28
|
+
"lasso",
|
|
29
|
+
"elasticnet",
|
|
30
|
+
"logistic_regression",
|
|
31
|
+
"mlp",
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
def build(
|
|
35
|
+
self,
|
|
36
|
+
data: pd.DataFrame,
|
|
37
|
+
target: str,
|
|
38
|
+
model_type: str = "tree",
|
|
39
|
+
excluded_columns: Iterable[str] | None = None,
|
|
40
|
+
) -> ColumnTransformer:
|
|
41
|
+
"""
|
|
42
|
+
Build a preprocessing transformer.
|
|
43
|
+
|
|
44
|
+
Parameters
|
|
45
|
+
----------
|
|
46
|
+
data : pd.DataFrame
|
|
47
|
+
Input dataset.
|
|
48
|
+
|
|
49
|
+
target : str
|
|
50
|
+
Target column.
|
|
51
|
+
|
|
52
|
+
model_type : str
|
|
53
|
+
Model family. Determines whether numerical features
|
|
54
|
+
should be scaled.
|
|
55
|
+
|
|
56
|
+
excluded_columns : Iterable[str] | None
|
|
57
|
+
Columns to exclude from preprocessing.
|
|
58
|
+
|
|
59
|
+
Returns
|
|
60
|
+
-------
|
|
61
|
+
ColumnTransformer
|
|
62
|
+
Leakage-safe preprocessing transformer.
|
|
63
|
+
"""
|
|
64
|
+
|
|
65
|
+
if not isinstance(data, pd.DataFrame):
|
|
66
|
+
raise TypeError("data must be a pandas DataFrame.")
|
|
67
|
+
|
|
68
|
+
if data.empty:
|
|
69
|
+
raise ValueError("Cannot preprocess an empty dataset.")
|
|
70
|
+
|
|
71
|
+
if target not in data.columns:
|
|
72
|
+
raise ValueError(
|
|
73
|
+
f"Target column '{target}' does not exist."
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
excluded = set(excluded_columns or [])
|
|
77
|
+
|
|
78
|
+
feature_data = data.drop(
|
|
79
|
+
columns=[target],
|
|
80
|
+
errors="ignore",
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
feature_data = feature_data.drop(
|
|
84
|
+
columns=[
|
|
85
|
+
column
|
|
86
|
+
for column in excluded
|
|
87
|
+
if column in feature_data.columns
|
|
88
|
+
],
|
|
89
|
+
errors="ignore",
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
numerical_columns = feature_data.select_dtypes(
|
|
93
|
+
include="number"
|
|
94
|
+
).columns.tolist()
|
|
95
|
+
|
|
96
|
+
categorical_columns = feature_data.select_dtypes(
|
|
97
|
+
include=[
|
|
98
|
+
"object",
|
|
99
|
+
"category",
|
|
100
|
+
"bool",
|
|
101
|
+
"string",
|
|
102
|
+
]
|
|
103
|
+
).columns.tolist()
|
|
104
|
+
|
|
105
|
+
if not numerical_columns and not categorical_columns:
|
|
106
|
+
raise ValueError(
|
|
107
|
+
"No supported feature columns were found."
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
scale_numeric = self._should_scale(model_type)
|
|
111
|
+
|
|
112
|
+
numeric_steps = [
|
|
113
|
+
(
|
|
114
|
+
"imputer",
|
|
115
|
+
SimpleImputer(strategy="median"),
|
|
116
|
+
)
|
|
117
|
+
]
|
|
118
|
+
|
|
119
|
+
if scale_numeric:
|
|
120
|
+
numeric_steps.append(
|
|
121
|
+
(
|
|
122
|
+
"scaler",
|
|
123
|
+
StandardScaler(),
|
|
124
|
+
)
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
numeric_pipeline = Pipeline(
|
|
128
|
+
steps=numeric_steps
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
categorical_pipeline = Pipeline(
|
|
132
|
+
steps=[
|
|
133
|
+
(
|
|
134
|
+
"imputer",
|
|
135
|
+
SimpleImputer(
|
|
136
|
+
strategy="most_frequent"
|
|
137
|
+
),
|
|
138
|
+
),
|
|
139
|
+
(
|
|
140
|
+
"encoder",
|
|
141
|
+
OneHotEncoder(
|
|
142
|
+
handle_unknown="ignore",
|
|
143
|
+
sparse_output=True,
|
|
144
|
+
),
|
|
145
|
+
),
|
|
146
|
+
]
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
transformers = []
|
|
150
|
+
|
|
151
|
+
if numerical_columns:
|
|
152
|
+
transformers.append(
|
|
153
|
+
(
|
|
154
|
+
"numerical",
|
|
155
|
+
numeric_pipeline,
|
|
156
|
+
numerical_columns,
|
|
157
|
+
)
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
if categorical_columns:
|
|
161
|
+
transformers.append(
|
|
162
|
+
(
|
|
163
|
+
"categorical",
|
|
164
|
+
categorical_pipeline,
|
|
165
|
+
categorical_columns,
|
|
166
|
+
)
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
return ColumnTransformer(
|
|
170
|
+
transformers=transformers,
|
|
171
|
+
remainder="drop",
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
def _should_scale(self, model_type: str) -> bool:
|
|
175
|
+
"""Determine whether numerical features should be scaled."""
|
|
176
|
+
|
|
177
|
+
normalized_model = model_type.lower().strip()
|
|
178
|
+
|
|
179
|
+
return normalized_model in self.SCALE_MODELS
|
modelforge/profiler.py
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
import pandas as pd
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class DatasetProfiler:
|
|
5
|
+
"""Profile a dataset and generate structural and statistical information."""
|
|
6
|
+
|
|
7
|
+
def profile(self, data: pd.DataFrame) -> dict:
|
|
8
|
+
"""
|
|
9
|
+
Generate a complete dataset profile.
|
|
10
|
+
|
|
11
|
+
Parameters
|
|
12
|
+
----------
|
|
13
|
+
data : pd.DataFrame
|
|
14
|
+
Dataset to analyze.
|
|
15
|
+
|
|
16
|
+
Returns
|
|
17
|
+
-------
|
|
18
|
+
dict
|
|
19
|
+
Dataset profiling information.
|
|
20
|
+
|
|
21
|
+
Raises
|
|
22
|
+
------
|
|
23
|
+
TypeError
|
|
24
|
+
If data is not a pandas DataFrame.
|
|
25
|
+
|
|
26
|
+
ValueError
|
|
27
|
+
If the dataset is empty.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
if not isinstance(data, pd.DataFrame):
|
|
31
|
+
raise TypeError("data must be a pandas DataFrame.")
|
|
32
|
+
|
|
33
|
+
if data.empty:
|
|
34
|
+
raise ValueError("Cannot profile an empty dataset.")
|
|
35
|
+
|
|
36
|
+
column_info = {}
|
|
37
|
+
|
|
38
|
+
for column in data.columns:
|
|
39
|
+
series = data[column]
|
|
40
|
+
|
|
41
|
+
column_info[column] = {
|
|
42
|
+
"dtype": str(series.dtype),
|
|
43
|
+
"missing_values": int(series.isna().sum()),
|
|
44
|
+
"missing_percentage": float(
|
|
45
|
+
series.isna().mean() * 100
|
|
46
|
+
),
|
|
47
|
+
"unique_values": int(
|
|
48
|
+
series.nunique(dropna=True)
|
|
49
|
+
),
|
|
50
|
+
"is_numeric": bool(
|
|
51
|
+
pd.api.types.is_numeric_dtype(series)
|
|
52
|
+
),
|
|
53
|
+
"is_categorical": bool(
|
|
54
|
+
pd.api.types.is_object_dtype(series)
|
|
55
|
+
or pd.api.types.is_string_dtype(series)
|
|
56
|
+
or isinstance(
|
|
57
|
+
series.dtype,
|
|
58
|
+
pd.CategoricalDtype,
|
|
59
|
+
)
|
|
60
|
+
or pd.api.types.is_bool_dtype(series)
|
|
61
|
+
),
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
numeric_columns = data.select_dtypes(
|
|
65
|
+
include="number"
|
|
66
|
+
).columns.tolist()
|
|
67
|
+
|
|
68
|
+
categorical_columns = data.select_dtypes(
|
|
69
|
+
include=["object", "category", "bool", "string"]
|
|
70
|
+
).columns.tolist()
|
|
71
|
+
|
|
72
|
+
return {
|
|
73
|
+
"rows": int(data.shape[0]),
|
|
74
|
+
"columns": int(data.shape[1]),
|
|
75
|
+
"column_names": data.columns.tolist(),
|
|
76
|
+
"duplicate_rows": int(
|
|
77
|
+
data.duplicated().sum()
|
|
78
|
+
),
|
|
79
|
+
"memory_usage_bytes": int(
|
|
80
|
+
data.memory_usage(deep=True).sum()
|
|
81
|
+
),
|
|
82
|
+
"numeric_columns": numeric_columns,
|
|
83
|
+
"categorical_columns": categorical_columns,
|
|
84
|
+
"column_info": column_info,
|
|
85
|
+
}
|