autoforge-engine 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autoforge_engine-0.1.0.dist-info/METADATA +105 -0
- autoforge_engine-0.1.0.dist-info/RECORD +32 -0
- autoforge_engine-0.1.0.dist-info/WHEEL +5 -0
- autoforge_engine-0.1.0.dist-info/entry_points.txt +2 -0
- autoforge_engine-0.1.0.dist-info/licenses/LICENSE +0 -0
- autoforge_engine-0.1.0.dist-info/top_level.txt +1 -0
- modelforge/artifact_manager.py +485 -0
- modelforge/automl.py +1472 -0
- modelforge/cli.py +1258 -0
- modelforge/column_intelligence.py +404 -0
- modelforge/config.py +580 -0
- modelforge/cross_validation.py +749 -0
- modelforge/data_audit.py +392 -0
- modelforge/data_loader.py +76 -0
- modelforge/evaluation.py +397 -0
- modelforge/experiment_tracker.py +490 -0
- modelforge/explainability.py +346 -0
- modelforge/feature_engineering.py +393 -0
- modelforge/feature_selection.py +528 -0
- modelforge/hyperparameter_optimization.py +593 -0
- modelforge/model_registry.py +684 -0
- modelforge/model_screening.py +531 -0
- modelforge/persistence.py +456 -0
- modelforge/pipeline_generator.py +278 -0
- modelforge/prediction_validator.py +316 -0
- modelforge/preprocessing.py +179 -0
- modelforge/profiler.py +85 -0
- modelforge/ranking.py +351 -0
- modelforge/reproducibility.py +295 -0
- modelforge/reproducibility_integration.py +192 -0
- modelforge/run_manager.py +200 -0
- modelforge/target_selector.py +108 -0
|
@@ -0,0 +1,404 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pandas as pd
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class ColumnIntelligence:
|
|
10
|
+
"""
|
|
11
|
+
Analyze dataframe columns and infer useful semantic properties.
|
|
12
|
+
|
|
13
|
+
The output preserves the public fields expected by ModelForge while
|
|
14
|
+
exposing additional information for downstream AutoML components.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
def __init__(
|
|
18
|
+
self,
|
|
19
|
+
high_cardinality_threshold: float = 0.90,
|
|
20
|
+
id_like_threshold: float = 0.95,
|
|
21
|
+
text_unique_threshold: float = 0.50,
|
|
22
|
+
) -> None:
|
|
23
|
+
if not 0 < high_cardinality_threshold <= 1:
|
|
24
|
+
raise ValueError(
|
|
25
|
+
"high_cardinality_threshold must be between 0 and 1."
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
if not 0 < id_like_threshold <= 1:
|
|
29
|
+
raise ValueError(
|
|
30
|
+
"id_like_threshold must be between 0 and 1."
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
if not 0 < text_unique_threshold <= 1:
|
|
34
|
+
raise ValueError(
|
|
35
|
+
"text_unique_threshold must be between 0 and 1."
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
self.high_cardinality_threshold = high_cardinality_threshold
|
|
39
|
+
self.id_like_threshold = id_like_threshold
|
|
40
|
+
self.text_unique_threshold = text_unique_threshold
|
|
41
|
+
|
|
42
|
+
def analyze(self, data: pd.DataFrame) -> dict[str, Any]:
|
|
43
|
+
"""Analyze all columns in a dataframe."""
|
|
44
|
+
if not isinstance(data, pd.DataFrame):
|
|
45
|
+
raise TypeError("data must be a pandas DataFrame.")
|
|
46
|
+
|
|
47
|
+
if data.empty:
|
|
48
|
+
raise ValueError("data must not be empty.")
|
|
49
|
+
|
|
50
|
+
columns: dict[str, dict[str, Any]] = {}
|
|
51
|
+
|
|
52
|
+
for column in data.columns:
|
|
53
|
+
columns[str(column)] = self._analyze_column(
|
|
54
|
+
data[column],
|
|
55
|
+
str(column),
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
return {
|
|
59
|
+
"n_rows": int(len(data)),
|
|
60
|
+
"n_columns": int(len(data.columns)),
|
|
61
|
+
"columns": columns,
|
|
62
|
+
"numeric_columns": [
|
|
63
|
+
column
|
|
64
|
+
for column, info in columns.items()
|
|
65
|
+
if info["inferred_type"] == "numerical"
|
|
66
|
+
],
|
|
67
|
+
"categorical_columns": [
|
|
68
|
+
column
|
|
69
|
+
for column, info in columns.items()
|
|
70
|
+
if info["inferred_type"] == "categorical"
|
|
71
|
+
],
|
|
72
|
+
"text_columns": [
|
|
73
|
+
column
|
|
74
|
+
for column, info in columns.items()
|
|
75
|
+
if info["is_text_like"]
|
|
76
|
+
],
|
|
77
|
+
"datetime_columns": [
|
|
78
|
+
column
|
|
79
|
+
for column, info in columns.items()
|
|
80
|
+
if info["inferred_type"] == "datetime"
|
|
81
|
+
],
|
|
82
|
+
"boolean_columns": [
|
|
83
|
+
column
|
|
84
|
+
for column, info in columns.items()
|
|
85
|
+
if info["inferred_type"] == "boolean"
|
|
86
|
+
],
|
|
87
|
+
"id_like_columns": [
|
|
88
|
+
column
|
|
89
|
+
for column, info in columns.items()
|
|
90
|
+
if info["is_id_like"]
|
|
91
|
+
],
|
|
92
|
+
"high_cardinality_columns": [
|
|
93
|
+
column
|
|
94
|
+
for column, info in columns.items()
|
|
95
|
+
if info["is_high_cardinality"]
|
|
96
|
+
],
|
|
97
|
+
"constant_columns": [
|
|
98
|
+
column
|
|
99
|
+
for column, info in columns.items()
|
|
100
|
+
if info["inferred_type"] == "constant"
|
|
101
|
+
],
|
|
102
|
+
"missing_columns": [
|
|
103
|
+
column
|
|
104
|
+
for column, info in columns.items()
|
|
105
|
+
if info["missing_count"] > 0
|
|
106
|
+
],
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
def _analyze_column(
|
|
110
|
+
self,
|
|
111
|
+
series: pd.Series,
|
|
112
|
+
column_name: str,
|
|
113
|
+
) -> dict[str, Any]:
|
|
114
|
+
"""Analyze one dataframe column."""
|
|
115
|
+
|
|
116
|
+
total_count = int(len(series))
|
|
117
|
+
missing_count = int(series.isna().sum())
|
|
118
|
+
non_missing_count = total_count - missing_count
|
|
119
|
+
|
|
120
|
+
unique_count = int(series.nunique(dropna=True))
|
|
121
|
+
|
|
122
|
+
unique_ratio = (
|
|
123
|
+
unique_count / non_missing_count
|
|
124
|
+
if non_missing_count > 0
|
|
125
|
+
else 0.0
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
dtype_name = str(series.dtype)
|
|
129
|
+
|
|
130
|
+
is_numeric = pd.api.types.is_numeric_dtype(series)
|
|
131
|
+
is_boolean = pd.api.types.is_bool_dtype(series)
|
|
132
|
+
is_datetime = pd.api.types.is_datetime64_any_dtype(series)
|
|
133
|
+
|
|
134
|
+
datetime_like = False
|
|
135
|
+
|
|
136
|
+
if not is_datetime and (
|
|
137
|
+
pd.api.types.is_object_dtype(series)
|
|
138
|
+
or pd.api.types.is_string_dtype(series)
|
|
139
|
+
):
|
|
140
|
+
datetime_like = self._looks_like_datetime(series)
|
|
141
|
+
|
|
142
|
+
is_text_like = self._looks_like_text(
|
|
143
|
+
series,
|
|
144
|
+
unique_ratio,
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
is_constant = unique_count <= 1
|
|
148
|
+
|
|
149
|
+
if is_constant:
|
|
150
|
+
inferred_type = "constant"
|
|
151
|
+
|
|
152
|
+
elif self._is_id_name(column_name):
|
|
153
|
+
inferred_type = "id"
|
|
154
|
+
|
|
155
|
+
elif is_boolean:
|
|
156
|
+
inferred_type = "boolean"
|
|
157
|
+
|
|
158
|
+
elif is_datetime or datetime_like:
|
|
159
|
+
inferred_type = "datetime"
|
|
160
|
+
|
|
161
|
+
elif is_numeric:
|
|
162
|
+
inferred_type = "numerical"
|
|
163
|
+
|
|
164
|
+
elif is_text_like:
|
|
165
|
+
inferred_type = "text"
|
|
166
|
+
|
|
167
|
+
else:
|
|
168
|
+
inferred_type = "categorical"
|
|
169
|
+
|
|
170
|
+
is_high_cardinality = (
|
|
171
|
+
unique_ratio >= self.high_cardinality_threshold
|
|
172
|
+
and unique_count >= 5
|
|
173
|
+
and inferred_type in {"categorical", "text"}
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
is_id_like = self._is_id_like(
|
|
177
|
+
series=series,
|
|
178
|
+
column_name=column_name,
|
|
179
|
+
unique_ratio=unique_ratio,
|
|
180
|
+
inferred_type=inferred_type,
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
statistics = (
|
|
184
|
+
self._numeric_statistics(series)
|
|
185
|
+
if is_numeric
|
|
186
|
+
else {}
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
return {
|
|
190
|
+
# Existing/public contract.
|
|
191
|
+
"name": column_name,
|
|
192
|
+
"dtype": dtype_name,
|
|
193
|
+
"inferred_type": inferred_type,
|
|
194
|
+
"is_text_like": bool(is_text_like),
|
|
195
|
+
"is_id_like": bool(is_id_like),
|
|
196
|
+
"is_constant": bool(is_constant),
|
|
197
|
+
|
|
198
|
+
# Additional intelligence.
|
|
199
|
+
"role": self._role_from_type(inferred_type),
|
|
200
|
+
"total_count": total_count,
|
|
201
|
+
"non_missing_count": non_missing_count,
|
|
202
|
+
"missing_count": missing_count,
|
|
203
|
+
"missing_ratio": (
|
|
204
|
+
missing_count / total_count
|
|
205
|
+
if total_count > 0
|
|
206
|
+
else 0.0
|
|
207
|
+
),
|
|
208
|
+
"unique_count": unique_count,
|
|
209
|
+
"unique_ratio": float(unique_ratio),
|
|
210
|
+
"is_numeric": bool(is_numeric),
|
|
211
|
+
"is_boolean": bool(is_boolean),
|
|
212
|
+
"is_datetime": bool(is_datetime or datetime_like),
|
|
213
|
+
"is_high_cardinality": bool(is_high_cardinality),
|
|
214
|
+
"statistics": statistics,
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
def _role_from_type(self, inferred_type: str) -> str:
|
|
218
|
+
"""Map the public inferred type to a semantic role."""
|
|
219
|
+
mapping = {
|
|
220
|
+
"numerical": "numeric",
|
|
221
|
+
"categorical": "categorical",
|
|
222
|
+
"text": "text",
|
|
223
|
+
"datetime": "datetime",
|
|
224
|
+
"boolean": "boolean",
|
|
225
|
+
"id": "id",
|
|
226
|
+
"constant": "constant",
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
return mapping.get(inferred_type, inferred_type)
|
|
230
|
+
|
|
231
|
+
def _is_id_name(self, column_name: str) -> bool:
|
|
232
|
+
"""Check whether a column name strongly indicates an identifier."""
|
|
233
|
+
normalized_name = (
|
|
234
|
+
column_name.strip()
|
|
235
|
+
.lower()
|
|
236
|
+
.replace("-", "_")
|
|
237
|
+
.replace(" ", "_")
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
explicit_id_names = {
|
|
241
|
+
"id",
|
|
242
|
+
"uuid",
|
|
243
|
+
"guid",
|
|
244
|
+
"identifier",
|
|
245
|
+
"record_id",
|
|
246
|
+
"row_id",
|
|
247
|
+
"customer_id",
|
|
248
|
+
"user_id",
|
|
249
|
+
"employee_id",
|
|
250
|
+
"transaction_id",
|
|
251
|
+
"account_id",
|
|
252
|
+
"order_id",
|
|
253
|
+
"product_id",
|
|
254
|
+
"patient_id",
|
|
255
|
+
"student_id",
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
return (
|
|
259
|
+
normalized_name in explicit_id_names
|
|
260
|
+
or normalized_name.endswith("_id")
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
def _looks_like_text(
|
|
264
|
+
self,
|
|
265
|
+
series: pd.Series,
|
|
266
|
+
unique_ratio: float,
|
|
267
|
+
) -> bool:
|
|
268
|
+
"""Detect likely free-form text columns."""
|
|
269
|
+
if not (
|
|
270
|
+
pd.api.types.is_object_dtype(series)
|
|
271
|
+
or pd.api.types.is_string_dtype(series)
|
|
272
|
+
):
|
|
273
|
+
return False
|
|
274
|
+
|
|
275
|
+
non_null = series.dropna()
|
|
276
|
+
|
|
277
|
+
if non_null.empty:
|
|
278
|
+
return False
|
|
279
|
+
|
|
280
|
+
values = non_null.astype(str)
|
|
281
|
+
|
|
282
|
+
average_length = float(values.str.len().mean())
|
|
283
|
+
|
|
284
|
+
return bool(
|
|
285
|
+
average_length >= 30
|
|
286
|
+
and unique_ratio >= self.text_unique_threshold
|
|
287
|
+
)
|
|
288
|
+
|
|
289
|
+
def _looks_like_datetime(self, series: pd.Series) -> bool:
|
|
290
|
+
"""Detect object/string columns that appear to contain dates."""
|
|
291
|
+
non_null = series.dropna()
|
|
292
|
+
|
|
293
|
+
if non_null.empty:
|
|
294
|
+
return False
|
|
295
|
+
|
|
296
|
+
sample = non_null.astype(str).head(100)
|
|
297
|
+
|
|
298
|
+
if sample.empty:
|
|
299
|
+
return False
|
|
300
|
+
|
|
301
|
+
parsed = pd.to_datetime(
|
|
302
|
+
sample,
|
|
303
|
+
errors="coerce",
|
|
304
|
+
format="mixed",
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
parse_ratio = float(parsed.notna().mean())
|
|
308
|
+
|
|
309
|
+
return parse_ratio >= 0.90
|
|
310
|
+
|
|
311
|
+
def _is_id_like(
|
|
312
|
+
self,
|
|
313
|
+
series: pd.Series,
|
|
314
|
+
column_name: str,
|
|
315
|
+
unique_ratio: float,
|
|
316
|
+
inferred_type: str,
|
|
317
|
+
) -> bool:
|
|
318
|
+
"""Detect columns that are likely identifiers."""
|
|
319
|
+
|
|
320
|
+
if self._is_id_name(column_name):
|
|
321
|
+
return True
|
|
322
|
+
|
|
323
|
+
if inferred_type in {"categorical", "text"}:
|
|
324
|
+
return bool(unique_ratio >= self.id_like_threshold)
|
|
325
|
+
|
|
326
|
+
if inferred_type == "numerical":
|
|
327
|
+
if unique_ratio >= self.id_like_threshold:
|
|
328
|
+
values = series.dropna()
|
|
329
|
+
|
|
330
|
+
if not values.empty:
|
|
331
|
+
try:
|
|
332
|
+
numeric_values = values.astype(float)
|
|
333
|
+
|
|
334
|
+
integer_like = np.all(
|
|
335
|
+
np.isclose(
|
|
336
|
+
numeric_values,
|
|
337
|
+
np.round(numeric_values),
|
|
338
|
+
)
|
|
339
|
+
)
|
|
340
|
+
|
|
341
|
+
if integer_like and self._looks_sequential(
|
|
342
|
+
values
|
|
343
|
+
):
|
|
344
|
+
return True
|
|
345
|
+
except (TypeError, ValueError):
|
|
346
|
+
return False
|
|
347
|
+
|
|
348
|
+
return False
|
|
349
|
+
|
|
350
|
+
def _looks_sequential(self, series: pd.Series) -> bool:
|
|
351
|
+
"""Check whether numeric values resemble a sequential identifier."""
|
|
352
|
+
if series.empty:
|
|
353
|
+
return False
|
|
354
|
+
|
|
355
|
+
values = np.sort(series.unique())
|
|
356
|
+
|
|
357
|
+
if len(values) < 5:
|
|
358
|
+
return False
|
|
359
|
+
|
|
360
|
+
differences = np.diff(values)
|
|
361
|
+
|
|
362
|
+
if len(differences) == 0:
|
|
363
|
+
return False
|
|
364
|
+
|
|
365
|
+
positive_differences = differences[differences > 0]
|
|
366
|
+
|
|
367
|
+
if len(positive_differences) == 0:
|
|
368
|
+
return False
|
|
369
|
+
|
|
370
|
+
return bool(
|
|
371
|
+
np.all(
|
|
372
|
+
np.isclose(
|
|
373
|
+
positive_differences,
|
|
374
|
+
positive_differences[0],
|
|
375
|
+
)
|
|
376
|
+
)
|
|
377
|
+
)
|
|
378
|
+
|
|
379
|
+
def _numeric_statistics(
|
|
380
|
+
self,
|
|
381
|
+
series: pd.Series,
|
|
382
|
+
) -> dict[str, Any]:
|
|
383
|
+
"""Return safe summary statistics for numeric columns."""
|
|
384
|
+
numeric = pd.to_numeric(
|
|
385
|
+
series,
|
|
386
|
+
errors="coerce",
|
|
387
|
+
).dropna()
|
|
388
|
+
|
|
389
|
+
if numeric.empty:
|
|
390
|
+
return {}
|
|
391
|
+
|
|
392
|
+
return {
|
|
393
|
+
"min": float(numeric.min()),
|
|
394
|
+
"max": float(numeric.max()),
|
|
395
|
+
"mean": float(numeric.mean()),
|
|
396
|
+
"median": float(numeric.median()),
|
|
397
|
+
"std": (
|
|
398
|
+
float(numeric.std())
|
|
399
|
+
if len(numeric) > 1
|
|
400
|
+
else 0.0
|
|
401
|
+
),
|
|
402
|
+
"zero_count": int((numeric == 0).sum()),
|
|
403
|
+
"negative_count": int((numeric < 0).sum()),
|
|
404
|
+
}
|