autoforge-engine 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,404 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Any
4
+
5
+ import numpy as np
6
+ import pandas as pd
7
+
8
+
9
+ class ColumnIntelligence:
10
+ """
11
+ Analyze dataframe columns and infer useful semantic properties.
12
+
13
+ The output preserves the public fields expected by ModelForge while
14
+ exposing additional information for downstream AutoML components.
15
+ """
16
+
17
+ def __init__(
18
+ self,
19
+ high_cardinality_threshold: float = 0.90,
20
+ id_like_threshold: float = 0.95,
21
+ text_unique_threshold: float = 0.50,
22
+ ) -> None:
23
+ if not 0 < high_cardinality_threshold <= 1:
24
+ raise ValueError(
25
+ "high_cardinality_threshold must be between 0 and 1."
26
+ )
27
+
28
+ if not 0 < id_like_threshold <= 1:
29
+ raise ValueError(
30
+ "id_like_threshold must be between 0 and 1."
31
+ )
32
+
33
+ if not 0 < text_unique_threshold <= 1:
34
+ raise ValueError(
35
+ "text_unique_threshold must be between 0 and 1."
36
+ )
37
+
38
+ self.high_cardinality_threshold = high_cardinality_threshold
39
+ self.id_like_threshold = id_like_threshold
40
+ self.text_unique_threshold = text_unique_threshold
41
+
42
+ def analyze(self, data: pd.DataFrame) -> dict[str, Any]:
43
+ """Analyze all columns in a dataframe."""
44
+ if not isinstance(data, pd.DataFrame):
45
+ raise TypeError("data must be a pandas DataFrame.")
46
+
47
+ if data.empty:
48
+ raise ValueError("data must not be empty.")
49
+
50
+ columns: dict[str, dict[str, Any]] = {}
51
+
52
+ for column in data.columns:
53
+ columns[str(column)] = self._analyze_column(
54
+ data[column],
55
+ str(column),
56
+ )
57
+
58
+ return {
59
+ "n_rows": int(len(data)),
60
+ "n_columns": int(len(data.columns)),
61
+ "columns": columns,
62
+ "numeric_columns": [
63
+ column
64
+ for column, info in columns.items()
65
+ if info["inferred_type"] == "numerical"
66
+ ],
67
+ "categorical_columns": [
68
+ column
69
+ for column, info in columns.items()
70
+ if info["inferred_type"] == "categorical"
71
+ ],
72
+ "text_columns": [
73
+ column
74
+ for column, info in columns.items()
75
+ if info["is_text_like"]
76
+ ],
77
+ "datetime_columns": [
78
+ column
79
+ for column, info in columns.items()
80
+ if info["inferred_type"] == "datetime"
81
+ ],
82
+ "boolean_columns": [
83
+ column
84
+ for column, info in columns.items()
85
+ if info["inferred_type"] == "boolean"
86
+ ],
87
+ "id_like_columns": [
88
+ column
89
+ for column, info in columns.items()
90
+ if info["is_id_like"]
91
+ ],
92
+ "high_cardinality_columns": [
93
+ column
94
+ for column, info in columns.items()
95
+ if info["is_high_cardinality"]
96
+ ],
97
+ "constant_columns": [
98
+ column
99
+ for column, info in columns.items()
100
+ if info["inferred_type"] == "constant"
101
+ ],
102
+ "missing_columns": [
103
+ column
104
+ for column, info in columns.items()
105
+ if info["missing_count"] > 0
106
+ ],
107
+ }
108
+
109
+ def _analyze_column(
110
+ self,
111
+ series: pd.Series,
112
+ column_name: str,
113
+ ) -> dict[str, Any]:
114
+ """Analyze one dataframe column."""
115
+
116
+ total_count = int(len(series))
117
+ missing_count = int(series.isna().sum())
118
+ non_missing_count = total_count - missing_count
119
+
120
+ unique_count = int(series.nunique(dropna=True))
121
+
122
+ unique_ratio = (
123
+ unique_count / non_missing_count
124
+ if non_missing_count > 0
125
+ else 0.0
126
+ )
127
+
128
+ dtype_name = str(series.dtype)
129
+
130
+ is_numeric = pd.api.types.is_numeric_dtype(series)
131
+ is_boolean = pd.api.types.is_bool_dtype(series)
132
+ is_datetime = pd.api.types.is_datetime64_any_dtype(series)
133
+
134
+ datetime_like = False
135
+
136
+ if not is_datetime and (
137
+ pd.api.types.is_object_dtype(series)
138
+ or pd.api.types.is_string_dtype(series)
139
+ ):
140
+ datetime_like = self._looks_like_datetime(series)
141
+
142
+ is_text_like = self._looks_like_text(
143
+ series,
144
+ unique_ratio,
145
+ )
146
+
147
+ is_constant = unique_count <= 1
148
+
149
+ if is_constant:
150
+ inferred_type = "constant"
151
+
152
+ elif self._is_id_name(column_name):
153
+ inferred_type = "id"
154
+
155
+ elif is_boolean:
156
+ inferred_type = "boolean"
157
+
158
+ elif is_datetime or datetime_like:
159
+ inferred_type = "datetime"
160
+
161
+ elif is_numeric:
162
+ inferred_type = "numerical"
163
+
164
+ elif is_text_like:
165
+ inferred_type = "text"
166
+
167
+ else:
168
+ inferred_type = "categorical"
169
+
170
+ is_high_cardinality = (
171
+ unique_ratio >= self.high_cardinality_threshold
172
+ and unique_count >= 5
173
+ and inferred_type in {"categorical", "text"}
174
+ )
175
+
176
+ is_id_like = self._is_id_like(
177
+ series=series,
178
+ column_name=column_name,
179
+ unique_ratio=unique_ratio,
180
+ inferred_type=inferred_type,
181
+ )
182
+
183
+ statistics = (
184
+ self._numeric_statistics(series)
185
+ if is_numeric
186
+ else {}
187
+ )
188
+
189
+ return {
190
+ # Existing/public contract.
191
+ "name": column_name,
192
+ "dtype": dtype_name,
193
+ "inferred_type": inferred_type,
194
+ "is_text_like": bool(is_text_like),
195
+ "is_id_like": bool(is_id_like),
196
+ "is_constant": bool(is_constant),
197
+
198
+ # Additional intelligence.
199
+ "role": self._role_from_type(inferred_type),
200
+ "total_count": total_count,
201
+ "non_missing_count": non_missing_count,
202
+ "missing_count": missing_count,
203
+ "missing_ratio": (
204
+ missing_count / total_count
205
+ if total_count > 0
206
+ else 0.0
207
+ ),
208
+ "unique_count": unique_count,
209
+ "unique_ratio": float(unique_ratio),
210
+ "is_numeric": bool(is_numeric),
211
+ "is_boolean": bool(is_boolean),
212
+ "is_datetime": bool(is_datetime or datetime_like),
213
+ "is_high_cardinality": bool(is_high_cardinality),
214
+ "statistics": statistics,
215
+ }
216
+
217
+ def _role_from_type(self, inferred_type: str) -> str:
218
+ """Map the public inferred type to a semantic role."""
219
+ mapping = {
220
+ "numerical": "numeric",
221
+ "categorical": "categorical",
222
+ "text": "text",
223
+ "datetime": "datetime",
224
+ "boolean": "boolean",
225
+ "id": "id",
226
+ "constant": "constant",
227
+ }
228
+
229
+ return mapping.get(inferred_type, inferred_type)
230
+
231
+ def _is_id_name(self, column_name: str) -> bool:
232
+ """Check whether a column name strongly indicates an identifier."""
233
+ normalized_name = (
234
+ column_name.strip()
235
+ .lower()
236
+ .replace("-", "_")
237
+ .replace(" ", "_")
238
+ )
239
+
240
+ explicit_id_names = {
241
+ "id",
242
+ "uuid",
243
+ "guid",
244
+ "identifier",
245
+ "record_id",
246
+ "row_id",
247
+ "customer_id",
248
+ "user_id",
249
+ "employee_id",
250
+ "transaction_id",
251
+ "account_id",
252
+ "order_id",
253
+ "product_id",
254
+ "patient_id",
255
+ "student_id",
256
+ }
257
+
258
+ return (
259
+ normalized_name in explicit_id_names
260
+ or normalized_name.endswith("_id")
261
+ )
262
+
263
+ def _looks_like_text(
264
+ self,
265
+ series: pd.Series,
266
+ unique_ratio: float,
267
+ ) -> bool:
268
+ """Detect likely free-form text columns."""
269
+ if not (
270
+ pd.api.types.is_object_dtype(series)
271
+ or pd.api.types.is_string_dtype(series)
272
+ ):
273
+ return False
274
+
275
+ non_null = series.dropna()
276
+
277
+ if non_null.empty:
278
+ return False
279
+
280
+ values = non_null.astype(str)
281
+
282
+ average_length = float(values.str.len().mean())
283
+
284
+ return bool(
285
+ average_length >= 30
286
+ and unique_ratio >= self.text_unique_threshold
287
+ )
288
+
289
+ def _looks_like_datetime(self, series: pd.Series) -> bool:
290
+ """Detect object/string columns that appear to contain dates."""
291
+ non_null = series.dropna()
292
+
293
+ if non_null.empty:
294
+ return False
295
+
296
+ sample = non_null.astype(str).head(100)
297
+
298
+ if sample.empty:
299
+ return False
300
+
301
+ parsed = pd.to_datetime(
302
+ sample,
303
+ errors="coerce",
304
+ format="mixed",
305
+ )
306
+
307
+ parse_ratio = float(parsed.notna().mean())
308
+
309
+ return parse_ratio >= 0.90
310
+
311
+ def _is_id_like(
312
+ self,
313
+ series: pd.Series,
314
+ column_name: str,
315
+ unique_ratio: float,
316
+ inferred_type: str,
317
+ ) -> bool:
318
+ """Detect columns that are likely identifiers."""
319
+
320
+ if self._is_id_name(column_name):
321
+ return True
322
+
323
+ if inferred_type in {"categorical", "text"}:
324
+ return bool(unique_ratio >= self.id_like_threshold)
325
+
326
+ if inferred_type == "numerical":
327
+ if unique_ratio >= self.id_like_threshold:
328
+ values = series.dropna()
329
+
330
+ if not values.empty:
331
+ try:
332
+ numeric_values = values.astype(float)
333
+
334
+ integer_like = np.all(
335
+ np.isclose(
336
+ numeric_values,
337
+ np.round(numeric_values),
338
+ )
339
+ )
340
+
341
+ if integer_like and self._looks_sequential(
342
+ values
343
+ ):
344
+ return True
345
+ except (TypeError, ValueError):
346
+ return False
347
+
348
+ return False
349
+
350
+ def _looks_sequential(self, series: pd.Series) -> bool:
351
+ """Check whether numeric values resemble a sequential identifier."""
352
+ if series.empty:
353
+ return False
354
+
355
+ values = np.sort(series.unique())
356
+
357
+ if len(values) < 5:
358
+ return False
359
+
360
+ differences = np.diff(values)
361
+
362
+ if len(differences) == 0:
363
+ return False
364
+
365
+ positive_differences = differences[differences > 0]
366
+
367
+ if len(positive_differences) == 0:
368
+ return False
369
+
370
+ return bool(
371
+ np.all(
372
+ np.isclose(
373
+ positive_differences,
374
+ positive_differences[0],
375
+ )
376
+ )
377
+ )
378
+
379
+ def _numeric_statistics(
380
+ self,
381
+ series: pd.Series,
382
+ ) -> dict[str, Any]:
383
+ """Return safe summary statistics for numeric columns."""
384
+ numeric = pd.to_numeric(
385
+ series,
386
+ errors="coerce",
387
+ ).dropna()
388
+
389
+ if numeric.empty:
390
+ return {}
391
+
392
+ return {
393
+ "min": float(numeric.min()),
394
+ "max": float(numeric.max()),
395
+ "mean": float(numeric.mean()),
396
+ "median": float(numeric.median()),
397
+ "std": (
398
+ float(numeric.std())
399
+ if len(numeric) > 1
400
+ else 0.0
401
+ ),
402
+ "zero_count": int((numeric == 0).sum()),
403
+ "negative_count": int((numeric < 0).sum()),
404
+ }