autoforge-engine 0.1.3__tar.gz → 0.1.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/PKG-INFO +4 -1
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/README.md +3 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/autoforge_engine.egg-info/PKG-INFO +4 -1
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/automl.py +119 -42
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/cli.py +1 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/model_registry.py +1 -2
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/pyproject.toml +3 -2
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_automl.py +81 -1
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_cli.py +13 -1
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_config_automl_integration.py +3 -1
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_model_registry.py +14 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/LICENSE +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/autoforge_engine.egg-info/SOURCES.txt +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/autoforge_engine.egg-info/dependency_links.txt +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/autoforge_engine.egg-info/entry_points.txt +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/autoforge_engine.egg-info/requires.txt +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/autoforge_engine.egg-info/top_level.txt +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/__init__.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/artifact_manager.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/column_intelligence.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/config.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/cross_validation.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/data_audit.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/data_loader.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/evaluation.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/experiment_tracker.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/explainability.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/feature_engineering.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/feature_selection.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/hyperparameter_optimization.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/model_screening.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/persistence.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/pipeline_generator.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/prediction_validator.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/preprocessing.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/profiler.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/ranking.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/reproducibility.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/reproducibility_integration.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/run_manager.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/modelforge/target_selector.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/setup.cfg +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_artifact_manager.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_automl_reproducibility.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_automl_robustness.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_cli_workflow.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_column_intelligence.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_config.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_cross_validation.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_data_audit.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_data_loader.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_end_to_end.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_evaluation.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_experiment_artifacts.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_experiment_tracker.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_experiment_tracker_reproducibility.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_explainability.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_failure_isolation.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_feature_engineering.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_feature_selection.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_full_system.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_hyperparameter_optimization.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_model_screening.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_persistence.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_pipeline_generator.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_prediction_robustness.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_prediction_validator.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_preprocessing.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_profiler.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_ranking.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_reproducibility.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_reproducibility_integration.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_robustness.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_run_manager.py +0 -0
- {autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_target_selector.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: autoforge-engine
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.4
|
|
4
4
|
Summary: A transparent, local-first AutoML framework for automated ML pipeline discovery.
|
|
5
5
|
Author: Aditya Kumar Singh
|
|
6
6
|
Author-email: Aditya Kumar Singh <adityasingh45245@gmail.com>
|
|
@@ -248,6 +248,9 @@ Pass options such as `task_type="classification"`, `cv=5`, or
|
|
|
248
248
|
|
|
249
249
|
Pass a pandas DataFrame to `AutoML.fit`, name the target column, then save the
|
|
250
250
|
fitted pipeline. Replace the example path with the path to your own dataset.
|
|
251
|
+
`fit` prints a report with dataset details, all evaluated model metrics, the
|
|
252
|
+
selected pipeline, and experiment identifiers by default. Pass
|
|
253
|
+
`print_report=False` to suppress it.
|
|
251
254
|
|
|
252
255
|
```python
|
|
253
256
|
from pathlib import Path
|
|
@@ -216,6 +216,9 @@ Pass options such as `task_type="classification"`, `cv=5`, or
|
|
|
216
216
|
|
|
217
217
|
Pass a pandas DataFrame to `AutoML.fit`, name the target column, then save the
|
|
218
218
|
fitted pipeline. Replace the example path with the path to your own dataset.
|
|
219
|
+
`fit` prints a report with dataset details, all evaluated model metrics, the
|
|
220
|
+
selected pipeline, and experiment identifiers by default. Pass
|
|
221
|
+
`print_report=False` to suppress it.
|
|
219
222
|
|
|
220
223
|
```python
|
|
221
224
|
from pathlib import Path
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: autoforge-engine
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.4
|
|
4
4
|
Summary: A transparent, local-first AutoML framework for automated ML pipeline discovery.
|
|
5
5
|
Author: Aditya Kumar Singh
|
|
6
6
|
Author-email: Aditya Kumar Singh <adityasingh45245@gmail.com>
|
|
@@ -248,6 +248,9 @@ Pass options such as `task_type="classification"`, `cv=5`, or
|
|
|
248
248
|
|
|
249
249
|
Pass a pandas DataFrame to `AutoML.fit`, name the target column, then save the
|
|
250
250
|
fitted pipeline. Replace the example path with the path to your own dataset.
|
|
251
|
+
`fit` prints a report with dataset details, all evaluated model metrics, the
|
|
252
|
+
selected pipeline, and experiment identifiers by default. Pass
|
|
253
|
+
`print_report=False` to suppress it.
|
|
251
254
|
|
|
252
255
|
```python
|
|
253
256
|
from pathlib import Path
|
|
@@ -233,6 +233,7 @@ class AutoML:
|
|
|
233
233
|
task_type: str | None = None,
|
|
234
234
|
model_names: list[str] | None = None,
|
|
235
235
|
excluded_columns: list[str] | None = None,
|
|
236
|
+
print_report: bool = True,
|
|
236
237
|
) -> dict[str, Any]:
|
|
237
238
|
"""
|
|
238
239
|
Run the complete ModelForge AutoML workflow.
|
|
@@ -241,8 +242,12 @@ class AutoML:
|
|
|
241
242
|
recorded after successful or failed execution.
|
|
242
243
|
|
|
243
244
|
Reproducibility metadata is captured for every run.
|
|
245
|
+
Set print_report=False to suppress the completion report.
|
|
244
246
|
"""
|
|
245
247
|
|
|
248
|
+
if not isinstance(print_report, bool):
|
|
249
|
+
raise TypeError("print_report must be a boolean.")
|
|
250
|
+
|
|
246
251
|
target = (
|
|
247
252
|
target
|
|
248
253
|
if target is not None
|
|
@@ -357,6 +362,9 @@ class AutoML:
|
|
|
357
362
|
|
|
358
363
|
self.result = result
|
|
359
364
|
|
|
365
|
+
if print_report:
|
|
366
|
+
_print_automl_report(self, result)
|
|
367
|
+
|
|
360
368
|
return result
|
|
361
369
|
|
|
362
370
|
except Exception as exc:
|
|
@@ -1487,14 +1495,13 @@ def automl(
|
|
|
1487
1495
|
"""Fit AutoML, print a run report, and return the fitted engine."""
|
|
1488
1496
|
|
|
1489
1497
|
engine = AutoML(**options)
|
|
1490
|
-
|
|
1498
|
+
engine.fit(
|
|
1491
1499
|
data=data,
|
|
1492
1500
|
target=target,
|
|
1493
1501
|
task_type=task_type,
|
|
1494
1502
|
model_names=model_names,
|
|
1495
1503
|
excluded_columns=excluded_columns,
|
|
1496
1504
|
)
|
|
1497
|
-
_print_automl_report(engine, result)
|
|
1498
1505
|
return engine
|
|
1499
1506
|
|
|
1500
1507
|
|
|
@@ -1508,6 +1515,7 @@ def _print_automl_report(
|
|
|
1508
1515
|
profile = result["profile"]
|
|
1509
1516
|
audit = result["audit"]
|
|
1510
1517
|
ranking = result["ranking"]
|
|
1518
|
+
target_info = result["target"]
|
|
1511
1519
|
missing_values = sum(
|
|
1512
1520
|
column.get("missing_values", 0)
|
|
1513
1521
|
for column in profile["column_info"].values()
|
|
@@ -1525,9 +1533,8 @@ def _print_automl_report(
|
|
|
1525
1533
|
dataset_table.add_column("Value", style="white")
|
|
1526
1534
|
dataset_table.add_row("Rows", f"{profile['rows']:,}")
|
|
1527
1535
|
dataset_table.add_row("Columns", f"{profile['columns']:,}")
|
|
1528
|
-
dataset_table.add_row("
|
|
1529
|
-
dataset_table.add_row("
|
|
1530
|
-
dataset_table.add_row("Task", str(result["target"]["task_type"]))
|
|
1536
|
+
dataset_table.add_row("Target", str(target_info["target"]))
|
|
1537
|
+
dataset_table.add_row("Task", str(target_info["task_type"]))
|
|
1531
1538
|
dataset_table.add_row("Missing values", f"{missing_values:,}")
|
|
1532
1539
|
dataset_table.add_row(
|
|
1533
1540
|
"Duplicate rows",
|
|
@@ -1538,43 +1545,56 @@ def _print_automl_report(
|
|
|
1538
1545
|
Panel(dataset_table, title="Dataset", border_style="cyan")
|
|
1539
1546
|
)
|
|
1540
1547
|
|
|
1541
|
-
|
|
1542
|
-
|
|
1543
|
-
|
|
1544
|
-
|
|
1545
|
-
|
|
1546
|
-
metric_label = "CV R2" if engine.task_type == "regression" else "CV F1"
|
|
1547
|
-
ranking_columns = [
|
|
1548
|
-
("rank", "#"),
|
|
1549
|
-
("model", "Model"),
|
|
1550
|
-
("overall_score", "Overall"),
|
|
1551
|
-
(metric_column, metric_label),
|
|
1552
|
-
("speed_score", "Speed"),
|
|
1553
|
-
("status", "Status"),
|
|
1548
|
+
metric_columns = [
|
|
1549
|
+
column
|
|
1550
|
+
for column in ranking.columns
|
|
1551
|
+
if column.startswith("cv_mean_")
|
|
1552
|
+
and not column.endswith("_seconds")
|
|
1554
1553
|
]
|
|
1555
1554
|
ranking_table = Table(box=None, padding=(0, 1))
|
|
1556
|
-
|
|
1557
|
-
|
|
1558
|
-
|
|
1559
|
-
|
|
1560
|
-
for _, row in ranking.
|
|
1561
|
-
|
|
1562
|
-
|
|
1563
|
-
|
|
1564
|
-
|
|
1555
|
+
ranking_table.add_column("#", justify="right", no_wrap=True)
|
|
1556
|
+
ranking_table.add_column("Model", overflow="fold")
|
|
1557
|
+
ranking_table.add_column("CV Metrics", overflow="fold")
|
|
1558
|
+
|
|
1559
|
+
for _, row in ranking.sort_values(
|
|
1560
|
+
"rank",
|
|
1561
|
+
na_position="last",
|
|
1562
|
+
).iterrows():
|
|
1563
|
+
metric_details = []
|
|
1564
|
+
for column in metric_columns:
|
|
1565
|
+
metric_label = column.removeprefix("cv_mean_")
|
|
1566
|
+
metric_label = metric_label.replace("_", " ").upper()
|
|
1567
|
+
metric_label = metric_label.replace("ROC AUC", "ROC-AUC")
|
|
1565
1568
|
value = row[column]
|
|
1566
|
-
|
|
1567
|
-
|
|
1568
|
-
|
|
1569
|
-
|
|
1570
|
-
|
|
1569
|
+
formatted_value = (
|
|
1570
|
+
"N/A" if pd.isna(value) else f"{value:.4f}"
|
|
1571
|
+
)
|
|
1572
|
+
metric_details.append(
|
|
1573
|
+
f"CV {metric_label}: {formatted_value}"
|
|
1574
|
+
)
|
|
1575
|
+
if "overall_score" in ranking.columns:
|
|
1576
|
+
metric_details.append(
|
|
1577
|
+
f"Overall: {row['overall_score']:.4f}"
|
|
1578
|
+
)
|
|
1579
|
+
if "status" in ranking.columns:
|
|
1580
|
+
metric_details.append(f"Status: {row['status']}")
|
|
1581
|
+
model_name = engine.registry.get(str(row["model"])).name
|
|
1582
|
+
rank = row.get("rank")
|
|
1583
|
+
rank_label = "N/A" if pd.isna(rank) else str(int(rank))
|
|
1584
|
+
ranking_table.add_row(
|
|
1585
|
+
rank_label,
|
|
1586
|
+
model_name,
|
|
1587
|
+
" | ".join(metric_details),
|
|
1588
|
+
)
|
|
1571
1589
|
console.print(
|
|
1572
1590
|
Panel(ranking_table, title="Model Ranking", border_style="green")
|
|
1573
1591
|
)
|
|
1574
1592
|
|
|
1575
|
-
|
|
1576
|
-
|
|
1577
|
-
|
|
1593
|
+
best_model = result.get("best_model", engine.best_model)
|
|
1594
|
+
best_matches = ranking.loc[
|
|
1595
|
+
ranking["model"].astype(str) == str(best_model)
|
|
1596
|
+
]
|
|
1597
|
+
best_row = best_matches.iloc[0] if not best_matches.empty else None
|
|
1578
1598
|
best_table = Table(
|
|
1579
1599
|
show_header=False,
|
|
1580
1600
|
box=None,
|
|
@@ -1582,13 +1602,25 @@ def _print_automl_report(
|
|
|
1582
1602
|
)
|
|
1583
1603
|
best_table.add_column("Property", style="cyan")
|
|
1584
1604
|
best_table.add_column("Value", style="white")
|
|
1585
|
-
|
|
1586
|
-
best_table.add_row(
|
|
1587
|
-
|
|
1588
|
-
|
|
1589
|
-
|
|
1590
|
-
|
|
1591
|
-
|
|
1605
|
+
best_model_name = engine.registry.get(str(best_model)).name
|
|
1606
|
+
best_table.add_row("Model", best_model_name)
|
|
1607
|
+
if best_row is not None:
|
|
1608
|
+
if "rank" in ranking.columns:
|
|
1609
|
+
best_table.add_row("Rank", str(int(best_row["rank"])))
|
|
1610
|
+
if "overall_score" in ranking.columns:
|
|
1611
|
+
best_table.add_row(
|
|
1612
|
+
"Overall score",
|
|
1613
|
+
f"{best_row['overall_score']:.4f}",
|
|
1614
|
+
)
|
|
1615
|
+
for column in metric_columns:
|
|
1616
|
+
metric_label = column.removeprefix("cv_mean_")
|
|
1617
|
+
metric_label = metric_label.replace("_", " ").upper()
|
|
1618
|
+
metric_label = metric_label.replace("ROC AUC", "ROC-AUC")
|
|
1619
|
+
value = best_row[column]
|
|
1620
|
+
best_table.add_row(
|
|
1621
|
+
f"CV {metric_label}",
|
|
1622
|
+
"N/A" if pd.isna(value) else f"{value:.4f}",
|
|
1623
|
+
)
|
|
1592
1624
|
best_table.add_row("Objective", str(engine.objective))
|
|
1593
1625
|
best_table.add_row("Cross-validation", f"{engine.cv} folds")
|
|
1594
1626
|
best_table.add_row("Holdout", f"{engine.test_size:.0%}")
|
|
@@ -1608,6 +1640,51 @@ def _print_automl_report(
|
|
|
1608
1640
|
Panel(best_table, title="Best Model & Settings", border_style="yellow")
|
|
1609
1641
|
)
|
|
1610
1642
|
|
|
1643
|
+
pipeline_table = Table(
|
|
1644
|
+
show_header=False,
|
|
1645
|
+
box=None,
|
|
1646
|
+
padding=(0, 1),
|
|
1647
|
+
)
|
|
1648
|
+
pipeline_table.add_column("Property", style="cyan")
|
|
1649
|
+
pipeline_table.add_column("Value", style="white")
|
|
1650
|
+
pipeline = result.get("best_pipeline")
|
|
1651
|
+
pipeline_steps = getattr(pipeline, "steps", [])
|
|
1652
|
+
for step_name, step in pipeline_steps:
|
|
1653
|
+
pipeline_table.add_row(
|
|
1654
|
+
str(step_name),
|
|
1655
|
+
type(step).__name__,
|
|
1656
|
+
)
|
|
1657
|
+
for transformer_name, transformer, _ in getattr(
|
|
1658
|
+
step,
|
|
1659
|
+
"transformers_",
|
|
1660
|
+
[],
|
|
1661
|
+
):
|
|
1662
|
+
if isinstance(transformer, str):
|
|
1663
|
+
description = transformer
|
|
1664
|
+
else:
|
|
1665
|
+
nested_steps = getattr(transformer, "steps", None)
|
|
1666
|
+
if nested_steps:
|
|
1667
|
+
description = " -> ".join(
|
|
1668
|
+
type(component).__name__
|
|
1669
|
+
for _, component in nested_steps
|
|
1670
|
+
)
|
|
1671
|
+
else:
|
|
1672
|
+
description = type(transformer).__name__
|
|
1673
|
+
pipeline_table.add_row(
|
|
1674
|
+
f" {transformer_name}",
|
|
1675
|
+
description,
|
|
1676
|
+
)
|
|
1677
|
+
for name, value in result.get("feature_selection", {}).items():
|
|
1678
|
+
pipeline_table.add_row(
|
|
1679
|
+
name.replace("_", " ").title(),
|
|
1680
|
+
"disabled" if value is None else str(value),
|
|
1681
|
+
)
|
|
1682
|
+
pipeline_table.add_row("Cross-validation", f"{engine.cv}-fold")
|
|
1683
|
+
pipeline_table.add_row("Random state", str(engine.random_state))
|
|
1684
|
+
console.print(
|
|
1685
|
+
Panel(pipeline_table, title="Pipeline Information", border_style="cyan")
|
|
1686
|
+
)
|
|
1687
|
+
|
|
1611
1688
|
audit_table = Table(box=None, padding=(0, 1))
|
|
1612
1689
|
audit_table.add_column("Severity", style="yellow")
|
|
1613
1690
|
audit_table.add_column("Finding")
|
|
@@ -487,10 +487,9 @@ class ModelRegistry:
|
|
|
487
487
|
task_type="classification",
|
|
488
488
|
category="svm",
|
|
489
489
|
requires_scaling=True,
|
|
490
|
-
supports_probability=
|
|
490
|
+
supports_probability=False,
|
|
491
491
|
default_params={
|
|
492
492
|
"kernel": "rbf",
|
|
493
|
-
"probability": True,
|
|
494
493
|
},
|
|
495
494
|
hyperparameter_space={
|
|
496
495
|
"C": [
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "autoforge-engine"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.4"
|
|
8
8
|
description = "A transparent, local-first AutoML framework for automated ML pipeline discovery."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.11"
|
|
@@ -54,4 +54,5 @@ modelforge = "modelforge.cli:app"
|
|
|
54
54
|
include = ["modelforge*"]
|
|
55
55
|
|
|
56
56
|
[tool.pytest.ini_options]
|
|
57
|
-
testpaths = ["tests"]
|
|
57
|
+
testpaths = ["tests"]
|
|
58
|
+
pythonpath = ["."]
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
3
|
import json
|
|
4
|
+
import warnings
|
|
4
5
|
|
|
5
6
|
import numpy as np
|
|
6
7
|
import pandas as pd
|
|
@@ -100,7 +101,7 @@ def test_automl_convenience_function_prints_report(
|
|
|
100
101
|
data_path,
|
|
101
102
|
"target",
|
|
102
103
|
task_type="regression",
|
|
103
|
-
model_names=["linear_regression"],
|
|
104
|
+
model_names=["linear_regression", "ridge"],
|
|
104
105
|
cv=2,
|
|
105
106
|
experiment_directory=tmp_path / "experiments",
|
|
106
107
|
)
|
|
@@ -113,7 +114,86 @@ def test_automl_convenience_function_prints_report(
|
|
|
113
114
|
assert "Best Model & Settings" in output
|
|
114
115
|
assert "Data Quality" in output
|
|
115
116
|
assert "Run Information" in output
|
|
117
|
+
assert "Pipeline Information" in output
|
|
116
118
|
assert "AUTOFORGE COMPLETE" in output
|
|
119
|
+
assert f"{fitted.result['profile']['rows']:,}" in output
|
|
120
|
+
assert f"{fitted.result['profile']['columns']:,}" in output
|
|
121
|
+
assert fitted.result["target"]["target"] in output
|
|
122
|
+
assert fitted.result["target"]["task_type"] in output
|
|
123
|
+
assert fitted.experiment_id in output
|
|
124
|
+
for model in fitted.result["ranking"]["model"]:
|
|
125
|
+
assert fitted.registry.get(model).name in output
|
|
126
|
+
metric_columns = [
|
|
127
|
+
column
|
|
128
|
+
for column in fitted.result["ranking"].columns
|
|
129
|
+
if column.startswith("cv_mean_")
|
|
130
|
+
and not column.endswith("_seconds")
|
|
131
|
+
]
|
|
132
|
+
assert metric_columns
|
|
133
|
+
for column in metric_columns:
|
|
134
|
+
metric_label = column.removeprefix("cv_mean_")
|
|
135
|
+
metric_label = metric_label.replace("_", " ").upper()
|
|
136
|
+
metric_label = metric_label.replace("ROC AUC", "ROC-AUC")
|
|
137
|
+
assert f"CV {metric_label}" in output
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def test_automl_fit_prints_report_by_default(
|
|
141
|
+
capsys,
|
|
142
|
+
tmp_path,
|
|
143
|
+
):
|
|
144
|
+
data = regression_data()
|
|
145
|
+
engine = AutoML(
|
|
146
|
+
cv=2,
|
|
147
|
+
experiment_directory=tmp_path / "experiments",
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
result = engine.fit(
|
|
151
|
+
data,
|
|
152
|
+
target="target",
|
|
153
|
+
task_type="regression",
|
|
154
|
+
model_names=["linear_regression", "ridge"],
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
output = capsys.readouterr().out
|
|
158
|
+
assert f"{result['profile']['rows']:,}" in output
|
|
159
|
+
assert f"{result['profile']['columns']:,}" in output
|
|
160
|
+
assert result["target"]["target"] in output
|
|
161
|
+
assert result["target"]["task_type"] in output
|
|
162
|
+
assert result["experiment_id"] in output
|
|
163
|
+
for model in result["ranking"]["model"]:
|
|
164
|
+
assert engine.registry.get(model).name in output
|
|
165
|
+
assert "Pipeline Information" in output
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def test_classification_report_shows_metrics_for_all_models(
|
|
169
|
+
capsys,
|
|
170
|
+
tmp_path,
|
|
171
|
+
):
|
|
172
|
+
engine = AutoML(
|
|
173
|
+
cv=2,
|
|
174
|
+
experiment_directory=tmp_path / "experiments",
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
with warnings.catch_warnings():
|
|
178
|
+
warnings.simplefilter("error", FutureWarning)
|
|
179
|
+
result = engine.fit(
|
|
180
|
+
classification_data(),
|
|
181
|
+
target="target",
|
|
182
|
+
task_type="classification",
|
|
183
|
+
model_names=["logistic_regression", "svc"],
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
output = capsys.readouterr().out
|
|
187
|
+
assert "CV F1" in output
|
|
188
|
+
assert "CV ROC-AUC" in output
|
|
189
|
+
assert engine.registry.get(result["best_model"]).name in output
|
|
190
|
+
assert result["experiment_id"] in output
|
|
191
|
+
for model in result["ranking"]["model"]:
|
|
192
|
+
assert engine.registry.get(model).name in output
|
|
193
|
+
for column in ("cv_mean_f1", "cv_mean_roc_auc"):
|
|
194
|
+
for value in result["ranking"][column]:
|
|
195
|
+
rendered_value = "N/A" if pd.isna(value) else f"{value:.4f}"
|
|
196
|
+
assert rendered_value in output
|
|
117
197
|
|
|
118
198
|
|
|
119
199
|
def test_automl_rejects_invalid_variance_threshold():
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import re
|
|
2
|
+
|
|
1
3
|
from typer.testing import CliRunner
|
|
2
4
|
|
|
3
5
|
from modelforge.cli import app
|
|
@@ -244,7 +246,17 @@ def test_train_config_help():
|
|
|
244
246
|
)
|
|
245
247
|
|
|
246
248
|
assert result.exit_code == 0
|
|
247
|
-
|
|
249
|
+
|
|
250
|
+
# Typer/Rich can add ANSI escape sequences to
|
|
251
|
+
# help output in CI environments. Remove them
|
|
252
|
+
# before checking the actual CLI text.
|
|
253
|
+
clean_output = re.sub(
|
|
254
|
+
r"\x1b\[[0-9;]*m",
|
|
255
|
+
"",
|
|
256
|
+
result.stdout,
|
|
257
|
+
)
|
|
258
|
+
|
|
259
|
+
assert "--config" in clean_output
|
|
248
260
|
|
|
249
261
|
|
|
250
262
|
def test_train_cli_target_overrides_config(
|
|
@@ -124,7 +124,8 @@ def test_fit_can_use_target_from_config(monkeypatch):
|
|
|
124
124
|
"price": [1, 2, 3],
|
|
125
125
|
"feature": [4, 5, 6],
|
|
126
126
|
}
|
|
127
|
-
)
|
|
127
|
+
),
|
|
128
|
+
print_report=False,
|
|
128
129
|
)
|
|
129
130
|
|
|
130
131
|
assert captured["target"] == "price"
|
|
@@ -187,6 +188,7 @@ def test_explicit_fit_values_override_config():
|
|
|
187
188
|
task_type="regression",
|
|
188
189
|
model_names=["ridge"],
|
|
189
190
|
excluded_columns=["explicit_column"],
|
|
191
|
+
print_report=False,
|
|
190
192
|
)
|
|
191
193
|
finally:
|
|
192
194
|
monkeypatch.undo()
|
|
@@ -153,6 +153,20 @@ def test_logistic_regression_requires_scaling():
|
|
|
153
153
|
assert spec.supports_probability is True
|
|
154
154
|
|
|
155
155
|
|
|
156
|
+
def test_svc_uses_decision_scores_without_probability_estimates():
|
|
157
|
+
registry = ModelRegistry()
|
|
158
|
+
|
|
159
|
+
spec = registry.get("svc")
|
|
160
|
+
model = registry.create("svc")
|
|
161
|
+
|
|
162
|
+
assert spec.supports_probability is False
|
|
163
|
+
assert "probability" not in spec.default_params
|
|
164
|
+
assert model.get_params(deep=False)["probability"] in {
|
|
165
|
+
False,
|
|
166
|
+
"deprecated",
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
|
|
156
170
|
def test_tree_models_do_not_require_scaling():
|
|
157
171
|
registry = ModelRegistry()
|
|
158
172
|
|
|
File without changes
|
|
File without changes
|
{autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/autoforge_engine.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/autoforge_engine.egg-info/entry_points.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{autoforge_engine-0.1.3 → autoforge_engine-0.1.4}/tests/test_experiment_tracker_reproducibility.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|