classifier-toolkit 0.3.6__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of classifier-toolkit might be problematic. Click here for more details.
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/.gitignore +5 -1
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/PKG-INFO +49 -14
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/README.md +47 -13
- classifier_toolkit-0.4.0/classifier_toolkit/datasets/__init__.py +3 -0
- classifier_toolkit-0.4.0/classifier_toolkit/datasets/_demo.py +178 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/explainability/__init__.py +1 -1
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/explainability/misclassification.py +3 -3
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/explainability/plots.py +1 -5
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_reduction/counter_intuitive.py +1 -1
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_reduction/drift.py +29 -18
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_reduction/reducer.py +17 -5
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/utils/scoring.py +8 -6
- classifier_toolkit-0.4.0/classifier_toolkit/feature_selection/wrapper_methods/combination_search.py +1131 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/wrapper_methods/recursive_feature_eliminator.py +538 -18
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/wrapper_methods/rfe.py +6 -5
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/model_training/hyper_parameter_tuning/tuner.py +219 -46
- classifier_toolkit-0.4.0/classifier_toolkit/model_validation/__init__.py +27 -0
- classifier_toolkit-0.4.0/classifier_toolkit/model_validation/evaluator.py +1089 -0
- classifier_toolkit-0.4.0/classifier_toolkit/risk_class/__init__.py +37 -0
- classifier_toolkit-0.4.0/classifier_toolkit/risk_class/risk_classes.py +1144 -0
- classifier_toolkit-0.4.0/classifier_toolkit/risk_class/risk_classes_dp.py +670 -0
- classifier_toolkit-0.4.0/examples/example_bayesian_search.ipynb +492 -0
- classifier_toolkit-0.4.0/examples/example_combination_feature_search.ipynb +404 -0
- classifier_toolkit-0.4.0/examples/example_explainability_catboost.ipynb +641 -0
- classifier_toolkit-0.4.0/examples/example_explainability_lgbm.ipynb +672 -0
- classifier_toolkit-0.4.0/examples/example_feature_reduction.ipynb +376 -0
- classifier_toolkit-0.4.0/examples/example_grid_search.ipynb +588 -0
- classifier_toolkit-0.4.0/examples/example_model_training_catboost.ipynb +194 -0
- classifier_toolkit-0.4.0/examples/example_model_training_lgbm.ipynb +186 -0
- classifier_toolkit-0.4.0/examples/example_model_validation.ipynb +369 -0
- classifier_toolkit-0.4.0/examples/example_recursive_feature_eliminator.ipynb +539 -0
- classifier_toolkit-0.4.0/examples/example_risk_classes_dp.ipynb +290 -0
- classifier_toolkit-0.4.0/examples/example_train_test_partition.ipynb +276 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/pyproject.toml +16 -1
- classifier_toolkit-0.4.0/tests/datasets/test_demo_data.py +65 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/explainability/test_catboost_e2e.py +2 -3
- classifier_toolkit-0.4.0/tests/explainability/test_interactions.py +40 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_reduction/test_drift.py +112 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_reduction/test_reducer.py +50 -0
- classifier_toolkit-0.4.0/tests/feature_selection/test_combination_search.py +842 -0
- classifier_toolkit-0.4.0/tests/feature_selection/test_recursive_feature_eliminator.py +1537 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_selection/test_rfe.py +27 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_selection/test_scoring.py +19 -0
- classifier_toolkit-0.4.0/tests/model_training/hyper_parameter_tuning/test_tuner.py +877 -0
- classifier_toolkit-0.4.0/tests/model_validation/__init__.py +0 -0
- classifier_toolkit-0.4.0/tests/model_validation/test_compare_score_distributions.py +90 -0
- classifier_toolkit-0.4.0/tests/model_validation/test_evaluate_risk_classes_dp.py +46 -0
- classifier_toolkit-0.4.0/tests/model_validation/test_evaluator_threshold.py +33 -0
- classifier_toolkit-0.4.0/tests/model_validation/test_print_full_validation_report.py +101 -0
- classifier_toolkit-0.4.0/tests/model_validation/test_print_risk_class_validation_report.py +71 -0
- classifier_toolkit-0.4.0/tests/model_validation/test_score_distribution_psi.py +69 -0
- classifier_toolkit-0.4.0/tests/risk_class/__init__.py +0 -0
- classifier_toolkit-0.4.0/tests/risk_class/test_construct_bins_dp.py +281 -0
- classifier_toolkit-0.4.0/tests/risk_class/test_risk_classes.py +193 -0
- classifier_toolkit-0.4.0/tests/risk_class/test_validate_risk_classes.py +137 -0
- classifier_toolkit-0.3.6/.github/pull_request_template/default.md +0 -13
- classifier_toolkit-0.3.6/.github/workflows/checks.yaml +0 -148
- classifier_toolkit-0.3.6/.github/workflows/docs.yml +0 -31
- classifier_toolkit-0.3.6/.github/workflows/master.yaml +0 -48
- classifier_toolkit-0.3.6/.github/workflows/release.yaml +0 -51
- classifier_toolkit-0.3.6/.github/workflows/working-branch.yaml +0 -14
- classifier_toolkit-0.3.6/.python-version +0 -1
- classifier_toolkit-0.3.6/.sqlfluff +0 -38
- classifier_toolkit-0.3.6/Makefile +0 -22
- classifier_toolkit-0.3.6/classifier_toolkit/feature_selection/wrapper_methods/combination_search.py +0 -632
- classifier_toolkit-0.3.6/docs/CNAME +0 -1
- classifier_toolkit-0.3.6/docs/calibration/overview.md +0 -53
- classifier_toolkit-0.3.6/docs/changelog.md +0 -65
- classifier_toolkit-0.3.6/docs/data_partition/data_preprocess.md +0 -43
- classifier_toolkit-0.3.6/docs/data_partition/optimize_data.md +0 -30
- classifier_toolkit-0.3.6/docs/data_partition/overview.md +0 -30
- classifier_toolkit-0.3.6/docs/data_partition/split_train_test.md +0 -43
- classifier_toolkit-0.3.6/docs/eda/bivariate_analysis.md +0 -38
- classifier_toolkit-0.3.6/docs/eda/eda_toolkit.md +0 -65
- classifier_toolkit-0.3.6/docs/eda/feature_engineering.md +0 -60
- classifier_toolkit-0.3.6/docs/eda/first_glance.md +0 -54
- classifier_toolkit-0.3.6/docs/eda/overview.md +0 -128
- classifier_toolkit-0.3.6/docs/eda/univariate_analysis.md +0 -48
- classifier_toolkit-0.3.6/docs/eda/visualizations.md +0 -51
- classifier_toolkit-0.3.6/docs/eda/warnings/default_warnings.md +0 -45
- classifier_toolkit-0.3.6/docs/eda/warnings/warning_system.md +0 -32
- classifier_toolkit-0.3.6/docs/examples/eda_example.md +0 -71
- classifier_toolkit-0.3.6/docs/examples/feature_selection_advanced.md +0 -111
- classifier_toolkit-0.3.6/docs/examples/feature_selection_example.md +0 -123
- classifier_toolkit-0.3.6/docs/explainability/interactions.md +0 -48
- classifier_toolkit-0.3.6/docs/explainability/misclassification.md +0 -46
- classifier_toolkit-0.3.6/docs/explainability/overview.md +0 -87
- classifier_toolkit-0.3.6/docs/explainability/plots.md +0 -47
- classifier_toolkit-0.3.6/docs/explainability/toolkit.md +0 -59
- classifier_toolkit-0.3.6/docs/explainability/tree_explainer.md +0 -50
- classifier_toolkit-0.3.6/docs/feature_reduction/correlation.md +0 -34
- classifier_toolkit-0.3.6/docs/feature_reduction/counter_intuitive.md +0 -49
- classifier_toolkit-0.3.6/docs/feature_reduction/drift.md +0 -35
- classifier_toolkit-0.3.6/docs/feature_reduction/expert_rules.md +0 -26
- classifier_toolkit-0.3.6/docs/feature_reduction/low_variance.md +0 -22
- classifier_toolkit-0.3.6/docs/feature_reduction/overview.md +0 -49
- classifier_toolkit-0.3.6/docs/feature_reduction/predictive_power.md +0 -55
- classifier_toolkit-0.3.6/docs/feature_reduction/reducer.md +0 -46
- classifier_toolkit-0.3.6/docs/feature_selection/embedded_methods/elastic_net.md +0 -54
- classifier_toolkit-0.3.6/docs/feature_selection/feature_stability.md +0 -38
- classifier_toolkit-0.3.6/docs/feature_selection/meta_selector.md +0 -99
- classifier_toolkit-0.3.6/docs/feature_selection/overview.md +0 -89
- classifier_toolkit-0.3.6/docs/feature_selection/utils/data_handling.md +0 -60
- classifier_toolkit-0.3.6/docs/feature_selection/utils/scoring.md +0 -50
- classifier_toolkit-0.3.6/docs/feature_selection/wrapper_methods/bayesian_search.md +0 -44
- classifier_toolkit-0.3.6/docs/feature_selection/wrapper_methods/boruta.md +0 -49
- classifier_toolkit-0.3.6/docs/feature_selection/wrapper_methods/combination_search.md +0 -39
- classifier_toolkit-0.3.6/docs/feature_selection/wrapper_methods/recursive_feature_eliminator.md +0 -35
- classifier_toolkit-0.3.6/docs/feature_selection/wrapper_methods/rfe.md +0 -90
- classifier_toolkit-0.3.6/docs/feature_selection/wrapper_methods/sequential_selection.md +0 -54
- classifier_toolkit-0.3.6/docs/index.md +0 -92
- classifier_toolkit-0.3.6/docs/model_training/overview.md +0 -51
- classifier_toolkit-0.3.6/docs/model_training/tuner.md +0 -359
- classifier_toolkit-0.3.6/docs/reference/calibration/ovr_calibration.md +0 -3
- classifier_toolkit-0.3.6/docs/reference/calibration/reliability.md +0 -3
- classifier_toolkit-0.3.6/docs/reference/data_partition/data_preprocess.md +0 -46
- classifier_toolkit-0.3.6/docs/reference/data_partition/optimize_data.md +0 -36
- classifier_toolkit-0.3.6/docs/reference/data_partition/overview.md +0 -37
- classifier_toolkit-0.3.6/docs/reference/data_partition/split_train_test.md +0 -57
- classifier_toolkit-0.3.6/docs/reference/eda/bivariate_analysis.md +0 -42
- classifier_toolkit-0.3.6/docs/reference/eda/eda_toolkit.md +0 -47
- classifier_toolkit-0.3.6/docs/reference/eda/feature_engineering.md +0 -42
- classifier_toolkit-0.3.6/docs/reference/eda/first_glance.md +0 -42
- classifier_toolkit-0.3.6/docs/reference/eda/overview.md +0 -35
- classifier_toolkit-0.3.6/docs/reference/eda/univariate_analysis.md +0 -43
- classifier_toolkit-0.3.6/docs/reference/eda/visualizations.md +0 -39
- classifier_toolkit-0.3.6/docs/reference/eda/warnings/default_warnings.md +0 -52
- classifier_toolkit-0.3.6/docs/reference/eda/warnings/warning_system.md +0 -36
- classifier_toolkit-0.3.6/docs/reference/explainability/interactions.md +0 -7
- classifier_toolkit-0.3.6/docs/reference/explainability/misclassification.md +0 -3
- classifier_toolkit-0.3.6/docs/reference/explainability/overview.md +0 -42
- classifier_toolkit-0.3.6/docs/reference/explainability/plots.md +0 -7
- classifier_toolkit-0.3.6/docs/reference/explainability/toolkit.md +0 -3
- classifier_toolkit-0.3.6/docs/reference/explainability/tree_explainer.md +0 -7
- classifier_toolkit-0.3.6/docs/reference/feature_reduction/base.md +0 -5
- classifier_toolkit-0.3.6/docs/reference/feature_reduction/correlation.md +0 -49
- classifier_toolkit-0.3.6/docs/reference/feature_reduction/counter_intuitive.md +0 -40
- classifier_toolkit-0.3.6/docs/reference/feature_reduction/drift.md +0 -75
- classifier_toolkit-0.3.6/docs/reference/feature_reduction/expert_rules.md +0 -16
- classifier_toolkit-0.3.6/docs/reference/feature_reduction/low_variance.md +0 -25
- classifier_toolkit-0.3.6/docs/reference/feature_reduction/overview.md +0 -33
- classifier_toolkit-0.3.6/docs/reference/feature_reduction/predictive_power.md +0 -68
- classifier_toolkit-0.3.6/docs/reference/feature_reduction/reducer.md +0 -121
- classifier_toolkit-0.3.6/docs/reference/feature_selection/base.md +0 -3
- classifier_toolkit-0.3.6/docs/reference/feature_selection/embedded_methods/elastic_net.md +0 -40
- classifier_toolkit-0.3.6/docs/reference/feature_selection/feature_stability.md +0 -38
- classifier_toolkit-0.3.6/docs/reference/feature_selection/meta_selector.md +0 -41
- classifier_toolkit-0.3.6/docs/reference/feature_selection/overview.md +0 -28
- classifier_toolkit-0.3.6/docs/reference/feature_selection/utils/data_handling.md +0 -43
- classifier_toolkit-0.3.6/docs/reference/feature_selection/utils/plottings.md +0 -5
- classifier_toolkit-0.3.6/docs/reference/feature_selection/utils/scoring.md +0 -37
- classifier_toolkit-0.3.6/docs/reference/feature_selection/wrapper_methods/bayesian_search.md +0 -35
- classifier_toolkit-0.3.6/docs/reference/feature_selection/wrapper_methods/boruta.md +0 -34
- classifier_toolkit-0.3.6/docs/reference/feature_selection/wrapper_methods/combination_search.md +0 -111
- classifier_toolkit-0.3.6/docs/reference/feature_selection/wrapper_methods/recursive_feature_eliminator.md +0 -268
- classifier_toolkit-0.3.6/docs/reference/feature_selection/wrapper_methods/rfe.md +0 -62
- classifier_toolkit-0.3.6/docs/reference/feature_selection/wrapper_methods/sequential_selection.md +0 -55
- classifier_toolkit-0.3.6/docs/reference/model_training/params.md +0 -5
- classifier_toolkit-0.3.6/docs/reference/tuner/tuner.md +0 -3
- classifier_toolkit-0.3.6/examples/__init__.py +0 -1
- classifier_toolkit-0.3.6/examples/example_bayesian_search.ipynb +0 -4475
- classifier_toolkit-0.3.6/examples/example_combination_feature_search.ipynb +0 -4262
- classifier_toolkit-0.3.6/examples/example_explainability_catboost.ipynb +0 -1565
- classifier_toolkit-0.3.6/examples/example_explainability_lgbm.ipynb +0 -1625
- classifier_toolkit-0.3.6/examples/example_feature_reduction.ipynb +0 -2280
- classifier_toolkit-0.3.6/examples/example_grid_search.ipynb +0 -1975
- classifier_toolkit-0.3.6/examples/example_model_training_catboost.ipynb +0 -133
- classifier_toolkit-0.3.6/examples/example_model_training_lgbm.ipynb +0 -133
- classifier_toolkit-0.3.6/examples/example_recursive_feature_eliminator.ipynb +0 -862
- classifier_toolkit-0.3.6/examples/example_train_test_partition.ipynb +0 -446
- classifier_toolkit-0.3.6/main.py +0 -6
- classifier_toolkit-0.3.6/mkdocs.yml +0 -200
- classifier_toolkit-0.3.6/notebooks/paylater_removed.json +0 -244
- classifier_toolkit-0.3.6/ruff.toml +0 -46
- classifier_toolkit-0.3.6/tests/feature_selection/test_combination_search.py +0 -403
- classifier_toolkit-0.3.6/tests/feature_selection/test_recursive_feature_eliminator.py +0 -638
- classifier_toolkit-0.3.6/tests/model_training/hyper_parameter_tuning/test_tuner.py +0 -418
- classifier_toolkit-0.3.6/uv.lock +0 -3998
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/LICENSE +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/calibration/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/calibration/base.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/calibration/ovr_calibration.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/calibration/reliability.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/data_partition/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/data_partition/data_preprocess.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/data_partition/optimize_data.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/data_partition/split_train_test.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/eda/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/eda/bivariate_analysis.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/eda/eda_toolkit.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/eda/feature_engineering.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/eda/first_glance.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/eda/univariate_analysis.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/eda/visualizations.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/eda/warnings/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/eda/warnings/automated_warnings.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/eda/warnings/default_warnings.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/explainability/base.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/explainability/interactions.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/explainability/toolkit.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/explainability/tree_explainer.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_reduction/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_reduction/base.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_reduction/correlation.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_reduction/expert_rules.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_reduction/low_variance.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_reduction/predictive_power.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/base.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/embedded_methods/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/embedded_methods/elastic_net.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/feature_stability.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/meta_selector.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/utils/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/utils/data_handling.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/utils/feature_type_constraints.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/utils/plottings.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/wrapper_methods/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/wrapper_methods/bayesian_search.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/wrapper_methods/boruta.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/wrapper_methods/rfe_catboost.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_selection/wrapper_methods/sequential_selection.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/model_training/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/model_training/hyper_parameter_tuning/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/model_training/models/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/model_training/models/base.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/model_training/models/ensemble_methods.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/model_training/utils/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/model_training/utils/params.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/calibration/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/calibration/test_ovr_calibration.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/calibration/test_reliability.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/conftest.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/data_partition/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/data_partition/test_data_preprocess.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/data_partition/test_optimize_data.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/data_partition/test_split_train_test.py +0 -0
- {classifier_toolkit-0.3.6/tests/eda → classifier_toolkit-0.4.0/tests/datasets}/__init__.py +0 -0
- {classifier_toolkit-0.3.6/tests/explainability → classifier_toolkit-0.4.0/tests/eda}/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/eda/test_bivariate_analysis.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/eda/test_feature_engineering.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/eda/test_first_glance.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/eda/test_univariate_analysis.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/eda/test_visualizations.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/eda/test_warnings.py +0 -0
- {classifier_toolkit-0.3.6/tests/model_training → classifier_toolkit-0.4.0/tests/explainability}/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/explainability/test_misclassification.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/explainability/test_multiclass_e2e.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/explainability/test_plots.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/explainability/test_smoke.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/explainability/test_toolkit.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/explainability/test_tree_explainer.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_reduction/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_reduction/test_correlation.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_reduction/test_counter_intuitive.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_reduction/test_expert_rules.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_reduction/test_low_variance.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_reduction/test_predictive_power.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_reduction/test_smoke.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_selection/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_selection/test_bayesian_search.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_selection/test_boruta.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_selection/test_elastic_net.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_selection/test_feature_stability.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_selection/test_rfe_catboost.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/feature_selection/test_sequential_selection.py +0 -0
- {classifier_toolkit-0.3.6/tests/model_training/hyper_parameter_tuning → classifier_toolkit-0.4.0/tests/model_training}/__init__.py +0 -0
- {classifier_toolkit-0.3.6/tests/model_training/models → classifier_toolkit-0.4.0/tests/model_training/hyper_parameter_tuning}/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/model_training/hyper_parameter_tuning/test_params.py +0 -0
- /classifier_toolkit-0.3.6/docs/stylesheets/extra.css → /classifier_toolkit-0.4.0/tests/model_training/models/__init__.py +0 -0
- {classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/tests/model_training/models/test_ensemble_methods.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: classifier-toolkit
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Project-URL: Documentation, https://supreme-adventure-jg5qkyr.pages.github.io/
|
|
5
5
|
Author-email: "senih.yilmaz" <senih.yilmaz@qonto.com>, "jeremy.fraoua" <jeremy.fraoua@qonto.com>, "gauthier.marquand" <gauthier.marquand@qonto.com>, "arnaud.alepee" <arnaud.alepee@qonto.com>
|
|
6
6
|
License-File: LICENSE
|
|
@@ -18,6 +18,7 @@ Requires-Dist: polars<2.0.0,>=1.2.1
|
|
|
18
18
|
Requires-Dist: pyarrow>=18.0.0
|
|
19
19
|
Requires-Dist: scikit-learn<2.0.0,>=1.4.0
|
|
20
20
|
Requires-Dist: scipy>=1.11.0
|
|
21
|
+
Requires-Dist: seaborn<0.14.0,>=0.13.0
|
|
21
22
|
Requires-Dist: shap>=0.46.0
|
|
22
23
|
Requires-Dist: statsmodels<0.15.0,>=0.14.2
|
|
23
24
|
Requires-Dist: tabulate<0.10.0,>=0.9.0
|
|
@@ -60,13 +61,17 @@ This library is published in the PyPI directory. To install, users can run pip i
|
|
|
60
61
|
|
|
61
62
|
### Usage
|
|
62
63
|
|
|
63
|
-
This library automates binary classification
|
|
64
|
+
This library automates binary and multiclass classification workflows. It is independent of the modelled problem: the class of interest is configured through `pos_label` (the positive class for binary targets, the class of interest for multiclass ones). It includes several packages designed to address the main steps in any machine learning/data science task:
|
|
64
65
|
|
|
65
|
-
1. **EDA**: accessible via `
|
|
66
|
+
1. **EDA**: accessible via `EDAToolkit`. Provides EDA and feature engineering functionality with all necessary visualizations.
|
|
66
67
|
2. **Feature Reduction**: filter-style pre-selection pipeline (expert rules, low variance, drift, predictive power, counter-intuitive direction, high correlation).
|
|
67
68
|
3. **Feature Selection**: wrapper and embedded methods (RFE, Boruta, Sequential, Bayesian, ElasticNet, MetaSelector).
|
|
68
|
-
4. **Model Training**: accessible via `Tuner`. Hyperparameter optimization (grid search, Bayesian via Optuna) for LightGBM and CatBoost with train/val/test evaluation.
|
|
69
|
-
5.
|
|
69
|
+
4. **Model Training**: accessible via `Tuner`. Hyperparameter optimization (grid search, Bayesian via Optuna) for LightGBM and CatBoost with train/val/test evaluation, including multiclass objectives and class weights.
|
|
70
|
+
5. **Risk Class**: `RiskClassBuilderDP` / `RiskClassBuilder` turn model scores into risk classes with statistically validated, ordered event rates.
|
|
71
|
+
6. **Model Validation**: `Evaluator` computes metrics, calibration and stability checks for any model exposing `predict_proba`.
|
|
72
|
+
7. **Calibration**: `OVRHistogramCalibrator` and `plot_reliability_curves` for One-vs-Rest probability calibration.
|
|
73
|
+
8. **Explainability**: SHAP-based explanations, interactions and misclassification diagnostics.
|
|
74
|
+
9. **Data Partition**: temporal-aware train/test splitting, preprocessing and dtype optimization.
|
|
70
75
|
|
|
71
76
|
For detailed usage, refer to the documentation.
|
|
72
77
|
|
|
@@ -139,16 +144,46 @@ For detailed usage, refer to the documentation.
|
|
|
139
144
|
- **Model Training**: Hyperparameter optimization for LightGBM and CatBoost, with support for grid search and Bayesian optimization (via Optuna).
|
|
140
145
|
|
|
141
146
|
```python
|
|
142
|
-
from classifier_toolkit.model_training.hyper_parameter_tuning import Tuner
|
|
147
|
+
from classifier_toolkit.model_training.hyper_parameter_tuning.tuner import Tuner
|
|
148
|
+
|
|
149
|
+
tuner = Tuner(
|
|
150
|
+
X=X_train, y=y_train,
|
|
151
|
+
model_name="lightgbm",
|
|
152
|
+
X_val=X_val, y_val=y_val,
|
|
153
|
+
X_test=X_test, y_test=y_test,
|
|
154
|
+
search_method="bayesian",
|
|
155
|
+
n_trials=50,
|
|
156
|
+
optimization_metric="prauc",
|
|
157
|
+
)
|
|
158
|
+
result = tuner.tune()
|
|
159
|
+
|
|
160
|
+
best_model = result["best_model"]
|
|
161
|
+
result["trials_results"] # full trial results (DataFrame)
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
Reported metrics are `auc`, `prauc`, `ks`, `log_loss` and `brier` (`ks`/`brier` are binary-only). Custom parameter search spaces can be defined via `ModelParams` and `ParamRange`.
|
|
165
|
+
|
|
166
|
+
- **Risk Class**: Builds risk classes from model scores. `RiskClassBuilderDP` searches bin edges with dynamic programming so that each class' observed event rate falls in a target band (`target_ranges`, required), then checks that adjacent classes are statistically distinguishable; `RiskClassBuilder` discovers classes with KMeans.
|
|
143
167
|
|
|
144
|
-
|
|
145
|
-
|
|
168
|
+
```python
|
|
169
|
+
from classifier_toolkit.risk_class import RiskClassBuilderDP
|
|
146
170
|
|
|
147
|
-
|
|
148
|
-
|
|
171
|
+
builder = RiskClassBuilderDP(
|
|
172
|
+
target_col="target",
|
|
173
|
+
target_ranges=[(0.00, 0.02), (0.02, 0.05), (0.05, 0.10)],
|
|
174
|
+
min_obs_per_bin=200,
|
|
175
|
+
)
|
|
176
|
+
result = builder.build(train_proba, y_train)
|
|
177
|
+
print(result.bins, result.n_classes)
|
|
149
178
|
```
|
|
150
179
|
|
|
151
|
-
|
|
180
|
+
- **Model Validation**: `Evaluator` evaluates any model with `predict_proba` (metrics, ROC/PR/calibration/threshold plots), `score_distribution_psi` and `compare_score_distributions` compare a reference score distribution with a current one, and `evaluate_risk_classes_dp` checks calibration within each risk class.
|
|
181
|
+
|
|
182
|
+
- **Calibration**: `OVRHistogramCalibrator` calibrates multiclass (or binary) probabilities One-vs-Rest with histogram binning; `plot_reliability_curves` shows raw vs calibrated reliability per class.
|
|
183
|
+
|
|
184
|
+
- **Explainability**: `TreeSHAPExplainer`, `ExplainabilityToolkit`, SHAP plots, pairwise interaction analysis and a `MisclassificationAnalyzer` that explains confusion-matrix quadrants (per-class SHAP for multiclass models).
|
|
185
|
+
|
|
186
|
+
- **Data Partition**: Temporal-aware train/test splitting, preprocessing helpers and dtype optimization.
|
|
152
187
|
|
|
153
188
|
### Development & CI/CD
|
|
154
189
|
|
|
@@ -162,8 +197,8 @@ This project uses modern tooling for fast and efficient development workflows:
|
|
|
162
197
|
#### CI/CD Pipeline
|
|
163
198
|
Our CI/CD pipeline is optimized for speed and efficiency:
|
|
164
199
|
|
|
165
|
-
- **Parallel Test Execution**:
|
|
166
|
-
- **Shared Caching**:
|
|
200
|
+
- **Parallel Test Execution**: One test job per directory under `tests/` (discovered automatically, so new test groups are picked up without editing the workflow), all running simultaneously
|
|
201
|
+
- **Shared Caching**: The parallel jobs share the same dependency cache, avoiding duplicate downloads
|
|
167
202
|
- **Smart Test Reruns**: Failed tests run first (`pytest --lf --ff`) for faster feedback on fixes
|
|
168
203
|
- **Master Protection**: Build tests only run on `master` branch and PRs targeting `master`, saving CI resources on feature branches
|
|
169
204
|
- **Automatic Linting**: Code quality checks (Ruff, SQLFluff) run on every push
|
|
@@ -188,5 +223,5 @@ uv build
|
|
|
188
223
|
|
|
189
224
|
### Future Work
|
|
190
225
|
The next planned improvements and additions to the library include:
|
|
191
|
-
*
|
|
226
|
+
* Extending the evaluation and reporting tools in `model_validation` (more checks and report formats).
|
|
192
227
|
* Expanding documentation to include architecture diagrams and detailed usage examples.
|
|
@@ -29,13 +29,17 @@ This library is published in the PyPI directory. To install, users can run pip i
|
|
|
29
29
|
|
|
30
30
|
### Usage
|
|
31
31
|
|
|
32
|
-
This library automates binary classification
|
|
32
|
+
This library automates binary and multiclass classification workflows. It is independent of the modelled problem: the class of interest is configured through `pos_label` (the positive class for binary targets, the class of interest for multiclass ones). It includes several packages designed to address the main steps in any machine learning/data science task:
|
|
33
33
|
|
|
34
|
-
1. **EDA**: accessible via `
|
|
34
|
+
1. **EDA**: accessible via `EDAToolkit`. Provides EDA and feature engineering functionality with all necessary visualizations.
|
|
35
35
|
2. **Feature Reduction**: filter-style pre-selection pipeline (expert rules, low variance, drift, predictive power, counter-intuitive direction, high correlation).
|
|
36
36
|
3. **Feature Selection**: wrapper and embedded methods (RFE, Boruta, Sequential, Bayesian, ElasticNet, MetaSelector).
|
|
37
|
-
4. **Model Training**: accessible via `Tuner`. Hyperparameter optimization (grid search, Bayesian via Optuna) for LightGBM and CatBoost with train/val/test evaluation.
|
|
38
|
-
5.
|
|
37
|
+
4. **Model Training**: accessible via `Tuner`. Hyperparameter optimization (grid search, Bayesian via Optuna) for LightGBM and CatBoost with train/val/test evaluation, including multiclass objectives and class weights.
|
|
38
|
+
5. **Risk Class**: `RiskClassBuilderDP` / `RiskClassBuilder` turn model scores into risk classes with statistically validated, ordered event rates.
|
|
39
|
+
6. **Model Validation**: `Evaluator` computes metrics, calibration and stability checks for any model exposing `predict_proba`.
|
|
40
|
+
7. **Calibration**: `OVRHistogramCalibrator` and `plot_reliability_curves` for One-vs-Rest probability calibration.
|
|
41
|
+
8. **Explainability**: SHAP-based explanations, interactions and misclassification diagnostics.
|
|
42
|
+
9. **Data Partition**: temporal-aware train/test splitting, preprocessing and dtype optimization.
|
|
39
43
|
|
|
40
44
|
For detailed usage, refer to the documentation.
|
|
41
45
|
|
|
@@ -108,16 +112,46 @@ For detailed usage, refer to the documentation.
|
|
|
108
112
|
- **Model Training**: Hyperparameter optimization for LightGBM and CatBoost, with support for grid search and Bayesian optimization (via Optuna).
|
|
109
113
|
|
|
110
114
|
```python
|
|
111
|
-
from classifier_toolkit.model_training.hyper_parameter_tuning import Tuner
|
|
115
|
+
from classifier_toolkit.model_training.hyper_parameter_tuning.tuner import Tuner
|
|
116
|
+
|
|
117
|
+
tuner = Tuner(
|
|
118
|
+
X=X_train, y=y_train,
|
|
119
|
+
model_name="lightgbm",
|
|
120
|
+
X_val=X_val, y_val=y_val,
|
|
121
|
+
X_test=X_test, y_test=y_test,
|
|
122
|
+
search_method="bayesian",
|
|
123
|
+
n_trials=50,
|
|
124
|
+
optimization_metric="prauc",
|
|
125
|
+
)
|
|
126
|
+
result = tuner.tune()
|
|
127
|
+
|
|
128
|
+
best_model = result["best_model"]
|
|
129
|
+
result["trials_results"] # full trial results (DataFrame)
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
Reported metrics are `auc`, `prauc`, `ks`, `log_loss` and `brier` (`ks`/`brier` are binary-only). Custom parameter search spaces can be defined via `ModelParams` and `ParamRange`.
|
|
133
|
+
|
|
134
|
+
- **Risk Class**: Builds risk classes from model scores. `RiskClassBuilderDP` searches bin edges with dynamic programming so that each class' observed event rate falls in a target band (`target_ranges`, required), then checks that adjacent classes are statistically distinguishable; `RiskClassBuilder` discovers classes with KMeans.
|
|
112
135
|
|
|
113
|
-
|
|
114
|
-
|
|
136
|
+
```python
|
|
137
|
+
from classifier_toolkit.risk_class import RiskClassBuilderDP
|
|
115
138
|
|
|
116
|
-
|
|
117
|
-
|
|
139
|
+
builder = RiskClassBuilderDP(
|
|
140
|
+
target_col="target",
|
|
141
|
+
target_ranges=[(0.00, 0.02), (0.02, 0.05), (0.05, 0.10)],
|
|
142
|
+
min_obs_per_bin=200,
|
|
143
|
+
)
|
|
144
|
+
result = builder.build(train_proba, y_train)
|
|
145
|
+
print(result.bins, result.n_classes)
|
|
118
146
|
```
|
|
119
147
|
|
|
120
|
-
|
|
148
|
+
- **Model Validation**: `Evaluator` evaluates any model with `predict_proba` (metrics, ROC/PR/calibration/threshold plots), `score_distribution_psi` and `compare_score_distributions` compare a reference score distribution with a current one, and `evaluate_risk_classes_dp` checks calibration within each risk class.
|
|
149
|
+
|
|
150
|
+
- **Calibration**: `OVRHistogramCalibrator` calibrates multiclass (or binary) probabilities One-vs-Rest with histogram binning; `plot_reliability_curves` shows raw vs calibrated reliability per class.
|
|
151
|
+
|
|
152
|
+
- **Explainability**: `TreeSHAPExplainer`, `ExplainabilityToolkit`, SHAP plots, pairwise interaction analysis and a `MisclassificationAnalyzer` that explains confusion-matrix quadrants (per-class SHAP for multiclass models).
|
|
153
|
+
|
|
154
|
+
- **Data Partition**: Temporal-aware train/test splitting, preprocessing helpers and dtype optimization.
|
|
121
155
|
|
|
122
156
|
### Development & CI/CD
|
|
123
157
|
|
|
@@ -131,8 +165,8 @@ This project uses modern tooling for fast and efficient development workflows:
|
|
|
131
165
|
#### CI/CD Pipeline
|
|
132
166
|
Our CI/CD pipeline is optimized for speed and efficiency:
|
|
133
167
|
|
|
134
|
-
- **Parallel Test Execution**:
|
|
135
|
-
- **Shared Caching**:
|
|
168
|
+
- **Parallel Test Execution**: One test job per directory under `tests/` (discovered automatically, so new test groups are picked up without editing the workflow), all running simultaneously
|
|
169
|
+
- **Shared Caching**: The parallel jobs share the same dependency cache, avoiding duplicate downloads
|
|
136
170
|
- **Smart Test Reruns**: Failed tests run first (`pytest --lf --ff`) for faster feedback on fixes
|
|
137
171
|
- **Master Protection**: Build tests only run on `master` branch and PRs targeting `master`, saving CI resources on feature branches
|
|
138
172
|
- **Automatic Linting**: Code quality checks (Ruff, SQLFluff) run on every push
|
|
@@ -157,5 +191,5 @@ uv build
|
|
|
157
191
|
|
|
158
192
|
### Future Work
|
|
159
193
|
The next planned improvements and additions to the library include:
|
|
160
|
-
*
|
|
194
|
+
* Extending the evaluation and reporting tools in `model_validation` (more checks and report formats).
|
|
161
195
|
* Expanding documentation to include architecture diagrams and detailed usage examples.
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
"""Synthetic panel data for the examples and the documentation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pandas as pd
|
|
7
|
+
|
|
8
|
+
ID_COLS = ["entity_id", "observation_date"]
|
|
9
|
+
TARGET_COL = "target"
|
|
10
|
+
|
|
11
|
+
_SECTORS = {
|
|
12
|
+
"retail": 0.3,
|
|
13
|
+
"construction": 0.5,
|
|
14
|
+
"hospitality": 0.6,
|
|
15
|
+
"services": 0.0,
|
|
16
|
+
"tech": -0.4,
|
|
17
|
+
"health": -0.3,
|
|
18
|
+
"transport": 0.2,
|
|
19
|
+
"manufacturing": -0.1,
|
|
20
|
+
}
|
|
21
|
+
_LEGAL_FORMS = {"ltd": 0.0, "sole_trader": 0.3, "partnership": 0.1, "cooperative": -0.2}
|
|
22
|
+
_COUNTRIES = ["FR", "DE", "IT", "ES"]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def make_demo_data(
|
|
26
|
+
n_samples: int = 20_000,
|
|
27
|
+
event_rate: float = 0.05,
|
|
28
|
+
n_months: int = 24,
|
|
29
|
+
start_date: str = "2023-01-01",
|
|
30
|
+
random_state: int | None = 0,
|
|
31
|
+
) -> pd.DataFrame:
|
|
32
|
+
"""Generate a synthetic binary-classification panel for the examples.
|
|
33
|
+
|
|
34
|
+
Each row is one entity observed at one month. Entities have a hidden risk
|
|
35
|
+
level that drives both their features and the target, so the features carry
|
|
36
|
+
a realistic, noisy signal. The columns are built so that every tool in the
|
|
37
|
+
toolkit has something to find:
|
|
38
|
+
|
|
39
|
+
- informative numerical features, with the sign of their relation to the
|
|
40
|
+
target given by their name (``*_balance_*``, ``inflow_*`` lower the risk;
|
|
41
|
+
``declined_*``, ``low_balance_days_*`` raise it)
|
|
42
|
+
- near-duplicates (``low_balance_days_cnt_3m`` vs ``_1m``,
|
|
43
|
+
``min_balance_3m`` vs ``mean_balance_3m``) for correlation analysis
|
|
44
|
+
- a counter-intuitive feature: ``max_balance_3m`` *raises* the risk,
|
|
45
|
+
unlike the other balance features
|
|
46
|
+
- a drifting feature: ``card_spend_sum_1m`` grows over time, the target
|
|
47
|
+
relation doesn't
|
|
48
|
+
- near-constant features (``has_rare_flag``, ``reporting_version``,
|
|
49
|
+
``segment``) for variance filters
|
|
50
|
+
- pure noise (``noise_0`` … ``noise_4``)
|
|
51
|
+
- categorical features with an effect (``sector``, ``legal_form``) and
|
|
52
|
+
without one (``country``)
|
|
53
|
+
|
|
54
|
+
The data is entirely synthetic. It contains no real entity.
|
|
55
|
+
|
|
56
|
+
Parameters
|
|
57
|
+
----------
|
|
58
|
+
n_samples : int
|
|
59
|
+
Number of rows.
|
|
60
|
+
event_rate : float
|
|
61
|
+
Expected share of positive targets, between 0 and 1.
|
|
62
|
+
n_months : int
|
|
63
|
+
Number of monthly observation dates, starting at ``start_date``.
|
|
64
|
+
start_date : str
|
|
65
|
+
First observation date.
|
|
66
|
+
random_state : int or None
|
|
67
|
+
Seed, for reproducible data.
|
|
68
|
+
|
|
69
|
+
Returns
|
|
70
|
+
-------
|
|
71
|
+
pd.DataFrame
|
|
72
|
+
One row per (entity, observation date), sorted by date. ``entity_id``
|
|
73
|
+
and ``observation_date`` identify a row, ``target`` is 0/1, the other
|
|
74
|
+
columns are features.
|
|
75
|
+
|
|
76
|
+
Examples
|
|
77
|
+
--------
|
|
78
|
+
>>> from classifier_toolkit.datasets import make_demo_data
|
|
79
|
+
>>> df = make_demo_data(n_samples=1_000)
|
|
80
|
+
>>> df["target"].mean() # doctest: +SKIP
|
|
81
|
+
0.05
|
|
82
|
+
"""
|
|
83
|
+
if not 0 < event_rate < 1:
|
|
84
|
+
raise ValueError(f"event_rate must be between 0 and 1, got {event_rate}.")
|
|
85
|
+
if n_samples < 1 or n_months < 1:
|
|
86
|
+
raise ValueError("n_samples and n_months must be at least 1.")
|
|
87
|
+
|
|
88
|
+
rng = np.random.default_rng(random_state)
|
|
89
|
+
|
|
90
|
+
# Entities: about 4 observations each, at distinct months.
|
|
91
|
+
n_entities = -(-n_samples // min(4, n_months))
|
|
92
|
+
cell = rng.choice(n_entities * n_months, n_samples, replace=False)
|
|
93
|
+
entity, month = np.divmod(cell, n_months)
|
|
94
|
+
dates = pd.date_range(start_date, periods=n_months, freq="MS")
|
|
95
|
+
|
|
96
|
+
sector = rng.choice(list(_SECTORS), n_entities)
|
|
97
|
+
legal_form = rng.choice(list(_LEGAL_FORMS), n_entities, p=[0.5, 0.3, 0.15, 0.05])
|
|
98
|
+
country = rng.choice(_COUNTRIES, n_entities, p=[0.4, 0.3, 0.2, 0.1])
|
|
99
|
+
age_months = rng.gamma(2.0, 30.0, n_entities)
|
|
100
|
+
|
|
101
|
+
# Hidden risk: entity level plus a monthly shock.
|
|
102
|
+
entity_risk = rng.normal(0, 1, n_entities)
|
|
103
|
+
risk = entity_risk[entity] + rng.normal(0, 0.6, n_samples)
|
|
104
|
+
|
|
105
|
+
def noisy(scale: float) -> np.ndarray:
|
|
106
|
+
return rng.normal(0, scale, n_samples)
|
|
107
|
+
|
|
108
|
+
mean_balance = np.exp(9.0 - 0.6 * risk + noisy(0.5))
|
|
109
|
+
inflow = np.exp(10.0 - 0.4 * risk + noisy(0.6))
|
|
110
|
+
outflow = inflow * np.exp(0.15 * risk + noisy(0.2))
|
|
111
|
+
low_days_3m = rng.poisson(np.exp(1.0 + 0.7 * risk).clip(max=60))
|
|
112
|
+
low_days_1m = np.minimum(
|
|
113
|
+
np.round(low_days_3m * rng.uniform(0.25, 0.45, n_samples)), 31
|
|
114
|
+
).astype(int)
|
|
115
|
+
declined = rng.poisson(np.exp(-0.5 + 0.6 * risk + noisy(0.3)))
|
|
116
|
+
# Drift grows with the month index: about x1.75 after 24 months, x1.3 after 12.
|
|
117
|
+
inflation = 1.03 ** (month / 12 * 10)
|
|
118
|
+
|
|
119
|
+
df = pd.DataFrame(
|
|
120
|
+
{
|
|
121
|
+
"entity_id": pd.Series(entity).map(lambda i: f"E{i:06d}"),
|
|
122
|
+
"observation_date": dates[month],
|
|
123
|
+
"mean_balance_3m": mean_balance.round(2),
|
|
124
|
+
"min_balance_3m": (mean_balance * np.exp(-0.8 + noisy(0.15))).round(2),
|
|
125
|
+
"max_balance_3m": np.exp(9.5 + 0.25 * risk + noisy(0.5)).round(2),
|
|
126
|
+
"inflow_sum_3m": inflow.round(2),
|
|
127
|
+
"outflow_sum_3m": outflow.round(2),
|
|
128
|
+
"inflow_outflow_ratio_3m": (inflow / outflow).round(4),
|
|
129
|
+
"declined_payments_cnt_3m": declined,
|
|
130
|
+
"low_balance_days_cnt_3m": low_days_3m,
|
|
131
|
+
"low_balance_days_cnt_1m": low_days_1m,
|
|
132
|
+
"card_spend_sum_1m": (np.exp(7.0 + noisy(0.7)) * inflation).round(2),
|
|
133
|
+
"account_age_months": (
|
|
134
|
+
age_months[entity] + month - 0.15 * risk * 10 + noisy(2)
|
|
135
|
+
)
|
|
136
|
+
.clip(0)
|
|
137
|
+
.round(1),
|
|
138
|
+
"n_users": rng.poisson(1.5, n_samples) + 1,
|
|
139
|
+
"has_rare_flag": (rng.random(n_samples) < 0.003).astype(int),
|
|
140
|
+
"reporting_version": np.ones(n_samples, dtype=int),
|
|
141
|
+
**{f"noise_{i}": noisy(1).round(4) for i in range(5)},
|
|
142
|
+
"sector": sector[entity],
|
|
143
|
+
"legal_form": legal_form[entity],
|
|
144
|
+
"country": country[entity],
|
|
145
|
+
"segment": np.where(rng.random(n_samples) < 0.97, "standard", "pilot"),
|
|
146
|
+
}
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
logit = (
|
|
150
|
+
1.1 * risk
|
|
151
|
+
+ pd.Series(sector[entity]).map(_SECTORS).to_numpy()
|
|
152
|
+
+ pd.Series(legal_form[entity]).map(_LEGAL_FORMS).to_numpy()
|
|
153
|
+
+ noisy(0.3)
|
|
154
|
+
)
|
|
155
|
+
df[TARGET_COL] = (
|
|
156
|
+
rng.random(n_samples) < _sigmoid(logit + _intercept(logit, event_rate))
|
|
157
|
+
).astype(int)
|
|
158
|
+
|
|
159
|
+
for col in ["sector", "legal_form", "country", "segment"]:
|
|
160
|
+
df[col] = df[col].astype("category")
|
|
161
|
+
|
|
162
|
+
return df.sort_values(["observation_date", "entity_id"]).reset_index(drop=True)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _sigmoid(x: np.ndarray) -> np.ndarray:
|
|
166
|
+
return 1 / (1 + np.exp(-x))
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _intercept(logit: np.ndarray, event_rate: float) -> float:
|
|
170
|
+
"""Shift that makes the mean predicted probability equal ``event_rate``."""
|
|
171
|
+
low, high = -30.0, 30.0
|
|
172
|
+
for _ in range(60):
|
|
173
|
+
mid = (low + high) / 2
|
|
174
|
+
if _sigmoid(logit + mid).mean() < event_rate:
|
|
175
|
+
low = mid
|
|
176
|
+
else:
|
|
177
|
+
high = mid
|
|
178
|
+
return (low + high) / 2
|
|
@@ -40,8 +40,8 @@ _OUTPUT_COLUMNS = ("y_true", "y_pred_proba", "y_pred")
|
|
|
40
40
|
class MisclassificationAnalyzer:
|
|
41
41
|
"""Filter confusion-matrix quadrants and explain individual predictions.
|
|
42
42
|
|
|
43
|
-
|
|
44
|
-
|
|
43
|
+
The quadrants are false negatives, false positives, true positives and
|
|
44
|
+
true negatives.
|
|
45
45
|
|
|
46
46
|
For a multiclass target, the confusion-matrix-quadrant concept only
|
|
47
47
|
makes sense once collapsed to a single yes/no question — so this
|
|
@@ -158,7 +158,7 @@ class MisclassificationAnalyzer:
|
|
|
158
158
|
For a multiclass model, ``y_true``/``y_pred``/``y_pred_proba`` are
|
|
159
159
|
the pos_label-vs-rest collapse described in the class docstring.
|
|
160
160
|
|
|
161
|
-
``extra_columns`` (e.g. identifiers such as ``
|
|
161
|
+
``extra_columns`` (e.g. identifiers such as ``entity_id``) are attached
|
|
162
162
|
for traceability and never passed to the model. Rows are matched to
|
|
163
163
|
``X`` by index; names may not clash with model features or the
|
|
164
164
|
output columns.
|
{classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/explainability/plots.py
RENAMED
|
@@ -99,11 +99,7 @@ def plot_shap_dependence_grid(
|
|
|
99
99
|
is_categorical: bool = False,
|
|
100
100
|
show: bool = False,
|
|
101
101
|
) -> Optional[plt.Figure]:
|
|
102
|
-
"""Plot SHAP dependence plots for a list of features in a grid layout.
|
|
103
|
-
|
|
104
|
-
Ported from the paylater production workflow to provide the same
|
|
105
|
-
grid-layout dependence analysis inside classifier_toolkit.
|
|
106
|
-
"""
|
|
102
|
+
"""Plot SHAP dependence plots for a list of features in a grid layout."""
|
|
107
103
|
if isinstance(feature_list, str):
|
|
108
104
|
feature_list = [feature_list]
|
|
109
105
|
feature_list = list(feature_list)
|
|
@@ -7,7 +7,7 @@ a business prior. Each prior is expressed as a *direction rule*:
|
|
|
7
7
|
|
|
8
8
|
{
|
|
9
9
|
"name": "Balance features", # human-readable label
|
|
10
|
-
"patterns": ["
|
|
10
|
+
"patterns": ["balance"], # substrings matched against feature names
|
|
11
11
|
"expected_direction": "negative", # "positive" or "negative"
|
|
12
12
|
"exclude_patterns": ["pct", "ratio"], # optional — substrings that veto a match
|
|
13
13
|
}
|
{classifier_toolkit-0.3.6 → classifier_toolkit-0.4.0}/classifier_toolkit/feature_reduction/drift.py
RENAMED
|
@@ -36,6 +36,7 @@ Consider this a sensible starting point, not a formal statistical test.
|
|
|
36
36
|
"""
|
|
37
37
|
|
|
38
38
|
import logging
|
|
39
|
+
import warnings
|
|
39
40
|
from typing import Dict, List, Optional, Tuple
|
|
40
41
|
|
|
41
42
|
import numpy as np
|
|
@@ -160,24 +161,17 @@ def _js_numerical(
|
|
|
160
161
|
baseline: pd.Series,
|
|
161
162
|
current: pd.Series,
|
|
162
163
|
bins: int,
|
|
163
|
-
bin_method: str,
|
|
164
|
+
bin_method: str,
|
|
164
165
|
min_obs_per_bin: int,
|
|
165
166
|
auto_adjust_bins: bool,
|
|
166
167
|
) -> float:
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
try:
|
|
173
|
-
n_bins = _effective_bins(baseline, bins, min_obs_per_bin, auto_adjust_bins)
|
|
174
|
-
b_hist, _ = np.histogram(baseline, bins=n_bins, density=True)
|
|
175
|
-
c_hist, _ = np.histogram(current, bins=n_bins, density=True)
|
|
176
|
-
b_hist = b_hist / (b_hist.sum() + 1e-10)
|
|
177
|
-
c_hist = c_hist / (c_hist.sum() + 1e-10)
|
|
178
|
-
return float(jensenshannon(b_hist, c_hist))
|
|
179
|
-
except Exception:
|
|
168
|
+
"""Jensen-Shannon distance via binned histograms. Drift if >= threshold."""
|
|
169
|
+
p_b, p_c = _binned_probs(
|
|
170
|
+
baseline, current, bins, bin_method, min_obs_per_bin, auto_adjust_bins
|
|
171
|
+
)
|
|
172
|
+
if p_b is None:
|
|
180
173
|
return np.nan
|
|
174
|
+
return float(jensenshannon(p_b, p_c))
|
|
181
175
|
|
|
182
176
|
|
|
183
177
|
def _ks_numerical(baseline: pd.Series, current: pd.Series) -> Tuple[float, float]:
|
|
@@ -267,13 +261,30 @@ def _hellinger_categorical(baseline: pd.Series, current: pd.Series) -> float:
|
|
|
267
261
|
return np.nan
|
|
268
262
|
|
|
269
263
|
|
|
270
|
-
def _ad_numerical(
|
|
264
|
+
def _ad_numerical(baseline: pd.Series, current: pd.Series) -> float:
|
|
271
265
|
"""Anderson-Darling 2-sample p-value (numerical only). Drift if p_value < threshold.
|
|
272
266
|
|
|
273
|
-
|
|
274
|
-
|
|
267
|
+
SciPy interpolates the p-value from tabulated critical values, so it is
|
|
268
|
+
capped to ``[0.001, 0.25]``.
|
|
275
269
|
"""
|
|
276
|
-
|
|
270
|
+
baseline = baseline.dropna()
|
|
271
|
+
current = current.dropna()
|
|
272
|
+
# SciPy raises IndexError (not ValueError) when each sample has a single
|
|
273
|
+
# value, so samples that small are reported as undefined up front.
|
|
274
|
+
if len(baseline) < 2 or len(current) < 2:
|
|
275
|
+
return np.nan
|
|
276
|
+
try:
|
|
277
|
+
with warnings.catch_warnings():
|
|
278
|
+
# SciPy warns whenever the p-value hits the cap or the floor.
|
|
279
|
+
warnings.simplefilter("ignore")
|
|
280
|
+
result = scipy_stats.anderson_ksamp([baseline.values, current.values])
|
|
281
|
+
return float(result.pvalue)
|
|
282
|
+
except ValueError as exc:
|
|
283
|
+
# Raised for degenerate samples, e.g. no distinct observations. Any
|
|
284
|
+
# other exception propagates: a blanket `except Exception` here is
|
|
285
|
+
# what used to hide an always-NaN bug.
|
|
286
|
+
logger.debug("Anderson-Darling drift undefined: %s", exc)
|
|
287
|
+
return np.nan
|
|
277
288
|
|
|
278
289
|
|
|
279
290
|
def _bhattacharyya_numerical(
|
|
@@ -1,8 +1,7 @@
|
|
|
1
1
|
"""FeatureReducer — orchestrates all six filter steps in sequence.
|
|
2
2
|
|
|
3
|
-
The six steps
|
|
4
|
-
|
|
5
|
-
``fit`` / ``transform`` API instead of mutating datasets in place.
|
|
3
|
+
The six steps run in a fixed order behind a standard sklearn-compatible
|
|
4
|
+
``fit`` / ``transform`` API, without mutating the input datasets.
|
|
6
5
|
|
|
7
6
|
Steps:
|
|
8
7
|
|
|
@@ -93,7 +92,8 @@ class FeatureReducer(BaseFeatureReducer):
|
|
|
93
92
|
Tie-break metric for categorical correlation (step 6): ``"prauc"``
|
|
94
93
|
(default), ``"auc"``, or ``"target_assoc"``.
|
|
95
94
|
numerical_features : List[str], optional
|
|
96
|
-
Override auto-detected numerical columns (steps 2, 3, 4, 6).
|
|
95
|
+
Override auto-detected numerical columns (steps 2, 3, 4, 6). Step 4
|
|
96
|
+
scores numerical columns only; the other columns are kept by it.
|
|
97
97
|
categorical_features : List[str], optional
|
|
98
98
|
Override auto-detected categorical columns (steps 2, 3, 6).
|
|
99
99
|
apply_expert_rules : bool, optional
|
|
@@ -304,8 +304,18 @@ class FeatureReducer(BaseFeatureReducer):
|
|
|
304
304
|
stacklevel=2,
|
|
305
305
|
)
|
|
306
306
|
else:
|
|
307
|
+
num_features = _num(X_current)
|
|
308
|
+
if num_features is None:
|
|
309
|
+
# Score numeric columns only: left to auto-detect, the
|
|
310
|
+
# filter would also drop every non-numeric column unscored.
|
|
311
|
+
cat_features = set(_cat(X_current) or [])
|
|
312
|
+
num_features = [
|
|
313
|
+
c
|
|
314
|
+
for c in X_current.select_dtypes(include="number").columns
|
|
315
|
+
if c not in cat_features
|
|
316
|
+
]
|
|
307
317
|
f = WeakPredictiveFilter(
|
|
308
|
-
numerical_features=
|
|
318
|
+
numerical_features=num_features,
|
|
309
319
|
gini_threshold=self.gini_threshold,
|
|
310
320
|
prauc_lift_threshold=self.prauc_lift_threshold,
|
|
311
321
|
)
|
|
@@ -362,6 +372,8 @@ class FeatureReducer(BaseFeatureReducer):
|
|
|
362
372
|
self.removed_by_step_["low_variance_categorical"] = list(low_var_cat)
|
|
363
373
|
self.steps_["correlation"] = f
|
|
364
374
|
X_current = f.transform(X_current)
|
|
375
|
+
# The correlation filter only reports these (it keeps them).
|
|
376
|
+
X_current = X_current.drop(columns=low_var_cat)
|
|
365
377
|
self._log_step("correlation", f.dropped_features_)
|
|
366
378
|
if low_var_cat:
|
|
367
379
|
self._log_step("low_variance_categorical", low_var_cat)
|
|
@@ -132,7 +132,10 @@ def macro_average_precision_score(
|
|
|
132
132
|
y_pred_proba = np.asarray(y_pred_proba)
|
|
133
133
|
if y_pred_proba.ndim == 1 or y_pred_proba.shape[1] <= 2:
|
|
134
134
|
proba = _positive_column(y_pred_proba, classes)
|
|
135
|
-
|
|
135
|
+
# The positive class is the greater label (as for ROC-AUC); sklearn's
|
|
136
|
+
# default pos_label=1 would pick the *lesser* label for {1, 2} targets.
|
|
137
|
+
positive = np.unique(classes if classes is not None else np.asarray(y_true))[-1]
|
|
138
|
+
return float(average_precision_score(y_true, proba, pos_label=positive))
|
|
136
139
|
|
|
137
140
|
if classes is not None:
|
|
138
141
|
scores = [
|
|
@@ -233,11 +236,11 @@ def make_d_class_auc_scorer(d_class_label: Union[str, int]) -> Callable:
|
|
|
233
236
|
|
|
234
237
|
.. deprecated:: 0.3.6
|
|
235
238
|
Use :func:`make_pos_label_auc_scorer` instead; this alias will be
|
|
236
|
-
removed in
|
|
239
|
+
removed in 0.5.0.
|
|
237
240
|
"""
|
|
238
241
|
warnings.warn(
|
|
239
242
|
"make_d_class_auc_scorer is deprecated; use make_pos_label_auc_scorer "
|
|
240
|
-
"instead. It will be removed in
|
|
243
|
+
"instead. It will be removed in 0.5.0.",
|
|
241
244
|
DeprecationWarning,
|
|
242
245
|
stacklevel=2,
|
|
243
246
|
)
|
|
@@ -249,12 +252,11 @@ def make_d_class_pr_auc_scorer(d_class_label: Union[str, int]) -> Callable:
|
|
|
249
252
|
|
|
250
253
|
.. deprecated:: 0.3.6
|
|
251
254
|
Use :func:`make_pos_label_pr_auc_scorer` instead; this alias will be
|
|
252
|
-
removed in
|
|
255
|
+
removed in 0.5.0.
|
|
253
256
|
"""
|
|
254
257
|
warnings.warn(
|
|
255
258
|
"make_d_class_pr_auc_scorer is deprecated; use "
|
|
256
|
-
"make_pos_label_pr_auc_scorer instead. It will be removed in
|
|
257
|
-
"minor release.",
|
|
259
|
+
"make_pos_label_pr_auc_scorer instead. It will be removed in 0.5.0.",
|
|
258
260
|
DeprecationWarning,
|
|
259
261
|
stacklevel=2,
|
|
260
262
|
)
|