ogboost 0.5.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ogboost-0.5.5/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2024 asmahani
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
ogboost-0.5.5/PKG-INFO ADDED
@@ -0,0 +1,180 @@
1
+ Metadata-Version: 2.2
2
+ Name: ogboost
3
+ Version: 0.5.5
4
+ Summary: Ordinal Gradient Boosting
5
+ Author-email: "Alireza S. Mahani, Mansour T.A. Sharabiani" <alireza.s.mahani@gmail.com>
6
+ License: MIT License
7
+
8
+ Copyright (c) 2024 asmahani
9
+
10
+ Permission is hereby granted, free of charge, to any person obtaining a copy
11
+ of this software and associated documentation files (the "Software"), to deal
12
+ in the Software without restriction, including without limitation the rights
13
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
+ copies of the Software, and to permit persons to whom the Software is
15
+ furnished to do so, subject to the following conditions:
16
+
17
+ The above copyright notice and this permission notice shall be included in all
18
+ copies or substantial portions of the Software.
19
+
20
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
+ SOFTWARE.
27
+
28
+ Keywords: ordinal regression,gradient boosting,machine learning,scikit-learn
29
+ Description-Content-Type: text/markdown
30
+ License-File: LICENSE
31
+ Requires-Dist: numpy
32
+ Requires-Dist: pandas
33
+ Requires-Dist: scikit-learn
34
+ Requires-Dist: scipy
35
+ Requires-Dist: pydantic
36
+ Requires-Dist: pytest
37
+ Requires-Dist: matplotlib
38
+
39
+ # Ordinal Gradient Boosting (`OGBoost`)
40
+
41
+ ## Overview
42
+
43
+ `OGBoost` is a scikit-learn-compatible, Python package for gradient boosting tailored to ordinal regression problems. It does so by alternating between:
44
+ 1. Fitting a Machine Learning (ML) regression model - such as a decision tree - to predict a latent score that specifies the mean of a probability density function (PDF), and
45
+ 1. Fitting a set of thresholds that generate discrete outcomes from the PDF.
46
+
47
+ In other words, `OGBoost` implements coordinate-descent optimization that combines functional gradient descent - for updating the regression function - with ordinary gradient descent - for updating the threshold vector.
48
+
49
+ The main class of the package, `GradientBoostingOrdinal`, is designed to have the same look and feel as `scikit-learn`'s `GradientBoostingClassifier`. It includes many of the same features such as custom link functions, sample weighting, early stopping using a validation set, and staged predictions.
50
+
51
+ There are, however, important differences as well.
52
+
53
+ ## Unique Features of `OGBoost`
54
+
55
+ ### Latent-Score Prediction
56
+
57
+ The `decision_function` method of the `GradientBoostingOrdinal` behaves differently from `scikit-learn`'s classifiers. Assuming the target variable has `K` distinct classes, a nominal classifier's decision function would return `K` values for each sample. On the other hand, `decision_function` in `ogboost` would return the latent score for each sample, which is a single value. This latent score can be considered a high-resolution alternative to class labels, and thus may have superior ranking performance.
58
+
59
+ ### Early Stopping using Cross-Validation (CV)
60
+
61
+ In addition to using a single validation set for early stopping, similar to `GradientBoostingClassifier`, `ogboost` early stopping using CV, which means error/performance over the entire data is used for calculating out-of-sample performance. This can improve the robustness of the early-stopping strategy, especially for small and/or imbalanced datasets.
62
+
63
+ ### Heterogeneous Ensemble
64
+
65
+ While most gradient-boosting software packages exclusively use decision trees with a predetermined set of hyperparameters as the base learner in all boosting iterations, `ogboost` offers significantly more flexibility.
66
+
67
+ 1. Users can pass in a `base_learner` parameter to the class initializer to override the default choice of a `DecisionTreeRegressor`. This can be any regression algorithm such as a feed-forward neural network (`MLPRegressor`), or a K-nearest-neighbor regressor (`KNeighborsRegressor`), etc.
68
+ 1. Rather than a single base learner, users can specify a list of base learners, which will be drawn from in that order in each boosting iteration. This amounts to creating a *heterogeneous* ensemble as opposed to a *homogeneous* ensemble.
69
+
70
+ ## Installation
71
+ ```bash
72
+ pip install ogboost
73
+ ```
74
+
75
+ ## Quick Start
76
+ ### Load the Wine Quality Dataset
77
+ The package includes a utility to load the wine quality dataset (red and white) from the UCI repository. Note that `load_wine_quality` shifts the target variable (`quality`) to start from `0`. (This is required by the `GradientBoostingOrdinal` class.)
78
+
79
+ ```python
80
+ from ogboost import load_wine_quality
81
+ red_wine, white_wine = load_wine_quality()
82
+ X, y = red_wine.drop(columns="quality"), red_wine["quality"]
83
+ ```
84
+
85
+ ### Training, Prediction and Evaluation
86
+ ```python
87
+ from ogboost import GradientBoostingOrdinal
88
+
89
+ ## training ##
90
+ model = GradientBoostingOrdinal(n_estimators=100, link_function='logit', verbose=1)
91
+ model.fit(X, y)
92
+
93
+ ## prediction ##
94
+ # class labels
95
+ predicted_labels = model.predict(X)
96
+ # class probabilities
97
+ predicted_probabilities = model.predict_proba(X)
98
+ # latent score
99
+ predicted_latent = model.decision_function(X)
100
+
101
+ # evaluation
102
+ concordance_latent = model.score(X, y) # concordance using latent scores
103
+ concordance_label = model.score(X, y, pred_type = 'labels') # concordance using class labels
104
+ print(f"Concordance - class labels: {concordance_label:.3f}")
105
+ print(f"Concordance - latent scores: {concordance_latent:.3f}")
106
+ ```
107
+
108
+ ### Early-Stopping using Cross-Validation
109
+
110
+ ```python
111
+ from sklearn.model_selection import cross_val_score
112
+ from sklearn.model_selection import RepeatedKFold
113
+ import time
114
+
115
+ n_splits = 10
116
+ n_repeats = 10
117
+ kf = RepeatedKFold(n_splits=n_splits, n_repeats=n_repeats)
118
+
119
+ # early-stopping using a simple holdout set
120
+ model_earlystop_simple = GradientBoostingOrdinal(n_iter_no_change=10, validation_fraction=0.2)
121
+ start = time.time()
122
+ c_index_simple = cross_val_score(model_earlystop_simple, X, y, cv=kf, n_jobs=-1)
123
+ end = time.time()
124
+ print(f'Simple early stopping: {c_index_simple.mean():.3f} ({end - start:.1f} seconds)')
125
+
126
+ # early-stopping using cross-validation
127
+ model_earlystop_cv = GradientBoostingOrdinal(n_iter_no_change=10, cv_early_stopping_splits=5)
128
+ start = time.time()
129
+ c_index_cv = cross_val_score(model_earlystop_cv, X, y, cv=kf, n_jobs=-1)
130
+ end = time.time()
131
+ print(f'CV early stopping: {c_index_cv.mean():.3f} ({end - start:.1f} seconds)')
132
+
133
+ # statistical comparison of the two methods
134
+ from scipy.stats import ttest_rel
135
+ t_stat, p_value = ttest_rel(c_index_simple, c_index_cv)
136
+ print(f't-statistic: {t_stat}, p-value: {p_value}')
137
+ ```
138
+
139
+ ### Heterogeneous Ensemble
140
+
141
+ Generating a random set of hyperparameter combinations for `DecisionTreeRegressor`:
142
+
143
+ ```python
144
+ import numpy as np
145
+ from sklearn.tree import DecisionTreeRegressor
146
+
147
+ # Number of samples to generate
148
+ n_samples = 10
149
+
150
+ max_depth_choices = [3, 6, 9, None]
151
+ max_depths = np.random.choice(max_depth_choices, size=n_samples, replace=True)
152
+ max_leaf_nodes_choices = [10, 20, 30, None]
153
+ max_leaf_nodes = np.random.choice(max_leaf_nodes_choices, size=n_samples, replace=True)
154
+
155
+ params = list(zip(max_depths, max_leaf_nodes))
156
+
157
+ # Create list of DecisionTreeRegressor models with sampled parameters
158
+ models = [
159
+ DecisionTreeRegressor(
160
+ max_depth=max_depth,
161
+ max_leaf_nodes=max_leaf_nodes
162
+ )
163
+ for max_depth, max_leaf_nodes in params
164
+ ]
165
+ ```
166
+
167
+ Quantifying the performance of the heterogeneous ensemble:
168
+ ```python
169
+ learning_rate = 0.1
170
+
171
+ model_heter = GradientBoostingOrdinal(
172
+ base_learner=models,
173
+ n_estimators=n_samples
174
+ )
175
+ cv_heter = cross_val_score(model_heter, X, y, cv=kf, n_jobs=-1)
176
+ print(f'average cv score of heteogeneous ensemble: {np.mean(cv_heter):.3f}')
177
+ ```
178
+
179
+ ## License
180
+ This package is licensed under the [MIT License](./LICENSE).
@@ -0,0 +1,142 @@
1
+ # Ordinal Gradient Boosting (`OGBoost`)
2
+
3
+ ## Overview
4
+
5
+ `OGBoost` is a scikit-learn-compatible, Python package for gradient boosting tailored to ordinal regression problems. It does so by alternating between:
6
+ 1. Fitting a Machine Learning (ML) regression model - such as a decision tree - to predict a latent score that specifies the mean of a probability density function (PDF), and
7
+ 1. Fitting a set of thresholds that generate discrete outcomes from the PDF.
8
+
9
+ In other words, `OGBoost` implements coordinate-descent optimization that combines functional gradient descent - for updating the regression function - with ordinary gradient descent - for updating the threshold vector.
10
+
11
+ The main class of the package, `GradientBoostingOrdinal`, is designed to have the same look and feel as `scikit-learn`'s `GradientBoostingClassifier`. It includes many of the same features such as custom link functions, sample weighting, early stopping using a validation set, and staged predictions.
12
+
13
+ There are, however, important differences as well.
14
+
15
+ ## Unique Features of `OGBoost`
16
+
17
+ ### Latent-Score Prediction
18
+
19
+ The `decision_function` method of the `GradientBoostingOrdinal` behaves differently from `scikit-learn`'s classifiers. Assuming the target variable has `K` distinct classes, a nominal classifier's decision function would return `K` values for each sample. On the other hand, `decision_function` in `ogboost` would return the latent score for each sample, which is a single value. This latent score can be considered a high-resolution alternative to class labels, and thus may have superior ranking performance.
20
+
21
+ ### Early Stopping using Cross-Validation (CV)
22
+
23
+ In addition to using a single validation set for early stopping, similar to `GradientBoostingClassifier`, `ogboost` early stopping using CV, which means error/performance over the entire data is used for calculating out-of-sample performance. This can improve the robustness of the early-stopping strategy, especially for small and/or imbalanced datasets.
24
+
25
+ ### Heterogeneous Ensemble
26
+
27
+ While most gradient-boosting software packages exclusively use decision trees with a predetermined set of hyperparameters as the base learner in all boosting iterations, `ogboost` offers significantly more flexibility.
28
+
29
+ 1. Users can pass in a `base_learner` parameter to the class initializer to override the default choice of a `DecisionTreeRegressor`. This can be any regression algorithm such as a feed-forward neural network (`MLPRegressor`), or a K-nearest-neighbor regressor (`KNeighborsRegressor`), etc.
30
+ 1. Rather than a single base learner, users can specify a list of base learners, which will be drawn from in that order in each boosting iteration. This amounts to creating a *heterogeneous* ensemble as opposed to a *homogeneous* ensemble.
31
+
32
+ ## Installation
33
+ ```bash
34
+ pip install ogboost
35
+ ```
36
+
37
+ ## Quick Start
38
+ ### Load the Wine Quality Dataset
39
+ The package includes a utility to load the wine quality dataset (red and white) from the UCI repository. Note that `load_wine_quality` shifts the target variable (`quality`) to start from `0`. (This is required by the `GradientBoostingOrdinal` class.)
40
+
41
+ ```python
42
+ from ogboost import load_wine_quality
43
+ red_wine, white_wine = load_wine_quality()
44
+ X, y = red_wine.drop(columns="quality"), red_wine["quality"]
45
+ ```
46
+
47
+ ### Training, Prediction and Evaluation
48
+ ```python
49
+ from ogboost import GradientBoostingOrdinal
50
+
51
+ ## training ##
52
+ model = GradientBoostingOrdinal(n_estimators=100, link_function='logit', verbose=1)
53
+ model.fit(X, y)
54
+
55
+ ## prediction ##
56
+ # class labels
57
+ predicted_labels = model.predict(X)
58
+ # class probabilities
59
+ predicted_probabilities = model.predict_proba(X)
60
+ # latent score
61
+ predicted_latent = model.decision_function(X)
62
+
63
+ # evaluation
64
+ concordance_latent = model.score(X, y) # concordance using latent scores
65
+ concordance_label = model.score(X, y, pred_type = 'labels') # concordance using class labels
66
+ print(f"Concordance - class labels: {concordance_label:.3f}")
67
+ print(f"Concordance - latent scores: {concordance_latent:.3f}")
68
+ ```
69
+
70
+ ### Early-Stopping using Cross-Validation
71
+
72
+ ```python
73
+ from sklearn.model_selection import cross_val_score
74
+ from sklearn.model_selection import RepeatedKFold
75
+ import time
76
+
77
+ n_splits = 10
78
+ n_repeats = 10
79
+ kf = RepeatedKFold(n_splits=n_splits, n_repeats=n_repeats)
80
+
81
+ # early-stopping using a simple holdout set
82
+ model_earlystop_simple = GradientBoostingOrdinal(n_iter_no_change=10, validation_fraction=0.2)
83
+ start = time.time()
84
+ c_index_simple = cross_val_score(model_earlystop_simple, X, y, cv=kf, n_jobs=-1)
85
+ end = time.time()
86
+ print(f'Simple early stopping: {c_index_simple.mean():.3f} ({end - start:.1f} seconds)')
87
+
88
+ # early-stopping using cross-validation
89
+ model_earlystop_cv = GradientBoostingOrdinal(n_iter_no_change=10, cv_early_stopping_splits=5)
90
+ start = time.time()
91
+ c_index_cv = cross_val_score(model_earlystop_cv, X, y, cv=kf, n_jobs=-1)
92
+ end = time.time()
93
+ print(f'CV early stopping: {c_index_cv.mean():.3f} ({end - start:.1f} seconds)')
94
+
95
+ # statistical comparison of the two methods
96
+ from scipy.stats import ttest_rel
97
+ t_stat, p_value = ttest_rel(c_index_simple, c_index_cv)
98
+ print(f't-statistic: {t_stat}, p-value: {p_value}')
99
+ ```
100
+
101
+ ### Heterogeneous Ensemble
102
+
103
+ Generating a random set of hyperparameter combinations for `DecisionTreeRegressor`:
104
+
105
+ ```python
106
+ import numpy as np
107
+ from sklearn.tree import DecisionTreeRegressor
108
+
109
+ # Number of samples to generate
110
+ n_samples = 10
111
+
112
+ max_depth_choices = [3, 6, 9, None]
113
+ max_depths = np.random.choice(max_depth_choices, size=n_samples, replace=True)
114
+ max_leaf_nodes_choices = [10, 20, 30, None]
115
+ max_leaf_nodes = np.random.choice(max_leaf_nodes_choices, size=n_samples, replace=True)
116
+
117
+ params = list(zip(max_depths, max_leaf_nodes))
118
+
119
+ # Create list of DecisionTreeRegressor models with sampled parameters
120
+ models = [
121
+ DecisionTreeRegressor(
122
+ max_depth=max_depth,
123
+ max_leaf_nodes=max_leaf_nodes
124
+ )
125
+ for max_depth, max_leaf_nodes in params
126
+ ]
127
+ ```
128
+
129
+ Quantifying the performance of the heterogeneous ensemble:
130
+ ```python
131
+ learning_rate = 0.1
132
+
133
+ model_heter = GradientBoostingOrdinal(
134
+ base_learner=models,
135
+ n_estimators=n_samples
136
+ )
137
+ cv_heter = cross_val_score(model_heter, X, y, cv=kf, n_jobs=-1)
138
+ print(f'average cv score of heteogeneous ensemble: {np.mean(cv_heter):.3f}')
139
+ ```
140
+
141
+ ## License
142
+ This package is licensed under the [MIT License](./LICENSE).
@@ -0,0 +1,8 @@
1
+ # ogboost/__init__.py
2
+
3
+ # Re-export key components from submodules
4
+ from .data import load_wine_quality
5
+ from .main import GradientBoostingOrdinal, concordance_index, LinkFunctions
6
+
7
+ # Define the public API
8
+ __all__ = ["load_wine_quality", "GradientBoostingOrdinal", "concordance_index", "LinkFunctions"]
@@ -0,0 +1,103 @@
1
+ """
2
+ Module for loading the Wine Quality Dataset.
3
+
4
+ This module provides functionality to download and load the Wine Quality Dataset from the
5
+ UCI Machine Learning Repository. The dataset includes information about physicochemical
6
+ tests (e.g., pH, alcohol content) and quality scores for red and white wines.
7
+
8
+ The module contains the following functionality:
9
+ - Automatically downloads the datasets if not found locally.
10
+ - Loads the red and white wine datasets into pandas DataFrames.
11
+ - Rescales the 'quality' column to start from 0 and ensures it is of integer type.
12
+
13
+ Functions
14
+ ---------
15
+ load_wine_quality():
16
+ Loads the wine quality dataset. Automatically downloads the datasets
17
+ if they are not found locally and returns them as pandas DataFrames.
18
+
19
+ _download_wine_quality_datasets(destination: str):
20
+ Downloads the wine quality datasets (red and white) from the UCI repository
21
+ to the specified destination folder.
22
+
23
+ Examples
24
+ --------
25
+ >>> from mymodule import load_wine_quality
26
+ >>> red_wine, white_wine = load_wine_quality()
27
+ >>> print(red_wine.head())
28
+ >>> print(white_wine.head())
29
+ """
30
+ import os
31
+ import urllib.request
32
+ import pandas as pd
33
+
34
+ def load_wine_quality():
35
+ """
36
+ Loads the wine quality dataset from the UCI repository.
37
+ If the datasets are not already downloaded,
38
+ they will be downloaded automatically.
39
+
40
+ Returns
41
+ -------
42
+ tuple of pandas.DataFrame
43
+ A tuple containing two DataFrames:
44
+ - The first for red wine data.
45
+ - The second for white wine data.
46
+ """
47
+ # Define the local paths for the datasets
48
+ dataset_folder = os.path.join(os.path.dirname(__file__), "data")
49
+ red_wine_path = os.path.join(dataset_folder, "winequality-red.csv")
50
+ white_wine_path = os.path.join(dataset_folder, "winequality-white.csv")
51
+
52
+ # Check if the datasets exist locally; if not, download them
53
+ if not os.path.exists(red_wine_path) or not os.path.exists(white_wine_path):
54
+ print("Datasets not found locally. Downloading from UCI repository...")
55
+ os.makedirs(dataset_folder, exist_ok=True)
56
+ _download_wine_quality_datasets(dataset_folder)
57
+
58
+ # Load the datasets into pandas DataFrames
59
+ red_wine_df = pd.read_csv(red_wine_path, sep=";")
60
+ white_wine_df = pd.read_csv(white_wine_path, sep=";")
61
+
62
+ # rescale quality to start from 0
63
+ red_wine_df["quality"] -= red_wine_df["quality"].min()
64
+ white_wine_df["quality"] -= white_wine_df["quality"].min()
65
+
66
+ # ensure data type for response is integer
67
+ red_wine_df["quality"] = red_wine_df["quality"].astype(int)
68
+ white_wine_df["quality"] = white_wine_df["quality"].astype(int)
69
+
70
+ return red_wine_df, white_wine_df
71
+
72
+
73
+ def _download_wine_quality_datasets(destination: str):
74
+ """
75
+ Downloads the wine quality datasets from the UCI repository.
76
+
77
+ Parameters
78
+ ----------
79
+ destination : str
80
+ Directory where the datasets will be saved.
81
+ """
82
+ # UCI repository URLs
83
+ red_wine_url = (
84
+ "https://archive.ics.uci.edu/ml/machine-learning-databases/"
85
+ "wine-quality/winequality-red.csv"
86
+ )
87
+ white_wine_url = (
88
+ "https://archive.ics.uci.edu/ml/machine-learning-databases/"
89
+ "wine-quality/winequality-white.csv"
90
+ )
91
+
92
+ # Define file paths
93
+ red_wine_path = os.path.join(destination, "winequality-red.csv")
94
+ white_wine_path = os.path.join(destination, "winequality-white.csv")
95
+
96
+ # Download the datasets
97
+ print(f"Downloading red wine dataset to {red_wine_path}...")
98
+ urllib.request.urlretrieve(red_wine_url, red_wine_path)
99
+ print("Red wine dataset downloaded.")
100
+
101
+ print(f"Downloading white wine dataset to {white_wine_path}...")
102
+ urllib.request.urlretrieve(white_wine_url, white_wine_path)
103
+ print("White wine dataset downloaded.")