ogboost 0.5.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ogboost-0.5.5/LICENSE +21 -0
- ogboost-0.5.5/PKG-INFO +180 -0
- ogboost-0.5.5/README.md +142 -0
- ogboost-0.5.5/ogboost/__init__.py +8 -0
- ogboost-0.5.5/ogboost/data.py +103 -0
- ogboost-0.5.5/ogboost/main.py +1125 -0
- ogboost-0.5.5/ogboost/tests/test_data.py +36 -0
- ogboost-0.5.5/ogboost/tests/test_main.py +62 -0
- ogboost-0.5.5/ogboost.egg-info/PKG-INFO +180 -0
- ogboost-0.5.5/ogboost.egg-info/SOURCES.txt +13 -0
- ogboost-0.5.5/ogboost.egg-info/dependency_links.txt +1 -0
- ogboost-0.5.5/ogboost.egg-info/requires.txt +7 -0
- ogboost-0.5.5/ogboost.egg-info/top_level.txt +1 -0
- ogboost-0.5.5/pyproject.toml +26 -0
- ogboost-0.5.5/setup.cfg +4 -0
ogboost-0.5.5/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 asmahani
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
ogboost-0.5.5/PKG-INFO
ADDED
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
Metadata-Version: 2.2
|
|
2
|
+
Name: ogboost
|
|
3
|
+
Version: 0.5.5
|
|
4
|
+
Summary: Ordinal Gradient Boosting
|
|
5
|
+
Author-email: "Alireza S. Mahani, Mansour T.A. Sharabiani" <alireza.s.mahani@gmail.com>
|
|
6
|
+
License: MIT License
|
|
7
|
+
|
|
8
|
+
Copyright (c) 2024 asmahani
|
|
9
|
+
|
|
10
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
11
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
12
|
+
in the Software without restriction, including without limitation the rights
|
|
13
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
14
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
15
|
+
furnished to do so, subject to the following conditions:
|
|
16
|
+
|
|
17
|
+
The above copyright notice and this permission notice shall be included in all
|
|
18
|
+
copies or substantial portions of the Software.
|
|
19
|
+
|
|
20
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
21
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
22
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
23
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
24
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
25
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
26
|
+
SOFTWARE.
|
|
27
|
+
|
|
28
|
+
Keywords: ordinal regression,gradient boosting,machine learning,scikit-learn
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
License-File: LICENSE
|
|
31
|
+
Requires-Dist: numpy
|
|
32
|
+
Requires-Dist: pandas
|
|
33
|
+
Requires-Dist: scikit-learn
|
|
34
|
+
Requires-Dist: scipy
|
|
35
|
+
Requires-Dist: pydantic
|
|
36
|
+
Requires-Dist: pytest
|
|
37
|
+
Requires-Dist: matplotlib
|
|
38
|
+
|
|
39
|
+
# Ordinal Gradient Boosting (`OGBoost`)
|
|
40
|
+
|
|
41
|
+
## Overview
|
|
42
|
+
|
|
43
|
+
`OGBoost` is a scikit-learn-compatible, Python package for gradient boosting tailored to ordinal regression problems. It does so by alternating between:
|
|
44
|
+
1. Fitting a Machine Learning (ML) regression model - such as a decision tree - to predict a latent score that specifies the mean of a probability density function (PDF), and
|
|
45
|
+
1. Fitting a set of thresholds that generate discrete outcomes from the PDF.
|
|
46
|
+
|
|
47
|
+
In other words, `OGBoost` implements coordinate-descent optimization that combines functional gradient descent - for updating the regression function - with ordinary gradient descent - for updating the threshold vector.
|
|
48
|
+
|
|
49
|
+
The main class of the package, `GradientBoostingOrdinal`, is designed to have the same look and feel as `scikit-learn`'s `GradientBoostingClassifier`. It includes many of the same features such as custom link functions, sample weighting, early stopping using a validation set, and staged predictions.
|
|
50
|
+
|
|
51
|
+
There are, however, important differences as well.
|
|
52
|
+
|
|
53
|
+
## Unique Features of `OGBoost`
|
|
54
|
+
|
|
55
|
+
### Latent-Score Prediction
|
|
56
|
+
|
|
57
|
+
The `decision_function` method of the `GradientBoostingOrdinal` behaves differently from `scikit-learn`'s classifiers. Assuming the target variable has `K` distinct classes, a nominal classifier's decision function would return `K` values for each sample. On the other hand, `decision_function` in `ogboost` would return the latent score for each sample, which is a single value. This latent score can be considered a high-resolution alternative to class labels, and thus may have superior ranking performance.
|
|
58
|
+
|
|
59
|
+
### Early Stopping using Cross-Validation (CV)
|
|
60
|
+
|
|
61
|
+
In addition to using a single validation set for early stopping, similar to `GradientBoostingClassifier`, `ogboost` early stopping using CV, which means error/performance over the entire data is used for calculating out-of-sample performance. This can improve the robustness of the early-stopping strategy, especially for small and/or imbalanced datasets.
|
|
62
|
+
|
|
63
|
+
### Heterogeneous Ensemble
|
|
64
|
+
|
|
65
|
+
While most gradient-boosting software packages exclusively use decision trees with a predetermined set of hyperparameters as the base learner in all boosting iterations, `ogboost` offers significantly more flexibility.
|
|
66
|
+
|
|
67
|
+
1. Users can pass in a `base_learner` parameter to the class initializer to override the default choice of a `DecisionTreeRegressor`. This can be any regression algorithm such as a feed-forward neural network (`MLPRegressor`), or a K-nearest-neighbor regressor (`KNeighborsRegressor`), etc.
|
|
68
|
+
1. Rather than a single base learner, users can specify a list of base learners, which will be drawn from in that order in each boosting iteration. This amounts to creating a *heterogeneous* ensemble as opposed to a *homogeneous* ensemble.
|
|
69
|
+
|
|
70
|
+
## Installation
|
|
71
|
+
```bash
|
|
72
|
+
pip install ogboost
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
## Quick Start
|
|
76
|
+
### Load the Wine Quality Dataset
|
|
77
|
+
The package includes a utility to load the wine quality dataset (red and white) from the UCI repository. Note that `load_wine_quality` shifts the target variable (`quality`) to start from `0`. (This is required by the `GradientBoostingOrdinal` class.)
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
from ogboost import load_wine_quality
|
|
81
|
+
red_wine, white_wine = load_wine_quality()
|
|
82
|
+
X, y = red_wine.drop(columns="quality"), red_wine["quality"]
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
### Training, Prediction and Evaluation
|
|
86
|
+
```python
|
|
87
|
+
from ogboost import GradientBoostingOrdinal
|
|
88
|
+
|
|
89
|
+
## training ##
|
|
90
|
+
model = GradientBoostingOrdinal(n_estimators=100, link_function='logit', verbose=1)
|
|
91
|
+
model.fit(X, y)
|
|
92
|
+
|
|
93
|
+
## prediction ##
|
|
94
|
+
# class labels
|
|
95
|
+
predicted_labels = model.predict(X)
|
|
96
|
+
# class probabilities
|
|
97
|
+
predicted_probabilities = model.predict_proba(X)
|
|
98
|
+
# latent score
|
|
99
|
+
predicted_latent = model.decision_function(X)
|
|
100
|
+
|
|
101
|
+
# evaluation
|
|
102
|
+
concordance_latent = model.score(X, y) # concordance using latent scores
|
|
103
|
+
concordance_label = model.score(X, y, pred_type = 'labels') # concordance using class labels
|
|
104
|
+
print(f"Concordance - class labels: {concordance_label:.3f}")
|
|
105
|
+
print(f"Concordance - latent scores: {concordance_latent:.3f}")
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
### Early-Stopping using Cross-Validation
|
|
109
|
+
|
|
110
|
+
```python
|
|
111
|
+
from sklearn.model_selection import cross_val_score
|
|
112
|
+
from sklearn.model_selection import RepeatedKFold
|
|
113
|
+
import time
|
|
114
|
+
|
|
115
|
+
n_splits = 10
|
|
116
|
+
n_repeats = 10
|
|
117
|
+
kf = RepeatedKFold(n_splits=n_splits, n_repeats=n_repeats)
|
|
118
|
+
|
|
119
|
+
# early-stopping using a simple holdout set
|
|
120
|
+
model_earlystop_simple = GradientBoostingOrdinal(n_iter_no_change=10, validation_fraction=0.2)
|
|
121
|
+
start = time.time()
|
|
122
|
+
c_index_simple = cross_val_score(model_earlystop_simple, X, y, cv=kf, n_jobs=-1)
|
|
123
|
+
end = time.time()
|
|
124
|
+
print(f'Simple early stopping: {c_index_simple.mean():.3f} ({end - start:.1f} seconds)')
|
|
125
|
+
|
|
126
|
+
# early-stopping using cross-validation
|
|
127
|
+
model_earlystop_cv = GradientBoostingOrdinal(n_iter_no_change=10, cv_early_stopping_splits=5)
|
|
128
|
+
start = time.time()
|
|
129
|
+
c_index_cv = cross_val_score(model_earlystop_cv, X, y, cv=kf, n_jobs=-1)
|
|
130
|
+
end = time.time()
|
|
131
|
+
print(f'CV early stopping: {c_index_cv.mean():.3f} ({end - start:.1f} seconds)')
|
|
132
|
+
|
|
133
|
+
# statistical comparison of the two methods
|
|
134
|
+
from scipy.stats import ttest_rel
|
|
135
|
+
t_stat, p_value = ttest_rel(c_index_simple, c_index_cv)
|
|
136
|
+
print(f't-statistic: {t_stat}, p-value: {p_value}')
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
### Heterogeneous Ensemble
|
|
140
|
+
|
|
141
|
+
Generating a random set of hyperparameter combinations for `DecisionTreeRegressor`:
|
|
142
|
+
|
|
143
|
+
```python
|
|
144
|
+
import numpy as np
|
|
145
|
+
from sklearn.tree import DecisionTreeRegressor
|
|
146
|
+
|
|
147
|
+
# Number of samples to generate
|
|
148
|
+
n_samples = 10
|
|
149
|
+
|
|
150
|
+
max_depth_choices = [3, 6, 9, None]
|
|
151
|
+
max_depths = np.random.choice(max_depth_choices, size=n_samples, replace=True)
|
|
152
|
+
max_leaf_nodes_choices = [10, 20, 30, None]
|
|
153
|
+
max_leaf_nodes = np.random.choice(max_leaf_nodes_choices, size=n_samples, replace=True)
|
|
154
|
+
|
|
155
|
+
params = list(zip(max_depths, max_leaf_nodes))
|
|
156
|
+
|
|
157
|
+
# Create list of DecisionTreeRegressor models with sampled parameters
|
|
158
|
+
models = [
|
|
159
|
+
DecisionTreeRegressor(
|
|
160
|
+
max_depth=max_depth,
|
|
161
|
+
max_leaf_nodes=max_leaf_nodes
|
|
162
|
+
)
|
|
163
|
+
for max_depth, max_leaf_nodes in params
|
|
164
|
+
]
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
Quantifying the performance of the heterogeneous ensemble:
|
|
168
|
+
```python
|
|
169
|
+
learning_rate = 0.1
|
|
170
|
+
|
|
171
|
+
model_heter = GradientBoostingOrdinal(
|
|
172
|
+
base_learner=models,
|
|
173
|
+
n_estimators=n_samples
|
|
174
|
+
)
|
|
175
|
+
cv_heter = cross_val_score(model_heter, X, y, cv=kf, n_jobs=-1)
|
|
176
|
+
print(f'average cv score of heteogeneous ensemble: {np.mean(cv_heter):.3f}')
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
## License
|
|
180
|
+
This package is licensed under the [MIT License](./LICENSE).
|
ogboost-0.5.5/README.md
ADDED
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
# Ordinal Gradient Boosting (`OGBoost`)
|
|
2
|
+
|
|
3
|
+
## Overview
|
|
4
|
+
|
|
5
|
+
`OGBoost` is a scikit-learn-compatible, Python package for gradient boosting tailored to ordinal regression problems. It does so by alternating between:
|
|
6
|
+
1. Fitting a Machine Learning (ML) regression model - such as a decision tree - to predict a latent score that specifies the mean of a probability density function (PDF), and
|
|
7
|
+
1. Fitting a set of thresholds that generate discrete outcomes from the PDF.
|
|
8
|
+
|
|
9
|
+
In other words, `OGBoost` implements coordinate-descent optimization that combines functional gradient descent - for updating the regression function - with ordinary gradient descent - for updating the threshold vector.
|
|
10
|
+
|
|
11
|
+
The main class of the package, `GradientBoostingOrdinal`, is designed to have the same look and feel as `scikit-learn`'s `GradientBoostingClassifier`. It includes many of the same features such as custom link functions, sample weighting, early stopping using a validation set, and staged predictions.
|
|
12
|
+
|
|
13
|
+
There are, however, important differences as well.
|
|
14
|
+
|
|
15
|
+
## Unique Features of `OGBoost`
|
|
16
|
+
|
|
17
|
+
### Latent-Score Prediction
|
|
18
|
+
|
|
19
|
+
The `decision_function` method of the `GradientBoostingOrdinal` behaves differently from `scikit-learn`'s classifiers. Assuming the target variable has `K` distinct classes, a nominal classifier's decision function would return `K` values for each sample. On the other hand, `decision_function` in `ogboost` would return the latent score for each sample, which is a single value. This latent score can be considered a high-resolution alternative to class labels, and thus may have superior ranking performance.
|
|
20
|
+
|
|
21
|
+
### Early Stopping using Cross-Validation (CV)
|
|
22
|
+
|
|
23
|
+
In addition to using a single validation set for early stopping, similar to `GradientBoostingClassifier`, `ogboost` early stopping using CV, which means error/performance over the entire data is used for calculating out-of-sample performance. This can improve the robustness of the early-stopping strategy, especially for small and/or imbalanced datasets.
|
|
24
|
+
|
|
25
|
+
### Heterogeneous Ensemble
|
|
26
|
+
|
|
27
|
+
While most gradient-boosting software packages exclusively use decision trees with a predetermined set of hyperparameters as the base learner in all boosting iterations, `ogboost` offers significantly more flexibility.
|
|
28
|
+
|
|
29
|
+
1. Users can pass in a `base_learner` parameter to the class initializer to override the default choice of a `DecisionTreeRegressor`. This can be any regression algorithm such as a feed-forward neural network (`MLPRegressor`), or a K-nearest-neighbor regressor (`KNeighborsRegressor`), etc.
|
|
30
|
+
1. Rather than a single base learner, users can specify a list of base learners, which will be drawn from in that order in each boosting iteration. This amounts to creating a *heterogeneous* ensemble as opposed to a *homogeneous* ensemble.
|
|
31
|
+
|
|
32
|
+
## Installation
|
|
33
|
+
```bash
|
|
34
|
+
pip install ogboost
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Quick Start
|
|
38
|
+
### Load the Wine Quality Dataset
|
|
39
|
+
The package includes a utility to load the wine quality dataset (red and white) from the UCI repository. Note that `load_wine_quality` shifts the target variable (`quality`) to start from `0`. (This is required by the `GradientBoostingOrdinal` class.)
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
from ogboost import load_wine_quality
|
|
43
|
+
red_wine, white_wine = load_wine_quality()
|
|
44
|
+
X, y = red_wine.drop(columns="quality"), red_wine["quality"]
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
### Training, Prediction and Evaluation
|
|
48
|
+
```python
|
|
49
|
+
from ogboost import GradientBoostingOrdinal
|
|
50
|
+
|
|
51
|
+
## training ##
|
|
52
|
+
model = GradientBoostingOrdinal(n_estimators=100, link_function='logit', verbose=1)
|
|
53
|
+
model.fit(X, y)
|
|
54
|
+
|
|
55
|
+
## prediction ##
|
|
56
|
+
# class labels
|
|
57
|
+
predicted_labels = model.predict(X)
|
|
58
|
+
# class probabilities
|
|
59
|
+
predicted_probabilities = model.predict_proba(X)
|
|
60
|
+
# latent score
|
|
61
|
+
predicted_latent = model.decision_function(X)
|
|
62
|
+
|
|
63
|
+
# evaluation
|
|
64
|
+
concordance_latent = model.score(X, y) # concordance using latent scores
|
|
65
|
+
concordance_label = model.score(X, y, pred_type = 'labels') # concordance using class labels
|
|
66
|
+
print(f"Concordance - class labels: {concordance_label:.3f}")
|
|
67
|
+
print(f"Concordance - latent scores: {concordance_latent:.3f}")
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
### Early-Stopping using Cross-Validation
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from sklearn.model_selection import cross_val_score
|
|
74
|
+
from sklearn.model_selection import RepeatedKFold
|
|
75
|
+
import time
|
|
76
|
+
|
|
77
|
+
n_splits = 10
|
|
78
|
+
n_repeats = 10
|
|
79
|
+
kf = RepeatedKFold(n_splits=n_splits, n_repeats=n_repeats)
|
|
80
|
+
|
|
81
|
+
# early-stopping using a simple holdout set
|
|
82
|
+
model_earlystop_simple = GradientBoostingOrdinal(n_iter_no_change=10, validation_fraction=0.2)
|
|
83
|
+
start = time.time()
|
|
84
|
+
c_index_simple = cross_val_score(model_earlystop_simple, X, y, cv=kf, n_jobs=-1)
|
|
85
|
+
end = time.time()
|
|
86
|
+
print(f'Simple early stopping: {c_index_simple.mean():.3f} ({end - start:.1f} seconds)')
|
|
87
|
+
|
|
88
|
+
# early-stopping using cross-validation
|
|
89
|
+
model_earlystop_cv = GradientBoostingOrdinal(n_iter_no_change=10, cv_early_stopping_splits=5)
|
|
90
|
+
start = time.time()
|
|
91
|
+
c_index_cv = cross_val_score(model_earlystop_cv, X, y, cv=kf, n_jobs=-1)
|
|
92
|
+
end = time.time()
|
|
93
|
+
print(f'CV early stopping: {c_index_cv.mean():.3f} ({end - start:.1f} seconds)')
|
|
94
|
+
|
|
95
|
+
# statistical comparison of the two methods
|
|
96
|
+
from scipy.stats import ttest_rel
|
|
97
|
+
t_stat, p_value = ttest_rel(c_index_simple, c_index_cv)
|
|
98
|
+
print(f't-statistic: {t_stat}, p-value: {p_value}')
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
### Heterogeneous Ensemble
|
|
102
|
+
|
|
103
|
+
Generating a random set of hyperparameter combinations for `DecisionTreeRegressor`:
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
import numpy as np
|
|
107
|
+
from sklearn.tree import DecisionTreeRegressor
|
|
108
|
+
|
|
109
|
+
# Number of samples to generate
|
|
110
|
+
n_samples = 10
|
|
111
|
+
|
|
112
|
+
max_depth_choices = [3, 6, 9, None]
|
|
113
|
+
max_depths = np.random.choice(max_depth_choices, size=n_samples, replace=True)
|
|
114
|
+
max_leaf_nodes_choices = [10, 20, 30, None]
|
|
115
|
+
max_leaf_nodes = np.random.choice(max_leaf_nodes_choices, size=n_samples, replace=True)
|
|
116
|
+
|
|
117
|
+
params = list(zip(max_depths, max_leaf_nodes))
|
|
118
|
+
|
|
119
|
+
# Create list of DecisionTreeRegressor models with sampled parameters
|
|
120
|
+
models = [
|
|
121
|
+
DecisionTreeRegressor(
|
|
122
|
+
max_depth=max_depth,
|
|
123
|
+
max_leaf_nodes=max_leaf_nodes
|
|
124
|
+
)
|
|
125
|
+
for max_depth, max_leaf_nodes in params
|
|
126
|
+
]
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Quantifying the performance of the heterogeneous ensemble:
|
|
130
|
+
```python
|
|
131
|
+
learning_rate = 0.1
|
|
132
|
+
|
|
133
|
+
model_heter = GradientBoostingOrdinal(
|
|
134
|
+
base_learner=models,
|
|
135
|
+
n_estimators=n_samples
|
|
136
|
+
)
|
|
137
|
+
cv_heter = cross_val_score(model_heter, X, y, cv=kf, n_jobs=-1)
|
|
138
|
+
print(f'average cv score of heteogeneous ensemble: {np.mean(cv_heter):.3f}')
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
## License
|
|
142
|
+
This package is licensed under the [MIT License](./LICENSE).
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# ogboost/__init__.py
|
|
2
|
+
|
|
3
|
+
# Re-export key components from submodules
|
|
4
|
+
from .data import load_wine_quality
|
|
5
|
+
from .main import GradientBoostingOrdinal, concordance_index, LinkFunctions
|
|
6
|
+
|
|
7
|
+
# Define the public API
|
|
8
|
+
__all__ = ["load_wine_quality", "GradientBoostingOrdinal", "concordance_index", "LinkFunctions"]
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Module for loading the Wine Quality Dataset.
|
|
3
|
+
|
|
4
|
+
This module provides functionality to download and load the Wine Quality Dataset from the
|
|
5
|
+
UCI Machine Learning Repository. The dataset includes information about physicochemical
|
|
6
|
+
tests (e.g., pH, alcohol content) and quality scores for red and white wines.
|
|
7
|
+
|
|
8
|
+
The module contains the following functionality:
|
|
9
|
+
- Automatically downloads the datasets if not found locally.
|
|
10
|
+
- Loads the red and white wine datasets into pandas DataFrames.
|
|
11
|
+
- Rescales the 'quality' column to start from 0 and ensures it is of integer type.
|
|
12
|
+
|
|
13
|
+
Functions
|
|
14
|
+
---------
|
|
15
|
+
load_wine_quality():
|
|
16
|
+
Loads the wine quality dataset. Automatically downloads the datasets
|
|
17
|
+
if they are not found locally and returns them as pandas DataFrames.
|
|
18
|
+
|
|
19
|
+
_download_wine_quality_datasets(destination: str):
|
|
20
|
+
Downloads the wine quality datasets (red and white) from the UCI repository
|
|
21
|
+
to the specified destination folder.
|
|
22
|
+
|
|
23
|
+
Examples
|
|
24
|
+
--------
|
|
25
|
+
>>> from mymodule import load_wine_quality
|
|
26
|
+
>>> red_wine, white_wine = load_wine_quality()
|
|
27
|
+
>>> print(red_wine.head())
|
|
28
|
+
>>> print(white_wine.head())
|
|
29
|
+
"""
|
|
30
|
+
import os
|
|
31
|
+
import urllib.request
|
|
32
|
+
import pandas as pd
|
|
33
|
+
|
|
34
|
+
def load_wine_quality():
|
|
35
|
+
"""
|
|
36
|
+
Loads the wine quality dataset from the UCI repository.
|
|
37
|
+
If the datasets are not already downloaded,
|
|
38
|
+
they will be downloaded automatically.
|
|
39
|
+
|
|
40
|
+
Returns
|
|
41
|
+
-------
|
|
42
|
+
tuple of pandas.DataFrame
|
|
43
|
+
A tuple containing two DataFrames:
|
|
44
|
+
- The first for red wine data.
|
|
45
|
+
- The second for white wine data.
|
|
46
|
+
"""
|
|
47
|
+
# Define the local paths for the datasets
|
|
48
|
+
dataset_folder = os.path.join(os.path.dirname(__file__), "data")
|
|
49
|
+
red_wine_path = os.path.join(dataset_folder, "winequality-red.csv")
|
|
50
|
+
white_wine_path = os.path.join(dataset_folder, "winequality-white.csv")
|
|
51
|
+
|
|
52
|
+
# Check if the datasets exist locally; if not, download them
|
|
53
|
+
if not os.path.exists(red_wine_path) or not os.path.exists(white_wine_path):
|
|
54
|
+
print("Datasets not found locally. Downloading from UCI repository...")
|
|
55
|
+
os.makedirs(dataset_folder, exist_ok=True)
|
|
56
|
+
_download_wine_quality_datasets(dataset_folder)
|
|
57
|
+
|
|
58
|
+
# Load the datasets into pandas DataFrames
|
|
59
|
+
red_wine_df = pd.read_csv(red_wine_path, sep=";")
|
|
60
|
+
white_wine_df = pd.read_csv(white_wine_path, sep=";")
|
|
61
|
+
|
|
62
|
+
# rescale quality to start from 0
|
|
63
|
+
red_wine_df["quality"] -= red_wine_df["quality"].min()
|
|
64
|
+
white_wine_df["quality"] -= white_wine_df["quality"].min()
|
|
65
|
+
|
|
66
|
+
# ensure data type for response is integer
|
|
67
|
+
red_wine_df["quality"] = red_wine_df["quality"].astype(int)
|
|
68
|
+
white_wine_df["quality"] = white_wine_df["quality"].astype(int)
|
|
69
|
+
|
|
70
|
+
return red_wine_df, white_wine_df
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _download_wine_quality_datasets(destination: str):
|
|
74
|
+
"""
|
|
75
|
+
Downloads the wine quality datasets from the UCI repository.
|
|
76
|
+
|
|
77
|
+
Parameters
|
|
78
|
+
----------
|
|
79
|
+
destination : str
|
|
80
|
+
Directory where the datasets will be saved.
|
|
81
|
+
"""
|
|
82
|
+
# UCI repository URLs
|
|
83
|
+
red_wine_url = (
|
|
84
|
+
"https://archive.ics.uci.edu/ml/machine-learning-databases/"
|
|
85
|
+
"wine-quality/winequality-red.csv"
|
|
86
|
+
)
|
|
87
|
+
white_wine_url = (
|
|
88
|
+
"https://archive.ics.uci.edu/ml/machine-learning-databases/"
|
|
89
|
+
"wine-quality/winequality-white.csv"
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
# Define file paths
|
|
93
|
+
red_wine_path = os.path.join(destination, "winequality-red.csv")
|
|
94
|
+
white_wine_path = os.path.join(destination, "winequality-white.csv")
|
|
95
|
+
|
|
96
|
+
# Download the datasets
|
|
97
|
+
print(f"Downloading red wine dataset to {red_wine_path}...")
|
|
98
|
+
urllib.request.urlretrieve(red_wine_url, red_wine_path)
|
|
99
|
+
print("Red wine dataset downloaded.")
|
|
100
|
+
|
|
101
|
+
print(f"Downloading white wine dataset to {white_wine_path}...")
|
|
102
|
+
urllib.request.urlretrieve(white_wine_url, white_wine_path)
|
|
103
|
+
print("White wine dataset downloaded.")
|